[
 {
  "id": "2608.27186",
  "slug": "task-space-model-based-control-of-pneumatic-soft-actuators",
  "title": "Task-space model-based control of pneumatic soft actuators",
  "abstract": "Soft actuators enable dexterous and compliant interaction, but closed-loop task-space control remains challenging due to strong nonlinearities, distributed deformation, and uncertainty in their dynamics. This paper presents a real-time dynamic-model-based task-space feedback and estimation framework based on a non-minimal coordinate discrete elastic rod model formulated in absolute coordinates with holonomic constraints. The resulting structure preserves distributed mechanics while maintaining computational efficiency through sparse system matrices, enabling real-time control with up to 10 discretized rods. A quasi-static feedforward inverse model is combined with a task-space PI controller and a dynamic observer that fuses measurement residuals as virtual forces, enabling full-state estimation from sparse sensing. The approach is experimentally validated on three planar pneumatic soft actuators with varying geometries. Across five tasks, including drawing the digits 0-9 across the workspace (3-18 mm/s tip speed), tracking periodic motion (up to 37 cm/s), cross-platform generalization, reduced sensing conditions, and real-time user-defined references, our method achieves 1.5-2.3 mm root mean square error (RMSE) for precision motions and 5.5-12.4 mm RMSE at 1-2 Hz. Results demonstrate that structured, non-minimal dynamic models can enable real-time, high-precision, moderate-bandwidth task-space control of planar soft pneumatic actuators in free space.",
  "published": "2026-08-27",
  "updated": "2026-08-27",
  "year": "2026",
  "authors": [
   "Nithin S. Kumar",
   "Joshua Gaston",
   "D. Caleb Rucker",
   "Eric J. Barth"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "11 pages, 14 figures",
  "topics": [
   "dexterous-manipulation",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.27186v1",
  "pdf_url": "https://arxiv.org/pdf/2608.27186v1",
  "html_url": "https://arxiv.org/html/2608.27186v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.27151",
  "slug": "planning-a-shared-modular-fixture-layout-across-robotic-disassembly-st",
  "title": "Planning a Shared Modular Fixture Layout Across Robotic Disassembly Stages",
  "abstract": "Stable support remains challenging in robotic disassembly of irregularly shaped products. As components are progressively removed, the available support surfaces, mass distribution, and task loads change throughout the process. A fixture layout designed for one workpiece state may therefore become infeasible at later stages, motivating unified support planning over the complete disassembly sequence. This paper presents a modular vacuum-based fixturing system that plans one shared support configuration for the complete disassembly sequence of a screwdriver or shaver, allowing each sequence to proceed without fixture reconfiguration. To search the mixed continuous--discrete layout space under repeated cross-stage evaluation, a denoising diffusion probabilistic model generates physics-informed initial configurations that are refined through Bayesian optimization. Robotic screw and component-removal experiments verified the disassembly feasibility of the planned layouts, while 11 directional-load tests quantified their stability. Comparisons between the measured operational loads and directional responses yielded mean empirical stability margins of 66.9% for the screwdriver and 81.6% for the shaver. These results demonstrate that a product-specific shared layout can provide stable support throughout the tested robotic disassembly sequence.",
  "published": "2026-08-27",
  "updated": "2026-08-27",
  "year": "2026",
  "authors": [
   "Haohui Pan",
   "Takuya Kiyokawa",
   "Kensuke Harada"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "15 pages, 9 figures. Submitted to IEEE Transactions on Automation Science and Engineering (T-ASE)",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.27151v1",
  "pdf_url": "https://arxiv.org/pdf/2608.27151v1",
  "html_url": "https://arxiv.org/html/2608.27151v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.27088",
  "slug": "active-sensing-to-characterize-the-heterogeneity-of-plant-stress",
  "title": "Active sensing to characterize the heterogeneity of plant stress",
  "abstract": "While most phenotyping platforms rely primarily on image-based measurements, advanced plant characterization requires the integration of active physiological sensing modali- ties such as chlorophyll fluorescence. We present an autonomous robotic platform designed to perform targeted fluorescence measurements on plant leaves. The system combines 3D plant reconstruction, geometric analysis, and motion planning to localize suitable measurement points and generate collision-free trajectories for a robotic manipulator. A dense 3D model of the plant is reconstructed from multi-view data and used to extract candidate leaf surfaces based on orientation, accessibility, and sensing constraints. These targets are then integrated into a task-level planning framework that guides the end-effector to precise contact or near-contact configurations required for point-based fluorescence acquisition. The platform enables automated, repeatable, and spatially resolved physiological measurements that go beyond passive imaging. By tightly coupling perception, geometric reasoning, and manipulation, the proposed system provides a robotics-driven approach to high-resolution plant phenotyping and opens new directions for autonomous agricultural inspection and plant-aware manipulation.",
  "published": "2026-08-27",
  "updated": "2026-08-27",
  "year": "2026",
  "authors": [
   "Ayman Laaroussi",
   "Peter Hanappe",
   "David Colliaux"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "UR2026",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.27088v1",
  "pdf_url": "https://arxiv.org/pdf/2608.27088v1",
  "html_url": "https://arxiv.org/html/2608.27088v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.27085",
  "slug": "pass-the-bucket-efficient-robust-local-load-balancing-for-teams-of-het",
  "title": "Pass the Bucket: Efficient, Robust, Local Load Balancing for Teams of Heterogeneous Robots",
  "abstract": "We study the problem of decentralized, self-organized task sharing for a swarm of heterogeneous robots that collaborate in transportation or other objectives that require coordinated motion planning. To this end, we present theoretical and practical results for the simple but effective mechanism of \\emph{bucket brigades} for load balancing, in which a team of heterogenous robots share a spatial task in a confined, one-dimensional space, while only being able to sense collisions with neighbors or walls. The goal is to optimize throughput of the overall system, without central control or information, aiming at an interval partition proportional to robot velocities. We address possible chaotic system behavior by developing a stabilization mechanism based on simple local aid, a ``token'', that temporarily decelerates robots after an encounter. This purely local change eliminates persistent oscillations, resulting in convergence towards a stable system state. We accelerate system convergence by comparing a single boundary token to ubiquitous two-directional tokens and optimizing the deceleration factor. Event-driven simulations report convergence times and robustness: For a large variety of perturbations (such as robot deletion, position or velocity jittering), the system reliably re-converges. The results suggest a local, practical mechanism for robust load balancing for heterogeneous teams of robots that promises an effective tool as basis for more complex scenarios.",
  "published": "2026-08-27",
  "updated": "2026-08-27",
  "year": "2026",
  "authors": [
   "Tobias Wallner",
   "Dominik Krupke",
   "Arne Schmidt",
   "S\u00e1ndor P. Fekete"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "This paper was submitted to IROS 2026 on March 2nd and accepted on June 17th",
  "topics": [
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.27085v1",
  "pdf_url": "https://arxiv.org/pdf/2608.27085v1",
  "html_url": "https://arxiv.org/html/2608.27085v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.27079",
  "slug": "graft-grounded-and-efficient-online-reinforcement-adaptation-for-fine",
  "title": "GRAFT: Grounded and Efficient Online Reinforcement Adaptation for Fine-Grained Robot Manipulation",
  "abstract": "Pretrained vision-language-action (VLA) policies provide strong priors for robot manipulation, yet adapting them online to fine-grained biomedical tasks remains challenging. Task success often hinges on subtle, view-dependent visual cues, while task-level rewards provide little guidance about which regions matter, making it difficult to learn task-relevant visual grounding from limited real-robot interaction. Online adaptation is further constrained by the computational cost of VLA inference and replay-based updates. We introduce GRAFT (Grounded Reinforcement Adaptation for Fast Task Learning), a framework for efficient online VLA adaptation through grounded perception. GRAFT uses region-level supervision to learn view-specific visual anchors that focus perception on task-relevant local cues without requiring region proposals at deployment. It further combines single-step action generation with cached visual-language prefix reuse to accelerate online learning. Across four biomedical manipulation tasks, GRAFT improves success rates by 25 percentage points under matched adaptation budgets, while reducing the computational overhead of online policy updates.",
  "published": "2026-08-27",
  "updated": "2026-08-27",
  "year": "2026",
  "authors": [
   "Yibo Qiu",
   "Haoliang Ye",
   "Shu'ang Sun",
   "Zan Huang",
   "Ronald X Xu",
   "Mingzhai Sun"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "",
  "topics": [
   "vla"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.27079v1",
  "pdf_url": "https://arxiv.org/pdf/2608.27079v1",
  "html_url": "https://arxiv.org/html/2608.27079v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.27073",
  "slug": "spatialcrafter-single-image-world-modeling-with-generative-3d-proxies",
  "title": "SpatialCrafter: Single Image World Modeling with Generative 3D Proxies",
  "abstract": "Explorable image-to-scene generation is essential for applications in gaming, robotics, and virtual reality. Existing methods based on video diffusion model (VDM) commonly rely on incomplete conditioning signals such as sparse point clouds or 2D panoramas, leading to stochastic hallucinations, long-term drifts and suboptimal 3D consistency. We present SpatialCrafter, a novel two-stage framework that addresses these issues by introducing a global 3D proxy for high-fidelity image-to-scene generation. Specifically, we decompose the generation process into global proxy generation and appearance refinement. For proxy generation, we propose a Point-anchored Sparse Structure~(PaSS) Flow module that predicts a spatially aligned and geometrically consistent 3D proxy. For appearance refinement, we re-frame the VDM as a Generative Deferred Refiner which synthesizes high-frequency photorealistic details upon proxy-defined scene geometry. To better integrate the proxy with the pre-trained VDM, we introduce Parallel Geometry Injection and Proxy-Aware Corruption training strategies, which improve robustness to proxy artifacts without disrupting the pretrained generative manifold. Furthermore, as no suitable dataset exists for this explorable scene generation task, we construct a new large-scale dataset of 115K scenes. To the best of our knowledge, it is the first hybrid dataset for image-to-scene generation. Extensive experiments on both synthetic and real-world datasets show that SpatialCrafter outperforms state-of-the-art methods, mitigates long-term drift, and remains robust and consistent under rapid camera motion and extreme viewpoint changes. Code, models, and the newly constructed dataset will be publicly released. See more at https://fangchuan.github.io/SpatialCrafter/.",
  "published": "2026-08-27",
  "updated": "2026-08-27",
  "year": "2026",
  "authors": [
   "Chuan Fang",
   "Lingteng Qiu",
   "Yixun Liang",
   "Rui Chen",
   "Kunming Luo",
   "Zhaohua Zheng",
   "Tongyuan Bai",
   "Feipeng Tian",
   "Zilong Dong",
   "Zihan Zhou",
   "Ping Tan"
  ],
  "author_count": 11,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "12 pages",
  "topics": [
   "world-models",
   "spatial-3d",
   "video-generation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.27073v1",
  "pdf_url": "https://arxiv.org/pdf/2608.27073v1",
  "html_url": "https://arxiv.org/html/2608.27073v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.27033",
  "slug": "riemann-1-0-an-embodied-world-action-model-for-physical-ai",
  "title": "Riemann-1.0: An Embodied World Action Model for Physical AI",
  "abstract": "We introduce Riemann-1.0, a fully causal autoregressive World Action Model for embodied intelligence. Riemann-1.0 jointly models multi-view visual observations, robot states, and embodiment-specific actions within a unified causal autoregressive sequence, representing robot actions and world evolution as causal state transitions. Unlike existing WAMs based on joint generation, video-first prediction, or decoupled modeling paradigms, Riemann-1.0 unifies online robot policy execution and action-conditioned world simulation within a single model, enabling it to function as both an executable robot policy and a multi-embodiment visual world simulator. To scale embodied experience across heterogeneous data sources, we further develop a progressive embodied pretraining framework that unifies learning from egocentric human videos, handheld-gripper demonstrations, and heterogeneous robot trajectories under a shared World Action Modeling objective. Built upon 200K+ hours of interaction data, Riemann-1.0 progressively transfers large-scale embodied experience into executable robot manipulation capabilities. Riemann-1.0 achieves state-of-the-art performance across both simulation benchmarks and real-world manipulation tasks. It achieves success rates of 94.3% on RoboTwin2.0, 99.0% on LIBERO, and 62.6% on the long-horizon compositional benchmark RoboCasa-365, outperforming the previous best method by 8.4% On long-horizon real-world manipulation tasks, Riemann-1.0 achieves a Success Rate (SR) of 85.0% and a Progress Success Rate (PSR) of 94.4%, exceeding the strongest open-source baseline by 15% in SR. These results demonstrate that unified World Action Modeling together with progressive embodied pretraining effectively transforms large-scale embodied experience into generalizable robot manipulation capabilities.",
  "published": "2026-08-27",
  "updated": "2026-08-27",
  "year": "2026",
  "authors": [
   "Haofeng Sun",
   "Jiangbo Pei",
   "Fei Kang",
   "Zexiang Liu",
   "Yaokun Li",
   "Boyi Jiang",
   "Hua Xue",
   "Cindy Zhou",
   "Wei Li",
   "Yichen Wei",
   "Mengyin An",
   "Fanliang Zhao",
   "Biao Jiang",
   "Zile Wang",
   "Yang Liu",
   "Yangguang Li"
  ],
  "author_count": 16,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "",
  "topics": [
   "egocentric-data",
   "sim2real",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.27033v1",
  "pdf_url": "https://arxiv.org/pdf/2608.27033v1",
  "html_url": "https://arxiv.org/html/2608.27033v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.27000",
  "slug": "arbitrary-order-hermite-interpolation-of-rigid-motion-jets-via-hyper-m",
  "title": "Arbitrary-Order Hermite Interpolation of Rigid-Motion Jets via Hyper-Multidual Quaternions",
  "abstract": "We study bilateral interpolation of finite-order rigid-motion jets represented by unit dual quaternions. An order-$n$ multidual (MD) algebra is the truncated polynomial algebra $\\mathbb{R}[\\varepsilon]/(\\varepsilon^{n+1})$; hyper-multidual (HMD) quaternions are dual quaternions with coefficients in this algebra. Temporal HMD transforms encode a pose and its derivatives, whereas a generic HMD curve need not be the temporal jet of its pose projection; we call this requirement holonomicity. We show that a temporal transform and its relative descriptor are unitary and derive recursive coefficient constraints, together with a local realizability converse in an admissible logarithm chart. We then extend screw linear interpolation (ScLERP) algebraically to unit HMD quaternions. Although it matches complete endpoint transforms, direct HMD--ScLERP is generically non-holonomic for arbitrary endpoint jets. We give a coefficient criterion and explicit endpoint and first-order interior contact defects. A holonomic alternative is obtained by mapping endpoint transforms to logarithmic dual-quaternion coordinates, applying the degree-$(2n+1)$ Hermite polynomial that matches derivatives through order $n$, and lifting by the exponential. HMD arithmetic also recovers higher-order rigid-motion acceleration fields without explicit differentiation of $\\mathrm{dexp}$. Rotation and full $\\mathrm{SE}(3)$ tests through second order, with an additional third-order polynomial check, reproduce the stated defects and endpoint jets.",
  "published": "2026-08-27",
  "updated": "2026-08-27",
  "year": "2026",
  "authors": [
   "Daniel Condurache"
  ],
  "author_count": 1,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "21 pages, 1 figure",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.27000v1",
  "pdf_url": "https://arxiv.org/pdf/2608.27000v1",
  "html_url": "https://arxiv.org/html/2608.27000v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.26947",
  "slug": "4dsynth-controllable-procedural-world-synthesis-for-dynamic-embodied-s",
  "title": "4DSynth: Controllable Procedural World Synthesis for Dynamic Embodied Simulation",
  "abstract": "Embodied agents need environments that are visually diverse, physically interactive, and changing over time. Procedural simulators can generate large interactive scene collections, and recent 4D generators produce compelling visual dynamics. Combining these properties in one environment, however, still demands extensive manual effort, and the result is rarely editable or controllable enough to reuse at scale. We present 4DSynth, a controllable procedural system that turns a natural-language description, a blueprint mask, or a single photograph into an editable 4D environment with explicit geometry, animated actors, collision-free trajectories, and physics-ready simulation state. Multiple scene routes share one geometry-grounded representation, so the same pipeline handles animation, camera planning, rendering, and task generation. To validate the full pipeline, we construct 4DSynth-Nav, an interactive navigation benchmark generated entirely from 4DSynth's procedural scenes. Two vision-language models evaluated across three difficulty tiers both fail the majority of tasks and stall after early subtasks. The same procedural controllability that produces these environments also makes each failure reproducible and each difficulty axis independently tunable. This paper presents both a controllable generation pipeline and the scalable benchmark it enables, offering a practical foundation for developing and evaluating embodied agents.",
  "published": "2026-08-27",
  "updated": "2026-08-27",
  "year": "2026",
  "authors": [
   "Zehao Qi",
   "Haochen Luo",
   "Jia-Wang Bian",
   "Zeyu Ma",
   "Shuyang Sun"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "",
  "topics": [
   "sim2real",
   "navigation",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.26947v1",
  "pdf_url": "https://arxiv.org/pdf/2608.26947v1",
  "html_url": "https://arxiv.org/html/2608.26947v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.26939",
  "slug": "dynamic-haven-selection-for-multi-agent-pickup-and-delivery-in-constra",
  "title": "Dynamic Haven Selection for Multi-Agent Pickup and Delivery in Constrained Warehouses",
  "abstract": "Space-efficient warehouse layouts often contain single-agent-width aisles and dead-end workstations where robots have few places to wait without blocking others. In Multi-Agent Pickup and Delivery (MAPD) on such constrained layouts, robots must accept online pickup-delivery tasks while preserving protected waiting locations called Havens. The Safe HAven Retreat Planner (SHARP) introduced a mechanism that extends each committed task path with a validated retreat to the agent's dedicated initial Haven, but fixed-Haven commitments can send agents toward distant Havens after deliveries. We present A-sharp (Adaptive SHARP), which changes an agent's retreat target at task assignment time. A naive switch can cause two agents to rely on the same waiting location or let another committed path pass through a location that is still occupied or reserved. A-sharp prevents these failures with an availability test for candidate Havens and a pending-release rule that keeps the previous Haven protected until the agent departs. Under explicit Haven-structure and Safe Interval Path Planning (SIPP) assumptions, we prove invariant preservation and finite-release completeness: every task in any finite release sequence is delivered in finite time. Across 72,000 runs on 14,400 paired map-agent-count-rate-seed cases over four maps, both SHARP and A-sharp complete their respective 14,400 runs. For makespan (final delivery time), a prespecified paired comparison with Holm correction over all 138 configurations with more Havens than agents finds A-sharp significantly better in 107 configurations and never significantly worse than SHARP; on the tested tree map, the median reduction is 16.7%.",
  "published": "2026-08-27",
  "updated": "2026-08-27",
  "year": "2026",
  "authors": [
   "Taisei Hirayama",
   "Kohei Yoshida",
   "Hiroki Sakaji",
   "Itsuki Noda"
  ],
  "author_count": 4,
  "categories": [
   "cs.MA",
   "cs.RO"
  ],
  "primary_category": "cs.MA",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "19 pages, 4 figures, and 3 tables. Accepted at the Joint Workshop on Planning for Complex Real-World Applications (CAIPI) and Bridging the Gap Between AI Planning and (Reinforcement) Learning (PRL), co-located with IJCAI-ECAI 2026",
  "topics": [
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.26939v1",
  "pdf_url": "https://arxiv.org/pdf/2608.26939v1",
  "html_url": "https://arxiv.org/html/2608.26939v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.26932",
  "slug": "contact-aided-factor-graph-localization-for-underwater-sampling",
  "title": "Contact-Aided Factor-Graph Localization for Underwater Sampling",
  "abstract": "Accurate state estimation for autonomous underwater vehicles performing close-range seafloor sampling remains challenging. In low-altitude operation, down-looking cameras over featureless planar seabeds produce scale ambiguity, lateral degeneracy, and inconsistent feature tracking. Meanwhile, inertial-Doppler Velocity Log (DVL) fusion alone provides no mechanism for structural drift correction. We propose a Contact-Aided Factor-Graph Localization framework that treats physical interaction as an informative geometric constraint within a smoothing-based localization formulation. The method tightly fuses suction-based manipulator contact events with adaptive visual odometry, learned object detections, and on-board sensors. Visual odometry relative-pose factors and landmark bearing-range factors are uncertainty-scaled according to inlier statistics to prevent visually weak frames from destabilizing the estimator, while contact events are modeled as high-confidence factors that induce implicit loop closures without appearance-based place recognition. Furthermore, the system can fully initialize online during motion. Experimental evaluation in tanks, harbor, and simulation environments demonstrates that contact-induced constraints significantly reduce trajectory drift and improve object revisit accuracy compared to filtering-based navigation and contact-free graph formulations. These results highlight the role of embodied physical interaction as a localization primitive in perception-degraded underwater environments",
  "published": "2026-08-27",
  "updated": "2026-08-27",
  "year": "2026",
  "authors": [
   "Michele Grimaldi",
   "Yosaku Maeda",
   "Hitoshi Kakami",
   "Ignacio Carlucho",
   "Yvan R. Petillot",
   "Tomoya Inoue"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "",
  "topics": [
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.26932v1",
  "pdf_url": "https://arxiv.org/pdf/2608.26932v1",
  "html_url": "https://arxiv.org/html/2608.26932v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.26888",
  "slug": "beyond-shallow-water-photorealism-physically-and-sensor-grounded-simul",
  "title": "Beyond Shallow-Water Photorealism: Physically and Sensor-Grounded Simulation for Deep-Sea Robotics",
  "abstract": "Many recent underwater simulators emphasize visual realism at the expense of physical fidelity, focusing on shallow-water effects with limited relevance in deep-water environments and high computational cost. In this work, we shift the focus toward deep-sea physical and sensor realism. We present a physics- and sensor-grounded extension of the Stonefish simulator that augments its hydrodynamic models with stochastic IMU and DVL drift, magnetometer disturbances, higher-order hydrodynamics, terramechanics, pressure-driven environmental variability, and physically based underwater optics. These additions are designed to better capture the forces and measurements shaping the behavior of deep-ocean AUVs, ROVs, landers, ASVs, and gliders, while remaining compatible with real-time simulation. This work advances underwater simulation toward more representative deep-sea operating conditions, which is particularly relevant for long-duration navigation and learning-based autonomy, where inaccurate sensor and environmental models introduce non-physical artifacts and overly optimistic performance. While challenges remain, including complex fluid-structure interactions and full environmental stochasticity, the proposed framework provides a practical foundation for navigation, perception, and autonomy research under deep-sea conditions.",
  "published": "2026-08-27",
  "updated": "2026-08-27",
  "year": "2026",
  "authors": [
   "Michele Grimaldi",
   "Enrico Di Maria",
   "Ignacio Carlucho",
   "Yvan R. Petillot"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "",
  "topics": [
   "sim2real",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.26888v1",
  "pdf_url": "https://arxiv.org/pdf/2608.26888v1",
  "html_url": "https://arxiv.org/html/2608.26888v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.26883",
  "slug": "active-surface-driven-reconfigurable-gripper-robust-grasping-and-seque",
  "title": "Active Surface-Driven Reconfigurable Gripper: Robust Grasping and Sequential Manipulation of Thin Objects",
  "abstract": "Robotic grippers face substantial challenges in grasping and manipulating thin objects. Most existing grippers rely on highly precise approach and grasp motions, which limits robustness and reduces applicability. This paper explores thin-object grasping using books as a representative example. Here, we propose a novel solution that integrates an active surface with underactuated compliance to achieve stable grasping of thin objects without complex control. First, an underactuated gripper with an active surface is designed. The active-surface thumb performs in-hand repositioning of the target book without requiring adjustments of the robot arm or the other fingers, while the underactuated fingers establish compliant contact conditions with the environment, and the reconfigurable structure enables reliable grasping of books under different configurations. Second, we establish a kinematic model of the gripper, and determine the initial grasp postures for two representative scenarios (books lying flat on a desktop and books vertically packed in a shelf). Third, by analyzing the physical model of a book lying on a table and its interaction with the gripper and the environment, we systematically optimize the structural parameters and grasping strategy. Finally, extensive experiments validate the effectiveness of the proposed gripper and strategy. The results demonstrate strong robustness and adaptability when grasping thin objects placed flat (including books, paper, fabric, plastic film, and mouse pad), as well as a high success rate when grasping vertically packed books. Moreover, the proposed gripper can reliably complete long sequential \"grasp-place\" tasks.",
  "published": "2026-08-27",
  "updated": "2026-08-27",
  "year": "2026",
  "authors": [
   "Ziyi Zheng",
   "Keqi Zhu",
   "Hao Wu",
   "Yanzhe Wang",
   "Huixu Dong"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "Accepted by RSS2026",
  "topics": [
   "dexterous-manipulation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.26883v1",
  "pdf_url": "https://arxiv.org/pdf/2608.26883v1",
  "html_url": "https://arxiv.org/html/2608.26883v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.26868",
  "slug": "cgs-slam-collaborative-gaussian-splatting-based-slam-for-multi-agent-r",
  "title": "CGS-SLAM: Collaborative Gaussian Splatting based SLAM for Multi-Agent Reconstruction",
  "abstract": "Recent advances in SLAM have leveraged 3DGS for photorealistic reconstruction and novel view synthesis. However, most methods rely on RGB-D input, which is unavailable on consumer-grade smartphones, and few integrate 3DGS within a collaborative framework. Therefore, we present CGS-SLAM, a hybrid decentralized/centralized system enabling multi-agent 3DGS SLAM using only RGB and inertial data. Each agent performs local tracking with inertial data as a motion prior and reconstructs a scaled map using a metric monocular depth estimator (Depth Pro). Keyframe encodings are shared among agents, enabling dynamic keyframing in regions of spatial overlaps with other agents, enhancing submap alignment. Afterwards, a central server aligns submaps using VGGT as a view alignment model. This bidirectional communication keeps communication cost low during mapping and global reconstruction in difficult GNSS-denied environments. Experiments on multiple datasets demonstrate competitive tracking performance, improved rendering quality over state-of-the-art methods, and accurate submap alignment.",
  "published": "2026-08-27",
  "updated": "2026-08-27",
  "year": "2026",
  "authors": [
   "Jean-Daniel de Ambrogi",
   "Aladine Chetouani",
   "Vincent Nguyen",
   "Aur\u00e9lien Chateigner"
  ],
  "author_count": 4,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "",
  "topics": [
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.26868v1",
  "pdf_url": "https://arxiv.org/pdf/2608.26868v1",
  "html_url": "https://arxiv.org/html/2608.26868v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.26821",
  "slug": "temporalflow-vla-learning-physically-grounded-execution-history-for-lo",
  "title": "TemporalFlow-VLA: Learning Physically Grounded Execution History for Long-Horizon Robot Manipulation",
  "abstract": "Vision-language-action (VLA) models leverage pretrained vision-language representations for robot control, yet simply adding historical frames does not reliably capture recent physical change. This is especially problematic in multi-stage manipulation, where visually similar states may require different actions depending on prior execution. To address this challenge, we present TemporalFlow-VLA, which learns compact execution history through physically grounded temporal supervision. Using recorded robot states, robot geometry, and calibrated cameras, we construct robot-surface temporal flow as a training-only target and supervise two execution-aligned temporal queries that provide structured history to the action expert. The geometric supervision path is not evaluated at deployment. TemporalFlow-VLA achieves 97.63 +/- 0.26% average success on LIBERO, including 96.60 +/- 0.87% on LIBERO Long, and 85.5%/84.2% Clean/Randomized success across 12 RoboTwin tasks. It shows its clearest advantage over prior methods on longer-horizon, multi-stage manipulation. Controlled history interventions show that action prediction depends on both historical content and temporal order. With asynchronous feature caching, temporal conditioning maintains single-frame-level server-side sampling latency without additional historical-encoding overhead. Overall, TemporalFlow-VLA provides a compact, physically grounded interface for exploiting ordered execution history without explicit motion estimation or geometric processing at deployment.",
  "published": "2026-08-27",
  "updated": "2026-08-27",
  "year": "2026",
  "authors": [
   "Jiarui Yang",
   "Yehao Lu",
   "Yuning Su",
   "Yu Zhong",
   "Yufeng Xie",
   "Yazhou Zhang",
   "Haiyu Lan",
   "Kaixiang Lu",
   "Peiwen Lin",
   "Chuang Wang",
   "Junwei Liang",
   "Enyu Li"
  ],
  "author_count": 12,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "",
  "topics": [
   "vla"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.26821v1",
  "pdf_url": "https://arxiv.org/pdf/2608.26821v1",
  "html_url": "https://arxiv.org/html/2608.26821v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.26819",
  "slug": "clipper-replayable-shortlisted-optimization-for-repeated-spatial-cover",
  "title": "CLIPPER: Replayable Shortlisted Optimization for Repeated Spatial Coverage Planning",
  "abstract": "Operational requirements developed with the City of Braunschweig frame municipal micromobility planning under geofenced exclusions, mandatory retained sites, spacing rules, and area-level caps. Each policy edit requires a new feasible plan; full-set greedy takes tens of seconds per alternative at city scale. We present CLIPPER (Constraint-exact Low-latency Iterative Planning with Pooled Evaluation and Replay). It forms bounded candidate pools but recomputes exact current gains and checks every active constraint before selection. Coverage from each candidate alone sets the initial order. Offline full-set scans measure gains omitted by the pool; online, a conservative bound triggers expansion or audit. CLIPPER-F gives each proposal group the same number of candidate slots. Across Braunschweig, Munich, and Berlin, its mean coverage over complete chains stays within 0.245 percentage points of full-set greedy under the same policy, with 13.6--28.9 times lower mean rollout time. CLIPPER-A instead distributes one shared candidate budget across the groups. Under its coverage-prioritized policy, it uses 9--15% of full-set greedy's rollout time under the same policy, with mean gaps of 1.82 percentage points in Braunschweig, 0.12 in Munich, and 0.27 in Berlin. Together, CLIPPER enables rapid, replayable comparison of recorded city-scale planning states while enforcing every encoded model constraint.",
  "published": "2026-08-27",
  "updated": "2026-08-27",
  "year": "2026",
  "authors": [
   "Julian Teusch",
   "J\u00f6rg Philipp M\u00fcller",
   "Monika Sester"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.CG",
   "cs.DS"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "Accepted at ACM SIGSPATIAL 2026. 6 pages, 2 figures, 2 tables",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.26819v1",
  "pdf_url": "https://arxiv.org/pdf/2608.26819v1",
  "html_url": "https://arxiv.org/html/2608.26819v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.26800",
  "slug": "rapid-on-robot-learning-for-dynamic-manipulation-skills-robot-juggling",
  "title": "Rapid On-Robot Learning for Dynamic Manipulation Skills: Robot Juggling",
  "abstract": "We present an online learning framework that enables a bimanual robot to acquire diverse juggling patterns directly on physical hardware within minutes, even with a significant sim2real gap. One of the most important lessons from this work is that a model, even when far from reality, can be extremely useful for learning. This motivates a central philosophy of our approach: learning should build upon the robot's current knowledge rather than replace it. Our regularized memory-based learning puts this principle into practice by learning a local model from accumulated experience while retaining the global prior model to extrapolate where experience is sparse. This enables efficient and stable online learning from each new experience without resorting to uninformed exploration over a vast space of possible behaviors. Equally important to continual on-robot learning is safety, allowing the robot to repeatedly practice and improve in the real world. We construct a mutually reachable set that allows safe transitions between successive throws and catches, without driving either arm into a state from which its next action would require violating the robot's joint or actuator limits. Together, these ideas enable a bimanual robot with multi-fingered hands and onboard vision to safely learn and compose five canonical three-ball juggling patterns, including cascade, tennis, half-shower, shower, and box, within less than 5 minutes of real-world interaction. More broadly, this work points toward robots that build upon imperfect prior knowledge and continually refine their behavior through their own real-world experience.",
  "published": "2026-08-27",
  "updated": "2026-08-27",
  "year": "2026",
  "authors": [
   "Taeyoon Lee",
   "Chunpeng Wang",
   "Christopher G. Atkeson",
   "Alfred A. Rizzi",
   "Nicolas Rojas"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "sim2real",
   "navigation",
   "hardware-codesign",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.26800v1",
  "pdf_url": "https://arxiv.org/pdf/2608.26800v1",
  "html_url": "https://arxiv.org/html/2608.26800v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.26789",
  "slug": "online-joint-calibration-of-steering-offset-and-planar-lidar-extrinsic",
  "title": "Online Joint Calibration of Steering Offset and Planar LiDAR Extrinsics for Wheeled Mobile Robots",
  "abstract": "Accurate steering sensing and LiDAR-to-vehicle extrinsics are crucial for reliable path tracking in warehouse mobile robots (WMRs); miscalibration often leads to snaking, weaving, and elevated cross-track error (CTE). In practice, steering ``zero'' is commonly set manually (e.g., eyeballing straightness via a PS4 joystick), while LiDAR extrinsics are assumed from CAD and may drift after maintenance. Such static, manual procedures frequently cause miscalibration in safety-critical environments. This paper presents an Extended Kalman Filter (EKF)--based method for online estimation of steering offset and planar LiDAR extrinsics within a bicycle-kinematics model, providing a principled alternative to manual calibration. Experiments on real datasets show that correcting steering offset reduces CTE substantially, validating the effectiveness of the proposed approach.",
  "published": "2026-08-27",
  "updated": "2026-08-27",
  "year": "2026",
  "authors": [
   "Subodh Mishra",
   "Arindam Dhar",
   "Suprotim Majumdar",
   "Naveen Arulselvan"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "",
  "topics": [
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.26789v1",
  "pdf_url": "https://arxiv.org/pdf/2608.26789v1",
  "html_url": "https://arxiv.org/html/2608.26789v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.26788",
  "slug": "decoupling-planning-and-control-for-instructable-agents",
  "title": "Decoupling Planning and Control for Instructable Agents",
  "abstract": "Recent work shows that pre-trained, instruction-tuned vision-language models (VLMs) perform well at mapping from instructions and observations to high-level plans, but struggle to realize such plans as reliable low-latency action sequences in unfamiliar environments. At the same time, world-model controllers excel at fast observation-to-action control, but lack open-ended task guidance. In this work, we combine these strengths into a single system, Instruct-to-Act, where we train a world-model controller to act autonomously at high frequency when conditioned on sparse, higher-latency, and high-level text instructions generated by a VLM planner. To train controllers to be language-instructable, we relabel segments of controller policy rollouts with synthetic instructions and jointly optimize a behavior-cloning objective along with existing reward-maximizing and world-modeling objectives. We evaluate our proposed approach across seven embodied environments, including three multi-agent environments where VLM planners coordinate through language while trained controllers serve as their actuators. Under matched observation and action spaces, our decoupled approach consistently outperforms controller-only and direct VLM action-generation variants, preserves fast control, and lets us swap in different pretrained VLM planners without fine-tuning, while remaining competitive with strong vision-language-action and multi-agent RL baselines on six of seven tasks.",
  "published": "2026-08-27",
  "updated": "2026-08-27",
  "year": "2026",
  "authors": [
   "Zineng Tang",
   "Kelsey R. Allen",
   "Sjoerd van Steenkiste",
   "Ishita Dasgupta",
   "Alane Suhr"
  ],
  "author_count": 5,
  "categories": [
   "cs.AI",
   "cs.CL",
   "cs.MA",
   "cs.RO"
  ],
  "primary_category": "cs.AI",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "Published as a conference paper at COLM 2026. Project page: https://zinengtang.github.io/instruct-to-act/",
  "topics": [
   "world-models",
   "vla",
   "hardware-codesign",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.26788v1",
  "pdf_url": "https://arxiv.org/pdf/2608.26788v1",
  "html_url": "https://arxiv.org/html/2608.26788v1",
  "code_url": "https://zinengtang.github.io/instruct-to-act/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2608.26766",
  "slug": "meshpriordit-hierarchical-modeling-for-action-conditioned-cloth-dynami",
  "title": "MeshPriorDiT: Hierarchical Modeling for Action-Conditioned Cloth Dynamics",
  "abstract": "Action-conditioned cloth dynamics prediction requires both locally plausible deformation and long-range coordination. Existing approaches largely follow two paradigms. Mesh-based GNNs capture local physical responses through material connectivity. However, their finite message-passing range limits coordination between topologically distant regions, while autoregressive rollouts tend to accumulate prediction errors. Transformer-based dynamics models capture long-range interactions through global attention, but often operate without explicit material connectivity and must learn local topological responses directly from data. We propose MeshPriorDiT, a hierarchical dynamics model that decomposes future cloth motion into a structured mesh prior and a generative residual. An action-conditioned mesh GNN first predicts multi-step vertex displacements, yielding a reference trajectory that respects material topology and grasp constraints. Conditioned on historical states, planned actions, and the mesh prior, a Residual DiT then uses conditional flow matching to jointly generate the residual motion not captured by the prior. The generated residual is further rescaled and decoded using material adjacency to coordinate corrections across neighboring vertices. We evaluate MeshPriorDiT on 15-step autoregressive rollouts across three cloth manipulation tasks. Averaged over the three tasks, MeshPriorDiT reduces average Global MSE by 43.42% relative to the GNN-Only baseline and by 75.03% relative to the DiT-DDPM baseline, while maintaining a favorable Edge-strain MSE comparable to that of GNN-Only.",
  "published": "2026-08-27",
  "updated": "2026-08-27",
  "year": "2026",
  "authors": [
   "Zihang Wang",
   "Jianming Hu",
   "Shang Su",
   "Hao Huang",
   "Mengkai Shi",
   "Jun Gao",
   "Shuo Feng"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "",
  "topics": [
   "dexterous-manipulation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.26766v1",
  "pdf_url": "https://arxiv.org/pdf/2608.26766v1",
  "html_url": "https://arxiv.org/html/2608.26766v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.26759",
  "slug": "fixed-haven-reservation-for-online-multi-agent-pickup-and-delivery-in",
  "title": "Fixed-Haven Reservation for Online Multi-Agent Pickup and Delivery in Dense Warehouses",
  "abstract": "Dense warehouses often contain single-lane aisles, dead ends, and tree-like guidepaths that leave little room for idle agents to wait without blocking others. Existing Multi-Agent Pickup and Delivery (MAPD) guarantees for completing all finitely released tasks typically rely on extra waiting endpoints that planned paths can avoid, or on biconnected topology; these assumptions may fail in such layouts. We study fixed-Haven reservation for online MAPD, where pickup-delivery tasks are released over time. Each agent owns a fixed Safe Haven (Haven for short), usually its start cell, that only the owner may occupy and that other agents treat as blocked. For finite task releases, we prove that this fixed-Haven contract completes all released tasks under Haven-Reachability and explicit planning/progress assumptions. We implement the contract in SHARP, a Safe-Haven Retreat Planner that keeps every busy or retreating agent on a collision-free reserved route ending at its Haven. We compare SHARP with representative TP and PIBT-family MAPD baselines: Token Passing (TP), Priority Inheritance with Backtracking (PIBT), and PIBT with Temporary Priority and Temporary Avoidance (PIBTTP-TA) for biconnected main areas with attached trees. In the robustness sweep, SHARP is the only method with 100% success on all tested configurations, at substantially higher centralized planning cost on tree-like layouts. A TP-style fixed-home-return counterfactual with full-route validation also recovers robustness on tested tree-like layouts, suggesting that fixed return is a central robustness mechanism there. A no-overwrite variant shows that disabling mid-retreat reassignment worsens service time (release-to-delivery latency) by 1.89 times and makespan by 1.53 times in the tested high-load tree condition.",
  "published": "2026-08-27",
  "updated": "2026-08-27",
  "year": "2026",
  "authors": [
   "Taisei Hirayama",
   "Kohei Yoshida",
   "Hiroki Sakaji",
   "Itsuki Noda"
  ],
  "author_count": 4,
  "categories": [
   "cs.MA",
   "cs.RO"
  ],
  "primary_category": "cs.MA",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "11 pages, 9 figures. Accepted at the 14th Workshop on Planning and Robotics (PlanRob), co-located with ICAPS 2026",
  "topics": [
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.26759v1",
  "pdf_url": "https://arxiv.org/pdf/2608.26759v1",
  "html_url": "https://arxiv.org/html/2608.26759v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.26739",
  "slug": "residual-deep-reinforcement-learning-based-computed-torque-control-for",
  "title": "Residual Deep Reinforcement Learning-Based Computed Torque Control for a Cable-Driven Lower-Limb Rehabilitation Robot under Disturbances and Parametric Uncertainties",
  "abstract": "Accurate trajectory tracking in cable-driven lower-limb rehabilitation robots is challenging because model uncertainty, external disturbances, joint constraints, and pull-only cable actuation can degrade nominal control performance. Conventional model-based controllers provide an interpretable control structure but remain sensitive to model mismatch, whereas fully learning-based control can reduce transparency and complicate constraint-aware operation. This study proposes a residual deep reinforcement learning-enhanced computed torque control framework in which computed torque control generates the nominal command and a bounded Deep Deterministic Policy Gradient policy supplies only an additional compensating torque. The approach is evaluated in simulation under nominal, uncertain, disturbed, combined, and generalization conditions, together with trajectory-tracking, joint-limit, cable-demand, workspace-feasibility, and cable-Jacobian diagnostics. Across the evaluated conditions, the residual controller improves tracking and disturbance rejection relative to computed torque control while preserving the interpretable model-based command structure and satisfying the reported feasibility checks in the representative evaluation. Broader tests indicate that tracking improvements can persist beyond the representative case while also exposing trajectory-dependent constraint limitations. These results support bounded residual learning as a practical robustness-enhancement strategy for simulation-based rehabilitation robot control and motivate further constraint-aware and experimental validation.",
  "published": "2026-08-27",
  "updated": "2026-08-27",
  "year": "2026",
  "authors": [
   "Mohammad-Hossein Fakouri",
   "Ali Keymasi-Khalaji"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "32 pages, 24 figures, 13 tables. Preprint",
  "topics": [
   "rl-control",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.26739v1",
  "pdf_url": "https://arxiv.org/pdf/2608.26739v1",
  "html_url": "https://arxiv.org/html/2608.26739v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.26737",
  "slug": "generative-semantic-scene-completion",
  "title": "Generative Semantic Scene Completion",
  "abstract": "Outdoor LiDAR semantic scene completion (SSC) recovers a dense semantic voxel grid from a scan observing 1% of the target volume, under class imbalance beyond 7,000x. We recast SSC as generative semantic scene completion (GSSC): a single discrete-diffusion formulation in three roles. First, paired sparse-dense scene synthesis (PS$^3$) generates matched sparse LiDAR observations with their dense semantic completions, addressing the long tail at its source and yielding the PS$^3$-SemanticKITTI corpus we train on alongside SemanticKITTI. Second, semantic-guided generative scene completion (SGSC) generates the scene from noise with multinomial discrete diffusion, conditioned on the sparse scan through a bird's-eye-view semantic map and a sparse 3D feature stream. Third, the same framework instead refines an existing completion in one flow-matching step: structured source discrete diffusion (S$^2$D$^2$). S$^2$D$^2$ improves the mIoU of SGSC's own output and every external SSC base tested, without base retraining or test-time adaptation. On the strongest base, one step without test-time augmentation reaches 38.8% mIoU on the SemanticKITTI hidden test. To our knowledge that is the best causal, single-sweep, single-sample result on that leaderboard, +2.1 pp over the previous best published score under the same restriction. Four correction steps with eight-view test-time augmentation reach 39.2%, outside that restriction.",
  "published": "2026-08-27",
  "updated": "2026-08-27",
  "year": "2026",
  "authors": [
   "Shi Chen",
   "Weifeng Ge"
  ],
  "author_count": 2,
  "categories": [
   "cs.CV",
   "cs.LG",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "18 pages, 12 figures, 4 tables. Supplementary material (29 pages) is included as an ancillary file. Project page: https://shichen.world/GSSC-project-page/ - Code, models and the PS$^3$ dataset: https://github.com/BillyChern/GSSC-S2D2",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.26737v1",
  "pdf_url": "https://arxiv.org/pdf/2608.26737v1",
  "html_url": "https://arxiv.org/html/2608.26737v1",
  "code_url": "https://github.com/BillyChern/GSSC-S2D2",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2608.26673",
  "slug": "predvla-a-sub-million-parameter-predictive-coding-policy-for-robot-man",
  "title": "PredVLA: A Sub-Million-Parameter Predictive-Coding Policy for Robot Manipulation",
  "abstract": "Large pretrained vision-language-action models dominate modern robot-manipulation benchmarks, but it remains unclear how much model scale is necessary for strong language-conditioned control, or whether fundamentally different control architectures can remain competitive at much smaller parameter budgets. We present PredVLA, a language-conditioned predictive-coding policy with only 0.68 million trainable network parameters and no robot-data pretraining, whose hierarchical generative recurrent dynamics predict visual features and proprioception while observations influence latent state only through online inference from the resulting sensory prediction errors. On LIBERO, PredVLA achieves an 86.9% mean success rate across the three short-horizon suites and 75.4% when the long-horizon suite is included. Under a controlled comparison using the same frozen front end, demonstrations, action decoder, and evaluation protocol, PredVLA achieves 3.7x and 7.4x mean success rates of parameter-matched Transformer and LSTM policies, respectively. The predictive-coding formulation also makes the contribution of observation-driven correction directly measurable: because observations influence the recurrent state only through prediction-error-based latent inference, disabling this inference yields an exact open-loop control condition. Together, these results show that a sub-million-parameter recurrent generative policy can achieve strong performance on modern language-conditioned manipulation benchmarks while providing an explicit mechanism for prediction-error-driven online state correction.",
  "published": "2026-08-27",
  "updated": "2026-08-27",
  "year": "2026",
  "authors": [
   "Hiroki Sawada",
   "Shunichi Kasahara"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "",
  "topics": [
   "vla",
   "sim2real",
   "imitation-diffusion",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.26673v1",
  "pdf_url": "https://arxiv.org/pdf/2608.26673v1",
  "html_url": "https://arxiv.org/html/2608.26673v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.26669",
  "slug": "beyond-the-proving-ground-independent-public-road-testing-of-assisted",
  "title": "Beyond the Proving Ground: Independent Public-Road Testing of Assisted Lane Change Systems using LiDAR",
  "abstract": "Testing of commercial Advanced Driver Assistance Systems is essential to ensure safety and compliance during type approval and in service operation. However, proving ground scenarios may not reflect real world driving complexity, while geo fencing can require manufacturer collaboration and limit assessment independence. This work presents a methodology for independently testing Assisted Lane Change systems on public roads. A campaign on the A31 French motorway used a test vehicle equipped with a LiDAR based vehicle detection and tracking system. Tests covered combinations of inter vehicle distance and speed between the test vehicle and the take over vehicle. Real time kinematic global navigation satellite system receivers assessed detection and tracking performance. Recorded lane change trajectories were compared with the lane change suppression requirements of UNECE Regulation Number 79. Of 27 predefined lane change manoeuvres, 18 were completed and 9 suppressed. In 6 cases, the system allowed manoeuvres that did not meet regulatory minimum distance requirements. In 3 cases, the deviation remained statistically significant after accounting for measurement uncertainty. To the authors knowledge, this is the first public road campaign designed to assess Assisted Lane Change compliance with Regulation Number 79 safety distance requirements. The results demonstrate the suitability of LiDAR based sensing for this purpose. The methodology can support market surveillance and future regulatory revisions by revealing real world behaviours not covered by approval procedures.",
  "published": "2026-08-27",
  "updated": "2026-08-27",
  "year": "2026",
  "authors": [
   "Marcello Cellina",
   "Akos Kriston",
   "Antonio Migneco",
   "Davide Maggi",
   "Stefano Favelli",
   "Fabrizio Re",
   "Fabrizio Minarini",
   "Andrea Nuovo",
   "Riccardo Dona",
   "Biagio Ciuffo"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "",
  "topics": [
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.26669v1",
  "pdf_url": "https://arxiv.org/pdf/2608.26669v1",
  "html_url": "https://arxiv.org/html/2608.26669v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.26645",
  "slug": "flare-a-failure-aware-framework-for-autonomous-correction-and-recovery",
  "title": "FLARE: A Failure-Aware Framework for Autonomous Correction and Recovery in Visual-Language Robotic Manipulation",
  "abstract": "Vision-Language-Action Models~(VLAs) have demonstrated significant promise in generalizing to complex, long-horizon robotic manipulation tasks. However, their performance remains brittle, as they are typically trained on trajectory-monotonic, failure-free demonstrations. This reliance on ``perfect\" data leaves them unable to recover from common execution errors, such as a missed grasp, a dropped object, or an unexpected collision. In this paper, we propose FLARE, a novel framework that endows VLAs with robust error recovery capabilities through a ``Retry\" and ``Reset\" paradigm. First, we introduce a ``Retry\" mechanism by injecting perturbation and bridging segments that decouple robot pose from environment state into demonstrations, enabling the policy to autonomously handle execution deviations. Second, to address critical, state-breaking (OOD) failures, we introduce a ``Reset\" pipeline. We leverage an MLLM for offline failure analysis to automatically identify OOD states from execution videos. This analysis enables the efficient, targeted collection of a small library of object-centric ``Reset\" skills, which are trained to restore the environment to a task-valid state. Our full framework integrates these learned policies. At inference, an online MLLM monitor arbitrates between task execution and ``Reset\" skills. Experiments on challenging, contact-rich manipulation tasks show our approach significantly improves task success and robustness.",
  "published": "2026-08-27",
  "updated": "2026-08-27",
  "year": "2026",
  "authors": [
   "Ganlong Zhao",
   "Zijia Tang",
   "Xingping Chen",
   "Zhanghui Kuang",
   "Ye Tian",
   "Guanbin Li"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "CVPR 2026",
  "venue_source": "arxiv-comment",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "Accepted to CVPR 2026",
  "topics": [
   "vla",
   "dexterous-manipulation",
   "tactile",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.26645v1",
  "pdf_url": "https://arxiv.org/pdf/2608.26645v1",
  "html_url": "https://arxiv.org/html/2608.26645v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.26622",
  "slug": "relaxation-aware-multimodal-sensing-of-soft-gripper-driven-by-structur",
  "title": "Relaxation-Aware Multimodal Sensing of Soft Gripper Driven by Structure-Perception-Learning",
  "abstract": "Achieving stable, sustained grasping with soft robotic hands remains a fundamental challenge. Compliance enables safe and adaptive contact, yet the intrinsic viscoelasticity of soft polymers leads to stress relaxation and a continuous decay of grasping force during holding. Inspired by human grasping, which combines phase-dependent stiffness regulation with continuous sensing and feedback, this paper presents an integrated structure--perception--learning framework. We develop a variable-stiffness soft gripper that uses onboard vision and infrared thermography to track deformation and the temperature field in real time, preserving continuous tracking of the interaction state. To mitigate relaxation-induced force decay, we propose a temperature-coupled viscoelastic force representation, together with a physics-informed learning model, to reconstruct the force trend and provide explicit compensation during holding. Experiments show that, in a 280s force-controlled grasp-and-hold task, the proposed method maintains the desired force with a mean absolute error of 0.066N, outperforming fixed-aperture and instantaneous-only baselines by 80% and 95%, respectively. Overall, the results support a mechanism--AI co-design view: mechanisms shape feasible interactions, while learning compensates remaining uncertainty in viscoelastic dynamics, together enabling stable, sustained grasping.",
  "published": "2026-08-27",
  "updated": "2026-08-27",
  "year": "2026",
  "authors": [
   "Yanzhe Wang",
   "Hao Wu",
   "Ziyi Zheng",
   "Huixu Dong"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "RSS 2026",
  "venue_source": "arxiv-comment",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "11 pages, 9 figures. Published in Robotics: Science and Systems (RSS 2026)",
  "topics": [
   "dexterous-manipulation",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.26622v1",
  "pdf_url": "https://arxiv.org/pdf/2608.26622v1",
  "html_url": "https://arxiv.org/html/2608.26622v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.26583",
  "slug": "solo-stable-omni-terrain-long-horizon-perceptive-humanoid-locomotion",
  "title": "SOLO: Stable Omni-terrain Long-Horizon Perceptive Humanoid Locomotion",
  "abstract": "Humans traverse complex terrain over long distances without losing balance, whereas perceptive humanoid policies become fragile as perception and control errors accumulate. We present SOLO, a unified framework addressing two compounding causes of this long-horizon fragility: dense terrain reconstruction smooths action-critical details, and pointwise imitation lacks temporal credit assignment. Its Query Reconstructor (QR) uses Fourier-encoded cell queries to retrieve spatially specific evidence from depth-proprioception tokens, preserving sharp terrain boundaries. Trajectory-Aware MSE (TA-MSE) Distillation adds next-state teacher-student disagreement to the PPO reward, enabling Generalized Advantage Estimation to propagate future disagreement penalties to preceding actions. In simulation, QR reduces height-map L1 error by factors of 3.3-4.0, while TA-MSE surpasses PPO and MSE+PPO in curriculum progression. On stress-test terrains, SOLO achieves 97.5% mean traversal success and 96% stepping-stone success, versus 75.0-75.6% and 0-3% for dense-reconstructor variants. Deployed zero-shot with only a chest-mounted depth camera and proprioception, SOLO completes a continuous 1.5-km outdoor route and an indoor mixed-terrain course. Project page: https://sunpihai-up.github.io/solo/",
  "published": "2026-08-27",
  "updated": "2026-08-27",
  "year": "2026",
  "authors": [
   "Pihai Sun",
   "Gang Han",
   "Jingkai Sun",
   "Jiahao Ma",
   "Zeran Su",
   "Zelin Tao",
   "Peiran Liu",
   "Shuai Shi",
   "Wei Cui",
   "Zifan Wang",
   "Jialin Yu",
   "Wen Zhao",
   "Kangning Yin",
   "Jiaxu Wang",
   "Jiahang Cao",
   "Lingfeng Zhang",
   "Hao Cheng",
   "Jian Tang",
   "Yijie Guo",
   "Qiang Zhang"
  ],
  "author_count": 20,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "",
  "topics": [
   "humanoids"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.26583v1",
  "pdf_url": "https://arxiv.org/pdf/2608.26583v1",
  "html_url": "https://arxiv.org/html/2608.26583v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.26578",
  "slug": "trapvla-trapping-vision-language-action-models-in-configured-failure-m",
  "title": "TrapVLA: Trapping Vision-Language-Action Models in Configured Failure Modes",
  "abstract": "This work introduces Configured Failure Trapping, a novel backdoor attack task against Vision-Language-Action (VLA) models, which aims to activate attacks through stealthy textual triggers and induce configured failure modes. Unlike prior backdoor attacks that treat any task failure as a successful attack, Configured Failure Trapping requires the attacker to control how the robot fails (e.g., causing the robot to grasp with a specified positional offset), making it substantially more challenging and hard to detect. To support the new task, we propose an effective data engine for synthesizing high-quality target trajectories and an automated suite for measuring configured-failure fidelity. Then, based on this foundation, we construct two new benchmarks, namely Trap-LIBERO and Trap-RoboTwin, that instantiate Configured Failure Trapping across four representative failure modes. To address this task, we identify sparse action deviation as a critical challenge and accordingly propose a novel method named TrapVLA, which explicitly learns trigger-induced action residuals to steer the policy toward the configured failure behavior. Extensive experiments across simulation benchmarks and real-world robotic settings show that TrapVLA effectively injects configured failure modes into VLA models while largely preserving performance on clean data. Project page: https://john-liua.github.io/TrapVLA/",
  "published": "2026-08-27",
  "updated": "2026-08-27",
  "year": "2026",
  "authors": [
   "Jun-Hui Liu",
   "Kun-Yu Lin",
   "Yi-Lin Wei",
   "Xu-Han Chen",
   "Yinghao Li",
   "Zhuohao Li",
   "Yuan-Ming Li",
   "Qing Zhang",
   "Xiaoyi Fan",
   "Dongmei Jiang",
   "Yan Li",
   "Wei-Shi Zheng"
  ],
  "author_count": 12,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "",
  "topics": [
   "vla",
   "dexterous-manipulation",
   "sim2real",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.26578v1",
  "pdf_url": "https://arxiv.org/pdf/2608.26578v1",
  "html_url": "https://arxiv.org/html/2608.26578v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.26571",
  "slug": "arrive-and-survive-scaling-safe-goal-conditioned-policy-learning-from",
  "title": "Arrive and Survive: Scaling Safe Goal-Conditioned Policy Learning from One-Bit Failure Signals",
  "abstract": "Contrastive reinforcement learning (CRL) scales effectively in goal-conditioned tasks by casting policy learning into a self-supervised contrastive objective. However, in a failure-terminated Markov decision process, established CRL considers pre-failure future goals only when constructing positive samples, without accounting for the probability mass removed by failure termination. Our theoretical analysis shows that this omission induces a systematic overestimation bias in goal-reaching values. Consequently, near-failure trajectories provide disproportionately strong supervision of success despite retaining little future occupancy. Unsafe actions can thereby be reinforced through catastrophic failure bootstrapping, leading to failed policy learning and unsustainable goal-reaching behaviours. To address this problem, we introduce two minimal yet strong corrections: mass-weighted InfoNCE corrects the overweighting of short surviving futures in critic learning, and a log-survival-mass score restores the missing survival mass in policy optimization. The resulting method, Safe Contrastive Reinforcement Learning (Safe-CRL), requires only the one-bit signal provided by failure termination to scale safe goal-conditioned policy learning. Across twelve failure-prone robot navigation and locomotion tasks, Safe-CRL consistently improves survival and substantially outperforms the Scaling-CRL baseline in goal-reaching performance. Additionally, deep Safe-CRL policies exhibit complex failure-avoidance behaviours. This study completes the CRL theory under failure termination and provides a scalable safe RL framework. The code is available via https://github.com/RomainLITUD/safe-crl.",
  "published": "2026-08-27",
  "updated": "2026-08-27",
  "year": "2026",
  "authors": [
   "Guopeng Li",
   "Yiyang Duan",
   "Yiru Jiao",
   "Chengcheng Xu"
  ],
  "author_count": 4,
  "categories": [
   "cs.LG",
   "cs.RO"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "21 pages, 14 figures, 5 tables, Code: https://github.com/RomainLITUD/safe-crl",
  "topics": [
   "humanoids",
   "rl-control",
   "spatial-3d",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.26571v1",
  "pdf_url": "https://arxiv.org/pdf/2608.26571v1",
  "html_url": "https://arxiv.org/html/2608.26571v1",
  "code_url": "https://github.com/RomainLITUD/safe-crl",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2608.26545",
  "slug": "memory-anchors-for-continual-robot-learning",
  "title": "Memory Anchors for Continual Robot Learning",
  "abstract": "Robot policies deployed in the wild should have the capability to continually learn new tasks without forgetting existing behaviors. A common approach to combat such catastrophic forgetting is to train on new task data with a replay buffer of previously learned task data. Although this buffer is commonly sampled randomly from all prior experiences, we show that a small set of these experiences contributes greatly in anchoring past performance. We call these experiences Memory Anchors. We identify Memory Anchors in regions where representations of new-task observations collapse onto those of old-task observations even though the tasks require conflicting actions, like when a familiar object must be manipulated in a new way. Rehearsing old data in this region plays a key role in preventing destructive overwriting of past task knowledge, serving as this critical Memory Anchor role. Excluding only 10% Memory Anchors before sampling the buffer leads to more than a 4.5x increase in catastrophic forgetting on the LIBERO benchmark suites. Conversely, enriching the replay buffer with Memory Anchors can decrease high-conflict task forgetting by 63% and enables successful continual learning of two task sequences on a real robot. Videos and additional visualizations can be found at https://robot-adaptation.github.io/MemoryAnchors",
  "published": "2026-08-27",
  "updated": "2026-08-27",
  "year": "2026",
  "authors": [
   "Maximilian Du",
   "Zhanyi Sun",
   "Chen Xu",
   "Paarth Shah",
   "Masha Itkina",
   "Shuran Song"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "22 pages, 15 figures",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.26545v1",
  "pdf_url": "https://arxiv.org/pdf/2608.26545v1",
  "html_url": "https://arxiv.org/html/2608.26545v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.26533",
  "slug": "barrier-function-conformal-safety-clearance-certification-with-cvar-fo",
  "title": "Barrier Function Conformal Safety Clearance Certification with CVaR for Driving Trajectory Selection",
  "abstract": "Autonomous driving motion planners generate and select candidate trajectories while accounting for interactions with surrounding agents. However, these evaluations do not certify the actual safety clearance of the selected trajectory. The framework evaluates the trajectory selected by ant planners and calibrates the gap between its plan time margin and realized safety clearance. A differentiable separating axis barrier margin deterministically lower bounds exact signed oriented-bounding-box (OBB) safety clearance, connecting the statistical certificate to safety margin. At plan time, the margin is evaluated using either a nominal prediction and sampled lower tail Conditional Value-at-Risk (CVaR), while post-selection conformal calibration over exchangeable drive sessions absorbs prediction and sampling errors. Conformal calibration provides statistical validity independently of predictor correctness. The method is evaluated on a frozen 300 session nuPlan study using native Predictive Driver Model (PDM) Closed loop proposals. At 10% target miscoverage, sampled lower CVaR reduces the conformal correction from 1.43m to 0.03m and increases the rate of nonnegative safety clearance certificates from 68.7% to 87.3%. Across all evaluated statistics, exact-clearance coverage remains above the 90% target at 93.3--96.7%.",
  "published": "2026-08-27",
  "updated": "2026-08-27",
  "year": "2026",
  "authors": [
   "Pei Yu Chang",
   "Qadeer Ahmed"
  ],
  "author_count": 2,
  "categories": [
   "eess.SY",
   "cs.RO"
  ],
  "primary_category": "eess.SY",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "",
  "topics": [
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.26533v1",
  "pdf_url": "https://arxiv.org/pdf/2608.26533v1",
  "html_url": "https://arxiv.org/html/2608.26533v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.26505",
  "slug": "closing-the-loop-on-the-poppy-humanoid-bipedal-locomotion-with-linear",
  "title": "Closing the Loop on the Poppy Humanoid: Bipedal Locomotion with Linear-Quadratic Control and Learned Cost Functions",
  "abstract": "The Poppy Humanoid is an open-source, low-cost robot suitable for research and education in artificial intelligence. However, we are unaware of any published methodology that achieves reliable, unassisted bipedal locomotion on the standard Poppy hardware. This paper contributes a functional closed-loop walking controller for Poppy, based on the linear-quadratic regulator (LQR) framework for trajectory tracking. Starting with data collected from open-loop playback of a nominal walking trajectory, our proposed method learns a quadratic cost function for an LQR controller that substantially improves the reliability of the motion. The closed-loop controller is validated empirically, demonstrating statistically significant improvements in walking performance compared to open-loop trajectory playback.",
  "published": "2026-08-27",
  "updated": "2026-08-27",
  "year": "2026",
  "authors": [
   "Xulin Chen",
   "Borui He",
   "Ruipeng Liu",
   "Naveed Tahir",
   "Zhenyu Gan",
   "Garrett E. Katz"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "Accepted by International Conference on the AI Revolution: Research, Ethics, and Society (AIR-RES 2026)",
  "topics": [
   "humanoids"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.26505v1",
  "pdf_url": "https://arxiv.org/pdf/2608.26505v1",
  "html_url": "https://arxiv.org/html/2608.26505v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.26496",
  "slug": "rtnav-towards-real-time-zero-shot-object-navigation",
  "title": "RTNav: Towards Real-Time Zero-Shot Object Navigation",
  "abstract": "Navigation in unknown environments to find unforeseen objects has become increasingly feasible with capable vision and language foundation models. However, these models also introduce non-negligible inference latency, which becomes an important concern when agents must operate continuously in the real world. Most state-of-the-art methods are still developed in synchronous simulators, where the environment waits for the agent to act and inference time is effectively free. As a result, agents are often designed around the sequential execution of perception, reasoning, and action, with little regard for time constraints. Under real-time execution, where wall-clock time counts towards the task budget, the inefficiencies of these architectures become clear. We show that recent zero-shot object navigation methods suffer consistent performance degradation under such realistic timing conditions. Motivated by this observation, we propose RTNav, a simple but effective architecture that treats inference latency, asynchronous environment stepping, and bounded compute as explicit design considerations. Evaluated on real-time variants of HM3D-v1, HM3D-v2, and HM3D-OVON, RTNav improves the success rate by up to 11% and the Success weighted by Completion Time by up to 5.1 points over prior work.",
  "published": "2026-08-27",
  "updated": "2026-08-27",
  "year": "2026",
  "authors": [
   "Easop Lee",
   "Lingyu Zhang",
   "Boyuan Chen"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "",
  "topics": [
   "sim2real",
   "navigation",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.26496v1",
  "pdf_url": "https://arxiv.org/pdf/2608.26496v1",
  "html_url": "https://arxiv.org/html/2608.26496v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.26383",
  "slug": "cross-platform-benchmark-of-neural-3d-reconstruction-for-autonomous-la",
  "title": "Cross-Platform Benchmark of Neural 3D Reconstruction for Autonomous Laboratory Robots",
  "abstract": "Autonomous robots performing laboratory tasks depend on 3D reconstruction pipelines that can turn raw camera streams into actionable object representations within the latency budget of a physical control loop. Neural 3D reconstruction methods have demonstrated high-quality view synthesis, but their real-time viability across the compute platforms on which laboratory robots actually run remains poorly characterized. In this work, we present a systematic compute-platform benchmark of neural 3D reconstruction methods, evaluating NeRF and 3D Gaussian Splatting training and rendering on GPU-enabled computing devices ranging from single-board computers to server-class nodes, and place Meta's SAM3D single-image reconstruction on the same axes to quantify its latency and fidelity gap relative to per-scene optimization. Our results show that Gaussian Splatting yields higher rendering quality than NeRF at greater GPU cost, and that onboard compute is insufficient for full per-scene optimization at interactive rates. Our preliminary assessment on SAM3D indicates that it delivers plausible object geometry within seconds, but with detail mismatches that can compromise downstream manipulation. Together, these findings motivate tiered pipelines in which lightweight feed-forward reconstruction sustains the real-time perception-and-tracking loop for laboratory robots, while heavier neural reconstruction is scheduled selectively on suitable compute.",
  "published": "2026-08-26",
  "updated": "2026-08-26",
  "year": "2026",
  "authors": [
   "Yongho Kim",
   "Mengjiao Han",
   "Victor Mateevitsi",
   "Silvio Rizzi",
   "Michael E. Papka",
   "Nicola Ferrier"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "This manuscript is peer-reviewed from the committees in the workshop \"VAxAutoSci: Visual Analytics in the Age of Autonomous Scientific Discovery\" in conjunction with 2026 IEEE Visualization & Visual Analytics",
  "topics": [
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.26383v1",
  "pdf_url": "https://arxiv.org/pdf/2608.26383v1",
  "html_url": "https://arxiv.org/html/2608.26383v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.26314",
  "slug": "dispersive-forward-tree-search-for-optimal-control-coverage-complexity",
  "title": "Dispersive Forward Tree Search for Optimal Control: Coverage, Complexity, and Computation",
  "abstract": "Steering-based planners require solutions to state-to-state boundary value problems, which can be inaccessible for nonlinear platforms. Forward propagation evades the steering requirement, but the finite-sample behavior of the associated planners remains uncharacterized and their implementations underperform in practice. This paper develops a propagation-based kinodynamic planner with deterministic finite-sample near-optimality guarantees. We work within the large class of differentially flat nonlinear systems and show that a forward tree of locally dispersive control commands contains a near-optimal trajectory at a certified tree size. We provide a general mechanism to construct dispersive command sets for control-affine systems, which are necessary to implement the search algorithm prescribed by the theory. We show that covering the certified trajectory class irrespective of cost provably demands a tree exponentially sized in the problem horizon, and present a cost-conditioned dominance pruning procedure that retains near-optimality at a tree size polynomial in the horizon. We implement the resulting search algorithm, Dispersive Forward Tree search (DFT*), as breadth-first expansion of the forward tree, which maps naturally onto parallel hardware. We design efficient dispersive samplers for the unicycle, the trailer car, and the quadrotor and evaluate challenging planning tasks for these platforms. DFT* delivers consistently competitive and often substantially better solution quality than state-of-the-art kinodynamic planners at comparable solution times on embedded-tier processors, accelerating further as parallel compute is scaled. We also implement DFT* in a receding-horizon loop to demonstrate real-time planning in dynamic environments at embedded-tier compute budgets.",
  "published": "2026-08-26",
  "updated": "2026-08-26",
  "year": "2026",
  "authors": [
   "Shashank A. Deshpande",
   "Jonathan P. How"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO",
   "math.OC"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "28 pages. Code: https://github.com/croshank/DFTSearch",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.26314v1",
  "pdf_url": "https://arxiv.org/pdf/2608.26314v1",
  "html_url": "https://arxiv.org/html/2608.26314v1",
  "code_url": "https://github.com/croshank/DFTSearch",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2608.26273",
  "slug": "constraint-aware-physics-informed-neural-networks-for-static-shape-est",
  "title": "Constraint-Aware Physics-Informed Neural Networks for Static Shape Estimation of Co-Manipulative Continuum Robots",
  "abstract": "Static shape estimation of co-manipulative continuum robots (CCRs) is challenging because the continuum arms and manipulated flexible object form a closed chain that must satisfy both static equilibrium and geometric loop-closure constraints. This paper presents a constraint-aware physics-informed neural network (PINN) for static shape estimation of a tendon-driven CCR modeled using the geometric variable strain formulation. The proposed method incorporates a projected static equilibrium residual and a configuration-level geometric residual to enforce the governing mechanics and closed-chain geometry. In simulation, the PINN is compared with a purely data-driven artificial neural network (ANN) under limited and noisy training data. With 140 samples and 50% label noise, the PINN reduces the relative configuration error, equilibrium residual, and closed-chain residual by 67.88%, 67.35%, and 88.06%, respectively. Using the full dataset, the PINN achieves 0.1597% relative configuration error with an inference time of 0.1773 ms, compared with 17.97 s for an iterative nonlinear solver. Experimental fine-tuning reduces the marker RMSE from 2.657 mm to 0.497 mm and increases R2 from -0.788 to 0.937. These results demonstrate accurate, physically consistent, and computationally efficient static shape estimation of closed-chain CCRs.",
  "published": "2026-08-26",
  "updated": "2026-08-26",
  "year": "2026",
  "authors": [
   "Rana Danesh",
   "Pari Qarehdaghi",
   "Farrokh Janabi-Sharifi"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.26273v1",
  "pdf_url": "https://arxiv.org/pdf/2608.26273v1",
  "html_url": "https://arxiv.org/html/2608.26273v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.26239",
  "slug": "wall-ss-scaling-long-horizon-world-models-via-next-scale-autoregressio",
  "title": "WALL-SS: Scaling Long-horizon World Models via Next-Scale Autoregression",
  "abstract": "Generative world models provide robots with predictive models of how the world evolves under interaction, with growing potential for simulation, planning, policy evaluation, and robot learning. Beyond clip-level future prediction, a unified generative formulation should relate actions to consequences, support flexible horizons and continuous interaction, and enable reward-driven optimization. We introduce WALL-SS, a world model that generates visual futures through Scale-wise autoregressive Scaling, enabling action-controllable and long-horizon robotic simulation. WALL-SS represents embodied trajectories as causal sequences of temporally interleaved observations and actions, making action-dependent state transitions explicit while naturally supporting variable-length generation, streaming extension through reusable causal states, and direct optimization through sequence probabilities. To make this formulation effective over long horizons, we generate each future observation in a coarse-to-fine manner and develop three complementary components within the same hierarchy. Action-conditioned next-scale prediction injects scale-aligned action representations to improve action-future coupling and model both successful and failed behaviors. Scale-compressed long-horizon memory retains recent interactions at fine resolution while compressing distant observations and actions, with scale-wise dream forcing enhancing robustness to self-generated context. Finally, on-policy alignment optimizes autoregressive visual dynamics with action-following and long-term consistency rewards while preserving the pretrained visual distribution. Experiments show that WALL-SS improves action following and trajectory accuracy, supports coherent minute-long streaming rollout under bounded memory, and consistently benefits from on-policy alignment in reducing action drift and long-horizon inconsistency.",
  "published": "2026-08-26",
  "updated": "2026-08-26",
  "year": "2026",
  "authors": [
   "Maeve Zhang",
   "Rain Sun",
   "Xiang Wang",
   "Cyril Zhang",
   "Shalfun Li",
   "Meng Cao",
   "Howard Lu",
   "Ethan Chen",
   "Harry Jhou",
   "KZ Zheng",
   "Lights Shi",
   "Regis Cheng",
   " Lorenzin",
   "Robert Wang",
   "Victor Yao",
   "Gody Li",
   "Elise Mon",
   "Yohann Tang",
   "Ryan Yu",
   "PS Zhang",
   "Vincent Chen",
   "Hang Su",
   "Roy Gan",
   "Hao Wang",
   "Qian Wang"
  ],
  "author_count": 25,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "",
  "topics": [
   "world-models",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.26239v1",
  "pdf_url": "https://arxiv.org/pdf/2608.26239v1",
  "html_url": "https://arxiv.org/html/2608.26239v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.26105",
  "slug": "vbvr-pro-a-scalable-and-verifiable-suite-for-native-visual-reasoning",
  "title": "VBVR-Pro: A Scalable and Verifiable Suite for Native Visual Reasoning",
  "abstract": "Native visual reasoning treats visual generation as the medium of reasoning itself: visual states (i.e. images and videos) are not merely inputs to be understood or outputs to be rendered, but first-class substrates for problem solving beyond language. Yet progress remains bottlenecked by the lack of scalable training tasks, reliable feedback, and controlled comparisons across generative substrates. In this work, we introduce VBVR-Pro, a closed-loop testbed that makes native visual reasoning through generation trainable, verifiable, optimizable, and experimentally controllable. 1) Task scaling. VBVR-Pro turns visual reasoning into a controlled task space of 300 procedurally generated tasks. Models trained on VBVR-Pro show strong transfer beyond the proposed suite across seven external visual reasoning benchmarks such as RISE-Video, MME-CoF-Pro, and BabyVision. 2) Verifiable rewards. VBVR-Pro provides verifiable reward scorers for task-grounded evaluation. Through a systematic study of leading MLLMs as judges, we identify recurring failure modes of the prevalent VLM-as-a-judge paradigm. In contrast, the proposed scorers are grounded in deterministic, task-specific rules, achieve fine-grained alignment with human judgments. Importantly, they serve as reliable reward signals for large-scale multi-task reinforcement learning and demonstrate stronger post-RL performance across visual reasoning tasks. 3) Mechanism study. VBVR-Pro enables controlled modality studies across more than 30 image, video, and interleaved generators. Our analysis shows that video generation remains strongest for tasks requiring persistent spatiotemporal state tracking, while interleaved generation provides a compute-efficient alternative. Critically, ablations and probing suggest the presence of vision-native trajectories that are crucial to visual reasoning. We release all data, models, scorers, and code.",
  "published": "2026-08-26",
  "updated": "2026-08-26",
  "year": "2026",
  "authors": [
   "Junxiang Xu",
   "Ruisi Wang",
   "Fanyi Pu",
   "Maijunxian Wang",
   "Ran Ji",
   "Tongxi Zhou",
   "Chenyang Gu",
   "Jing Zuo",
   "Hongcan Xiao",
   "Yimeng Geng",
   "Wanqi Yin",
   "Wei Chen",
   "Oscar Qian",
   "Zhengan Yan",
   "Ziqi Huang",
   "Haiwen Diao",
   "Liang Pan",
   "Bo Li",
   "Xiangyu Fan",
   "Dezhi Luo",
   "Fengyuan Yu",
   "Zehong Zhao",
   "Qingying Gao",
   "Tinghui Zhu",
   "Yilan Zhang",
   "Jingqi Tong",
   "Pinyuan Feng",
   "Zhengze Jiang",
   "Letian Wang",
   "Ziyu Guo",
   "Renrui Zhang",
   "Jieneng Chen",
   "Sonia Joseph",
   "Constantin Venhoff",
   "Saman Motamed",
   "Mengyue Yang",
   "Chandra Sripada",
   "Alan Yuille",
   "Philip Torr",
   "Lvmin Zhang",
   "Vikash Kumar",
   "Daniel Khashabi",
   "Nikolaus Kriegeskorte",
   "Rapha\u00ebl Milli\u00e8re",
   "Vincent C. M\u00fcller",
   "Anyi Rao",
   "Quan Wang",
   "Ziwei Liu",
   "Dahua Lin",
   "Lei Yang",
   "Hokin Deng",
   "Zhongang Cai"
  ],
  "author_count": 52,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.LG",
   "cs.MM",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "Homepage: https://video-reason.com/",
  "topics": [
   "rl-control",
   "video-generation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.26105v1",
  "pdf_url": "https://arxiv.org/pdf/2608.26105v1",
  "html_url": "https://arxiv.org/html/2608.26105v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.26103",
  "slug": "zero-wam-in-context-world-action-modeling-from-human-videos-for-open-e",
  "title": "Zero-WAM: In-Context World-Action Modeling from Human Videos for Open-Ended Task Generalization",
  "abstract": "Zero-shot cross-task generalization, where a policy must execute manipulation tasks never seen during training, remains a central challenge in robot learning. In large language models, a novel task can be performed simply by specifying it in the context, without any parameter update. This form of in-context learning (ICL) turns generalization into a problem of task specification. To achieve cross-task generalization, we bring this paradigm to robotic manipulation, and argue that the natural task specification for manipulation is a human video: unlike language, it provides rich visual cues about the intended task evolution. We present Zero-WAM, a causal video-action model that executes unseen tasks by following in-context human video guidance. To address the scarcity of task-rich paired human-robot data, we propose an automatic pipeline that converts task-sampled robot trajectories into semantically matched human videos, yielding HumanGen, a dataset of 74.2K human-robot ICL pairs across 8.6K tasks. For model training, we further introduce an in-context future chunk prediction (IFP) objective that suppresses shortcuts learned from seen tasks and forces the policy to draw task information from the video prompt. On seven unseen tasks in RoboTwin 2.0 simulation, Zero-WAM achieves a 47.0% average success rate, an absolute improvement of 29.5 percentage points over the strongest video-action baseline. In real-world evaluations, it follows human video guidance to generalize to unseen task configurations involving multi-object scenes, long-horizon manipulation, and fine-grained insertion.",
  "published": "2026-08-26",
  "updated": "2026-08-27",
  "year": "2026",
  "authors": [
   "Jiaming Zhou",
   "Qihang Zhang",
   "Gangwei Xu",
   "Cunxin Fan",
   "Yujie Zhao",
   "Ruilin Wang",
   "Yiming Luo",
   "Shuai Yang",
   "Xing Zhu",
   "Yujun Shen",
   "Junwei Liang",
   "Yinghao Xu"
  ],
  "author_count": 12,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "https://robbyant-research.github.io/Zero-WAM/",
  "topics": [
   "egocentric-data"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.26103v2",
  "pdf_url": "https://arxiv.org/pdf/2608.26103v2",
  "html_url": "https://arxiv.org/html/2608.26103v2",
  "code_url": "https://robbyant-research.github.io/Zero-WAM/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2608.26076",
  "slug": "fast-generative-grasping-via-lie-group-constrained-meanflow",
  "title": "Fast Generative Grasping via Lie Group-Constrained MeanFlow",
  "abstract": "Grasp synthesis is a core task in robotic manipulation, for which the solution typically forms a multimodal distribution rather than a point estimate. Generative robotic grasping aims to learn this distribution with deep generative models such as diffusion and flow-based approaches. The iterative nature of such generative models makes them flexible and generalizable; however, multi-step sampling impedes the time-critical operation required in robotics. We devise an approach to fast generative grasping based on MeanFlow on the product Lie group $\\mathcal{G} = \\mathrm{SO}(3) \\times \\mathbb{R}^3$. The training objective couples a purely algebraic semigroup consistency condition with Riemannian Conditional Flow Matching on $\\mathcal{G}$ that anchors the average velocity to the data distribution. The resulting Lie Group-constrained MeanFlow formulation samples reliable grasps in $\\leq 5$ network evaluations, matching the grasp generation performance of state-of-the-art diffusion and flow-based models on the ACRONYM dataset at millisecond-scale inference latency (up to $39\\times$ speed-up). We further demonstrate that the approach directly translates to real-world robotic grasping without additional training or domain adaptation, exhibiting robust grasp synthesis under observation noise.",
  "published": "2026-08-26",
  "updated": "2026-08-26",
  "year": "2026",
  "authors": [
   "S. Talha Bukhari",
   "Yi Wei",
   "Ruiqi Ni",
   "Zachary Kingston",
   "Aniket Bera"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "",
  "topics": [
   "dexterous-manipulation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.26076v1",
  "pdf_url": "https://arxiv.org/pdf/2608.26076v1",
  "html_url": "https://arxiv.org/html/2608.26076v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.26074",
  "slug": "gating-before-commitment-anticipating-intent-divergence-to-prevent-pos",
  "title": "Gating Before Commitment: Anticipating Intent Divergence to Prevent Post-Interaction Decision Failures in Autonomous Driving",
  "abstract": "Intent misinterpretation during vehicle interactions causes recurring planning failures. We study a decision layer in which a language-guided intent module reads structured descriptors, computes a smoothed intent-geometry divergence score, and gates the planned maneuver before commitment, upstream of a corridor envelope. On a replayed off-road departure and four crash clips under a frozen, disclosed implementation, gating is the only layer that repairs the plan: on the main case it fires 72 ms after the drift onset but 161 ms before the corridor exit, keeping the trajectory in the corridor in all ten replays. The first calibration draws nine false triggers in 5.9 minutes, each from scoring uncertainty as half a conflict; a preregistered redesign treating uncertainty as abstention cuts this to 0.341 per minute. Two ablations bound the model's contribution: the full score detects fastest on four of five failures under the deployed eligibility, three of five against the unvetoed rule (000871 by one cycle; 000228 by a pre-onset fire on an uncertain stretch that five clips cannot classify as signal or coincidence; dropping the confidence term costs two detections), while on in-domain tracks at equal false positives the geometric rule more than triples its detection. The evidence supports the gating mechanism; the model's demonstrated roles are the fastest detection on these failures and an uncertainty veto on the geometric rule.",
  "published": "2026-08-26",
  "updated": "2026-08-26",
  "year": "2026",
  "authors": [
   "Cong Xu",
   "Ravi Sankar"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "8 pages, 4 figures. Submitted to the 16th Workshop on Planning, Perception and Navigation for Intelligent Vehicles (PPNIV) at IROS 2026. Supplementary video included as ancillary material",
  "topics": [
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.26074v1",
  "pdf_url": "https://arxiv.org/pdf/2608.26074v1",
  "html_url": "https://arxiv.org/html/2608.26074v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.26066",
  "slug": "virtoos-a-ros-2-unity-virtualization-toolkit-for-fleet-management-of-a",
  "title": "VirTooS: A ROS 2 - Unity Virtualization Toolkit for Fleet Management of Autonomous Mobile Robots",
  "abstract": "In this paper, we present VirTooS, a Python/C# toolkit designed to implement fleet-management tasks on teams of Autonomous Mobile Robots (AMRs). VirTooS leverages the Robot Operating System (ROS) 2 and Unity game engine to provide realistic, scalable virtual experiments in a mixed-reality environment. The toolbox allows users to easily generate and customize virtual scenarios for realistic simulations. Virtual and real sensors as, e.g., LiDARs, can be exploited to map and safely navigate in the mixed-reality environment. To enable distributed robotics experiments, we propose a set of tailored routines leveraging the ChoiRbot framework. As a motivating example, we show a set of experiments for task assignment problems in a virtual environment, allowing seamless interaction among real and virtual robots. Moreover, the package comes with a containerized suite to easily deploy it on different machines. The source code will be made publicly available on GitHub.",
  "published": "2026-08-26",
  "updated": "2026-08-26",
  "year": "2026",
  "authors": [
   "Andrea Drudi",
   "Lorenzo Pichierri",
   "Andrea Testa",
   "Giuseppe Notarstefano"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.26066v1",
  "pdf_url": "https://arxiv.org/pdf/2608.26066v1",
  "html_url": "https://arxiv.org/html/2608.26066v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.26058",
  "slug": "one-policy-many-embodiments-unified-camera-centric-action-geometry-pre",
  "title": "One Policy, Many Embodiments: Unified Camera-Centric Action Geometry Pre-training for Heterogeneous Embodied Manipulation",
  "abstract": "Scaling generalist vision-language-action (VLA) policies is severely bottlenecked by the inherent heterogeneity of embodied data, which spans diverse robot morphologies, camera configurations, and low-level action spaces. Existing paradigms typically address this mismatch through explicit action retargeting, human-to-robot video synthesis, or dataset-specific adaptation branches, fundamentally hindering the joint learning of a unified policy. We introduce UCAG-P, a camera-centric unified action formulation that structurally aligns heterogeneous embodied datasets into a shared geometric action space. Rather than treating robot-specific commands as the shared policy target, UCAG-P represents manipulation through camera-observable anchor motion in image and camera-frame coordinates, treating robot arms, humanoids, and human hands as different embodiments of a common action schema. A geometry-conditioned action translator combines predicted motion with target-embodiment kinematics to produce executable controls. The resulting decoupled architecture allows a shared VLA policy to learn transferable manipulation geometry while retaining embodiment-specific controllability. UCAG-P is trained on 4.03K hours of robot and simulation data and 2.34K hours of human demonstrations. A single checkpoint reaches 98.3% on LIBERO, 88.7% and 89.2% on RoboTwin Easy and Hard, 82.0% zero-shot on LIBERO-Plus, and 62.0% on RoboCasa GR-1, without benchmark-specific fine-tuning.",
  "published": "2026-08-26",
  "updated": "2026-08-26",
  "year": "2026",
  "authors": [
   " Xiaomi Embodied Intelligence Team",
   "University of Macau",
   " :",
   "Shaoqing Xu",
   "Fang Li",
   "Guozhi Zhan",
   "Zhixiang Duan",
   "Yuhan Wang",
   "Yuechen Luo",
   "Shengyin Jiang",
   "Hanbing Li",
   "Zhiying Du",
   "Longlong Wang",
   "Longmei Jiang",
   "Weixiang Liang",
   "Ying Gong",
   "Yong Pan",
   "Ziping Zhao",
   "Zhiyuan Chen",
   "Yangwei You",
   "Kun Ma",
   "Qinyuan Liu",
   "Hangjun Ye",
   "Zhi-xin Yang"
  ],
  "author_count": 24,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "Technical Report,Project page: https://public-bots.github.io/UCAG-P",
  "topics": [
   "vla",
   "humanoids",
   "egocentric-data",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.26058v1",
  "pdf_url": "https://arxiv.org/pdf/2608.26058v1",
  "html_url": "https://arxiv.org/html/2608.26058v1",
  "code_url": "https://public-bots.github.io/UCAG-P",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2608.26053",
  "slug": "r-3-training-robots-to-reason-in-natural-language-via-reinforcement-le",
  "title": "$R^3$: Training Robots to Reason in Natural Language via Reinforcement Learning",
  "abstract": "Reasoning in language allows foundation models to spend more test-time compute on hard problems, such as those requiring decomposition, constraint tracking, and prediction of future consequences. Whether this mechanism can improve robotic manipulation remains unclear, where long-horizon tasks require tracking partial progress, reasoning about object relations, recovering from mistakes, and steering noisy low-level policies. In this paper, we study whether VLMs can be trained to reason directly in natural language to guide low-level manipulation policies. We introduce $R^3$, a simple post-training recipe that turns off-the-shelf VLMs into robotic reasoners: it first mid-trains a VLM on expert-generated reasoning traces to initialize the desired reasoning style, then improves the reasoner with single-step rubric-based RL from offline action data. Unlike prior robotic reasoning methods that mostly use structured traces as auxiliary supervision, $R^3$ trains free-form language reasoning to produce test-time guidance for action. We instantiate $R^3$ on Language Table and simulated bimanual grocery packing, two controlled testbeds for studying robotic reasoning and long-horizon manipulation. $R^3$ improves exploration and generalization across unseen tasks and significantly outperforms instruction-only imitation learning baselines on both benchmarks. Our analyses suggest that free-form language reasoning can function as a test-time compute mechanism for steering low-level policies. Our project page is available at https://robotic-reasoner.github.io/.",
  "published": "2026-08-26",
  "updated": "2026-08-26",
  "year": "2026",
  "authors": [
   "Lehong Wu",
   "Yuxiao Qu",
   "Zheyuan Hu",
   "Ivan Zhang",
   "Limin Wei",
   "Zackory Erickson",
   "Aviral Kumar"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CL",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "42 pages, 23 figures",
  "topics": [
   "imitation-diffusion",
   "rl-control",
   "navigation",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.26053v1",
  "pdf_url": "https://arxiv.org/pdf/2608.26053v1",
  "html_url": "https://arxiv.org/html/2608.26053v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.26050",
  "slug": "when-obstacles-bend-modeling-vegetation-deformation-in-the-context-of",
  "title": "When Obstacles Bend: Modeling Vegetation Deformation in the context of Field Robotics",
  "abstract": "Autonomous robots operating in natural environments must often interact with vegetation rather than simply avoid it. In this context, traversability is typically defined from the robot's perspective, by measuring how a specific platform responds when moving through the environment. While practical, this viewpoint entangles the assessment of the environment with the robot's own dynamics, making the resulting characterization difficult to transfer across different platforms. More importantly, it does not directly reflect the properties of the vegetation itself, which are the true source of interaction and potential damage in applications such as agriculture and environmental monitoring. To address this limitation, we propose to characterize vegetation through its intrinsic mechanical properties, independently of any specific robot. By combining deformation measurements with contact force data, we estimate the underlying mechanical parameters and reconstruct the vegetation's response to interaction. This enables vegetation-aware navigation based on intrinsic environmental properties rather than platform-dependent metrics.",
  "published": "2026-08-26",
  "updated": "2026-08-26",
  "year": "2026",
  "authors": [
   "Muhammad Hsaeeb Zaar Khizar",
   "Tom Montagnon",
   "Roland Lenain",
   "Romuald Aufr\u00e8re",
   "Johann Laconte"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "8 pages, 7 figures, submitted to IEEE Robotics and Automation Letters (RA-L)",
  "topics": [
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.26050v1",
  "pdf_url": "https://arxiv.org/pdf/2608.26050v1",
  "html_url": "https://arxiv.org/html/2608.26050v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.26011",
  "slug": "phantom-navigator-stealthy-and-precise-unmanned-aerial-vehicle-redirec",
  "title": "Phantom Navigator: Stealthy and Precise Unmanned Aerial Vehicle Redirection with Real-Time Tracking and GPS Spoofing",
  "abstract": "Redirecting unmanned aerial vehicles (UAVs) from their intended mission trajectories has been an active area of research. However, existing UAV redirection attacks lack reliability, precision, and covertness for a targeted diversion. They primarily rely on physical capture, communication hijacking, or sensor spoofing. Yet, physical interception is costly, offers only a single opportunity for success, and poses a high risk of collateral damage; network-based attacks demand deep technical expertise and access to encrypted communication channels; and sensor spoofing techniques typically fall short in achieving the accuracy and robustness required to steer a UAV toward a specified target. Consequently, we propose Phantom Navigator, a UAV redirection attack to mislead drones to a designated spoofing target, covertly and precisely. Our approach combines offline pre-redirection reachability analysis, which provides high-fidelity estimates of achievable redirect ranges, with an online closed-loop, stealthy execution layer that ensures successful redirection in practice. Based on this approach, we build a physical attack platform equipped with a LiDAR--camera detection, tracking, and spoofing stack that performs real-time identification, pose estimation, and computation of targeted spoofing signals to covertly and accurately redirect victim UAVs to a designated location. We demonstrate the effectiveness of our redirection methodology and the attack implementation in real-world case studies.",
  "published": "2026-08-26",
  "updated": "2026-08-26",
  "year": "2026",
  "authors": [
   "Haocheng Meng",
   "Shaocheng Luo",
   "Songqiao Xie",
   "Miroslav Pajic"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "",
  "topics": [
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.26011v1",
  "pdf_url": "https://arxiv.org/pdf/2608.26011v1",
  "html_url": "https://arxiv.org/html/2608.26011v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.26002",
  "slug": "descent-directed-edge-scene-encoding-for-airport-surface-movement-pred",
  "title": "DESCENT: Directed Edge Scene Encoding for Airport Surface Movement Prediction",
  "abstract": "Advanced automation is a key technology for enhancing the safety of ground operations amidst the increasing density of commercial air traffic. While motion forecasting is a well-studied task in autonomous driving, its application to airport surface movements remains underexplored. To enable efficient and accurate prediction in this domain, we propose DESCENT, a transformer-based architecture designed to handle heterogeneous dynamics and strict topological constraints. Our approach features a Potential Reachable Set (PRS) context sampling mechanism that adaptively collects airfield environment context across diverse operational phases. Combined with a detection transformer-based decoder, DESCENT generates accurate trajectory forecasts. Extensive evaluations on the Amelia-10 benchmark demonstrate significant performance improvements over state-of-the-art baselines. These gains are especially pronounced in safety-critical scenarios, where our domain-aware sampling provides critical long-horizon context necessary for safe navigation.",
  "published": "2026-08-26",
  "updated": "2026-08-27",
  "year": "2026",
  "authors": [
   "Alexander Prutsch",
   "David Schinagl",
   "Horst Possegger"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "IROS 2026. Project page at https://a-pru.github.io/descent",
  "topics": [
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.26002v2",
  "pdf_url": "https://arxiv.org/pdf/2608.26002v2",
  "html_url": "https://arxiv.org/html/2608.26002v2",
  "code_url": "https://a-pru.github.io/descent",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.5
 },
 {
  "id": "2608.25940",
  "slug": "a-statistical-audit-of-physical-ai-benchmark-redundancy",
  "title": "A Statistical Audit of Physical AI Benchmark Redundancy",
  "abstract": "Physical AI models are evaluated on suites of benchmarks that differ across model reports, leaving the model-by-benchmark matrix sparse and the relationship between benchmarks unmeasured. We construct a matrix of 51 models on 12 physical AI benchmarks, selected from a registry of 51 benchmarks and 152 models by reporting density, combining scores from model cards and benchmark papers with our own evaluation runs under each benchmark's official protocol. We measure how much information the benchmarks share and show quantitative evidence of Redundancy. Redundancy affects reported rankings: collapsing the two substitute pairs into single columns moves 22 of 51 models by three or more places under an equally weighted average. We then select benchmarks greedily under a utility combining score dispersion with variance not explained by the already-selected set, and obtain a four-benchmark subset retaining 78.5\\% of the utility of all 12, on which we fit a Bradley--Terry ranking. The procedure requires only benchmark-level scores with sufficient overlap and is not specific to physical AI.",
  "published": "2026-08-26",
  "updated": "2026-08-26",
  "year": "2026",
  "authors": [
   "Zaruhi Navasardyan",
   "Hrant Davtyan"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.25940v1",
  "pdf_url": "https://arxiv.org/pdf/2608.25940v1",
  "html_url": "https://arxiv.org/html/2608.25940v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.25917",
  "slug": "choose-your-game-wisely-measuring-game-theoretic-structures-in-real-wo",
  "title": "Choose Your Game Wisely: Measuring Game-Theoretic Structures in Real-World Vehicle Interactions",
  "abstract": "Game-theoretic models provide principled frameworks for modeling vehicle interactions, but their underlying temporal assumptions have not been systematically examined against real-world driving behavior. In particular, it remains unclear how simultaneous, sequential, and asymmetric interaction structures can be measured from vehicle trajectories. This paper develops a trajectory-based interaction measurement framework to identify interaction events and quantify behavioral change onset, temporal organization, post-onset response dynamics, and ordering stability. The framework uses behavioral deviations to verify candidate interactions. We evaluate the framework on six real-world trajectory datasets, including INTERACTION, highD, inD, rounD, Waymo Open Motion, and nuPlan, covering diverse road geometries, traffic environments, and interaction types. The results show that concurrent and sequential behavioral changes both constitute substantial proportions of observed following, merging, and conflicting interactions. Among sequential interactions, stable ordering is more prevalent than alternating ordering, indicating that persistent asymmetric roles are a common interaction structure. Importantly, temporal precedence does not necessarily coincide with a measurable behavioral response, indicating that temporal ordering alone may not be sufficient to characterize behavioral dependence. These findings show that real-world interactions exhibit concurrent, sequential, and persistently ordered temporal structures. Different game-theoretic formulations are therefore better regarded as complementary modeling abstractions for different interaction regimes rather than as a universal structure governing all vehicle interactions.",
  "published": "2026-08-26",
  "updated": "2026-08-26",
  "year": "2026",
  "authors": [
   "Yueyuan Li",
   "Rongcheng Nie",
   "Weijie Xi",
   "Mingyang Jiang",
   "Songan Zhang",
   "Hanyang Zhuang",
   "Ming Yang"
  ],
  "author_count": 7,
  "categories": [
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.AI",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "8 pages, 3 figures, 3 tables",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.25917v1",
  "pdf_url": "https://arxiv.org/pdf/2608.25917v1",
  "html_url": "https://arxiv.org/html/2608.25917v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.25874",
  "slug": "low-resolution-perception-for-robotic-packing",
  "title": "Low-Resolution Perception for Robotic Packing",
  "abstract": "This work tackles the problem of scalable perception for robotic packing with low-cost, low-resolution depth sensing. We propose a framework where reconstruction cues drive next-view selection and grasp evidence updates a per-object stability estimate, jointly deciding what to acquire next and when to grasp. During the reconstruction, a low-resolution Next Best View (NBV) strategy explicitly avoids redundant views while preserving task-relevant geometry. We validate the approach in two steps: (i) an ablation study of the utility function under very low resolution, and (ii) a full end-to-end evaluation across policies, showing how low-resolution perception is a practical, scalable option for robotic packing.",
  "published": "2026-08-26",
  "updated": "2026-08-26",
  "year": "2026",
  "authors": [
   "Giuseppe Fabio Preziosa",
   "Federico Vignoni",
   "Chiara Castellano",
   "Marco Faroni",
   "Andrea Maria Zanchettin",
   "Paolo Rocco"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "6 pages, 5 figures. Accepted to IFAC World Congress 2026",
  "topics": [
   "dexterous-manipulation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.25874v1",
  "pdf_url": "https://arxiv.org/pdf/2608.25874v1",
  "html_url": "https://arxiv.org/html/2608.25874v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.25872",
  "slug": "vista-visually-inferred-spatial-contact-attention-for-contact-rich-man",
  "title": "VISTA: Visually Inferred Spatial ConTact Attention for Contact-Rich Manipulation",
  "abstract": "Contact-rich manipulation requires precise interaction feedback. While vision-centric imitation learning is prevalent, external visual observations provide indirect and ambiguous cues about contact states, particularly under occlusion or subtle object--gripper interactions; dedicated tactile or force sensors can provide rich contact information but introduce additional hardware complexity, calibration requirements, and deployment costs. To bridge this gap, we propose VISTA-Policy, an imitation learning paradigm that utilizes the Visual Deformation Field (VDF), a 3D displacement representation of a compliant gripper, as high-dimensional visuo-physical feedback. The framework integrates: 1) a Physics-Aware Encoding Engine for real-time VDF decoding; 2) an Energy Aggregation Denoising Mechanism to isolate true interaction signals; and 3) a Deformation-Augmented Policy Network with incremental gripper actions for precise closed-loop correction. Extensive evaluations on Cross-Scale Object Grasping, Cap Unscrewing, and Calligraphy Writing demonstrate that VISTA-Policy outperforms the strong pure-vision baseline 3D Diffusion Policy and the tactile baseline. VISTA-Policy further demonstrates substantial out-of-distribution generalization to unseen object scales and robustness against dynamic disturbances, offering a durable and cost-effective route toward general-purpose fine-grained manipulation in unstructured environments. Project videos and supplementary materials are available at: https://sites.google.com/view/vista-policy.",
  "published": "2026-08-26",
  "updated": "2026-08-26",
  "year": "2026",
  "authors": [
   "Jiayi Chen",
   "Wenlong Dong",
   "Yan Huang",
   "Xianglin Chen",
   "Zijian Lin",
   "Jiaqi Yin",
   "Yushan Liu",
   "Wenbo Ding"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "imitation-diffusion",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.25872v1",
  "pdf_url": "https://arxiv.org/pdf/2608.25872v1",
  "html_url": "https://arxiv.org/html/2608.25872v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.25864",
  "slug": "ma-vla-multi-arm-vision-language-action-model-for-collaboration-and-co",
  "title": "MA-VLA: Multi-Arm Vision-Language-Action Model for Collaboration and Compositional Generalization",
  "abstract": "Multi-arm collaboration is becoming a core capability in embodied manipulation. Recent vision-language-action (VLA) models integrate perception, language, and control, but most represent language as a single global instruction and do not provide an explicit mechanism for assigning and composing arm-specific behaviors. This design limits transfer to collaboration patterns that differ from those observed during training. We present MA-VLA, a unified framework for multi-arm collaboration via atomic action assignment. MA-VLA decomposes cooperative behavior into mid-level atomic prompts and allocates them to individual arms, enabling explicit subgoal specification and compositional reuse across tasks. To reduce reliance on fixed execution roles, we introduce Arm Shuffle, a training-time permutation of the observation, state, and assigned atomic prompts for each arm. This permutation enforces role-agnostic instruction following and supports recomposition into unseen coordination patterns, which we term multi-arm compositional generalization. We also construct a benchmark in which test-time collaboration patterns are absent in training set. Across simulation and real-world evaluations, prior state-of-the-art VLAs largely fail under these unseen collaborations, while MA-VLA consistently succeeds. These results indicate that structured, per-arm atomic action assignment offers a practical route to scalable generalization in multi-arm embodied systems. Code, models, and data are available at https://github.com/zhangzaibin/future-robots",
  "published": "2026-08-26",
  "updated": "2026-08-26",
  "year": "2026",
  "authors": [
   "Zaibin Zhang",
   "Junlan Xiao",
   "Zhongbo Zhang",
   "Yifan Wang",
   "Li Kang",
   "Yiran Qin",
   "Changxing Xia",
   "Heng Zhou",
   "Talas Fu",
   "Enshen Zhou",
   "Ruimao Zhang",
   "Zhenfei Yin",
   "Huchuan Lu",
   "Lijun Wang"
  ],
  "author_count": 14,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "ECCV 2026",
  "venue_source": "arxiv-comment",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "ECCV 2026",
  "topics": [
   "vla"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.25864v1",
  "pdf_url": "https://arxiv.org/pdf/2608.25864v1",
  "html_url": "https://arxiv.org/html/2608.25864v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.25830",
  "slug": "anytime-global-tensor-motion-planning",
  "title": "Anytime Global Tensor Motion Planning",
  "abstract": "Global Tensor Motion Planning (GTMP) solves motion planning with batched tensor operations over a layered multipartite graph. We generalize GTMP so that adjacent-layer edges are realized by any black-box local planner (e.g., linear interpolation, splines, sampling-based planning, trajectory optimization, or generative sampling). We provide two anytime policies on top of this generalization: Anytime GTMP with random restarts at a fixed budget, which covers every homotopy class almost surely, and AO-GTMP with informed expansion with growing budgets, which converges to the optimal cost. We prove that a single sampled graph covers every endpoint-fixed homotopy class admitting a \\(\u03b4\\)-clear representative of bounded length. We also prove that additional samples per layer reduce the per-layer miss probability exponentially, whereas stronger local planners reduce the required layer count only sublinearly. On manipulation benchmarks the method matches state-of-the-art performance, and on 2D navigation it returns batches of topologically diverse solutions, while the informed baselines concentrate on one or two classes.",
  "published": "2026-08-26",
  "updated": "2026-08-26",
  "year": "2026",
  "authors": [
   "Sai Coumar",
   "An T. Le",
   "Zachary Kingston"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "8 pages, 5 figures. Code: https://github.com/commalab/anytime_gtmp.git",
  "topics": [
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.25830v1",
  "pdf_url": "https://arxiv.org/pdf/2608.25830v1",
  "html_url": "https://arxiv.org/html/2608.25830v1",
  "code_url": "https://github.com/commalab/anytime_gtmp.git",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2608.25799",
  "slug": "agro-nav-autonomous-graph-based-orchard-navigation",
  "title": "AGRO-Nav: Autonomous Graph-based Orchard Navigation",
  "abstract": "Orchards form semi-structured environments in which parallel tree rows create natural driving corridors, yet narrow inter-row clearance and dense foliage lead geometry-agnostic grid planners to drift off the row center and risk trunk or canopy contact. We present AGRO-Nav, an automated framework for static graph-based global planning in orchards. From tree-row lines fitted to trunk clusters in a SLAM point cloud, it builds, without any manual waypoints, a sparse topological graph of intra- and inter-row connectivity; a global route is then found by Dijkstra search on this graph, connected to the start and goal by any-angle Theta* segments, and smoothed with a cubic B-spline. In real-orchard trials, AGRO-Nav follows the row center with a mean error of about 0.08 m, far below the A* (0.31 m) and Theta* (0.43 m) shortest-path baselines, while planning roughly four to five times faster. In Isaac Sim, it attains the lowest error among A*, Theta*, and a reproduced RANSAC midline baseline and remains stable as tree density drops to 70%, where the RANSAC baseline degrades. The resulting trajectories---straight row-centered segments joined by controlled turns---suit differential-drive and four-wheel-steering platforms.",
  "published": "2026-08-26",
  "updated": "2026-08-26",
  "year": "2026",
  "authors": [
   "Ho Young Yun",
   "Jaemin Yu",
   "Duksu Kim"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "8 pages, 4 figures",
  "topics": [
   "spatial-3d",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.25799v1",
  "pdf_url": "https://arxiv.org/pdf/2608.25799v1",
  "html_url": "https://arxiv.org/html/2608.25799v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.25798",
  "slug": "tacforcing-streaming-action-generation-with-execution-time-tactile-fee",
  "title": "TacForcing: Streaming Action Generation with Execution-Time Tactile Feedback",
  "abstract": "Contact-rich manipulation requires adapting to contact states that can evolve substantially within an action horizon. However, chunk-based vision-language-action models predict complete action chunks from observations collected before execution, leaving tactile conditioning stale during execution. Existing tactile-reactive approaches typically rely on separate high-frequency controllers, which increase both architectural and training complexity. In this paper, we introduce TacForcing, a streaming action-generation framework that effectively incorporates execution-time tactile feedback. Instead of employing a separate reactive controller, TacForcing replaces the standard action expert with a streaming action expert to generate actions conditioned on the evolving tactile observations acquired during execution. TacForcing also introduces Execution-Aware Tactile Attention (EATA), which restricts tactile conditioning to actions nearing execution, thereby reducing the temporal mismatch between tactile acquisition and action execution. Across six simulated UniVTAC tasks and three real-world contact-rich manipulation tasks, TacForcing achieves average success rates of 65% and 69%, respectively, outperforming strong baselines in both settings.",
  "published": "2026-08-26",
  "updated": "2026-08-26",
  "year": "2026",
  "authors": [
   "Jianbo Zhou",
   "Boyuan Zhao",
   "Yuzheng Zhang",
   "Yiyang Chen",
   "Wenxin Chen",
   "Qiuyue Li",
   "Xiangyang Gu",
   "Yuhan Cao",
   "Xiao Xia",
   "Yanzhe Hu",
   "Zhijie Deng"
  ],
  "author_count": 11,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "15 pages, 6 figures",
  "topics": [
   "vla",
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.25798v1",
  "pdf_url": "https://arxiv.org/pdf/2608.25798v1",
  "html_url": "https://arxiv.org/html/2608.25798v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.25757",
  "slug": "lm-x-explainable-action-modeling-with-progress-event-and-uncertainty-p",
  "title": "LM-X: Explainable Action Modeling with Progress, Event, and Uncertainty Prediction for Generalist Robot Manipulation",
  "abstract": "Generalist vision--language--action (VLA) policies learn long-horizon behavior mainly through short-horizon action prediction and reveal little beyond sampled commands. This creates two coupled bottlenecks: a single action target must implicitly absorb task progress, intermediate intent, and local reliability, while these control states remain hidden during execution. Inspired by functional principles of biological sensorimotor control, we introduce LM-X , which organizes prediction across task, event, and motor scales without claiming anatomical correspondence. Three explicitly supervised signals are emitted online and directly condition action generation: return-to-go (RTG) measures visible task progress, event-to-go (ETG) identifies the next semantic transition, and heteroscedastic action flow estimates local reliability through propagated variance. Explanation is therefore intrinsic to control rather than generated post hoc. Before a costly 20-day pretraining run on 64 NVIDIA B200 GPUs, a controlled five-task pretraining gate verifies the design: the complete model improves success by 16.0 points over the action-only backbone and by 10.8 points over the strongest single-head variant. We then train LM-X on more than 20,000 hours of real-robot trajectories, including over 1,000 hours of failed policy rollouts. LM-X achieves 74.1\\% across 50 randomized-hard RoboTwin2.0 tasks versus 55.4\\% for GR00T N1.7, and 68.6\\% versus 50.7\\% across seven real-robot tasks. RTG tracks semantic progress and visible regression, while variance rises during hesitation and oscillatory control. These results show that explicit multi-timescale predictive state can strengthen control while exposing interpretable internal estimates.",
  "published": "2026-08-26",
  "updated": "2026-08-27",
  "year": "2026",
  "authors": [
   "Jin Lou",
   "Zhiyuan Jing",
   "Andong Chen",
   "Xupeng Wang",
   "Yuan Xu",
   "Yuexuan Li",
   "Xingdong Zhu",
   "Zhijie Zhu",
   "Yingwei Ji",
   "Wenpeng Nie",
   "Yufei Liu",
   "Boyang Xing",
   "Lei Jiang",
   "Yan Cui",
   "Ying Chu",
   "Jingxuan Zhu",
   "Jingyi Li",
   "Liangliang Chen",
   "Jinyan Liu",
   "Zhiqi Song",
   "Jidong Zhang",
   "Hongming Li",
   "Yuchen Zhu"
  ],
  "author_count": 23,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "",
  "topics": [
   "foundation-pretraining"
  ],
  "orgs": [
   "NVIDIA"
  ],
  "abs_url": "https://arxiv.org/abs/2608.25757v2",
  "pdf_url": "https://arxiv.org/pdf/2608.25757v2",
  "html_url": "https://arxiv.org/html/2608.25757v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2608.25690",
  "slug": "trust-aware-sequential-decision-making-and-rollout-planning-for-resili",
  "title": "Trust-Aware Sequential Decision Making and Rollout Planning for Resilient Multi-Robot Systems",
  "abstract": "Sequential decision-making in multi-robot systems typically assumes that planning information is reliable and that agents execute the actions anticipated by the planner. Compromised agents can violate both assumptions, creating a mismatch between the planning model and physical execution. We study this problem in online multi-robot routing under localization spoofing. We introduce a distance-constrained spoofing model for monitor-aware adversaries, together with a tiered bipartite matching strategy that maximizes assignment influence while limiting spoofing magnitude. To mitigate such attacks, we develop a trust-aware monitor that combines probabilistic localization trust, calibrated using real GPS spoofing data, with behavioral evidence from task execution to classify agents and remove detected adversaries from subsequent planning. We further show that undetected adversaries can cause rollout to lose its expected cost-improvement behavior by violating planner-execution consistency. Trust-aware removal restores this consistency after detection, enabling stable routing and recovery of rollout's empirical advantage over the base policy. Experiments using real GPS spoofing datasets and San Francisco taxicab demand demonstrate effective detection and resilient routing across varying spoofing capabilities, adversarial fleet sizes, adaptive attacks, monitoring configurations, and rollout horizons.",
  "published": "2026-08-26",
  "updated": "2026-08-26",
  "year": "2026",
  "authors": [
   "Roee M. Francos",
   "Daniel Garces",
   "Orhan Eren Akg\u00fcn",
   "Nathaniel D. Bastian",
   "Stephanie Gil"
  ],
  "author_count": 5,
  "categories": [
   "cs.MA",
   "cs.RO"
  ],
  "primary_category": "cs.MA",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "20 pages, 17 figures",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.25690v1",
  "pdf_url": "https://arxiv.org/pdf/2608.25690v1",
  "html_url": "https://arxiv.org/html/2608.25690v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.25674",
  "slug": "opportunities-of-self-supervised-learning-for-gnss-evaluation-of-a-dee",
  "title": "Opportunities of Self Supervised Learning for GNSS: Evaluation of a Deep Learning-Enhanced PVT Algorithm",
  "abstract": "This work proposes a Deep Learning Enhanced PVT algorithm to mitigate multipath interference in dense urban areas. A supervised objective jointly predicts range corrections and uncertainty, while a JEPA-based self-supervised pretraining stage improves representation quality. The algorithm is evaluated over diverse driving scenarios, substantially improving PVT accuracy, particularly for unseen harsh urban conditions. These results highlight the potential of unlabelled GNSS data to improve generalization performance.",
  "published": "2026-08-26",
  "updated": "2026-08-26",
  "year": "2026",
  "authors": [
   "Thomas Barbero",
   "Bertrand Ekambi"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "European Navigation Conference 2026",
  "topics": [
   "world-models",
   "navigation",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.25674v1",
  "pdf_url": "https://arxiv.org/pdf/2608.25674v1",
  "html_url": "https://arxiv.org/html/2608.25674v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.25666",
  "slug": "prism-projection-integrated-sampling-based-mpc-with-bayesian-cost-tuni",
  "title": "PRISM: Projection-Integrated Sampling-Based MPC with Bayesian Cost Tuning for Bimanual Manipulation",
  "abstract": "Bimanual manipulation in cluttered, contact-rich environments remains challenging because it requires coordinated motion generation, interaction-aware planning, and reliable execution under tight kinematic constraints. We present PRISM, a projection-integrated sampling-based Model Predictive Control (MPC) framework that uses a GPU-accelerated physics simulator as an online world model for complex dual-arm manipulation. The main algorithmic contribution is a QP-guided control sampling strategy that decouples trajectory exploration from kinematic feasibility. At each MPC step, sampled joint-velocity trajectories are projected onto the set of motions satisfying joint position, velocity, acceleration, and jerk bounds, together with an initial-velocity boundary condition, before rollout evaluation. This enables broad yet feasible exploration of coordinated bimanual behaviors. To support efficient online execution, we derive a custom ADMM/Bregman-splitting QP solver that exploits joint-wise separability and reusable matrix factorizations. We further use Bayesian optimization to tune task-cost weights offline, reducing manual parameter selection. We evaluate PRISM on challenging variants of PerAct$^{2}$ tasks, including obstacle-constrained ball transport, tray transport, cube handover, and box lifting. Experiments show improved robustness and task success relative to representative sampling-based baselines, while maintaining real-time or near-real-time execution. We also demonstrate successful sim-to-real transfer on dual UR5e manipulators, highlighting the practical potential of physics-based online planning for contact-rich bimanual manipulation. Project details, including code and supplementary videos, are available at \\href{https://sites.google.com/view/prismbimanual}{\\texttt{https://sites.google.com/view/prismbimanual}}.",
  "published": "2026-08-26",
  "updated": "2026-08-26",
  "year": "2026",
  "authors": [
   "Alinjar Dan",
   "Iryna Hurova",
   "Karl Kruusam\u00e4e",
   "Arun Kumar Singh"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "",
  "topics": [
   "world-models",
   "tactile",
   "sim2real",
   "rl-control",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.25666v1",
  "pdf_url": "https://arxiv.org/pdf/2608.25666v1",
  "html_url": "https://arxiv.org/html/2608.25666v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.25659",
  "slug": "gaussiandream-efficient-3d-gaussian-world-modeling-for-robotic-manipul",
  "title": "GaussianDream++: Efficient 3D Gaussian World Modeling for Robotic Manipulation",
  "abstract": "Vision-Language-Action (VLA) policies have advanced language-conditioned robotic manipulation, yet action-imitation objectives provide only weak supervision for metric 3D structure and short-horizon physical evolution. Geometry-enhanced policies mainly improve current-scene grounding, whereas predictive policies often model future dynamics in RGB or latent spaces and may incur substantial deployment cost. GaussianDream demonstrates that training-time current Gaussian reconstruction and future Gaussian prediction provide effective 3D supervision, but its dense VGGT/TGE-based prefix jointly carries state, dynamics, and action-conditioning information. We present \\textbf{\\methodname}, a compact, policy-native extension that inserts \\textbf{World State Tokens} and \\textbf{World Prediction Tokens} directly into the VLA backbone. A training-only \\textbf{World Representation Head} decodes these tokens into a Current World and coupled Future Prediction over shared Gaussian primitives, while static--dynamic factorization preserves persistent structure and focuses residual motion on interaction-relevant regions. At inference, the head, renderer, auxiliary objectives, and VGGT/TGE pathway are removed, leaving only 20 world tokens without online Gaussian decoding or rollout. \\method achieves \\textbf{98.6\\%} on LIBERO and \\textbf{87.8\\%} on LIBERO-Plus, with clear gains under Camera and Layout shifts. Real-robot experiments further improve average success from 29.2\\% to 52.5\\% over reproduced $\u03c0_{0.5}$ while maintaining efficient closed-loop control.",
  "published": "2026-08-26",
  "updated": "2026-08-26",
  "year": "2026",
  "authors": [
   "Yuqing Jiang",
   "Zijian Zhang",
   "Weitao Zhou",
   "Jiawei Wang",
   "Junjie He",
   "Lei Yang",
   "Haifang Qing",
   "Si Liu",
   "Ding Zhao",
   "Ping Luo",
   "Haibao Yu"
  ],
  "author_count": 11,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "17 pages, 4 figures",
  "topics": [
   "world-models",
   "vla"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.25659v1",
  "pdf_url": "https://arxiv.org/pdf/2608.25659v1",
  "html_url": "https://arxiv.org/html/2608.25659v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.25642",
  "slug": "egonav-bridging-learned-waypoints-and-geometry-aware-local-control-for",
  "title": "EgoNav: Bridging Learned Waypoints and Geometry-Aware Local Control for Robust Indoor Navigation",
  "abstract": "Image-goal navigation using lightweight topological maps is a practical paradigm for indoor robot deployment: the map requires only geotagged images, and localization relies on visual matching rather than precise pose estimation. However, learned waypoint predictors can produce targets that violate geometric constraints or deviate from the global path. Executing these waypoints safely further requires a local planner capable of collision avoidance, yet existing systems either lack one or rely on fixed parameters that cannot adapt to confined spaces. To address these limitations while retaining the navigational intuition of the learned predictor, we present EgoNav, a hierarchical system that implements this idea by generating candidates from semantically segmented traversable regions and scoring them alongside the learned waypoint for geometric safety, directional coherence, and fidelity to the learned prior. An adaptive local path planner then executes the refined waypoint with parameters modulated based on the refinement outcome. Experiments in Habitat-sim and on a physical humanoid robot show that EgoNav consistently outperforms contemporary baselines in both success rate and path efficiency.",
  "published": "2026-08-26",
  "updated": "2026-08-26",
  "year": "2026",
  "authors": [
   "Jing Wang",
   "Shiqi Zhao",
   "Hairong Qu",
   "Peng Yin"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "EgoNav is a hierarchical system that implements EgoNav, a hierarchical system that implements this idea by generating candidates from semantically segmented traversable regions and scoring them alongside the learned waypoint for geometric safety, directional coherence, and fidelity to the learned prior.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jing Wang",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Shi-Qi Zhao",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Hai-Rong Qu",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Peng Yin",
    "id": null,
    "h_index": 0,
    "papers": 0
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.25642v1",
  "pdf_url": "https://arxiv.org/pdf/2608.25642v1",
  "html_url": "https://arxiv.org/html/2608.25642v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.25641",
  "slug": "leveraging-inter-object-affordances-for-efficient-planning-in-contact",
  "title": "Leveraging Inter-object Affordances for Efficient Planning in Contact-rich Tasks",
  "abstract": "Traditional task-and-motion planning (TAMP) approaches primarily focus on defining sequences of actions along with the necessary geometric and kinematic constraints to execute long-horizon tasks. However, their applicability in real-world settings is limited, as they typically assume simplified object models that overlook key physical properties critical for the successful execution of contact-rich tasks. Moreover, they often use sub-symbolic reasoning during motion planning, which drastically increases planning time and decreases overall success rates. We propose a method that leverages a TAMP approach, defining object-centric abstractions of execution constraints, called Unified TAMP (U-TAMP), to execute robotic tasks involving interactions among objects with heterogeneous shapes, sizes, and materials. Using a Vision-Language Model (VLM), we generate abstractions of inter-object affordances for characterizing physical interaction constraints between objects in contact-rich tasks, such as grasp and support constraints. These constraints are used to enrich the U-TAMP planning domain to deal with objects with variable physical properties. We perform experiments in simulated kitchen table organization scenarios and compare our results with those of the original U-TAMP, as well as a state-of-the-art VLM-based planner that leverages common sense knowledge of objects' affordances for plan generation. Our approach achieves significantly higher planning success rates and improves planning times by one to two orders of magnitude compared to other methods.",
  "published": "2026-08-26",
  "updated": "2026-08-26",
  "year": "2026",
  "authors": [
   "Pouya P. Niaz",
   "Justus Piater",
   "Alejandro Agostini"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes a method that leverages a TAMP approach, defining object-centric abstractions of execution constraints, called Unified TAMP (U-TAMP), to execute robotic tasks involving interactions among objects with heterogeneous shapes, sizes, and materials.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Pouya P. Niaz",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Justus Piater",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Alejandro Agostini",
    "id": null,
    "h_index": 0,
    "papers": 0
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.25641v1",
  "pdf_url": "https://arxiv.org/pdf/2608.25641v1",
  "html_url": "https://arxiv.org/html/2608.25641v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.25610",
  "slug": "advantage-driven-explicit-memory-for-social-navigation",
  "title": "Advantage-Driven Explicit Memory for Social Navigation",
  "abstract": "Robot policies are predominantly learned with classical parametric variants of imitation learning or RL, where training stores the agent's behavior exclusively in the policy's network parameters, putting a heavy burden on the representation learning algorithm. We propose a new navigation agent equipped with non-parametric memory which explicitly indexes prior steps leading to critical events. The advantages are twofold: first, it allows the policy to outsource some of its behavior into an explicit memory; second, it encourages a form of continual learning by allowing an agent to collect data from its testing episodes during deployment and therefore to better generalize to OOD situations. In the context of social navigation, we show that this improves the agent's capability to retain sparse, high-cost failures, such as human collisions. If the policy is trained in simulation, this also naturally addresses the sim-to-real gap, partially, by basing some of the decision making on real data. We integrate the explicit memory into a recurrent PPO architecture and use hidden states for memory retrieval to capture continuous spatiotemporal dynamics. The goal of exploiting rare, high-impact events is achieved by leveraging the RL agent's advantage signals. We train our agent in simulation with a combination of photorealistic rendering and non-visual crowd simulation and show that the agent is robust with respect to OOD social behavior.",
  "published": "2026-08-26",
  "updated": "2026-08-26",
  "year": "2026",
  "authors": [
   "Yeonsoo Park",
   "Mattia Racca",
   "Guillaume Bono",
   "Steeven Janny",
   "Gianluca Monaci",
   "Tomi Silander",
   "Christian Wolf"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A new navigation agent equipped with non-parametric memory which explicitly indexes prior steps leading to critical events is proposed, and is trained in simulation with a combination of photorealistic rendering and non-visual crowd simulation and is robust with respect to OOD social behavior.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yeonsoo Park",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Mattia Racca",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Guillaume Bono",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Steeven Janny",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Gianluca Monaci",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Tomi Silander",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Christian Wolf",
    "id": null,
    "h_index": 0,
    "papers": 0
   }
  ],
  "comment": "",
  "topics": [
   "sim2real",
   "imitation-diffusion",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.25610v1",
  "pdf_url": "https://arxiv.org/pdf/2608.25610v1",
  "html_url": "https://arxiv.org/html/2608.25610v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.25585",
  "slug": "ra-vla-retrieval-augmented-vla-for-test-time-adaptation",
  "title": "RA-VLA: Retrieval-Augmented VLA for Test-Time Adaptation",
  "abstract": "Vision-Language-Action (VLA) models provide a versatile foundation for general robotic manipulation, yet they exhibit significant brittleness when confronted with novel task distributions. While In-Context Imitation Learning (ICIL) offers a training-free alternative, existing frameworks suffer from an adaptation bottleneck that hinders the effective translation of expert context to executable actions. This failure originates from superficial retrieval mechanisms and an inherent behavioral inertia that anchors the policy to its pre-trained priors. To address these limitations, we present RA-VLA, a retrieval-augmented VLA framework that integrates behavior-aligned context retrieval with a grounded execution pipeline. By enforcing faithful adherence to functional cues within a scalable architecture, RA-VLA facilitates seamless task adaptation while preserving inference efficiency. Our empirical evaluations across the LIBERO benchmark and a real-world UR5e environment demonstrate that RA-VLA achieves superior success rates and computational efficiency, establishing a robust framework for training-free robotic adaptation.",
  "published": "2026-08-26",
  "updated": "2026-08-26",
  "year": "2026",
  "authors": [
   "Sanghwan Jang",
   "Minjin Jeon",
   "Minsoo Kim",
   "Seongjin Choi",
   "Dongha Kim",
   "Hwanjo Yu"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "ICML 2026",
  "venue_source": "arxiv-comment",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "RA-VLA is presented, a retrieval-augmented VLA framework that integrates behavior-aligned context retrieval with a grounded execution pipeline that facilitates seamless task adaptation while preserving inference efficiency.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Sanghwan Jang",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Minjin Jeon",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Minsoo Kim",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Seongjin Choi",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Dongha Kim",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Hwanjo Yu",
    "id": null,
    "h_index": 0,
    "papers": 0
   }
  ],
  "comment": "ICML 2026. Contact: s.jang@postech.ac.kr",
  "topics": [
   "vla",
   "imitation-diffusion"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.25585v1",
  "pdf_url": "https://arxiv.org/pdf/2608.25585v1",
  "html_url": "https://arxiv.org/html/2608.25585v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.8
 },
 {
  "id": "2608.25572",
  "slug": "confal-wm-confidence-guided-active-learning-for-action-conditioned-wor",
  "title": "ConfAL-WM: Confidence-Guided Active Learning for Action-Conditioned World Models",
  "abstract": "Action-conditioned world models have become an important foundation for embodied prediction, planning, and synthetic data generation, but their errors under new task and scene distributions are often concentrated in localized spatiotemporal regions such as robot arms, manipulated objects, contact areas, and occluded objects. This paper presents ConfAL-WM, a confidence-guided active learning framework for post-training embodied world models. Built upon EVAC, we attach a lightweight confidence probe to UNet decoder features and predict dense confidence maps in the latent space. These maps are aggregated into task-, frame-, and patch-level scores, enabling both efficient data selection and localized training enhancement. Our pipeline first retrains the confidence probe and warms up EVAC with a small subset of target-domain data, then performs task-level prescreening to allocate sampling budgets, and finally applies selected-data retraining with optional frame or patch weighted data enhancement. Experiments on RoboTwin2.0 show that confidence-guided selection improves post-training efficiency, while dense frame and patch weighting further enhances prediction quality and embodied trajectory consistency compared with scalar reward, progress, and judge-based scoring baselines. A quick visual overview of this work is available at https://ConfAL-WM.github.io.",
  "published": "2026-08-26",
  "updated": "2026-08-26",
  "year": "2026",
  "authors": [
   "Xiang Liu",
   "Sen Cui",
   "Changshui Zhang"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Experiments on RoboTwin2.0 show that confidence-guided selection improves post-training efficiency, while dense frame and patch weighting further enhances prediction quality and embodied trajectory consistency compared with scalar reward, progress, and judge-based scoring baselines.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xiang Liu",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Sen Cui",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Chang-Shui Zhang",
    "id": null,
    "h_index": 0,
    "papers": 0
   }
  ],
  "comment": "Project page: https://ConfAL-WM.github.io",
  "topics": [
   "world-models"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.25572v1",
  "pdf_url": "https://arxiv.org/pdf/2608.25572v1",
  "html_url": "https://arxiv.org/html/2608.25572v1",
  "code_url": "https://ConfAL-WM.github.io",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2608.25547",
  "slug": "a-tendon-driven-five-fingered-hand-with-distributed-tactile-perception",
  "title": "A Tendon-Driven Five-Fingered Hand with Distributed Tactile Perception for Dexterous Manipulation",
  "abstract": "To apply the techniques of embodied artificial intelligence to human-oid robots for complex manipulations, dexterous robotic hands are indispensable, which are restricted by the dexterity and tactile perception capability. In this work, we proposed a novel design of tendon-driven five-fingered hand with dis-tributed tactile perception. With a soft-rigid-hybrid structure employed, both compliance and operational force are endowed to the hand. Dual-modality tactile sensing elements are distributed on the distal and middle phalanges of all five fingers, enabling the simultaneous detection of static contact and dynamic force variations. Manipulation experiments, including counting gestures, finger-to-thumb pinching, object grasping, and bottle-grasp tactile recording, demonstrate the feasibility of the integrated actuation-perception system.",
  "published": "2026-08-26",
  "updated": "2026-08-26",
  "year": "2026",
  "authors": [
   "Huayang Chen",
   "Longhui Qin"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A novel design of tendon-driven five-fingered hand with dis-tributed tactile perception is proposed, with a soft-rigid-hybrid structure employed, and both compliance and operational force are endowed to the hand.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hua-Yang Chen",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Long-Hui Qin",
    "id": null,
    "h_index": 0,
    "papers": 0
   }
  ],
  "comment": "Accepted by International Conference on Service Robotics (ICoSR) 2026",
  "topics": [
   "dexterous-manipulation",
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.25547v1",
  "pdf_url": "https://arxiv.org/pdf/2608.25547v1",
  "html_url": "https://arxiv.org/html/2608.25547v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.25509",
  "slug": "dynamic-modeling-of-a-welding-torch-umbilical-and-its-impact-on-robot",
  "title": "Dynamic Modeling of a Welding Torch Umbilical and Its Impact on Robot Dynamics",
  "abstract": "Robotic welding is widely used in industrial manufacturing, where the welding torch is often connected to the generator through an external umbilical. With the increasing deployment of lightweight and collaborative robots, the dynamic influence of this umbilical can significantly affect the robot motion and the actuation forces. This paper proposes a constrained multibody dynamic model of a welding umbilical, represented as a serial chain of rigid bodies interconnected by passive joints with elastic and dissipative effects. Prescribed motions at the distal anchor point are introduced through holonomic kinematic constraints. The equations of motion are reduced by projecting the dynamics onto the subspace of admissible velocities, yielding an efficient formulation free of Lagrange multipliers. The reaction wrench exerted by the umbilical on the robot is explicitly recovered. A planar case study illustrates the approach.",
  "published": "2026-08-26",
  "updated": "2026-08-26",
  "year": "2026",
  "authors": [
   "Nicolas Gautier",
   "Yves Guillermit",
   "Mathieu Porez",
   "Fabien Rousset",
   "Damien Chablat"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Nicolas Gautier",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Yves Guillermit",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Mathieu Porez",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Fabien Rousset",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Damien Chablat",
    "id": null,
    "h_index": 0,
    "papers": 0
   }
  ],
  "comment": "",
  "topics": [
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.25509v1",
  "pdf_url": "https://arxiv.org/pdf/2608.25509v1",
  "html_url": "https://arxiv.org/html/2608.25509v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.25470",
  "slug": "transient-multimode-heat-transfer-of-an-industrial-automated-tape-layi",
  "title": "Transient multimode heat transfer of an industrial automated tape laying process under rapidly changing conditions",
  "abstract": "This work presents a transient heat-transfer model of an industrial automated tape laying (ATL) process designed to overcome the limitations of conventional thermal models in composite manufacturing. The model solves the heat-conduction equation with coupled advection, conduction, convection, and radiation. A key innovation is the implementation of an analytical view factor approach that accounts for finite emitter and tape widths, thereby correcting systematic overestimations of radiative heat flux inherent in 1.5D simplifications. Furthermore, a local convection assessment incorporates mixed convection effects characterized by the Richardson number, ensuring accuracy across a wide range of process speeds. The ATL system is represented by two interacting subsystems: the moving tape substrate and the infrared heat sources. The tape is discretized using a two-node model that resolves the physical phase shift between the heated and monitored surfaces. Numerical stability under high dynamics is ensured by a monolithic solution strategy using a high-order implicit integration scheme. Model predictions were validated on an industrial ATL line, demonstrating an overall deviation of only 1.08% (NRMSE) under rapid velocity and current modulations. This framework provides a high-fidelity, physics-based foundation for thermal state estimation, supporting consistent in-situ consolidation and improved part quality.",
  "published": "2026-08-26",
  "updated": "2026-08-26",
  "year": "2026",
  "authors": [
   "Bernhard Rameder",
   "Hubert Gattringer",
   "Andreas M\u00fcller",
   "Ronald Naderer"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "10.1016/j.compositesa.2026.110162",
  "oa_pdf": "https://doi.org/10.1016/j.compositesa.2026.110162",
  "s2_authors": [
   {
    "name": "Bernhard Rameder",
    "id": "2387215781",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "H. Gattringer",
    "id": "1770609",
    "h_index": 18,
    "papers": 175
   },
   {
    "name": "A. Mueller",
    "id": "30718827",
    "h_index": 29,
    "papers": 256
   },
   {
    "name": "Ronald Naderer",
    "id": "3335053",
    "h_index": 4,
    "papers": 18
   }
  ],
  "comment": "20 pages",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.25470v1",
  "pdf_url": "https://arxiv.org/pdf/2608.25470v1",
  "html_url": "https://arxiv.org/html/2608.25470v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.25459",
  "slug": "towards-safe-and-optimal-flight-viability-kernel-mpc-for-fully-actuate",
  "title": "Towards safe and optimal flight: Viability Kernel MPC for Fully Actuated Multirotor",
  "abstract": "Industrial aerial robotics demands safety guarantees for navigation in unstructured environments while optimizing performance and computational efficiency. This paper presents a method for generating safe pose trajectories for fully actuated multirotors within a Model Predictive Control (MPC) framework, leveraging both viability theory and data-driven methods. Obstacle avoidance is enforced through dynamically computed axis-aligned bounding boxes, providing formal safety guarantees without exhaustive offline reachability analysis. Numerical simulations on a fully actuated tilted hexarotor validate the approach, demonstrating successful navigation in cluttered environments with real-time computational performance.",
  "published": "2026-08-26",
  "updated": "2026-08-26",
  "year": "2026",
  "authors": [
   "Massimiliano Bertoni",
   "Alberto Piccina",
   "Gianni Lunardi",
   "Elias Fontanari",
   "Andrea Del Prete",
   "Angelo Cenedese",
   "Giulia Michieletto"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A method for generating safe pose trajectories for fully actuated multirotors within a Model Predictive Control (MPC) framework, leveraging both viability theory and data-driven methods is presented.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Massimiliano Bertoni",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Alberto Piccina",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Gianni Lunardi",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Elias Fontanari",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Andrea Del Prete",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Angelo Cenedese",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Giulia Michieletto",
    "id": null,
    "h_index": 0,
    "papers": 0
   }
  ],
  "comment": "",
  "topics": [
   "rl-control",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.25459v1",
  "pdf_url": "https://arxiv.org/pdf/2608.25459v1",
  "html_url": "https://arxiv.org/html/2608.25459v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.25435",
  "slug": "saliency-depth-conditioning-for-zero-shot-segmentation-of-communicatio",
  "title": "Saliency-Depth Conditioning for Zero-Shot Segmentation of Communication-Tower Components in Cluttered UAV Imagery",
  "abstract": "Fine-grained segmentation of communication-tower components in UAV imagery is essential for automated inspection, yet task-specific models are hard to develop due to limited instance-level annotations. Zero-shot segmentation models offer a promising alternative, but in cluttered scenes, visually similar background structures interfere with component localization, causing missed instances and false positives. We propose a model-agnostic saliency-depth foreground-conditioning strategy combining appearance-based saliency with monocular relative depth to construct a coarse tower prior and suppress irrelevant content. We integrate this module with Grounded-SAM and SAM 3, yielding SD-Grounded-SAM and SD-SAM 3. SD-Grounded-SAM further applies geometric and depth-aware box refinement before mask generation, while SD-SAM 3 relies on SAM 3's internal setup. On TOW-300, a dataset of 340 communication-tower UAV images, our strategy improves both baselines: SD-SAM 3 achieves the strongest instance-segmentation performance, while SD-Grounded-SAM produces fewer false positives. Ablations confirm complementary gains from saliency, depth, and box refinement, improving robustness in cluttered scenes.",
  "published": "2026-08-26",
  "updated": "2026-08-26",
  "year": "2026",
  "authors": [
   "Ali Lesani",
   "Chul Min Yeum",
   "Su-Min Kang"
  ],
  "author_count": 3,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A model-agnostic saliency-depth foreground-conditioning strategy combining appearance-based saliency with monocular relative depth to construct a coarse tower prior and suppress irrelevant content is proposed, improving robustness in cluttered scenes.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ali Lesani",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Chul Min Yeum",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Su-Min Kang",
    "id": null,
    "h_index": 0,
    "papers": 0
   }
  ],
  "comment": "",
  "topics": [
   "navigation",
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.25435v1",
  "pdf_url": "https://arxiv.org/pdf/2608.25435v1",
  "html_url": "https://arxiv.org/html/2608.25435v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.25427",
  "slug": "super-odometry-2-0-resilient-odometry-via-hierarchical-adaptation",
  "title": "SUPER ODOMETRY 2.0: Resilient Odometry via Hierarchical Adaptation",
  "abstract": "Resilient and robust odometry is crucial for autonomous systems operating in complex and dynamic environments. Existing odometry systems often struggle with severe sensory degradations and extreme conditions such as smoke, sandstorms, snow, or low-light conditions, threatening both the safety and functionality of robots. To address these challenges, we present Super Odometry, a sensor fusion framework that dynamically adapts to varying levels of environmental degradation. Super Odometry employs a hierarchical structure to integrate four core modules from lower-level to higher-level adaptability including adaptive feature selection, adaptive state direction selection, adaptive engine selection, and a novel learning- based inertial odometry. The inertial odometry, trained on over 100 hours of heterogeneous robotic platforms, captures comprehensive motion dynamics. Super Odometry elevates the inertial measurement unit (IMU) to equal importance with camera and LiDAR within the sensor fusion framework, providing a reliable fallback when exteroceptive sensors fail. Super Odometry has been validated across 200 kilometers and 800 operational hours on a fleet of aerial, wheeled, and legged robots, under diverse sensor configurations, environmental degradation, and aggressive motion profiles. It marks an important step towards safe and long-term robotic autonomy in all-degraded environments.",
  "published": "2026-08-26",
  "updated": "2026-08-26",
  "year": "2026",
  "authors": [
   "Shibo Zhao",
   "Sifan Zhou",
   "Yuchen Zhang",
   "Ji Zhang",
   "Chen Wang",
   "Wenshan Wang",
   "Sebastian Scherer"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "Sci. Robotics",
  "venue_source": "semantic-scholar",
  "citations": 12,
  "influential_citations": 0,
  "tldr": "Super Odometry elevates the inertial measurement unit to equal importance with camera and light detection and ranging (LiDAR) systems in the sensor fusion framework, providing a reliable fallback when exteroceptive sensors fail.",
  "doi": "10.1126/scirobotics.adv1818",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Shibo Zhao",
    "id": "2278582891",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Sifan Zhou",
    "id": "2372419438",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Yu-Chao Zhang",
    "id": "2129523498",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Ji Zhang",
    "id": "2397810256",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Chen Wang",
    "id": "2293357347",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Wenshan Wang",
    "id": "2296220021",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Sebastian Scherer",
    "id": "2333174025",
    "h_index": 1,
    "papers": 4
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.25427v1",
  "pdf_url": "https://arxiv.org/pdf/2608.25427v1",
  "html_url": "https://arxiv.org/html/2608.25427v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.61
 },
 {
  "id": "2608.25405",
  "slug": "lac-linear-and-angular-compliance-for-humanoid-whole-body-control",
  "title": "LAC: Linear and Angular Compliance for Humanoid Whole-body Control",
  "abstract": "Real-world humanoid tasks involve physical interaction with objects and humans, yet current controllers either reject external forces as disturbances or restrict compliance to limited body links while ignoring angular effects. We present LAC, a general whole-body controller that simultaneously realizes commanded Linear and Angular Compliance for wrenches applied to the upper body. First, we synthesize whole-body compliant responses into a large-scale augmented dataset. Sampled force and couple events are imposed on contact frames extracted from human interaction data. At each contact link, the external force and a virtual torque from the passively yielding kinematic chain drive a virtual admittance under the commanded stiffness. Subsequently, teacher-student reinforcement learning trains a single policy to track the compliant motions under external wrenches. Finally, extensive simulation and real-world experiments demonstrate whole-body compliant responses to wrenches across the upper body, monotonic modulation over the full range of both stiffness commands, and applicability to teleoperated loco-manipulation tasks. Project website: https://lac-humanoid.github.io/",
  "published": "2026-08-26",
  "updated": "2026-08-26",
  "year": "2026",
  "authors": [
   "Yang Liu",
   "Zhongkai Gu",
   "Wei Zhu",
   "Mitsuhiro Hayashibe"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "LAC is presented, a general whole-body controller that simultaneously realizes commanded Linear and Angular Compliance for wrenches applied to the upper body and monotonic modulation over the full range of both stiffness commands, and applicability to teleoperated loco-manipulation tasks.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yang Liu",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Zhong-Kai Gu",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Wei Zhu",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Mitsuhiro Hayashibe",
    "id": null,
    "h_index": 0,
    "papers": 0
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.25405v1",
  "pdf_url": "https://arxiv.org/pdf/2608.25405v1",
  "html_url": "https://arxiv.org/html/2608.25405v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.25395",
  "slug": "a-taxonomy-of-construction-task-activities-for-robot-workers",
  "title": "A Taxonomy of Construction Task Activities for Robot Workers",
  "abstract": "Recent vision-language-action models offer a path toward robots with broader repertoires than conventional task-specific systems. Construction deployment, however, requires a precise inventory of worker activities and the capabilities needed to execute them. We present TARCAT, an occupation-grounded taxonomy derived from 91 O*NET tasks across seven high-employment construction occupations and 30 instructional videos of physical work. TARCAT defines 41 action primitives in 12 groups and three classes and provides a mechanism for composing parameterized primitive sequences into reusable skills. This human-interpretable structure can organize demonstrations, specify robot requirements, and support coding agents that retrieve and extend skill libraries. We also demonstrate selected primitives on a DOBOT CR3 arm with a CRAFT hand. TARCAT thereby provides a common vocabulary for analyzing human work and developing general-purpose construction robots. Annotations are available at https://github.com/AICPS/TARCAT-Taxonomy.",
  "published": "2026-08-26",
  "updated": "2026-08-26",
  "year": "2026",
  "authors": [
   "Sadman Sakib",
   "Zhangyi None Peng",
   "Yujie Pang",
   "Yu Otsuki",
   "Mohammad Abdullah Al Faruque"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "TARCAT is presented, an occupation-grounded taxonomy derived from 91 O*NET tasks across seven high-employment construction occupations and 30 instructional videos of physical work that provides a common vocabulary for analyzing human work and developing general-purpose construction robots.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Sadman Sakib",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Zhangyi None Peng",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Yu-Jie Pang",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Yu Otsuki",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Mohammad Abdullah Al Faruque",
    "id": null,
    "h_index": 0,
    "papers": 0
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.25395v1",
  "pdf_url": "https://arxiv.org/pdf/2608.25395v1",
  "html_url": "https://arxiv.org/html/2608.25395v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.25366",
  "slug": "raem-robust-autonomous-exploration-for-multi-floor-environments-with-a",
  "title": "RAEM: Robust Autonomous Exploration for Multi-Floor Environments with a Quadruped Robot",
  "abstract": "In this paper, we propose RAEM, a robust autonomous exploration framework for quadruped robots operating in multi-floor environments. Most existing ground-robot exploration approaches rely on planar traversability representations, which cannot adequately represent the overlapping structures and cross-floor connectivity of multi-floor buildings. Although tomography-based representations provide effective traversability modeling for multi-floor navigation, maintaining a global tomography map incurs substantial computational overhead for online exploration with frequent replanning. Moreover, sparse and fragmented LiDAR observations in stairwells can degrade local traversability estimation, leading to irregular viewpoint placement and temporary topological disconnections. To address these challenges, RAEM adopts a hybrid local-global traversability representation, in which a local tomography map and an explicitly categorized local 3D grid map are used for online terrain analysis and connectivity evaluation, while an elevation-aware global topological graph is incrementally constructed from these local spatial representations for efficient cross-floor exploration planning. We further introduce a staircase center alignment strategy to reduce abrupt yaw variations during climbing and a dual path searching mechanism to recover guidance paths when the global topology is locally disconnected. Extensive simulation and real-world experiments demonstrate robust and computationally stable autonomous exploration across multi-floor structures, including continuous exploration of a five-floor stairwell.",
  "published": "2026-08-26",
  "updated": "2026-08-26",
  "year": "2026",
  "authors": [
   "Zikang Yuan",
   "Yuan Ren",
   "Yian Wang",
   "Yixue Wang",
   "Enze Fang",
   "Xuewei Zhang",
   "Junda Cheng",
   "Chi Chen",
   "Chin-Pang Ho",
   "Lijun Zhu",
   "Shaohang Xu",
   "Kwang-Ting Cheng",
   "Xin Yang"
  ],
  "author_count": 13,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "21 pages, 23 figures",
  "topics": [
   "humanoids",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.25366v1",
  "pdf_url": "https://arxiv.org/pdf/2608.25366v1",
  "html_url": "https://arxiv.org/html/2608.25366v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.25350",
  "slug": "beyond-pairwise-feedback-listwise-vision-language-supervision-for-pref",
  "title": "Beyond Pairwise Feedback: Listwise Vision-Language Supervision for Preference-Based Reward Learning",
  "abstract": "Vision-language models (VLMs) have emerged as a powerful source of supervision for reinforcement learning, enabling agents to leverage rich semantic knowledge during training. Inspired by the success of preference-based reward learning (PbRL) in reinforcement learning from human feedback (RLHF), vision-language model generated image-based preferences provide an effective source for learning reward functions. This can be done by visually comparing two outcomes through the Bradley-Terry (BT) model. However, this pairwise formulation utilizes only two observations at a time, despite VLMs being capable of ranking multiple candidates. The Plackett-Luce (PL) formulation can shape a reward model with listwise rankings as opposed to pairwise preferences, allowing for a more suited use of a VLM based ranking. In this work, to our knowledge, we introduce the first framework that combines VLM-generated preferences with the Plackett-Luce model for reward learning. We evaluate our approach on Meta-World manipulation tasks and show that Plackett-Luce (PL) reward models can train robotic policies from VLM-generated rankings as effectively as pairwise Bradley-Terry, $K$-wise Bradley-Terry, and RL-VLM-F baselines. Across all environments, at least one PL ranking size ($K \\in \\{3,4,5\\}$) consistently performs with or outperforms other methods in mean success rate. Unlike pairwise methods, which are restricted to $K=2$, PL supports different ranking sizes and can therefore be adapted to the environment and desired feedback format. Our best PL configuration achieves an 86% mean final success rate and matches the Oracle baseline on Drawer Open. Overall, these results demonstrate that listwise VLM preference supervision is a competitive and flexible approach to reward learning for reinforcement learning.",
  "published": "2026-08-26",
  "updated": "2026-08-26",
  "year": "2026",
  "authors": [
   "Srivalli Katkuri",
   "Maxwell Kawada",
   "Juan Wachs"
  ],
  "author_count": 3,
  "categories": [
   "cs.LG",
   "cs.RO"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "It is shown that Plackett-Luce (PL) reward models can train robotic policies from VLM-generated rankings as effectively as pairwise Bradley-Terry, $K$-wise Bradley-Terry, and RL-VLM-F baselines and demonstrate that listwise VLM preference supervision is a competitive and flexible approach to reward learning for reinforcement learning.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Srivalli Katkuri",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Maxwell Kawada",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Juan Wachs",
    "id": null,
    "h_index": 0,
    "papers": 0
   }
  ],
  "comment": "13 pages, 10 figures. Srivalli Katkuri and Maxwell Kawada contributed equally to this work",
  "topics": [
   "rl-control",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.25350v1",
  "pdf_url": "https://arxiv.org/pdf/2608.25350v1",
  "html_url": "https://arxiv.org/html/2608.25350v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.25319",
  "slug": "sonicnudge-controlled-displacement-of-hovering-uavs-via-estimator-cont",
  "title": "SonicNudge: Controlled Displacement of Hovering UAVs via Estimator-Controller Coupling",
  "abstract": "UAV displacement attacks have traditionally relied on spoofing sensors that directly report position or translational motion, such as GNSS and optical flow. In this work, we introduce SonicNudge, a new attack primitive that instead targets the gyroscope and shows that low-level inertial errors can be transformed into controlled displacement of hovering or slow-moving UAVs. The attack exploits estimator--controller coupling: a small gyroscope perturbation by ultrasonic resonance can persist as an attitude-estimation bias, and the flight controller can convert this biased estimate into a shifted hover point. This behavior is especially relevant to UAV tasks that require hovering, station-keeping, slow approach, or precise final alignment, such as perimeter denial, inspection, docking, landing alignment, and close-proximity operation, where meter-scale position errors can be operationally meaningful. We analyze this attack primitive in a PX4-style flight stack and validate it through 81 simulation runs and more than 10 indoor/outdoor physical experiments, showing that displacement is governed by estimator weighting, bias observability, and closed-loop position correction. Our study suggests that UAV and vehicle-system security should look beyond direct navigation spoofing and pay closer attention to low-level inertial errors and estimator--controller coupling as a subtle but important attack surface.",
  "published": "2026-08-26",
  "updated": "2026-08-26",
  "year": "2026",
  "authors": [
   "Shaocheng Luo",
   "Ashir Raza",
   "Haocheng Meng",
   "David Hunt",
   "Miroslav Pajic"
  ],
  "author_count": 5,
  "categories": [
   "eess.SY",
   "cs.RO"
  ],
  "primary_category": "eess.SY",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "SonicNudge is introduced, a new attack primitive that instead targets the gyroscope and shows that low-level inertial errors can be transformed into controlled displacement of hovering or slow-moving UAVs and vehicle-system security should look beyond direct navigation spoofing and pay closer attention to low-level inertial errors and estimator--controller coupling.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Shao-Cheng Luo",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Ashir Raza",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Haocheng Meng",
    "id": "2295764492",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "David Hunt",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Miroslav Pajic",
    "id": null,
    "h_index": 0,
    "papers": 0
   }
  ],
  "comment": "",
  "topics": [
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.25319v1",
  "pdf_url": "https://arxiv.org/pdf/2608.25319v1",
  "html_url": "https://arxiv.org/html/2608.25319v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.25284",
  "slug": "generative-action-chunk-sampling-for-adaptive-stiffness-control-in-phy",
  "title": "Generative Action-Chunk Sampling for Adaptive Stiffness Control in Physical Human-Robot Collaboration",
  "abstract": "Physical human-robot collaboration requires a robot to provide assistance when human intention is clear while remaining compliant when several future motions are plausible. We present an adaptive stiffness framework based on generative action-chunk sampling. Conditioned on an RGB image and external joint-torque estimates, the policy samples multiple future action chunks from an observation-conditioned prior. Variation among the sampled action chunks is used to continuously adapt joint stiffness and damping. Greater variation makes the robot more compliant to facilitate human guidance, whereas lower variation provides firmer assistance. In a real-world collaborative transport task with four possible directions, the proposed method achieved an average success rate of 0.95, compared with 0.83 for a fixed-stiffness ablation and 0.69 for a deterministic baseline. Near direction determination, variation among the sampled action chunks increased and the controller accordingly reduced stiffness. These results suggest that variation among actions sampled by a generative policy can serve as an online control signal for balancing assistance and compliance in physical human-robot interaction.",
  "published": "2026-08-26",
  "updated": "2026-08-26",
  "year": "2026",
  "authors": [
   "Aoi Otake",
   "Ferdinand Hartmann",
   "Ko Igari",
   "Shingo Murata"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "It is suggested that variation among actions sampled by a generative policy can serve as an online control signal for balancing assistance and compliance in physical human-robot interaction.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Aoi Otake",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Ferdinand Hartmann",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Ko Igari",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Shingo Murata",
    "id": null,
    "h_index": 0,
    "papers": 0
   }
  ],
  "comment": "Preprint version",
  "topics": [
   "imitation-diffusion",
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.25284v1",
  "pdf_url": "https://arxiv.org/pdf/2608.25284v1",
  "html_url": "https://arxiv.org/html/2608.25284v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.25275",
  "slug": "phaseshift-topology-aware-data-harmonization-and-model-consolidation-a",
  "title": "PhaseShift: Topology-Aware Data Harmonization and Model Consolidation Across Signalized Intersections",
  "abstract": "Learned traffic-behavior models are commonly trained separately for each intersection, creating model portfolios that cannot share evidence across sites. We present PhaseShift, a topology-aware framework that harmonizes heterogeneous roadside trajectories into a shared actor-centric representation and trains one reusable backbone. Ego-relative coordinates, trajectory-induced movement paths, normalized signal context, and variable-cardinality interaction tokens remove site conventions while preserving behaviorally relevant topology. The backbone supports pooled operation, zero-shot at a held-out intersection, and low-data adaptation. We evaluate five intersections in two Florida regions on balanced field data, 100k training windows and equal-sized test sets per site under a replay-conditioned, best-of-sampled-trajectory protocol. At 10s, one pooled model lowers both minADE and minFDE relative to trained local models at all five sites, with median reductions of 36.8% and 22.0%. Leave-one-intersection-out deployment, including one cross-region fold, beats local training on both 10-s metrics at four of five sites, although short-horizon performance is less uniform. Fine-tuning with 1,000 target update windows improves on zero-shot at three sites and is the strongest regime at one. At site 7, every cross-site mixture sharply lowers long-horizon error under a fixed 100k-window budget; test-likelihood gains argue against a best-of-sample dispersion-only explanation. Local models fall behind calibrated IDM at the two highest-flow sites after long autoregressive rollouts; pretrained-backbone regimes do not. Within this five-site evaluation, PhaseShift demonstrates consolidation across heterogeneous physical control settings while identifying sites that still require adaptation. The protocol measures conditional single-vehicle generation under replayed context, not closed-loop traffic simulation.",
  "published": "2026-08-26",
  "updated": "2026-08-26",
  "year": "2026",
  "authors": [
   "Yash Ranjan",
   "Artur Kumik",
   "Rahul Sengupta",
   "Anand Rangarajan",
   "Sanjay Ranka"
  ],
  "author_count": 5,
  "categories": [
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.AI",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "PhaseShift is presented, a topology-aware framework that harmonizes heterogeneous roadside trajectories into a shared actor-centric representation and trains one reusable backbone that supports pooled operation, zero-shot at a held-out intersection, and low-data adaptation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yash Ranjan",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Artur Kumik",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Rahul Sengupta",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Anand Rangarajan",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Sanjay Ranka",
    "id": null,
    "h_index": 0,
    "papers": 0
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.25275v1",
  "pdf_url": "https://arxiv.org/pdf/2608.25275v1",
  "html_url": "https://arxiv.org/html/2608.25275v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.25222",
  "slug": "development-of-a-voice-controlled-tendon-driven-bionic-hand",
  "title": "Development of a Voice-Controlled Tendon-Driven Bionic Hand",
  "abstract": "The impairment of the hands can seriously affect the abilities of every individual to perform the every-day activity, so the design of stable and controllable support devices is a significant field of study. This paper is about the design and implementation of an automated bionic hand which is dedicated to the coordinated finger movement through the simplified and efficient actuation mechanism. The method that the proposed system was designed on is the tendon-based method whereby the servo motors generate the movement of the fingers, with assistance of the angular control which is calibrated. An actuation is controlled by a microcontroller that will be programmed by use of an Arduino-based microcontroller to carry out programmed gestures that include open hand, fist, pinch and half flexion. It has an interface that is voice command enabled to make it easy to interact with a Bluetooth based sender receiver architecture which offers an option of executing trained commands which are immediately converted to finger actions. To explore the motions behavior, finger coordination and control response to the input, the behavior of the experiment system is tested. The actuation of the fingers was found to take a total of about 7-8 seconds to achieve full flexion of all fingers in a sequence. The system showed repetitive and constant motion throughout several actuation cycles without loss of any apparent tension or precision of control. There was a stable grasp of objects of different shapes and sizes, which implied consistent coordination between the fingers. These findings indicate that the proposed system offers predictable and steady control behavior and has a simple and efficient mechanical and control architecture.",
  "published": "2026-08-25",
  "updated": "2026-08-25",
  "year": "2026",
  "authors": [
   "Urja Kohli",
   "Shagata Chanda",
   "Kritika Gandhi",
   "Charu Nigam",
   "Aditi Surya Kamal",
   "Pooja Bhati"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.HC",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Urja Kohli",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Shagata Chanda",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Kritika Gandhi",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Charu Nigam",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Aditi Surya Kamal",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Pooja Bhati Department of Mechanical",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Automation Engineering",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Indira Gandhi Delhi Technical University for Women",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Delhi",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "India",
    "id": null,
    "h_index": 0,
    "papers": 0
   }
  ],
  "comment": "20 pages, 11 figures, 6 tables. Open-access preprint intended for journal or conference submission",
  "topics": [
   "dexterous-manipulation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.25222v1",
  "pdf_url": "https://arxiv.org/pdf/2608.25222v1",
  "html_url": "https://arxiv.org/html/2608.25222v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.25196",
  "slug": "longitudinal-robot-learning-from-demonstration-with-care-providers-in",
  "title": "Longitudinal Robot Learning from Demonstration with Care Providers in a Home Environment",
  "abstract": "Learning from demonstration (LfD) methods enable non-expert end users to teach robots novel skills without explicit programming. However most evaluations of the usability of LfD with non-experts has been conducted in controlled laboratory environments with a robotics experimenter present. In this work we identify non-expert end users' key barriers when teaching robots via demonstration without live robotics expert feedback in a home environment. In our human subjects experiment we support the non-expert end users through two forms of demonstrator guidance developed in prior work: pre-training and adaptive feedback. Towards the ecological validity of the evaluation, we conduct this experimentation over multiple visits, with a population of care providers. Finally, we propose to open source the resulting LfD dataset of care providers teaching a robot assistive tasks over multiple visits to a home environment.",
  "published": "2026-08-25",
  "updated": "2026-08-25",
  "year": "2026",
  "authors": [
   "Nina Moorman",
   "Julianna Schalkwyk",
   "Vriksha Srihari",
   "Qingyu Xiao",
   "Kamel Alrashedy",
   "Hongseok Jeong",
   "Kiersten Lange",
   "Matthew B. Luebbers",
   "Matthew Gombolay"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO",
   "cs.HC"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work identifies non-expert end users' key barriers when teaching robots via demonstration without live robotics expert feedback in a home environment and proposes to open source the resulting LfD dataset of care providers teaching a robot assistive tasks over multiple visits to a home environment.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Nina Moorman",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Julianna Schalkwyk",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Vriksha Srihari",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Qing-Yu Xiao",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Kamel Alrashedy",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Hongseok Jeong",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Kiersten Lange",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Matthew B. Luebbers",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Matthew Gombolay",
    "id": null,
    "h_index": 0,
    "papers": 0
   }
  ],
  "comment": "ICRA 2026 Workshop on Bridging the Gap between Robot Learning and Human-Robot Interaction",
  "topics": [
   "foundation-pretraining",
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.25196v1",
  "pdf_url": "https://arxiv.org/pdf/2608.25196v1",
  "html_url": "https://arxiv.org/html/2608.25196v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.25192",
  "slug": "cressim-neo-a-batched-gpu-simulation-engine-for-surgical-robotics-and",
  "title": "CRESSim-Neo: A Batched GPU Simulation Engine for Surgical Robotics and Robot Learning",
  "abstract": "We introduce CRESSim-Neo, a batched GPU simulation engine for surgical robotics and robot learning. CRESSim-Neo combines position-based simulation of rigid bodies, deformable tissues, fluids, and strands with batched rendering, surgery-specific sensing, and a GPU-resident data pipeline. The engine supports applications including tissue manipulation, fluid suction, suturing, cable-driven robots, and ultrasound image synthesis. Direct access to physics and rendering buffers enables GPU-resident robot learning and zero-copy PyTorch integration using DLPack. We demonstrate CRESSim-Neo across rigid-body, deformable-body, and fluid simulation tasks, including vision-based and surgical robot-learning scenarios. On an NVIDIA RTX 4090, the engine achieves up to 2.03 million environment steps per second for 8192 parallel CartPole environments, and scales to batched surgical scenarios involving tissue deformation, fluid interaction, and ultrasound sensing. Overall, CRESSim-Neo provides a unified and scalable platform for surgical simulation, synthetic data generation, and surgical robot learning.",
  "published": "2026-08-25",
  "updated": "2026-08-25",
  "year": "2026",
  "authors": [
   "Yafei Ou",
   "Ahnaf Naheen",
   "Tleukhan Mussin",
   "Hans Jarales",
   "Melwin Moncy",
   "Mahdi Tavakoli"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Overall, CRESSim-Neo provides a unified and scalable platform for surgical simulation, synthetic data generation, and surgical robot learning.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yafei Ou",
    "id": "2258631799",
    "h_index": 5,
    "papers": 17
   },
   {
    "name": "Ahnaf Naheen",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Tleukhan Mussin",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Hans Jarales",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Melwin Moncy",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Mahdi Tavakoli",
    "id": null,
    "h_index": 0,
    "papers": 0
   }
  ],
  "comment": "8 pages, 11 figures",
  "topics": [
   "data-teleop"
  ],
  "orgs": [
   "NVIDIA"
  ],
  "abs_url": "https://arxiv.org/abs/2608.25192v1",
  "pdf_url": "https://arxiv.org/pdf/2608.25192v1",
  "html_url": "https://arxiv.org/html/2608.25192v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2608.25171",
  "slug": "control-oriented-learning-for-dynamic-tracking-and-stability-analysis",
  "title": "Control-Oriented Learning for Dynamic Tracking and Stability Analysis of Soft Pneumatic Actuators",
  "abstract": "Soft pneumatic actuators offer inherent compliance and safe interaction but remain difficult to model and control because of their highly nonlinear, distributed dynamics. We present a control-oriented data-driven modeling and control framework that decomposes actuator behavior into a nonlinear static equilibrium model and a linear residual dynamics model identified using Extended Dynamic Mode Decomposition with control (EDMDc). This representation enables feedforward compensation, task-space feedback control, and local closed-loop stability analysis through an augmented linear model. Experiments achieve approximately 1 mm root mean square error (RMSE) during low-speed (approximately 10 mm/s) trajectory tracking and below 10 mm RMSE at higher speeds (approximately 100 mm/s). The framework further achieves stable tracking of highly dynamic user-generated references with peak accelerations exceeding 25 m/s^2 while simultaneously performing real-time obstacle avoidance. Finally, the proposed stability analysis is experimentally validated by accurately predicting stable, marginal, and unstable operating regimes. These results demonstrate that structured, control-oriented learning provides an accurate and practical framework for soft actuator control.",
  "published": "2026-08-25",
  "updated": "2026-08-25",
  "year": "2026",
  "authors": [
   "Nithin S. Kumar",
   "Eric J. Barth"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Nithin S. Kumar",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Eric J. Barth",
    "id": null,
    "h_index": 0,
    "papers": 0
   }
  ],
  "comment": "8 pages, 10 figures",
  "topics": [
   "navigation",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.25171v1",
  "pdf_url": "https://arxiv.org/pdf/2608.25171v1",
  "html_url": "https://arxiv.org/html/2608.25171v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.25162",
  "slug": "sequential-object-placement-optimization-with-convex-decomposition",
  "title": "Sequential Object Placement Optimization with Convex Decomposition",
  "abstract": "Robotic object packing has been a core challenge for robotic deployment in logistics, industry, etc., due to the curse of dimensionality in combinatorial search and the difficulty of dealing with dynamic and contact constraints for irregularly shaped objects. Current heuristic and learning-based methods assume a limited spatial discretization resolution of space, and computation becomes extremely inefficient as discretization accuracy increases. In this work, we eliminate these assumptions by introducing SOPO-CD, a sequential optimization framework that frames object placement as a differentiable nonlinear optimization problem in a decomposed free space. We prove that placing a convex object inside a convex hull is essentially constraining the vertices of the object inside the convex hull. The constraints and their derivatives can be written in closed form and calculated within $200$ns. We implement a custom solver that achieves optimal placement within tightly constrained space in milliseconds; a $100 \\times$ speedup compared to a classical grid search method. We generalize our framework to 2D Tangram, 2D Tetris, and 3D Bin Packing, and have demonstrated strong computational performance and packing utility. We also demonstrate solving a real-world Tangram puzzle online using an Allegro Hand and an Xarm.",
  "published": "2026-08-25",
  "updated": "2026-08-25",
  "year": "2026",
  "authors": [
   "Yuezhe Zhang",
   "Xiangyu Lyu",
   "Sohan Rudra",
   "Davide Tateo",
   "Georgia Chalvatzaki"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.25162v1",
  "pdf_url": "https://arxiv.org/pdf/2608.25162v1",
  "html_url": "https://arxiv.org/html/2608.25162v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.25142",
  "slug": "skydrive-learning-to-drive-in-a-new-city-from-aerial-traffic-monitorin",
  "title": "SkyDrive: Learning to Drive in a New City from Aerial Traffic Monitoring",
  "abstract": "Autonomous driving has made remarkable progress through imitation learning with massive human demonstration data. However, a trained planner often degrades severely when applied to a new environment zero-shot, because of domain shifts in traffic regulations, road layout and driving behaviors. Therefore, adapting a trajectory planner to a new city typically requires resource-demanding local data collection with a vehicle sensor suite. In this work, we show that driving behavior can be learned from a scalable and efficient alternative. We introduce \\emph{SkyDrive}, a framework that utilizes drone-based traffic monitoring to provide efficient supervision for autonomous driving agents in a new environment. While vehicle-based data collection logs the ego and its surroundings, an aerial platform naturally observes many road users simultaneously over an extended field of view. As a result, every vehicle can be a data source with grounded driving behavior, effectively scaling up the amount of supervision. Based on 137 hours of aerial traffic monitoring footage, we extract 650K driving samples and construct a benchmark for trajectory planners and motion predictors. Zero-shot experiments with multiple models reveal significant cross-city domain gaps, but many of them can be alleviated by limited supervision from the sky, e.g., 30 minutes of monitoring per location. Our findings show that aerial traffic monitoring is an efficient and scalable data source for adapting autonomous driving systems in new cities. Data and code will be made publicly available.",
  "published": "2026-08-25",
  "updated": "2026-08-25",
  "year": "2026",
  "authors": [
   "Weijiang Xiong",
   "Lan Feng",
   "Alexandre Alahi",
   "Nikolas Geroliminis"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work introduces \\emph{SkyDrive}, a framework that utilizes drone-based traffic monitoring to provide efficient supervision for autonomous driving agents in a new environment and shows that aerial traffic monitoring is an efficient and scalable data source for adapting autonomous driving systems in new cities.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Wei-Jiang Xiong",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Lan Feng",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Alexandre Alahi",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Nikolas Geroliminis",
    "id": null,
    "h_index": 0,
    "papers": 0
   }
  ],
  "comment": "",
  "topics": [
   "egocentric-data",
   "imitation-diffusion",
   "navigation",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.25142v1",
  "pdf_url": "https://arxiv.org/pdf/2608.25142v1",
  "html_url": "https://arxiv.org/html/2608.25142v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.25135",
  "slug": "extending-ground-constraint-lidar-imu-calibration-to-tilted-surfaces-i",
  "title": "Extending Ground-Constraint LiDAR-IMU Calibration to Tilted Surfaces in a Continuous-Time Framework",
  "abstract": "This paper presents a novel method that extends targetless LiDAR-IMU calibration for ground vehicles to non- flat environments. Calibration typically necessitates full exci- tation of the sensor rig, a requirement that is not fulfilled by ground vehicles in normal operation. To address the degenerate planar motion, state-of-the-art methods propose residuals that assume the colinearity of the gravity and physical surface normal vectors, restricting usage to cases where the ground is assumed flat. This paper proposes ground-plane residuals that do not require this assumption, and are applicable for planar motion on a tilted surface. Results are demonstrated on a dataset collected from a Husky ground vehicle, on the M2DGR dataset, as well as on an offroad vehicle dataset. Repeatability is shown to be improved both in tilted and flat-ground scenarios, with strong improvement demonstrated for the tilted case. The implementation and experiments are open-sourced at https://github.com/vkorotkine/licalib_tilted_ground.",
  "published": "2026-08-25",
  "updated": "2026-08-25",
  "year": "2026",
  "authors": [
   "Vassili Korotkine",
   "Pierre Chamoun",
   "Mohammed Ayman Shalaby",
   "James Richard Forbes"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Vassili Korotkine",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Pierre Chamoun",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Mohammed Ayman Shalaby",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "James Richard Forbes",
    "id": null,
    "h_index": 0,
    "papers": 0
   }
  ],
  "comment": "8 pages, 13 figures. Submitted to Robotics & Automation Letters",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.25135v1",
  "pdf_url": "https://arxiv.org/pdf/2608.25135v1",
  "html_url": "https://arxiv.org/html/2608.25135v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.25102",
  "slug": "ros2-connect-a-new-ros2-over-wan-solution",
  "title": "ROS2 Connect: A new ROS2 over WAN Solution",
  "abstract": "The Robot Operating System 2 (ROS2) has become a widely adopted framework for the development of distributed robotic systems. However, its communication architecture, based on DDS and RTPS, relies on multicast discovery mechanisms that are typically unavailable in wide-area network (WAN) environments, making remote operation challenging. This work presents ROS2 Connect, a WebSocket-based communication framework that enables transparent and secure ROS2 interaction across routed networks without requiring modifications to network infrastructure or DDS configurations. The proposed client-server architecture supports bidirectional exchange of topics, services, actions, and system data while integrating authentication and access control mechanisms. Experimental evaluation over a real WAN connection demonstrates significantly lower latency, higher stability, and improved scalability compared to existing solutions, including DDS Router, rosbridge and Zenoh. Initial results show that ROS2 Connect provides a reliable foundation for teleoperation and distributed robotics applications over wide-area networks.",
  "published": "2026-08-25",
  "updated": "2026-08-25",
  "year": "2026",
  "authors": [
   "Daniel Schott",
   "Lakshminarasimhan Srinivasan",
   "Christian Herrmann",
   "Andreas N\u00fcchter"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Initial results show that ROS2 Connect provides a reliable foundation for teleoperation and distributed robotics applications over wide-area networks and improved scalability compared to existing solutions, including DDS Router, rosbridge and Zenoh.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Daniel Schott",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Lakshminarasimhan Srinivasan",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Christian Herrmann",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Andreas N\u00fcchter",
    "id": null,
    "h_index": 0,
    "papers": 0
   }
  ],
  "comment": "Proceedings of the 8th International Workshop on Robotics Software Engineering co-located with the 2026 IEEE International Conference on Robotics and Automation (ROSE '26)",
  "topics": [
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.25102v1",
  "pdf_url": "https://arxiv.org/pdf/2608.25102v1",
  "html_url": "https://arxiv.org/html/2608.25102v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.24959",
  "slug": "gaussvla-geometry-aware-spatial-reasoning-for-vision-language-action-m",
  "title": "GaussVLA: Geometry-Aware Spatial Reasoning for Vision-Language-Action Model",
  "abstract": "Vision-Language-Action (VLA) models encode visual observations as flat 2D patch tokens that carry no intrinsic geometric structure, and augmenting them with dense monocular depth injects per-pixel scalar values that encode neither surface orientation nor geometric confidence. This leaves the policy with limited structured spatial reasoning for action prediction. We propose GaussVLA, a Mamba-based VLA that incorporates two custom modules: Gaussian Spatial Tokenizer (GST) to lift frozen semantic and depth features into compact 3D Gaussian tokens, pools geometrically salient regions with learned queries, and \\emph{Depth-Aware Chain-of-Thought (DA-CoT)} that performs structured, non-autoregressive geometric reasoning under language and flow-time conditioning. Across both simulation and real-world evaluations, GaussVLA demonstrates strong spatial-manipulation performance while remaining parameter-efficient. On LIBERO, it achieves 93.5% average success and 100.0% success on the Spatial suite with only 200M parameters, improving over SpatialVLA by 19.7% relative average success while remaining significantly more parameter-efficient.",
  "published": "2026-08-25",
  "updated": "2026-08-25",
  "year": "2026",
  "authors": [
   "Md Selim Sarowar",
   "Md Tanvir Islam",
   "Sungho Kim",
   "Sangtae Ahn"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "GaussVLA is proposed, a Mamba-based VLA that incorporates two custom modules: Gaussian Spatial Tokenizer (GST) to lift frozen semantic and depth features into compact 3D Gaussian tokens, and Depth-Aware Chain-of-Thought (DA-CoT) that performs structured, non-autoregressive geometric reasoning under language and flow-time conditioning.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Md Selim Sarowar",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Md Tanvir Islam",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Sungho Kim",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Sangtae Ahn",
    "id": null,
    "h_index": 0,
    "papers": 0
   }
  ],
  "comment": "Accepted to BMVC 2026",
  "topics": [
   "vla",
   "spatial-3d",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.24959v1",
  "pdf_url": "https://arxiv.org/pdf/2608.24959v1",
  "html_url": "https://arxiv.org/html/2608.24959v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.24885",
  "slug": "do-robotic-world-models-really-follow-actions-diagnosing-and-aligning",
  "title": "Do Robotic World Models Really Follow Actions? Diagnosing and Aligning Action-Conditioned Generation for Policy Learning",
  "abstract": "Action-conditioned world models are increasingly used as learned simulators for policy evaluation and improvement, yet their effectiveness rests on an unverified assumption: generated futures faithfully reflect arbitrary valid actions. Existing benchmarks are typically confined to expert demonstrations, leaving off-expert action following inadequately evaluated. To address this gap, we introduce WorldEcho, which probes action following over a broader action distribution using visual integrity and SE(3) trajectory alignment. Our diagnosis shows that current world models reasonably execute expert actions but struggle with diverse off-expert trajectories, either ignoring the commanded actions or producing visually invalid rollouts. We further propose WorldSync, which strengthens action following along three complementary axes: distributional coverage, representational grounding, and intervention-effect alignment. It broadens the training distribution over action consequences, grounds intermediate video representations in action-induced robot dynamics through an Action-Forcing Expert, and aligns predicted changes under action interventions with the corresponding changes in ground-truth futures. Experiments on RoboTwin benchmarks and real-robot tasks show that WorldSync improves WorldEcho metrics and serves as a more reliable simulator for iterative policy improvement, enabling policies to achieve higher success rates.",
  "published": "2026-08-25",
  "updated": "2026-08-25",
  "year": "2026",
  "authors": [
   "Sixiang Chen",
   "Jiaming Liu",
   "Jixian Wu",
   "Yichen Guo",
   "Tinghao Wang",
   "Siyuan Qian",
   "Hao Chen",
   "Jiajun Cao",
   "Jian Tang",
   "Shanghang Zhang"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "WorldEcho is introduced, which probes action following over a broader action distribution using visual integrity and SE(3) trajectory alignment, and WorldSync is proposed, which strengthens action following along three complementary axes: distributional coverage, representational grounding, and intervention-effect alignment.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Sixiang Chen",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Jiaming Liu",
    "id": "2258602418",
    "h_index": 14,
    "papers": 37
   },
   {
    "name": "Jixian Wu",
    "id": "2314746006",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Yichen Guo",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Tinghao Wang",
    "id": "2381068349",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Siyuan Qian",
    "id": "1610531198",
    "h_index": 5,
    "papers": 15
   },
   {
    "name": "Hao Chen",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Jiajun Cao",
    "id": "2268711797",
    "h_index": 9,
    "papers": 20
   },
   {
    "name": "Jian Tang",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Shanghang Zhang",
    "id": "2346116279",
    "h_index": 16,
    "papers": 52
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "sim2real"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.24885v1",
  "pdf_url": "https://arxiv.org/pdf/2608.24885v1",
  "html_url": "https://arxiv.org/html/2608.24885v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.24882",
  "slug": "latent-action-as-intention-enables-efficient-future-imagination-for-wo",
  "title": "Latent Action as Intention Enables Efficient Future Imagination for World Action Models",
  "abstract": "World action models (WAMs) improve robot control by modeling how observations evolve, but generating future observations at test time incurs substantial latency. Fast-WAM removes this process for efficiency; however, our matched implementations show lower generalization for Fast-WAM than for future-aware alternatives, especially with scarce robot demonstrations and in out-of-distribution scenarios. To bridge this gap, we introduce **LAWA**, a WAM architecture that uses compact latent actions as an operational representation of future intentions, enabling efficient test-time future imagination without generating future observations. Specifically, a discrete tokenizer enhanced by action-free pre-training produces manipulation-centric codebook targets. LAWA jointly denoises a continuous latent state anchored to these targets with executable action chunks while omitting the future-video branch at inference. On RoboCasa, LAWA achieves state-of-the-art average success rates of 65.6% and 80.8% in the few-shot and full data settings, improving over the matched Fast-WAM baseline by 9.6 and 4.5 points, respectively. It also preserves the performance level of the matched Joint-WAM variant while requiring 42.9% lower inference latency. LAWA also demonstrates competitive zero-shot robustness on LIBERO-Plus and superior performance on real-world tasks. These results show that future imagination need not be discarded: retaining it with compact latent actions yields an effective trade-off among performance, generalization, and latency. Code and models will be released.",
  "published": "2026-08-25",
  "updated": "2026-08-25",
  "year": "2026",
  "authors": [
   "Xiang Li",
   "Yupeng Zheng",
   "Songen Gu",
   "Huailiang Ma",
   "Feng Yu",
   "Xian Nie",
   "Shanshuai Yuan",
   "Yujie Zang",
   "Weize Li",
   "Shuai Tian",
   "Moyang Liu",
   "Ya-Qin Zhang",
   "Wenchao Ding"
  ],
  "author_count": 13,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xiang Li",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Yupeng Zheng",
    "id": "2357068526",
    "h_index": 16,
    "papers": 46
   },
   {
    "name": "Songen Gu",
    "id": "2291140987",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Huailiang Ma",
    "id": "2332528826",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Fengying Yu",
    "id": "2454150955",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Xiangli Nie",
    "id": "2356786010",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Shanshuai Yuan",
    "id": "2294378803",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Yujie Zang",
    "id": "3341619",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Weize Li",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Shuai Tian",
    "id": "2345921369",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Moyang Liu",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Ya-Qin Zhang",
    "id": "2383105350",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Wenchao Ding",
    "id": "2325819492",
    "h_index": 4,
    "papers": 16
   }
  ],
  "comment": "",
  "topics": [
   "foundation-pretraining",
   "video-generation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.24882v1",
  "pdf_url": "https://arxiv.org/pdf/2608.24882v1",
  "html_url": "https://arxiv.org/html/2608.24882v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.24743",
  "slug": "text-dnn-2-doubly-non-negative-relaxations-for-deep-neural-networks",
  "title": "$(\\text{DNN})^2$: Doubly Non-Negative Relaxations for Deep Neural Networks",
  "abstract": "Existing linear program (LP) and semidefinite program (SDP) relaxations for rectified linear unit (ReLU) neural network (NN) verification yield overly-conservative safety guarantees due to significant relaxation gaps. While the completely positive program (CPP) formulation closes this gap, it is NP-hard to solve. Its cheapest tractable relaxation, the doubly non-negative program (DNN), retains critical constraints as an SDP, but one whose size exceeds the reach of interior-point methods at practical scale. While Burer-Monteiro (BM) factorization has been applied to make SDP-based verification scalable, no such result exists for the strictly tighter DNN formulation. A key obstacle is that additional non-negativity constraints in the DNN cause dual multipliers for optimality certification to be non-unique, making standard certification methods inapplicable. We propose a novel eigenvalue maximization procedure that searches the non-unique multiplier space for a valid certificate, i.e. a global optimality guarantee. Experiments demonstrate that our approach $(\\text{DNN})^2$ produces bounds consistently tighter than the standard SDP method, often matching the exact solution, and that our certification procedure confirms global optimality when a valid certificate exists. These results are a key step toward providing tight, certifiable, and computationally scalable verification guarantees needed to deploy neural network controllers and perception modules in safety-critical autonomous systems.",
  "published": "2026-08-25",
  "updated": "2026-08-25",
  "year": "2026",
  "authors": [
   "Hanna Jiamei Zhang",
   "Alan Papalia",
   "Michael Everett",
   "David M. Rosen"
  ],
  "author_count": 4,
  "categories": [
   "cs.LG",
   "cs.RO",
   "eess.SY"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A novel eigenvalue maximization procedure that searches the non-unique multiplier space for a valid certificate, i.e. a global optimality guarantee, which is a key step toward providing tight, certifiable, and computationally scalable verification guarantees needed to deploy neural network controllers and perception modules in safety-critical autonomous systems.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Han Zhang",
    "id": "2304392523",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Alan Papalia",
    "id": "1824297799",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Michael Everett",
    "id": "2290069656",
    "h_index": 4,
    "papers": 16
   },
   {
    "name": "David M. Rosen",
    "id": "2251532935",
    "h_index": 2,
    "papers": 3
   }
  ],
  "comment": "6 pages, 3 figures, accepted and to be presented at 64th IEEE Conference on Decision and Control: CDC 2026",
  "topics": [
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.24743v1",
  "pdf_url": "https://arxiv.org/pdf/2608.24743v1",
  "html_url": "https://arxiv.org/html/2608.24743v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.24741",
  "slug": "one-shot-learning-from-demonstration-of-contact-rich-robotic-manipulat",
  "title": "One-Shot Learning from Demonstration of Contact-Rich Robotic Manipulation by Identifying Physical Interactions",
  "abstract": "Learning from Demonstration (LfD) allows robots to learn manipulation tasks directly from humans, thereby supporting the versatile application of robots. Most LfD methods do not explicitly model the physical interactions between a robot and its environment, such as the making and breaking of contact, while these are crucial during manipulation tasks. Because the same basic physical interactions recur often, they can be a basis for robust, generalizable, and adaptive task reproduction. We propose an LfD method that explicitly uses what physical interactions take place where and when. Using that information, a hybrid position-force controller tracks demonstrated trajectories until contact-based transition conditions from the demonstrations are met. We evaluate our method in real robot experiments consisting of opening doors and locks, bolt picking and screwing, dislodging, and surface contouring. We show that explicitly modeling physical interactions benefits LfD in four ways. First, by allowing reproduction of complex, sequential, and contact-rich manipulation tasks using only a single demonstration and no prior knowledge of the task. Second, by facilitating robustness to unknown geometric variations in the environment. Third, by facilitating generalization when geometric variations are known. Fourth, by facilitating online adaptation using geometric information explored during task reproduction. We discuss how robustness, generalization, and adaptivity can be explicitly implemented, which is generally lacking in the LfD literature. Thereby, our work aims to close a gap in interpretable few-shot LfD of robotic manipulation.",
  "published": "2026-08-25",
  "updated": "2026-08-25",
  "year": "2026",
  "authors": [
   "A. H. G. Overbeek",
   "H. van der Kooij",
   "M. Vlutters"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes an LfD method that explicitly uses what physical interactions take place where and when, and discusses how robustness, generalization, and adaptivity can be explicitly implemented, which is generally lacking in the LfD literature.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "A. H. G. Overbeek",
    "id": "2373134947",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "H. V. D. Kooij",
    "id": "10781775",
    "h_index": 37,
    "papers": 285
   },
   {
    "name": "M. Vlutters",
    "id": "4122322",
    "h_index": 12,
    "papers": 32
   }
  ],
  "comment": "",
  "topics": [
   "tactile",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.24741v1",
  "pdf_url": "https://arxiv.org/pdf/2608.24741v1",
  "html_url": "https://arxiv.org/html/2608.24741v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.24724",
  "slug": "fiber-bragg-grating-whiskers-for-bioinspired-hydrodynamic-perception-o",
  "title": "Fiber Bragg Grating Whiskers for Bioinspired Hydrodynamic Perception on Underwater Robots",
  "abstract": "Harbor seals track hydrodynamic trails with their vibrissae, enabling passive perception of moving targets in dark or turbid water. Inspired by this capability, we present compact fiber Bragg grating (FBG) whiskers for underwater robots. Like seal whiskers, they have a non-uniform taper and elliptical cross-section. Controlled towing experiments show a monotonic relative-flow response from 0.1 to 0.6 m/s, a strong reduction of self-induced oscillation relative to a cylindrical baseline, and a pronounced dependence on angle of attack. Experiments with a pitching foil show that the whiskers can detect the characteristic vortices shed by a stationary or moving source, detectable several seconds after the source has passed. Using this information, a single front-mounted whisker enabled a small underwater robot to distinguish between continuing straight and executing a turn, selecting the correct branch in 17 of 20 trials (85.0%) from whisker signals alone. These results connect bioinspired hydrodynamic sensing to robot action and suggest the utility of whiskers for tracking underwater objects.",
  "published": "2026-08-25",
  "updated": "2026-08-25",
  "year": "2026",
  "authors": [
   "Hao Li",
   "Tianyu Tu",
   "Siyue Yao",
   "Ziyang Chang",
   "Juhyun Jung",
   "Xiaochi Xie",
   "Long Yin Chung",
   "Tian-Ao Ren",
   "Genliang Chen",
   "Mark Cutkosky"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hao Li",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Tianyu Tu",
    "id": "2348444886",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Siyue Yao",
    "id": "2147300642",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Ziyang Chang",
    "id": "2457631843",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Juhyun Jung",
    "id": "2459471779",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Xiaochi Xie",
    "id": "2426813631",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Long Yin Chung",
    "id": "2425514722",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Tian-Ao Ren",
    "id": "2348443470",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Genliang Chen",
    "id": "2494841",
    "h_index": 23,
    "papers": 134
   },
   {
    "name": "Mark R. Cutkosky",
    "id": "2329168577",
    "h_index": 4,
    "papers": 21
   }
  ],
  "comment": "13 pages, 8 figures",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.24724v1",
  "pdf_url": "https://arxiv.org/pdf/2608.24724v1",
  "html_url": "https://arxiv.org/html/2608.24724v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.24714",
  "slug": "gaussianwam-distilling-geometry-and-semantics-from-3d-gaussian-fields",
  "title": "GaussianWAM: Distilling Geometry and Semantics from 3D Gaussian Fields into World-Action Models",
  "abstract": "World-Action Models (WAMs) jointly learn future visual prediction and action generation, using video dynamics as a representation-learning signal for robotic manipulation. However, their video latents are primarily optimized for visual prediction and are not explicitly encouraged to preserve cross-view geometric structure or spatially localized, object-relevant semantics. We propose \\textbf{GaussianWAM}, a training-time representation-enhancement framework that organizes geometric and semantic supervision through a 3D Gaussian field. Given synchronized multi-view observations, frozen geometry and vision foundation models provide depth, camera parameters, and dense semantic features. GaussianWAM binds these heterogeneous signals to shared Gaussian primitives and renders spatially aligned semantic, depth, and coverage targets, which are distilled into the current-observation representations of the WAM. All teacher models, Gaussian components, and auxiliary prediction heads are removed after training, leaving the original WAM inference path without additional modules or forward computation. On LIBERO-Plus, GaussianWAM improves FastWAM from 52.05\\% to 71.29\\% and Cosmos Policy from 71.52\\% to 77.30\\%. Direct CLIP and VGGT distillation already establishes a strong FastWAM baseline of 69.37\\%, while Gaussian-field unification further improves it to 71.29\\%, supporting the benefit of spatially organizing heterogeneous teacher signals. GaussianWAM also improves performance on standard LIBERO and shows positive transfer trends on RoboTwin and real-world manipulation. These results suggest that training-time Gaussian distillation provides a practical way to inject geometry- and semantics-related supervision into WAM representations without changing their deployment architecture.",
  "published": "2026-08-25",
  "updated": "2026-08-25",
  "year": "2026",
  "authors": [
   "Zijian Zhang",
   "Yuqing Jiang",
   "Weitao Zhou",
   "Minglei Li",
   "Jinhao Zhang",
   "Yao Mu",
   "Xiaofan Li",
   "Hao Zhao",
   "Haibao Yu"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "GaussianWAM is proposed, a training-time representation-enhancement framework that organizes geometric and semantic supervision through a 3D Gaussian field and improves performance on standard LIBERO and shows positive transfer trends on RoboTwin and real-world manipulation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zijian Zhang",
    "id": "2354634470",
    "h_index": 3,
    "papers": 14
   },
   {
    "name": "Yuqing Jiang",
    "id": "2345398013",
    "h_index": 3,
    "papers": 13
   },
   {
    "name": "Weitao Zhou",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Minglei Li",
    "id": "2448263547",
    "h_index": 0,
    "papers": 6
   },
   {
    "name": "Jinhao Zhang",
    "id": "2348679757",
    "h_index": 1,
    "papers": 10
   },
   {
    "name": "Yao Mu",
    "id": "2348161293",
    "h_index": 2,
    "papers": 18
   },
   {
    "name": "Xiaofan Li",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Hao Zhao",
    "id": "2293764606",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Haibao Yu",
    "id": null,
    "h_index": 0,
    "papers": 0
   }
  ],
  "comment": "13 pages, 5 figures",
  "topics": [
   "foundation-pretraining"
  ],
  "orgs": [
   "NVIDIA"
  ],
  "abs_url": "https://arxiv.org/abs/2608.24714v1",
  "pdf_url": "https://arxiv.org/pdf/2608.24714v1",
  "html_url": "https://arxiv.org/html/2608.24714v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2608.24618",
  "slug": "vip-variation-based-iterative-learning-planning-for-robotic-navigation",
  "title": "VIP: Variation-based Iterative-learning Planning for Robotic Navigation",
  "abstract": "Over the past decade, autonomous robotic systems have been increasingly deployed in applications such as surveying, search and rescue, and last-mile delivery. These applications require robots to generate safe and efficient motion plans in large, complex, and obstacle-dense environments, often under limited onboard computing resources. However, conventional planning methods commonly rely on finite-dimensional trajectory parameterization or increasingly long prediction horizons, leading to rapidly growing computational costs, particularly in multi-robot scenarios. This paper presents a novel variation-based iterative-learning planning (VIP) framework for efficient motion planning of both single robots and robotic swarms. Instead of optimizing a large number of discrete trajectory variables, VIP directly updates the planning command as a continuous function in an infinite-dimensional function space. The same variation-based update can be implemented in a model-in-the-loop manner for offline planning or in a robot-in-the-loop manner between online physical executions. By avoiding the computational burden associated with horizon expansion and high-dimensional trajectory discretization, VIP maintains a per-iteration computational complexity of $\\mathcal{O}(n)$, where $n$ denotes the number of spatial discretization points. Extensive simulations and real-world experiments demonstrate that the proposed framework can efficiently generate and iteratively improve motion plans for different planning objectives, robotic platforms, and swarm configurations, highlighting its effectiveness, computational efficiency, and scalability as a general planning methodology.",
  "published": "2026-08-25",
  "updated": "2026-08-25",
  "year": "2026",
  "authors": [
   "Shuli Lv",
   "Pengda Mao",
   "Chen Min",
   "Li Hong",
   "Runxiao Liu",
   "Shuai Wang",
   "Quan Quan"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Extensive simulations and real-world experiments demonstrate that the proposed framework can efficiently generate and iteratively improve motion plans for different planning objectives, robotic platforms, and swarm configurations, highlighting its effectiveness, computational efficiency, and scalability as a general planning methodology.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Shuli Lv",
    "id": "2310507995",
    "h_index": 2,
    "papers": 11
   },
   {
    "name": "Pengda Mao",
    "id": "2134661435",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Chen Min",
    "id": "2061285173",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Li Hong",
    "id": "2281037588",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Runxiao Liu",
    "id": "2298863800",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Shuai Wang",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Quan Quan",
    "id": "2296716154",
    "h_index": 3,
    "papers": 14
   }
  ],
  "comment": "",
  "topics": [
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.24618v1",
  "pdf_url": "https://arxiv.org/pdf/2608.24618v1",
  "html_url": "https://arxiv.org/html/2608.24618v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.24603",
  "slug": "gripper-aware-vision-language-action-models",
  "title": "Gripper-aware Vision Language Action Models",
  "abstract": "Vision language action models (VLAs) have advanced general purpose robotic grasping and manipulation by enabling robots to interpret visual observations and natural language instructions to generate executable action sequences. However, existing VLAs often implicitly assume gripper invariance, despite grasping strategies being inherently embodiment-dependent. Different gripper types, such as parallel-jaw and suction, usually require distinct interaction strategies to achieve the same grasping objective. Moreover, current datasets for VLAs predominantly rely on parallel-jaw grippers, limiting gripper-aware learning. To address this gap, we introduce MiGA, a multi-gripper-aware dataset spanning five distinct gripper types across multiple robots with 103,000 demonstrations, explicitly capturing strategy divergence under shared task objectives. We further propose GVLA, which combines a new multi-gripper tokenizer with adapter-based policy routing. Our new gripper encoding induces structured embedding information that balances parameter sharing and strategy differentiation, while layer-wise probing confirms meaningful gripper-conditioned representations for VLAs. Intensive experiments in both simulation and real-world robots show that our GVLA outperforms the current baselines across evaluated settings. Our method also improves zero-shot generalization or few-shot adaptation to new objects or unseen tasks, and enable more efficient gripper adaptation.",
  "published": "2026-08-25",
  "updated": "2026-08-25",
  "year": "2026",
  "authors": [
   "Hanyi Zhang",
   "Zihong Luo",
   "Tianyu Li",
   "Khang Nguyen",
   "Basu Hela",
   "Shreyas Kumar",
   "Ngoc Duy Tran",
   "Feng Dai",
   "Charith Munasinghe",
   "Jorge Pe\u00f1a Queralta",
   "Giovanni Toffetti",
   "Khoa Vo",
   "Ngan Le",
   "Ravi Prakash",
   "Quan Vuong",
   "Tung D. Ta",
   "Long Hu",
   "Anh Nguyen",
   "Baoru Huang"
  ],
  "author_count": 19,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work introduces MiGA, a multi-gripper-aware dataset spanning five distinct gripper types across multiple robots with 103,000 demonstrations, and proposes GVLA, which combines a new multi-gripper tokenizer with adapter-based policy routing that improves zero-shot generalization or few-shot adaptation to new objects or unseen tasks, and enable more efficient gripper adaptation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hanyi Zhang",
    "id": "2346426387",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Zihong Luo",
    "id": "2283421778",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Tianyu Li",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Khang Nguyen",
    "id": "2398778183",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Basu Hela",
    "id": "2367142348",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Shreyas Kumar",
    "id": "2391075837",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Ngoc Duy Tran",
    "id": "2459274383",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Feng Dai",
    "id": "2459275212",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Charith Munasinghe",
    "id": "2188873476",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "J. P. Queralta",
    "id": "101277445",
    "h_index": 31,
    "papers": 94
   },
   {
    "name": "Giovanni Toffetti",
    "id": "2459274969",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Khoa T. Vo",
    "id": "2065785558",
    "h_index": 11,
    "papers": 34
   },
   {
    "name": "Ngan Le",
    "id": "2323041712",
    "h_index": 3,
    "papers": 13
   },
   {
    "name": "Ravi Prakash",
    "id": "2297769213",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Quan Vuong",
    "id": "2288210223",
    "h_index": 19,
    "papers": 29
   },
   {
    "name": "Tung D. Ta",
    "id": "2322942638",
    "h_index": 2,
    "papers": 16
   },
   {
    "name": "Longze Hu",
    "id": "2399741544",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Anh Nguyen",
    "id": "2154623771",
    "h_index": 10,
    "papers": 27
   },
   {
    "name": "Baoru Huang",
    "id": "2243927673",
    "h_index": 10,
    "papers": 37
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "dexterous-manipulation",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.24603v1",
  "pdf_url": "https://arxiv.org/pdf/2608.24603v1",
  "html_url": "https://arxiv.org/html/2608.24603v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.24572",
  "slug": "fiber-optic-sensing-glove-for-high-performance-dexterous-manipulation",
  "title": "Fiber Optic Sensing Glove for High Performance Dexterous Manipulation Capture",
  "abstract": "Capturing hand pose during dexterous manipulation remains difficult: vision-based methods degrade under occlusion and challenging lighting, while sensorized gloves, though occlusion-free, are prone to drift and magnetic interference and rarely match motion-capture accuracy. We introduce a fiber optic sensing glove for full hand pose tracking that targets these failure modes, using multi-core shape-sensing fibers that capture each fiber's full 3D shape rather than curvature alone. A novel pipeline registers each reconstructed fiber shape to a common hand reference frame, and a new inverse-kinematics solver reconstructs full hand pose at 60 Hz using curve constraints. Benchmarked on a 2-hour dataset of dexterous object manipulation tasks across 5 subjects, the glove achieves 7.2 mm mean fingertip position error against motion capture ground truth, reduced to 4.9 mm by a one-time factory calibration of the fiber routing hub that transfers across users and sessions. These capabilities enable high-fidelity data capture and bimanual virtual teleoperation - both essential to advancing the robotics field.",
  "published": "2026-08-25",
  "updated": "2026-08-25",
  "year": "2026",
  "authors": [
   "J. D. Peiffer",
   "Taylor Niehues",
   "Li Guan",
   "Ziyi Kou",
   "Ergys Ristani"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work introduces a fiber optic sensing glove for full hand pose tracking that targets these failure modes, using multi-core shape-sensing fibers that capture each fiber's full 3D shape rather than curvature alone.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "J. Peiffer",
    "id": "2212025483",
    "h_index": 8,
    "papers": 14
   },
   {
    "name": "Taylor D. Niehues",
    "id": "21314068",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Li Guan",
    "id": "2287035186",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Ziyi Kou",
    "id": "2399163710",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Ergys Ristani",
    "id": "2788204",
    "h_index": 5,
    "papers": 11
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "egocentric-data",
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.24572v1",
  "pdf_url": "https://arxiv.org/pdf/2608.24572v1",
  "html_url": "https://arxiv.org/html/2608.24572v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.24563",
  "slug": "x-multi-vlm-based-imaging-factor-disentanglement-for-factor-aware-imag",
  "title": "X-MULTI: VLM-based Imaging Factor Disentanglement for Factor-Aware Image Synthesis",
  "abstract": "Imaging factor disentanglement in text-to-image generation aims to independently control image acquisition properties such as types of camera lenses, sensor types, viewpoints, and domains to enable combinatorial generalization. This should let the model synthesize novel factor combinations unobserved in the training data, such as pairing a fisheye lens with an event sensor never observed in training data. Recent work, MULTI, introduced learnable, factor-specific embeddings to disentangle imaging factors, along with the Factor Alignment Accuracy (FAA) metric to evaluate disentanglement quality. We identify and address two independent limitations. First, MULTI's pixel-level reconstruction objective supervises the model only on observed imaging factor combinations, providing no direct training signal for novel combinations. We therefore propose X-MULTI, which uses a pretrained vision-language model (VLM) to supervise novel factor combinations synthesized during training. Second, we show the FAA metric exhibits severe cross-factor correlation leakage, misrepresenting true disentanglement quality. We therefore propose Improved-FAA (I-FAA), which employs factor-specific augmentation strategies to break these correlations and enables more rigorous evaluation. Experiments demonstrate that X-MULTI achieves improved factor alignment on novel combinations compared to MULTI. Moreover, we show that correlation leakage in FAA distorts the evaluation of true factor disentanglement and I-FAA reduces this leakage and therefore provides a more robust assessment of factor alignment.",
  "published": "2026-08-25",
  "updated": "2026-08-25",
  "year": "2026",
  "authors": [
   "Sonali Godavarthy",
   "Matthias Neuwirth-Trapp",
   "Tim-Felix Faasch",
   "Maarten Bieshaar",
   "Michael Moeller",
   "Kristof Van Laerhoven",
   "Danda Pani Paudel"
  ],
  "author_count": 7,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "ECCV 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "It is shown that correlation leakage in FAA distorts the evaluation of true factor disentanglement and I-FAA reduces this leakage and therefore provides a more robust assessment of factor alignment.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Sonali Godavarthy",
    "id": "2434949453",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Matthias Neuwirth-Trapp",
    "id": "2376265720",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Tim-Felix Faasch",
    "id": "2434949445",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Maarten Bieshaar",
    "id": "2375064019",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Michael Moeller",
    "id": "2312049106",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Kristof Van Laerhoven",
    "id": "2256608275",
    "h_index": 3,
    "papers": 44
   },
   {
    "name": "D. Paudel",
    "id": "35268081",
    "h_index": 32,
    "papers": 203
   }
  ],
  "comment": "Accepted to the MUCG Workshop at ECCV 2026",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.24563v1",
  "pdf_url": "https://arxiv.org/pdf/2608.24563v1",
  "html_url": "https://arxiv.org/html/2608.24563v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.24525",
  "slug": "rog-dagger-rollout-guided-post-training-for-end-to-end-driving",
  "title": "RoG-DAgger: Rollout-Guided Post-Training for End-to-End Driving",
  "abstract": "Recent end-to-end driving systems demonstrate strong performance on closed-loop benchmarks, yet are still predominantly trained on fixed expert-collected data using open-loop imitation learning. This training-inference mismatch leaves the policy vulnerable in policy-induced states, where accumulated errors can lead to safety-critical failures. A promising post-training approach to overcome this issue is Dataset Aggregation (DAgger), which gathers expert demonstrations in policy-induced states and subsequently fine-tunes the policy on the resulting aggregated dataset. Existing driving DAgger pipelines, however, face three challenges: i) the expert is restricted to a limited trajectory-and-speed solution space, ii) takeover may occur too early or too late relative to impending failures, and iii) privileged expert decisions may rely on information unavailable to the student. To address this, we introduce RoG-DAgger, a post-training framework that uses short-horizon kinematic rollouts to construct high-quality expert demonstrations in safety-critical states. Specifically, RoG-DAgger expands the expert's trajectory-and-speed solution space and evaluates candidate plans through rollout to construct preventive supervision. Moreover, it uses rollout solvability to time the takeover near the estimated point of no return. Lastly, it aligns the expert's field of view with that of the student to provide student-compatible supervision. Across in-distribution (including long-horizon) and out-of-distribution evaluations, RoG-DAgger improves the end-to-end model SimLingo by 5.3 driving-score points and 6.2 percentage points in success rate on Bench2Drive, doubles its driving score from 22 to 44 on Longest6 v2, and improves out-of-distribution success rate from 55\\% to 66\\% on Fail2Drive.",
  "published": "2026-08-25",
  "updated": "2026-08-25",
  "year": "2026",
  "authors": [
   "Liangyu Zhong",
   "Joachim Sicking",
   "Fabian Hueger",
   "Hanno Gottschalk"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "RoG-DAgger is introduced, a post-training framework that uses short-horizon kinematic rollouts to construct high-quality expert demonstrations in safety-critical states and aligns the expert's field of view with that of the student to provide student-compatible supervision.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Liangyu Zhong",
    "id": "2323908047",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Joachim Sicking",
    "id": "35614188",
    "h_index": 7,
    "papers": 18
   },
   {
    "name": "Fabian Hueger",
    "id": "2459273312",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Hanno Gottschalk",
    "id": "2238628539",
    "h_index": 7,
    "papers": 32
   }
  ],
  "comment": "preprint, under review",
  "topics": [
   "imitation-diffusion",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.24525v1",
  "pdf_url": "https://arxiv.org/pdf/2608.24525v1",
  "html_url": "https://arxiv.org/html/2608.24525v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.24485",
  "slug": "neuralparker-a-reinforcement-learning-planner-for-irregular-parking-en",
  "title": "NeuralParker: A Reinforcement Learning Planner for Irregular Parking Environments",
  "abstract": "Automated parking commonly assumes marked slots and short approach maneuvers. Delivery and service vehicles, however, may need to reach an operator-specified pose in an irregular bounded environment from a distant start. Existing learning-based parking planners often rely on local observations, which can restrict long-range route reasoning. To address this problem, we present NeuralParker, a reinforcement learning-based hybrid planner for arbitrary-pose parking. NeuralParker encodes full-environment obstacle and boundary geometry in a target-relative vertex representation, allowing the policy to retain route-defining context throughout the approach. It further couples a learned curvature--length arc policy with an in-loop terminal ensemble that selects from diverse cubic Hermite connections using a curvature-regularized cost. We also establish factorial and long-range route-choice benchmarks to evaluate planning success and trajectory quality. Experiments on these benchmarks show that NeuralParker achieves higher planning success and better overall trajectory quality than the evaluated baselines, while ablation studies support the benefits of the target-relative global representation and terminal ensemble. Finally, a real-vehicle evaluation confirms that the planner transfers effectively to real delivery-vehicle perception at a working parking site, planning successfully at low computational cost.",
  "published": "2026-08-25",
  "updated": "2026-08-25",
  "year": "2026",
  "authors": [
   "Zihan Wang",
   "Bai Huang",
   "Yang Guan",
   "Xiao Li",
   "Haoyu Xu",
   "Naizheng Wang",
   "Shengbo Eben Li"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work presents NeuralParker, a reinforcement learning-based hybrid planner for arbitrary-pose parking that encodes full-environment obstacle and boundary geometry in a target-relative vertex representation, allowing the policy to retain route-defining context throughout the approach.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zihan Wang",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Baixiang Huang",
    "id": "2424391648",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yang Guan",
    "id": "50463545",
    "h_index": 16,
    "papers": 38
   },
   {
    "name": "Xiao Li",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Haoyu Xu",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Naizhen Wang",
    "id": "2408585039",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "S. Li",
    "id": "2311854016",
    "h_index": 4,
    "papers": 16
   }
  ],
  "comment": "",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.24485v1",
  "pdf_url": "https://arxiv.org/pdf/2608.24485v1",
  "html_url": "https://arxiv.org/html/2608.24485v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.24366",
  "slug": "variance-guided-spatial-attention-fusion-for-robust-end-to-end-driving",
  "title": "Variance-Guided Spatial Attention Fusion for Robust End-to-End Driving under Asymmetric Sensor Degradation",
  "abstract": "End-to-end multimodal driving has progressed rapidly by fusing camera and LiDAR streams. Existing pipelines remain fragile under asymmetric sensor degradation, where either an entire modality or only a localized region is corrupted while other regions remain useful. The key difficulty is not simply to add an uncertainty head, but to obtain dense reliability supervision, calibrate this reliability against physical fault severity, and use it before unreliable features bias the planner. We propose Variance-Guided Spatial Attention Fusion (VG-SAF), in which dense heteroscedastic reliability estimates act as interpretable spatial gates. The framework couples three components. First, a physically grounded augmentor simulates representative camera and LiDAR failures and emits a continuous spatial mask, providing dense supervision without additional annotation. Second, modality-specific experts predict per-pixel reliability scales through cross-branch dense distillation in log space, enforcing a monotone severity-to-scale response. Third, calibrated reliability maps drive a hybrid attention mechanism that suppresses unreliable cells with a local spatial gate and arbitrates between modalities through a cross-modal trust softmax. A Laplace uncertainty head emits a systemic waypoint uncertainty scale that signals severe or combined sensor degradation, including severities outside the training ranges. On the CARLA Longest6 benchmark, VG-SAF consistently improves closed-loop robustness over the baselines across camera-only, LiDAR-only, and joint degradation regimes, as measured by driving score, route completion, and infraction score.",
  "published": "2026-08-25",
  "updated": "2026-08-25",
  "year": "2026",
  "authors": [
   "Weizhi Tao",
   "Zengwang Jin",
   "Xiao Wang",
   "Hailong Huang"
  ],
  "author_count": 4,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Variance-Guided Spatial Attention Fusion (VG-SAF), in which dense heteroscedastic reliability estimates act as interpretable spatial gates, consistently improves closed-loop robustness over the baselines across camera-only, LiDAR-only, and joint degradation regimes, as measured by driving score, route completion, and infraction score.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Weizhi Tao",
    "id": "2267751183",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Zengwang Jin",
    "id": "2459437944",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Xiao Wang",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Hailong Huang",
    "id": "2382917000",
    "h_index": 1,
    "papers": 4
   }
  ],
  "comment": "17 pages, 9 figures, and 4 tables, including supplementary material. Submitted to IEEE Transactions on Vehicular Technology",
  "topics": [
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.24366v1",
  "pdf_url": "https://arxiv.org/pdf/2608.24366v1",
  "html_url": "https://arxiv.org/html/2608.24366v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.24282",
  "slug": "care-camera-residual-reserves-for-first-sightings-in-adaptive-lidar-se",
  "title": "CARE: Camera-Residual Reserves for First Sightings in Adaptive LiDAR Sensing",
  "abstract": "Adaptive LiDAR scanning concentrates a limited sensing budget on regions of interest predicted from past object tracks, lowering data volume in autonomous driving while maintaining detection accuracy. However, existing scanning policies face three challenges. First, history-driven approaches depend on past tracks, so unseen objects are detected late or missed. Second, random or uniform sampling outside the predicted regions has no awareness of where new objects appear. Third, camera-guided alternatives spend budget on all camera detections, resampling objects already covered, costing recall in crowded scenes and range when budgets are scarce. This paper introduces the CAmera-REsidual reserve (CARE), a training-free allocation rule that reserves part of a fixed ray budget for the directions of current camera detections that the track forecasts cannot explain; the rest follows the base history policy, and unused reserve returns to a random floor. The paper makes three contributions. First, a leakage-free ray-budget evaluation on nuScenes (150 scenes, 4,148 events) measuring the first-sighting loss of history-driven scanning, with a strict-causal variant using the preceding keyframe. Second, CARE raises first-sighting recall by 5.2, 5.2, and 4.3 points at 10%, 20%, and 35% budgets over the history policy, with paired intervals excluding zero; the camera cue drives this gain, and the first-sighting versus overall trade-off is a budget-dependent Pareto choice. Third, a safety-bounded forgetting module that releases budget from receding or static tracks beyond a speed-dependent guard distance; at tight budgets, forgetting without the guard significantly harms near-field recall, so the guard is what keeps it safe. The pipeline runs end to end on a real vehicle and, in closed-loop simulation, detects an occluded pedestrian earlier and brakes more reliably than history-driven scanning.",
  "published": "2026-08-25",
  "updated": "2026-08-25",
  "year": "2026",
  "authors": [
   "Jiachen Gong",
   "Yun Li",
   "Ehsan Javanmardi",
   "Wencan Mao",
   "Manabu Tsukada"
  ],
  "author_count": 5,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The CAmera-REsidual reserve (CARE), a training-free allocation rule that reserves part of a fixed ray budget for the directions of current camera detections that the track forecasts cannot explain; the rest follows the base history policy, and unused reserve returns to a random floor.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jiachen Gong",
    "id": "2451072392",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yun Li",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Ehsan Javanmardi",
    "id": "3365649",
    "h_index": 13,
    "papers": 92
   },
   {
    "name": "Wencan Mao",
    "id": "2332843099",
    "h_index": 5,
    "papers": 19
   },
   {
    "name": "Manabu Tsukada",
    "id": "2399034184",
    "h_index": 2,
    "papers": 7
   }
  ],
  "comment": "",
  "topics": [
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.24282v1",
  "pdf_url": "https://arxiv.org/pdf/2608.24282v1",
  "html_url": "https://arxiv.org/html/2608.24282v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.24242",
  "slug": "a-durable-vision-based-tactile-fingertip-for-robotic-manipulation",
  "title": "A Durable Vision-Based Tactile Fingertip for Robotic Manipulation",
  "abstract": "Currently available commercial vision-based tactile sensors provide rich contact information but remain vulnerable to abrasion and repeated concentrated loading, limiting their use in demanding robotic applications. This work presents a durable tactile fingertip comprising a soft silicone gel with a nonpigmented, textured, thin thermoplastic-polyurethane protective film and a replaceable sensing cartridge. Durability was evaluated using two accelerated laboratory procedures: a rotating-drum sanding test and a repetitive probe test applying 39.2 N (4.0 kgf) at 45 cycles per minute. Under the defined sanding conditions, the developed sensor reached the protective-film rupture endpoint after approximately 2-3 hours. During repetitive probe testing, all nine developed sensors remained functionally usable when testing was discontinued: seven after 5 days, one after 6 days, and one after 8 days. Commercial GelSight Mini and DIGIT specimens exhibited initial surface-film rupture after approximately 24-30 seconds of sanding and 25-35 minutes of repetitive loading. Damage to the developed sensor progressed gradually and produced little interference with tactile imaging at the test endpoints. These observations establish durability improvements of more than two orders of magnitude under the defined accelerated conditions. Combining increased durability, gradual degradation, and rapid cartridge replacement offers a practical approach to maintainable vision-based tactile sensing for demanding robotic applications.",
  "published": "2026-08-25",
  "updated": "2026-08-25",
  "year": "2026",
  "authors": [
   "F. Richard Cottrell",
   "Megha H. Tippur",
   "Edward H. Adelson"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "F. Cottrell",
    "id": "146697071",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Megha H. Tippur",
    "id": "2120213575",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Edward H. Adelson",
    "id": "2293172170",
    "h_index": 8,
    "papers": 18
   }
  ],
  "comment": "22 pages, 20 figures",
  "topics": [
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.24242v1",
  "pdf_url": "https://arxiv.org/pdf/2608.24242v1",
  "html_url": "https://arxiv.org/html/2608.24242v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.24223",
  "slug": "event-based-motion-estimation-via-oriented-distance-fields",
  "title": "Event-Based Motion Estimation via Oriented Distance Fields",
  "abstract": "Event-based motion estimation is central to tasks that demand high temporal resolution and robustness to fast motion. Existing methods typically rely on iterative optimization or repeated hypothesis comparison, offsetting the sensor's low-latency advantage. We propose Oriented Distance Field Motion Estimation (ODF Motion Estimation), which replaces this optimization with a single averaging step over a precomputed field of event distance vectors, combined with an adaptive event-count selection strategy and a parameter-free trail filter. On public and self-collected datasets, ODF motion estimation reaches sub-pixel accuracy at the lowest latency among compared methods. We validate its generality on two downstream applications rather than treating them as separate contributions. First, the estimated trajectory is converted into a blur kernel and paired with a compact iterative-unfolding network, trained on simulated motion-estimation noise, for real-time non-blind image deblurring, attaining competitive or superior PSNR/SSIM with under 1M parameters. Second, the same precomputed field is repurposed for directional event filtering in a low-power asynchronous pupil and glint tracker, sustaining stable tracking for tens of seconds while lowering a near-eye module's power draw.",
  "published": "2026-08-25",
  "updated": "2026-08-25",
  "year": "2026",
  "authors": [
   "Lei Sun",
   "Yuqin Ma",
   "Weilun Li",
   "Haoran Liang",
   "Runyi Yang",
   "Kaiwei Wang",
   "Danda Pani Paudel",
   "Luc Van Gool"
  ],
  "author_count": 8,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Oriented Distance Field Motion Estimation (ODF Motion Estimation), which replaces this optimization with a single averaging step over a precomputed field of event distance vectors, combined with an adaptive event-count selection strategy and a parameter-free trail filter, reaches sub-pixel accuracy at the lowest latency among compared methods.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Lei Sun",
    "id": "2247647289",
    "h_index": 6,
    "papers": 23
   },
   {
    "name": "Yuqin Ma",
    "id": "2290902582",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Weilun Li",
    "id": "2355902713",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Haoran Liang",
    "id": "2457737915",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Runyi Yang",
    "id": "2340465172",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Kaiwei Wang",
    "id": "2290695471",
    "h_index": 8,
    "papers": 17
   },
   {
    "name": "D. Paudel",
    "id": "35268081",
    "h_index": 32,
    "papers": 203
   },
   {
    "name": "L. V. Gool",
    "id": "2246990749",
    "h_index": 31,
    "papers": 217
   }
  ],
  "comment": "",
  "topics": [
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.24223v1",
  "pdf_url": "https://arxiv.org/pdf/2608.24223v1",
  "html_url": "https://arxiv.org/html/2608.24223v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.24217",
  "slug": "caro-contact-agnostic-residual-observation-for-zero-shot-robust-quadru",
  "title": "CARO: Contact-Agnostic Residual Observation for Zero-Shot Robust Quadruped Locomotion",
  "abstract": "We propose CARO, a contact-agnostic residual observation framework for policy adaptation. CARO embeds a fixed-base Euler--Lagrange model into the reinforcement learning control loop and constructs a torque-level residual observation without requiring torque sensors, explicit contact estimation, or vision-based measurements of the floating-base position and linear velocity. A disturbance observer extracts a structured signal representing dynamics mismatch, while the policy learns to exploit this feedback for online adaptation. CARO is trained under the same terrain, command, and domain-randomization conditions as the nominal policy, without specialized disturbance curricula or additional adaptation supervision. Nevertheless, it achieves substantially improved zero-shot robustness in simulation and sim-to-real transfer tasks involving out-of-distribution payloads, center-of-mass shifts, terrain geometries, abrupt dynamics changes, and elevated-platform landings.",
  "published": "2026-08-25",
  "updated": "2026-08-25",
  "year": "2026",
  "authors": [
   "Zihan Yang",
   "Shixuan Han",
   "Kexin Guo",
   "Xiang Yu"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "CARO achieves substantially improved zero-shot robustness in simulation and sim-to-real transfer tasks involving out-of-distribution payloads, center-of-mass shifts, terrain geometries, abrupt dynamics changes, and elevated-platform landings.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zihan Yang",
    "id": "2313686511",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Shixuan Han",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Kexin Guo",
    "id": "2301930605",
    "h_index": 6,
    "papers": 27
   },
   {
    "name": "Xiang Yu",
    "id": "2403003287",
    "h_index": 0,
    "papers": 3
   }
  ],
  "comment": "10 pages, 6 figures",
  "topics": [
   "humanoids",
   "sim2real",
   "rl-control",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.24217v1",
  "pdf_url": "https://arxiv.org/pdf/2608.24217v1",
  "html_url": "https://arxiv.org/html/2608.24217v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.24199",
  "slug": "nvidia-cosmos-h-dreams-real-time-generative-physics-simulation-for-sur",
  "title": "NVIDIA Cosmos-H-Dreams: Real-Time Generative Physics Simulation for Surgical Robotics",
  "abstract": "Generative simulation for surgical robotics still lacks real-time interaction. Physical-robot experiments, often involving animal or cadaver labs, are time-consuming, costly, and difficult to reproduce, while classical simulators struggle to capture photorealistic appearance and deformable-tissue dynamics. We address this gap with Cosmos-H-Dreams, an integrated real-time surgical world-model system combining an action-conditioned generative model, a teacher-to-student distillation recipe, and a deployment stack built on the NVIDIA FlashDreams streaming-inference library. Starting from Cosmos-H-Surgical-Simulator, a multi-embodiment action-conditioned surgical video world model fine-tuned on the large-scale Open-H-Embodiment corpus, we post-train this checkpoint on embodiment- and procedure-specific data. By distilling the resulting bidirectional teacher into a causal, few-step student with Self Forcing, we turn a passive video generator into a controllable surgical simulator that streams at $\\sim$160 inference FPS on a single NVIDIA RTX PRO 6000 Blackwell workstation GPU. Crucially, Cosmos-H-Dreams is controller-agnostic: any interface that emits a stream of robot kinematics can drive it. We demonstrate live control through a browser keyboard over WebRTC, a Meta Quest headset over WebXR, a commercial surgical robot console such as CMR Surgical's Versius, and learned policies operating in closed loop. To our knowledge, this is the first interactive surgical world model supporting live human and policy control. Human operators and policies alike can act inside the synthesized world and observe the consequences in real time. We release Cosmos-H-Dreams as an open surgical simulation system, providing a common foundation for surgical education, scalable synthetic data generation, and future intraoperative decision support.",
  "published": "2026-08-25",
  "updated": "2026-08-27",
  "year": "2026",
  "authors": [
   "Javier Gamazo Tejero",
   "Lukas Zbinden",
   "Keyur Sheth",
   "Raghavendra K M",
   "Nadim Daher",
   "Diego Granero Mara\u00f1a",
   "Filip Binkiewicz",
   "Patrick Thornycroft",
   "Mahdi Azizian",
   "Sean D. Huver"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Cosmos-H-Dreams is an integrated real-time surgical world-model system combining an action-conditioned generative model, a teacher-to-student distillation recipe, and a deployment stack built on the NVIDIA FlashDreams streaming-inference library, providing a common foundation for surgical education, scalable synthetic data generation, and future intraoperative decision support.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "J. G. Tejero",
    "id": "2365267837",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "L. Zbinden",
    "id": "6629817",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Keyur Sheth",
    "id": "2459246173",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "M. RaghavendraK",
    "id": "2459245874",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "N. Daher",
    "id": "2033914612",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Diego Granero Mara\u00f1a",
    "id": "2431187957",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "F. Binkiewicz",
    "id": "2083432587",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "P. Thornycroft",
    "id": "5754885",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Mahdi Azizian",
    "id": "2300285188",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Sean Huver",
    "id": "10419594",
    "h_index": 8,
    "papers": 30
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "sim2real",
   "data-teleop"
  ],
  "orgs": [
   "NVIDIA"
  ],
  "abs_url": "https://arxiv.org/abs/2608.24199v2",
  "pdf_url": "https://arxiv.org/pdf/2608.24199v2",
  "html_url": "https://arxiv.org/html/2608.24199v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2608.24162",
  "slug": "robust-slip-detection-and-material-classification-via-spatiotemporal-t",
  "title": "Robust Slip Detection and Material Classification via Spatiotemporal Transformers on a Uniformly-Illuminated Visuo-Tactile Sensor",
  "abstract": "Tactile sensing is central to robotic manipulation, among which slip detection stands out as a quintessential and critical task. However, existing slip datasets are predominantly limited to binary classification, lacking fine-grained directional perception. To address this limitation, we propose a visuo-tactile sensor featuring customized uniform RGB illumination, alongside a unified perception framework. At the hardware level, the sensor achieves high-precision, sub-millimeter depth reconstruction. Based on this capability, we collect a multi-task visuo-tactile dataset encompassing 15 objects, synchronously generating depth information for each data sample. Algorithmically, we design a dual-head TimeSformer network to process dynamic spatiotemporal slip. On unseen objects, this network achieves robust accuracies of 95.5% and 91.5% for 3-class contact state prediction and fine-grained 8-class slip direction classification, respectively. Furthermore, static tactile-based object class recognition utilizing a ResNet-50 backbone yields an outstanding accuracy of 98.8% across 15 categories. The proposed hardware-software framework provides high-fidelity feedback and a powerful multi-modal perception baseline for complex robotic manipulation.",
  "published": "2026-08-25",
  "updated": "2026-08-25",
  "year": "2026",
  "authors": [
   "Ziyang Ma",
   "Yuhao Sun",
   "Zichen Ai",
   "Xiangyang Ji",
   "Bin Fang"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ziyang Ma",
    "id": "2459455402",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Yuhao Sun",
    "id": "2189498375",
    "h_index": 10,
    "papers": 23
   },
   {
    "name": "Zichen Ai",
    "id": "2376448161",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Xiangyang Ji",
    "id": "2459267876",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Bin Fang",
    "id": "2345041551",
    "h_index": 3,
    "papers": 9
   }
  ],
  "comment": "Accepted at IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS) 2026. 8 pages, 10 figures",
  "topics": [
   "tactile",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.24162v1",
  "pdf_url": "https://arxiv.org/pdf/2608.24162v1",
  "html_url": "https://arxiv.org/html/2608.24162v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.24155",
  "slug": "coverage-planning-for-robotic-tooth-preparation-in-densely-constrained",
  "title": "Coverage Planning for Robotic Tooth Preparation in Densely Constrained Environments",
  "abstract": "Tooth preparation refers to the controlled removal of tooth structure to create an optimal substrate for fixed restorations and is a core procedure in restorative dentistry. Automating this task is particularly challenging for robots because the dental bur must operate within a densely constrained intraoral workspace, where even sub-millimeter deviations can compromise outcomes or damage adjacent structures. This paper presents a novel robotic system for autonomous full-crown tooth preparation. The proposed framework includes: 1) an anatomy-aware toolpath planning algorithm that conforms precisely to a technician-designed preparation model while protecting adjacent teeth, and 2) a clearance-oriented end-effector yaw assignment strategy that allows intraoral access while reducing the risk of soft-tissue interference. Together, these features enable the robot to accurately mill the irregular tooth surface with an average geometric deviation of 0.117 mm (RMSE), achieving both restoration quality and clinical safety. A series of simulations and phantom-head experiments validate the system's feasibility and effectiveness.",
  "published": "2026-08-25",
  "updated": "2026-08-25",
  "year": "2026",
  "authors": [
   "Yunwen Li",
   "Chen Chen",
   "Xiangjie Yan",
   "Chang Shu",
   "Jianxia Hou",
   "Shiji Song",
   "Xiang Li"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper presents a novel robotic system for autonomous full-crown tooth preparation that includes an anatomy-aware toolpath planning algorithm that conforms precisely to a technician-designed preparation model while protecting adjacent teeth, and a clearance-oriented end-effector yaw assignment strategy that allows intraoral access while reducing the risk of soft-tissue interference.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yunwen Li",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Chen Chen",
    "id": "2297434703",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Xiangjie Yan",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Chang Shu",
    "id": "2322523052",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Jianxia Hou",
    "id": "2115103272",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Shiji Song",
    "id": "2301516416",
    "h_index": 2,
    "papers": 17
   },
   {
    "name": "Xiang Li",
    "id": null,
    "h_index": 0,
    "papers": 0
   }
  ],
  "comment": "",
  "topics": [
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.24155v1",
  "pdf_url": "https://arxiv.org/pdf/2608.24155v1",
  "html_url": "https://arxiv.org/html/2608.24155v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.24115",
  "slug": "ponderpounce-a-pretrained-mllm-as-an-episode-context-engine-for-robot",
  "title": "PonderPounce: A Pretrained MLLM as an Episode Context Engine for Robot Control",
  "abstract": "Multimodal large language models (MLLMs) can integrate long visual histories, reason under partial observability, and infer behavior from a few examples. Yet vision-language-action (VLA) models generally inherit pretrained representations without using this contextual capacity as episode memory. Memory-dependent policies address this gap through purpose-built history mechanisms. PonderPounce instead reuses an MLLM's native causal context as robot memory. Ponder, a System2 MLLM, accumulates episode observations, demonstrations, and prior cognition in its native causal context and can generate subgoal text and demonstration reasoning for internal use. Pounce, a System1 VLA, receives the current observation, instruction, and proprioception directly; through the Ponder--Pounce interface, it asynchronously receives only the newest continuous cognition token and its age. Both are jointly trained end to end without a purpose-built memory module or separate bridge pretraining. Optimized serving achieves p50 latencies of 78ms for cognition refresh and 25ms for action-model invocation, supporting 20Hz action playback. On RoboMME with base-scale training data, PonderPounce reaches 60.83% with 9B and 50.04% with 0.8B under the same Pounce architecture and interface, versus 44.51% for FrameSamp+Modul and 17.93% for the current-observation \u03c0_{0.5}. With 9x data, it reaches 75.54% versus 57.88% for FrameSamp+Modul. On RoboCasa-DC, the same interface learns from action supervision alone and reaches 12.5% versus 11.6% for the strongest published demonstration-conditioned baseline, falling to 8.6% when cognition is replaced by a learned null state.",
  "published": "2026-08-25",
  "updated": "2026-08-25",
  "year": "2026",
  "authors": [
   "Suhwan Choi",
   "Jaeyoon Jung",
   "Sungkyung Kim",
   "Yunsung Lee",
   "Youngjae Yu"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "PonderPounce reuses an MLLM's native causal context as robot memory for vision-language-action models, and is jointly trained end to end without a purpose-built memory module or separate bridge pretraining.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Suhwan Choi",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Jaeyoon Jung",
    "id": "2352990194",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Sungkyung Kim",
    "id": "2268303693",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Yunsung Lee",
    "id": "2384469692",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Youngjae Yu",
    "id": "2385782381",
    "h_index": 2,
    "papers": 5
   }
  ],
  "comment": "Project page: https://worv-ai.github.io/ponderpounce/",
  "topics": [
   "vla",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.24115v1",
  "pdf_url": "https://arxiv.org/pdf/2608.24115v1",
  "html_url": "https://arxiv.org/html/2608.24115v1",
  "code_url": "https://worv-ai.github.io/ponderpounce/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2608.24111",
  "slug": "trajectory-level-continuous-action-representation-for-robotic-manipula",
  "title": "Trajectory-Level Continuous Action Representation for Robotic Manipulation",
  "abstract": "We propose CAT, a trajectory-level continuous action representation framework for robotic manipulation. Existing visuomotor systems often entangle action representation with control frequency or rely on fixed temporal parameterizations. This leads to representational redundancy at high sampling rates and limits the modeling of critical motion. CAT instead encodes action trajectories within a fixed real-time interval into a set of continuous latent tokens. To ensure temporal consistency across varying control frequencies, we further incorporate a frequency-aware positional encoding that establishs a shared temporal coordinate system. Trajectory-level regularization further stabilizes the latent representation. This approach prevents representation growth with timestep density and avoids reliance on predefined temporal parameterizations. Extensive system-level evaluations on LIBERO, MimicGen, and real-world long-horizon manipulation tasks demonstrate that CAT-based policies consistently outperform both competitive VQ-based and continuous visuomotor baselines under matched training settings. Across various model backbones and control frequencies, CAT consistently improves success rates. These results highlight the advantages of trajectory-level continuous action modeling for scalable robotic manipulation across varying control rates.",
  "published": "2026-08-25",
  "updated": "2026-08-26",
  "year": "2026",
  "authors": [
   "Tong Yang",
   "Jingkai Jia",
   "Yuecheng Xu",
   "Xueyao Chen",
   "Chi Zhang",
   "Wenqiang Zhang"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "CAT, a trajectory-level continuous action representation framework for robotic manipulation, is proposed and extensive system-level evaluations demonstrate that CAT-based policies consistently outperform both competitive VQ-based and continuous visuomotor baselines under matched training settings.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Tong Yang",
    "id": "2263762687",
    "h_index": 7,
    "papers": 17
   },
   {
    "name": "Jingkai Jia",
    "id": "2117241998",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Yuecheng Xu",
    "id": "2274203929",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Xue-Yao Chen",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Chi Zhang",
    "id": "2279678254",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Wenqiang Zhang",
    "id": "2342633447",
    "h_index": 2,
    "papers": 6
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.24111v2",
  "pdf_url": "https://arxiv.org/pdf/2608.24111v2",
  "html_url": "https://arxiv.org/html/2608.24111v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.24101",
  "slug": "tract-bridging-robot-control-and-visual-prediction-with-visual-tracks",
  "title": "TrAct: Bridging Robot Control and Visual Prediction with Visual Tracks",
  "abstract": "Robot actions are inherently embodiment-specific and only weakly aligned with image-space visual changes, limiting their effectiveness as conditioning signals for robot world models. In contrast, visual tracks provide an embodiment-agnostic representation of how task-relevant points move through a scene, offering dense image-space guidance for accurate and spatially precise future video prediction. Building on this observation, we propose TrAct, a world-model-based robot decision-making framework that uses visual tracks as an intermediate interface between control and prediction. TrAct consists of three components: a Vision-Language-Action-and-Track model (VLAT) that jointly predicts candidate actions and corresponding visual tracks from the current observation and language instruction; a track-conditioned world model (TWM) that predicts future visual outcomes conditioned on the proposed tracks; and a vision-language reward model (VLAC) that scores the predicted outcomes. At inference time, VLAT generates candidate action-track pairs, TWM rolls out their visual consequences, and VLAC selects the track whose predicted outcome best satisfies the instruction; the action paired with the selected track is then executed by the robot. Experiments on the proposed LIBERO-INTEGRAL benchmark and real-world Franka manipulation show that TrAct improves success rates from 27% to 55% in simulation and from 49% to 76% on real-world tasks compared with the strong VLA baseline $\u03c0_{0.5}$. Furthermore, TWM consistently improves video prediction quality over the action-conditioned world model (AWM). These results demonstrate that visual tracks provide an effective shared interface between robot control and visual prediction, enabling more accurate world modeling and stronger robot generalization.",
  "published": "2026-08-25",
  "updated": "2026-08-25",
  "year": "2026",
  "authors": [
   "Zhi Cao",
   "Howard Ji",
   "Kevin Zhang",
   "Kuangzhi Ge",
   "Li Fei-Fei",
   "Jiajun Wu",
   "Huang Huang"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "TrAct is proposed, a world-model-based robot decision-making framework that uses visual tracks as an intermediate interface between control and prediction, enabling more accurate world modeling and stronger robot generalization.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhihang Cao",
    "id": "2458268211",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Howard Ji",
    "id": "2459267375",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Kevin Zhang",
    "id": "2383305800",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Kuangzhi Ge",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Fei-Fei Li",
    "id": "2330589126",
    "h_index": 3,
    "papers": 15
   },
   {
    "name": "Jiajun Wu",
    "id": "2326444500",
    "h_index": 35,
    "papers": 214
   },
   {
    "name": "Huang Huang",
    "id": "2146053008",
    "h_index": 16,
    "papers": 36
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "vla",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.24101v1",
  "pdf_url": "https://arxiv.org/pdf/2608.24101v1",
  "html_url": "https://arxiv.org/html/2608.24101v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.24094",
  "slug": "siren-bench-behavior-driven-generation-and-evaluation-of-emergency-veh",
  "title": "SIREN-Bench: Behavior-Driven Generation and Evaluation of Emergency-Vehicle Interactions",
  "abstract": "Emergency vehicles (EMVs) can reorganize surrounding traffic as civilian vehicles brake, change lanes, or form rescue corridors in response to their passage. Evaluating these safety-critical interactions requires behavior-level control over both EMV privileges and civilian responses, together with consistent sensing and ground truth. Existing datasets and simulation benchmarks do not directly provide this combination. We present \\textbf{SIREN}, a behavior-driven SUMO--CARLA co-simulation platform for generating EMV--civilian interactions. SIREN couples SUMO's network-level traffic evolution and behavior logic with CARLA's continuous vehicle control and synchronized onboard sensing; depending on the active behavior, the interaction is controlled by SUMO, CARLA, or jointly. We instantiate the platform as \\textbf{SIREN-Bench-v1}, comprising seven parameterized interaction templates across emergency levels L1--L3 and three behavior families, with synchronized sensor observations and simulator-native annotations. We demonstrate the benchmark through three representative tasks: 3D object detection, trajectory prediction, and vision-language risk understanding. Evaluations of nine trajectory predictors, four LiDAR-based detectors, and five vision-language models reveal behavior-dependent failure modes. Traffic-clearance interactions are hardest for detection, privileged intersection traversal is hardest for prediction, and no learned predictor outperforms the constant-velocity reference on average. Vision-language models perform substantially better on normal traffic than on near-miss and collision events. These results demonstrate the value of behavior-centered benchmarking and establish SIREN as an extensible data-generation and evaluation platform for autonomous-driving and transportation safety research.",
  "published": "2026-08-25",
  "updated": "2026-08-25",
  "year": "2026",
  "authors": [
   "Yicheng Zhu",
   "Tianmu Zhao",
   "Haoxin Leng",
   "Fan Zuo",
   "Tao Li",
   "Zilin Bian"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The results demonstrate the value of behavior-centered benchmarking and establish SIREN as an extensible data-generation and evaluation platform for autonomous-driving and transportation safety research.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yichen Zhu",
    "id": "2383420535",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Tian Zhao",
    "id": "2444804444",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Haoxin Leng",
    "id": "2459245182",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Fan Zuo",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Tao Li",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Zilin Bian",
    "id": "152385441",
    "h_index": 12,
    "papers": 44
   }
  ],
  "comment": "",
  "topics": [
   "sim2real",
   "data-teleop",
   "video-generation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.24094v1",
  "pdf_url": "https://arxiv.org/pdf/2608.24094v1",
  "html_url": "https://arxiv.org/html/2608.24094v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.24042",
  "slug": "hierarchical-skill-retrieval-for-data-efficient-adaptation-of-vision-l",
  "title": "Hierarchical Skill Retrieval for Data-Efficient Adaptation of Vision-Language-Action Models",
  "abstract": "While Vision-Language-Action (VLA) models pretrained on large-scale robot datasets provide a strong foundation for robot manipulation, their performance can degrade when adapted to new tasks with limited task-specific demonstrations. Retrieval offers a practical way to reuse existing demonstrations for data-efficient adaptation, but existing methods often rely on visual similarity, state-action representations, or task-level language matching. These approaches may overlook the hierarchical structure of long-horizon manipulation tasks, where complete task matches are rare but reusable skills are often abundant. To address this challenge, we propose Hierarchical Skill Retrieval (HSR), a retrieval framework for data-efficient VLA adaptation. Specifically, HSR first decomposes a target task into candidate skill sequences. It evaluates each plan based on both semantic plausibility and skill reliability estimated from the prior dataset. The selected decomposition is then used for hybrid retrieval. This combines subtask-level language retrieval with behavior-feature reranking to identify demonstrations that are both semantically relevant and compatible with the target task. Finally, we adapt the policy through a two-stage pretraining and finetuning pipeline, which separates general skill acquisition from task-specific adaptation. Experiments on the LIBERO benchmark and several real-world robot manipulation tasks show that HSR improves the average success rate by 10.3% and 21.3% over the strongest baseline, respectively. These results demonstrate the effectiveness of structured skill-level retrieval for data-efficient VLA adaptation. Videos and code are available at https://hoar012.github.io/HSR-Project.",
  "published": "2026-08-25",
  "updated": "2026-08-25",
  "year": "2026",
  "authors": [
   "Haoran Hao",
   "Shahram Najam Syed",
   "Jeff Schneider",
   "Jeffrey Ichnowski"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "H Hierarchical Skill Retrieval first decomposes a target task into candidate skill sequences, then evaluates each plan based on both semantic plausibility and skill reliability estimated from the prior dataset, and combines subtask-level language retrieval with behavior-feature reranking for hybrid retrieval.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Haoran Hao",
    "id": "2326297036",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Shahram Najam Syed",
    "id": "35555523",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Jeff Schneider",
    "id": "2307561296",
    "h_index": 1,
    "papers": 9
   },
   {
    "name": "Jeffrey Ichnowski",
    "id": "2391709934",
    "h_index": 1,
    "papers": 10
   }
  ],
  "comment": "Project Page: https://hoar012.github.io/HSR-Project",
  "topics": [
   "vla",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.24042v1",
  "pdf_url": "https://arxiv.org/pdf/2608.24042v1",
  "html_url": "https://arxiv.org/html/2608.24042v1",
  "code_url": "https://hoar012.github.io/HSR-Project",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2608.24039",
  "slug": "design-to-plan-a-large-language-model-based-multi-agent-framework-for",
  "title": "Design-to-Plan: A Large Language Model-Based Multi-Agent Framework for Manufacturing Process Planning from 3D CAD Models and 2D Engineering Drawings",
  "abstract": "Manufacturing process planning transforms heterogeneous design information into coherent manufacturing decisions. However, existing approaches focus on isolated subtasks, such as feature recognition, drawing interpretation, or tool selection, and struggle to support the full reasoning chain from design artifacts to process plans. This is critical when planning must interpret 3D CAD models, 2D engineering drawings, materials, and domain-specific rules. To address this gap, this paper presents Design-to-Plan, a large language model (LLM)-based multi-agent framework for end-to-end manufacturing process planning. An orchestrator coordinates specialized agents for 3D feature recognition, 2D drawing analysis, 2D-3D context fusion, knowledge retrieval, process sequencing, tool selection, and report generation. Rather than using LLMs as standalone text generators, the framework deploys them as reasoning agents that interact with deterministic modules and knowledge sources to produce consistent and traceable decisions. In this hybrid design, deterministic modules and specialized agents extract structured information from CAD and drawing inputs, while LLM agents perform context-aware reasoning, retrieve manufacturing rules, resolve conflicts, and generate planning outputs. The framework is evaluated using 300 benchmark cases across three downstream ReAct-enabled agents, plus separate evaluations of CAD feature recognition, drawing analysis, and 2D-3D context fusion. The parallel architecture achieves 100% success across downstream agents, Tool F1 scores of 95.9%-97.6%, 90% source detection accuracy in conflict analysis, and a 60%-68% reduction in token usage for key planning tasks. Results show that structured LLM-based multi-agent coordination can bridge design representations and manufacturing knowledge, enabling scalable, efficient, and traceable design-to-plan automation.",
  "published": "2026-08-25",
  "updated": "2026-08-25",
  "year": "2026",
  "authors": [
   "Muhammad Tayyab Khan",
   "Lequn Chen",
   "Wenhe Feng",
   "Seung Ki Moon"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Results show that structured LLM-based multi-agent coordination can bridge design representations and manufacturing knowledge, enabling scalable, efficient, and traceable design-to-plan automation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Muhammad Tayyab Khan",
    "id": "2316099319",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Lequn Chen",
    "id": "32572493",
    "h_index": 18,
    "papers": 37
   },
   {
    "name": "Wenhe Feng",
    "id": "2298021526",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Seung Ki Moon",
    "id": "2297854935",
    "h_index": 6,
    "papers": 15
   }
  ],
  "comment": "Submitted to Elsevier Journal",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.24039v1",
  "pdf_url": "https://arxiv.org/pdf/2608.24039v1",
  "html_url": "https://arxiv.org/html/2608.24039v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.24026",
  "slug": "neurraft-robot-motion-planning-via-anchor-level-flow-matching-with-cle",
  "title": "NeurRAFT: Robot Motion Planning via Anchor-Level Flow Matching with Clearance-Aware Preference Tuning",
  "abstract": "Recent end-to-end neural motion planners generate trajectories from raw sensor observations, avoiding the privileged geometric models required by classical planners. However, collision-free planning in cluttered environments remains challenging. We present NeurRAFT, a generative planning framework based on anchor-level flow matching and clearance-aware preference tuning. Unlike prior neural planners that model dense waypoint sequences and spend capacity on redundant local details and smoothness, NeurRAFT operates on compact anchor waypoints. We train the planner using a Jacobian-weighted loss that accounts for the task-space impact of each anchor. At inference, the anchors are generated in two integration steps, followed by cubic-spline interpolation to recover a smooth, full-resolution trajectory. Since imitation learning from positive demonstrations cannot distinguish collision-free from near-collision trajectories, collision-prone behaviors persist at test time. Rather than relying on post-hoc corrections, we directly reshape the pretrained planner's distribution toward safer solutions without augmenting inference. Specifically, Direct Preference Optimization shifts probability mass toward trajectories with larger obstacle clearance, with the resulting improvement directly absorbed into the planner parameters. Experiments show substantial improvements over state-of-the-art planners, while real-world experiments demonstrate zero-shot transfer to a Franka robot under noisy and partially occluded depth observations. Video results available at https://neurraft.github.io/.",
  "published": "2026-08-25",
  "updated": "2026-08-25",
  "year": "2026",
  "authors": [
   "Sibo Tian",
   "Chang Liu",
   "Minghui Zheng",
   "Xiao Liang"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work presents NeurRAFT, a generative planning framework based on anchor-level flow matching and clearance-aware preference tuning that operates on compact anchor waypoints, and directly reshape the pretrained planner's distribution toward safer solutions without augmenting inference.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Sibo Tian",
    "id": "2180075125",
    "h_index": 8,
    "papers": 20
   },
   {
    "name": "Chang Liu",
    "id": "2382073084",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Minghui Zheng",
    "id": "40387914",
    "h_index": 15,
    "papers": 65
   },
   {
    "name": "Xiao Liang",
    "id": null,
    "h_index": 0,
    "papers": 0
   }
  ],
  "comment": "",
  "topics": [
   "imitation-diffusion"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.24026v1",
  "pdf_url": "https://arxiv.org/pdf/2608.24026v1",
  "html_url": "https://arxiv.org/html/2608.24026v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.24019",
  "slug": "trusted-polytopic-action-sets-for-fast-planning-in-underactuated-syste",
  "title": "Trusted Polytopic Action Sets for Fast Planning in Underactuated Systems",
  "abstract": "Underactuated systems pose a challenge for convex motion planning because their dynamically feasible motions lie on a manifold of trajectories in function space. Building on our earlier formulation of polytopic action sets (PAS), this paper presents a method for rapidly generating, online, trusted convex sets of short-horizon actions for underactuated and potentially nonlinear systems. Around a nominal trajectory, we construct local finite-dimensional action coordinates in which each parameter vector encodes a complete nearby motion through an affine trajectory map, rendering collision-avoidance and control bounds linear. To remain consistent with the nonlinear dynamics, we introduce a dynamics-violation metric and extract a trusted convex inner approximation using an IRIS-inspired inflation procedure directly in action space. The resulting PAS are reusable convex families of actions that can be queried and composed with linear programs, and a PAS-guided tree expansion treats nodes as composed reachable families rather than single trajectories, coupling local nonlinear fidelity with convex reuse for longer-horizon planning. The planner solves cluttered planar scenes in tens of milliseconds (14-78x faster than a kinodynamic RRT baseline) and reduces terminal error on a nonlinear underactuated benchmark by 26-86% over sampling and NLP baselines.",
  "published": "2026-08-25",
  "updated": "2026-08-25",
  "year": "2026",
  "authors": [
   "Akshay Jaitly",
   "Siavash Farzan"
  ],
  "author_count": 2,
  "categories": [
   "eess.SY",
   "cs.RO"
  ],
  "primary_category": "eess.SY",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A dynamics-violation metric is introduced and a trusted convex inner approximation is extracted using an IRIS-inspired inflation procedure directly in action space, resulting in reusable convex families of actions that can be queried and composed with linear programs.",
  "doi": "10.1109/LCSYS.2026.3718036",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "A. Jaitly",
    "id": "2292024506",
    "h_index": 2,
    "papers": 13
   },
   {
    "name": "Siavash Farzan",
    "id": "3059735",
    "h_index": 8,
    "papers": 27
   }
  ],
  "comment": "Accepted for publication in IEEE Control Systems Letters (L-CSS); to be presented at the 2026 IEEE Conference on Decision and Control. Code available at https://github.com/akshay5312/paamp_underactuated",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.24019v1",
  "pdf_url": "https://arxiv.org/pdf/2608.24019v1",
  "html_url": "https://arxiv.org/html/2608.24019v1",
  "code_url": "https://github.com/akshay5312/paamp_underactuated",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2608.23994",
  "slug": "bridging-teacher-expectations-and-robot-learning-via-coupling-dynamics",
  "title": "Bridging Teacher Expectations and Robot Learning via Coupling Dynamics",
  "abstract": "Human-robot teaching focuses on enabling nontechnical experts to customize robots according to their needs after deployment. With recent advances in machine learning, human-robot teaching is no longer confined to offline learning where the data gathering step from a human teacher is separated from when the robot learns. Instead, more recent approaches for human-robot teaching focus on coupling human teaching with robot learning. This coupling impacts the structure, timing, and content of the teaching and learning interaction. However, it is currently unclear how such coupling dynamics affect humanrobot teaching effectiveness and human perceptions towards the teaching process. Informed by human learning theories, in this paper we propose a new scale for classifying human-robot teaching interactions according to coupling dynamics present between the human teacher and robot learner. We apply this scale to a subset of the human-robot teaching literature to identify how coupling dynamics and human teacher mental model mismatches with the ground truth robot learning system affect teaching effectiveness and human perceptions towards the teaching process",
  "published": "2026-08-25",
  "updated": "2026-08-25",
  "year": "2026",
  "authors": [
   "Evan Dallas",
   "Sean Dallas",
   "Wing-Yue Geoffrey Louie"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.HC"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A new scale for classifying human-robot teaching interactions according to coupling dynamics present between the human teacher and robot learner is proposed and applied to a subset of the human-robot teaching literature to identify how coupling dynamics and human teacher mental model mismatches with the ground truth robot learning system affect teaching effectiveness and human perceptions towards the teaching process.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Evan Dallas",
    "id": "2186555504",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Sean Dallas",
    "id": "2331871454",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "W. Louie",
    "id": "1867074",
    "h_index": 12,
    "papers": 49
   }
  ],
  "comment": "8 pages, 1 figure",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.23994v1",
  "pdf_url": "https://arxiv.org/pdf/2608.23994v1",
  "html_url": "https://arxiv.org/html/2608.23994v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.23983",
  "slug": "sensorless-damage-safe-grasping",
  "title": "Sensorless damage-safe grasping",
  "abstract": "Robotic fruit harvesting must hold produce securely without bruising it, yet compression stiffness varies several-fold with ripeness within a single species, so no fixed grip force spans the range. Rather than tune force, we bound deformation: a controller closes the gripper until the object's estimated compression strain reaches a user-specified limit $\\varepsilon$, using only the encoder position and motor-effort signal on every servo gripper---no tactile or force-torque sensor. Dividing an effort-based contact force by a lower bound on object stiffness makes the stop provably conservative---true compression stays at or below $\\varepsilon$---for any $\\varepsilon$ above a contact-detection strain floor we identify and quantify: robust detection itself spends compression, linearly in closing speed, making speed an explicit throughput--gentleness knob. Unlike a hand-tuned force threshold, $\\varepsilon$ is a certified, size-scaling, operator-interpretable damage limit, and a ready safe-action parameter for learned grasping policies. In MuJoCo simulation over a realistic fruit-stiffness range, under a sensor-noise model calibrated to the real servo, the controller holds $\\ge 98\\,\\%$ grasp at $0\\,\\%$ damage across all medium-to-firm stiffnesses for the entire certified $\\varepsilon$ range, which neither fixed-force baseline attains; on stiffness-graded 3D-printed TPU cubes it matches baseline grasp success at roughly half the grip force and cuts soft-object damage from $100\\,\\%$ to $40\\,\\%$.",
  "published": "2026-08-25",
  "updated": "2026-08-25",
  "year": "2026",
  "authors": [
   "Yusei Shuto",
   "Danilo Vasconcellos Vargas"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yusei Shuto",
    "id": "145980263",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Danilo Vasconcellos Vargas",
    "id": "2459245447",
    "h_index": 0,
    "papers": 0
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.23983v1",
  "pdf_url": "https://arxiv.org/pdf/2608.23983v1",
  "html_url": "https://arxiv.org/html/2608.23983v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.23972",
  "slug": "safety-aware-model-predictive-path-integral-control-with-signal-tempor",
  "title": "Safety-aware Model Predictive Path Integral Control with Signal Temporal Logic",
  "abstract": "Safety-aware motion planning remains a challenge in robotics, especially when missions are time-critical and are under complex specifications. In this paper, we propose safety-aware-stl-mppi, a computationally efficient sampling-based receding-horizon planning framework designed to promote satisfaction of constraints expressed in Signal Temporal Logic (STL). Our approach encodes discrete-time STL formulas into candidate time-varying control barrier functions (CBF), which are integrated into a model predictive path integral (MPPI) controller. Our method inherits the benefits of low computational cost from an efficiently parallelizable sampling based planner and utilizes CBF for constraints expressed in STL. We compare against several MPPI baselines using four artificial Mars Rover planning case studies with a diverse environment and cost setups, where we show our method consistently achieving high safety and efficiency. We show a quadcopter planning experiment with NVIDIA Isaac Lab.",
  "published": "2026-08-25",
  "updated": "2026-08-25",
  "year": "2026",
  "authors": [
   "Yiqi Zhao",
   "Taekyung Kim",
   "Hideki Okamoto",
   "Bardh Hoxha",
   "Jyotirmoy V. Deshmukh",
   "Lars Lindemann",
   "Georgios Fainekos"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper proposes safety-aware-stl-mppi, a computationally efficient sampling-based receding-horizon planning framework designed to promote satisfaction of constraints expressed in Signal Temporal Logic (STL).",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yiqi Zhao",
    "id": "6825484",
    "h_index": 15,
    "papers": 71
   },
   {
    "name": "Taekyung Kim",
    "id": "2305753219",
    "h_index": 5,
    "papers": 17
   },
   {
    "name": "Hideki Okamoto",
    "id": "2256348492",
    "h_index": 6,
    "papers": 26
   },
   {
    "name": "Bardh Hoxha",
    "id": "1729953",
    "h_index": 19,
    "papers": 99
   },
   {
    "name": "J. Deshmukh",
    "id": "2256346716",
    "h_index": 2,
    "papers": 17
   },
   {
    "name": "Lars Lindemann",
    "id": "2268408784",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Georgios Fainekos",
    "id": "1682745",
    "h_index": 41,
    "papers": 203
   }
  ],
  "comment": "",
  "topics": [
   "rl-control",
   "safety-eval"
  ],
  "orgs": [
   "NVIDIA"
  ],
  "abs_url": "https://arxiv.org/abs/2608.23972v1",
  "pdf_url": "https://arxiv.org/pdf/2608.23972v1",
  "html_url": "https://arxiv.org/html/2608.23972v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2608.23939",
  "slug": "codrift-compositional-drifting-for-offline-reinforcement-learning",
  "title": "CoDrift: Compositional Drifting for Offline Reinforcement Learning",
  "abstract": "Offline reinforcement learning is intrinsically multi-objective: a policy must remain compatible with the behavioral support of a fixed dataset while preferentially selecting high-value actions. We recast these objectives in a common form by viewing each as an action-space motion field that specifies how generated actions should move. This perspective enables heterogeneous learning objectives to be combined directly through field composition. Inspired by drifting models, we propose CoDrift, a compositional framework for one-step generative policy learning. CoDrift combines three objective-level fields into a unified policy field. The conditional field preserves state-dependent behavioral structure, while the marginal field pools actions across states to provide a more stable generative signal in the single-positive-sample regime of continuous-control offline RL. The value field moves generated actions toward higher-value regions. The composed field is absorbed into a stochastic generator that produces an action with a single forward pass at deployment. We evaluate CoDrift on 73 tasks from OGBench and D4RL in both offline and offline-to-online settings. CoDrift compares favorably with state-of-the-art methods and achieves the best average rank in both settings.",
  "published": "2026-08-25",
  "updated": "2026-08-25",
  "year": "2026",
  "authors": [
   "Xiewei Ni",
   "Ruofeng Mei",
   "Xiangyu Xu"
  ],
  "author_count": 3,
  "categories": [
   "cs.LG",
   "cs.RO"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes CoDrift, a compositional framework for one-step generative policy learning that combines three objective-level fields into a unified policy field that compares favorably with state-of-the-art methods and achieves the best average rank in both settings.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xiewei Ni",
    "id": "2419572732",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Ruo-Syuan Mei",
    "id": "2326150673",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Xiangyu Xu",
    "id": "2333462875",
    "h_index": 2,
    "papers": 9
   }
  ],
  "comment": "",
  "topics": [
   "imitation-diffusion",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.23939v1",
  "pdf_url": "https://arxiv.org/pdf/2608.23939v1",
  "html_url": "https://arxiv.org/html/2608.23939v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.23924",
  "slug": "dynamical-system-based-imitation-learning-and-neuroadaptive-control-fo",
  "title": "Dynamical System-Based Imitation Learning and Neuroadaptive Control for Trajectory Recovery in Autonomous Ships",
  "abstract": "Repetitive maritime operations can be effectively learned using the Imitation Learning (IL) paradigm, which transfers human expertise directly to Unmanned Surface Vehicle (USV) control systems. Dynamical Systems (DS) are widely used to model non-linear human demonstrations while offering inherent stability guarantees. However, real-world execution under persistent marine perturbations reveals a critical trade-off: standard DS-based IL approaches prioritize global target convergence at the expense of localized trajectory reproduction fidelity. To address this limitation, we present a hybrid learning-control architecture that integrates a DS-based IL reference generator with a neuroadaptive controller. Our approach introduces a control action that drives the USV back to the demonstrated path following exogenous disturbances, enabling dynamic human-like reactive alignment-termed behavioral tracking. The proposed methodology is validated using the Marine Systems Simulator (MSS) toolbox. Simulation results confirm that the framework generalizes complex maneuvering tasks while substantially improving trajectory tracking fidelity under disturbances compared to alternative control strategies.",
  "published": "2026-08-25",
  "updated": "2026-08-25",
  "year": "2026",
  "authors": [
   "Yeyson A. Becerra-Mora",
   "Jos\u00e9 \u00c1ngel Acosta"
  ],
  "author_count": 2,
  "categories": [
   "eess.SY",
   "cs.RO"
  ],
  "primary_category": "eess.SY",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work presents a hybrid learning-control architecture that integrates a DS-based IL reference generator with a neuroadaptive controller, enabling dynamic human-like reactive alignment-termed behavioral tracking under persistent marine perturbations.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yeyson A. Becerra-Mora",
    "id": "2226009943",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Jos\u00e9 \u00c1ngel Acosta",
    "id": "2340684470",
    "h_index": 1,
    "papers": 3
   }
  ],
  "comment": "Preprint submitted to journal (under review). 22 pages, 8 figures, 3 tables",
  "topics": [
   "egocentric-data",
   "sim2real",
   "imitation-diffusion"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.23924v1",
  "pdf_url": "https://arxiv.org/pdf/2608.23924v1",
  "html_url": "https://arxiv.org/html/2608.23924v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.23887",
  "slug": "interpreting-control-latents-for-system-identification-via-conditional",
  "title": "Interpreting Control Latents for System Identification via Conditional Flow Matching",
  "abstract": "Latent-conditioned adaptive policies can control robots across changing dynamics, but their learned latents remain internal representations of the policy rather than physical models that can be inspected, rolled out, or used by other control modules. This limits closed-loop analysis, diagnosis, and further improvement of a fixed policy. A direct mapping from latent to physical parameters is also under-specified, because multiple systems can induce similar closed-loop behavior. We therefore decode each operational latent into a distribution of quadrotor models using conditional flow matching. The decoded distribution enables two downstream uses without modifying the policy: online predictive tuning of a high-level controller around the fixed low-level policy, and robustness analysis under specified disturbances. Under perturbed actuator dynamics, decoded-model predictive tuning reduces position tracking RMSE by $23\\%$ and heading RMSE by $45\\%$ relative to fixed gains. Under Gaussian force disturbances, decoded-model ensembles closely predict the lateral tracking-error evolution. Together, these results show that control latents can be converted into physical model ensembles for tuning, robustness analysis, and diagnosis of frozen adaptive policies.",
  "published": "2026-08-24",
  "updated": "2026-08-24",
  "year": "2026",
  "authors": [
   "Dingqi Zhang",
   "Ruiqi Zhang",
   "Mark W. Mueller"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Dingqi Zhang",
    "id": "2322093521",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Ruiqi Zhang",
    "id": "2151890280",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Mark W. Mueller",
    "id": "2257383835",
    "h_index": 7,
    "papers": 14
   }
  ],
  "comment": "",
  "topics": [
   "hardware-codesign",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.23887v1",
  "pdf_url": "https://arxiv.org/pdf/2608.23887v1",
  "html_url": "https://arxiv.org/html/2608.23887v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.23863",
  "slug": "dreamledger-where-to-refuse-world-model-imagination-using-execution-se",
  "title": "DreamLedger: Where to Refuse World-Model Imagination Using Execution-Settled Credit",
  "abstract": "Robots are beginning to act on world-model predictions, yet reliability is still expressed through instantaneous, model-internal signals that say whether a prediction looks trustworthy now, not where comparable imagination has already failed. DreamLedger instead treats reliability as a persistent deployment object: an execution-settled credit file recording how often consumed predictions are borne out, indexed by operating condition, region, and prediction horizon, and consulted before each use. Each consumed prediction is registered as a claim and settled against arriving reality without manual labels; the resulting credit gates consumption (low credit shortens reliance or triggers observation), and every reliance event remains auditable via dependency tickets and replayable logs. Persistent credit changes where the gate refuses rather than what the model gets wrong: 69% of denials land on cells that have already failed, episode-local reset triples off-target denials in healthy conditions, and under a localized recurrent degradation persistent credit halves burned imagination, at a cost in task completion. Across three simulated domains, unmodified DreamerV3, TD-MPC2, and V-JEPA 2-AC mounts, and a real Franka, paired quadrotor evaluation shows credit-gated planning reduces burned imagination by 62% (95% CI 43-81%) versus blind consumption; settlement-grounded calibration yields moderate, seed-consistent operating points where raw instantaneous gates collapse to extremes, while persistent books trade verification for reliance (manipulation probes 1.00 to 0.36/episode at success 0.98 vs. 0.94). The trust layer spans decoder-, latent-, and token-space interfaces. On hardware, settlement runs under real sensing and contact noise, both models are priced creditworthy at the frozen 9-cm tolerance, a failure loop is re-priced online, and all 1,062 registered spends replay from the audit logs.",
  "published": "2026-08-24",
  "updated": "2026-08-26",
  "year": "2026",
  "authors": [
   "Xianyao Li",
   "Ruitong Tian",
   "Rui Min",
   "Fang Xu",
   "Jing Du"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work evaluates DreamLedger in three simulated domains (indoor flight, tabletop manipulation, 2D navigation), via mounts on unmodified DreamerV3, TD-MPC2, and V-JEPA 2-AC, and on a real Franka manipulator.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xianyao Li",
    "id": "2459445108",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Ruitong Tian",
    "id": "2296072834",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Rui Min",
    "id": "2459245268",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Fang Xu",
    "id": "2156159672",
    "h_index": 9,
    "papers": 26
   },
   {
    "name": "Jing Du",
    "id": "2116518997",
    "h_index": 15,
    "papers": 35
   }
  ],
  "comment": "12 pages, 6 figures, 10 tables",
  "topics": [
   "world-models",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.23863v2",
  "pdf_url": "https://arxiv.org/pdf/2608.23863v2",
  "html_url": "https://arxiv.org/html/2608.23863v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.23839",
  "slug": "resilience-matters-for-embodied-agents-system-new-metrics-systematic-e",
  "title": "Resilience Matters for Embodied Agents System: New Metrics, Systematic Evaluation, and Optimization",
  "abstract": "Embodied Agents System (EAS) are increasingly deployed in open-world physical domains, where reliability directly dictates deployment quality and human-agent trust. However, existing evaluations rely on outcome-centric metrics as success rate or safety scores that collapse diverse execution trajectories into coarse scores, obscuring the dynamic processes underlying agent behavior. Therefore, they ignore a critical property of EAS -- which we define as the Resilience -- that reflects how EASs recover, stabilize, and extend under perturbations and across iterative updates. The lack of resilience is particularly critical in open-world environments due to continuous unexpected disruptions, thus directly affecting the quality of EAS deployment. To address this problem, we gain insight from the resilience-engineering concepts to EAS groundings and propose a novel resilience evaluation framework that can be flexibly applied to any EAS. Specifically, we define the first comprehensive resilience metrics suite for EASs system that exposes Rebound, Stability, and Graceful Extensibility across embodied tasks execution, providing a practical grounding for EAS resilience analysis. We further implement the resilience evaluation layer that transforms execution process into assessments for diagnosis and optimization. Across 400 household tasks with 10 EAS, we reveal the process-level distinction hidden by outcome metrics, including recovery cost differences among successful episodes ($\u0394C_{rec}=25.2$), increased instability and task-family degradation. Metrics-guided optimizations reduce recovery cost and increase stability, graceful extensibility completion, showing the diagnostic effect of resilience evaluation. Our results reveal a trade-off among resilience characteristics, suggesting that a resilient EAS construction should be configured according to deployment-specific requirements.",
  "published": "2026-08-24",
  "updated": "2026-08-24",
  "year": "2026",
  "authors": [
   "Yapeng Liu",
   "Yuanzhao Zhai",
   "Xudong Gong",
   "Dawei Feng",
   "Bo Ding",
   "Lin Wang",
   "Huaimin Wang"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The first comprehensive resilience metrics suite for EASs system that exposes Rebound, Stability, and Graceful Extensibility across embodied tasks execution is defined, providing a practical grounding for EAS resilience analysis and revealing a trade-off among resilience characteristics.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yapeng Liu",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Yuanzhao Zhai",
    "id": "1931511592",
    "h_index": 7,
    "papers": 39
   },
   {
    "name": "Xudong Gong",
    "id": "1630380410",
    "h_index": 5,
    "papers": 22
   },
   {
    "name": "Dawei Feng",
    "id": "2264007696",
    "h_index": 8,
    "papers": 55
   },
   {
    "name": "Bo Ding",
    "id": "2265321448",
    "h_index": 8,
    "papers": 43
   },
   {
    "name": "Lin Wang",
    "id": "2375286831",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Huaimin Wang",
    "id": "2113255325",
    "h_index": 11,
    "papers": 63
   }
  ],
  "comment": "12 pages, 5 figures",
  "topics": [
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.23839v1",
  "pdf_url": "https://arxiv.org/pdf/2608.23839v1",
  "html_url": "https://arxiv.org/html/2608.23839v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.23831",
  "slug": "learning-to-act-while-waiting-rl-finetuning-of-generalist-robot-polici",
  "title": "Learning to Act While Waiting: RL Finetuning of Generalist Robot Policies Under Inference Latency",
  "abstract": "While reinforcement learning (RL) allows generalist robot policies to continually improve during deployment, the large model size of modern generalist policies, such as VLAs, poses a fundamental obstacle to effective RL improvement. In particular, their severe inference latency---which can lead to pauses or jerky movements---can alter the effective environment dynamics and, if not correctly accounted for, break the Markov assumption that RL relies on, causing standard RL algorithms to fail completely. In this work, we introduce a latency-aware framework, Asynchronous RL with Intermediate Information (ARLI), that enables RL-based improvement of generalist policies under inference delays. Our framework builds on asynchronous inference approaches, which interleave action generation with execution to hide latency, and addresses its incompatibility with RL by providing a low-latency RL policy design that maximizes reactivity within the inference window through two contributions: state augmentations that restore near-Markovian structure by incorporating committed actions and a mid-inference observation. We evaluate our approach across simulated and real-world manipulation tasks, and find that it enables effective finetuning under inference delays where standard RL fails entirely, even matching or exceeding the performance of standard RL in idealized no-latency settings.",
  "published": "2026-08-24",
  "updated": "2026-08-26",
  "year": "2026",
  "authors": [
   "Brian Zhu",
   "Momen Khalil",
   "E Harrison",
   "Emanuele Poggi",
   "Philipp Schmitt",
   "Bernd Kast",
   "Philine Meister",
   "Pranav Atreya",
   "Qiyang Li",
   "Finn Ferchau",
   "Cesar Colmenero",
   "Yash Shahapurkar",
   "Gokul Narayanan",
   "Melih Erdogan",
   "Kai Wurm",
   "Georg von Wichert",
   "Oier Mees",
   "Eugen Solowjow",
   "Andrew Wagenmaker",
   "Sergey Levine"
  ],
  "author_count": 20,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work introduces a latency-aware framework, Asynchronous RL with Intermediate Information (ARLI), that enables RL-based improvement of generalist policies under inference delays where standard RL fails entirely, even matching or exceeding the performance of standard RL in idealized no-latency settings.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Brian Zhu",
    "id": "2161311133",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Momen Khalil",
    "id": "2378867439",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "E. Harrison",
    "id": "2244904635",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Emanuele Poggi",
    "id": "2459246006",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Philipp S. Schmitt",
    "id": "34411952",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Bernd Kast",
    "id": "46948785",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Philine Meister",
    "id": "74474885",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "P. Atreya",
    "id": "1643955309",
    "h_index": 9,
    "papers": 13
   },
   {
    "name": "Qiyang Li",
    "id": "8194287",
    "h_index": 15,
    "papers": 34
   },
   {
    "name": "Finn Ferchau",
    "id": "2449641934",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Cesar Colmenero",
    "id": "2459246276",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Yash Shahapurkar",
    "id": "74153478",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Gokul Narayanan",
    "id": "1685448096",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Melih Erdogan",
    "id": "2133936265",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Kai M. Wurm",
    "id": "32322345",
    "h_index": 17,
    "papers": 26
   },
   {
    "name": "G. V. Wichert",
    "id": "2359250437",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Oier Mees",
    "id": "7264115",
    "h_index": 28,
    "papers": 45
   },
   {
    "name": "Eugen Solowjow",
    "id": "2419277",
    "h_index": 16,
    "papers": 48
   },
   {
    "name": "Andrew Wagenmaker",
    "id": "9041933",
    "h_index": 16,
    "papers": 33
   },
   {
    "name": "Sergey Levine",
    "id": "2249615151",
    "h_index": 32,
    "papers": 53
   }
  ],
  "comment": "25 pages, 12 figures, project website: https://async-rl-intermediate-information.github.io/",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.23831v2",
  "pdf_url": "https://arxiv.org/pdf/2608.23831v2",
  "html_url": "https://arxiv.org/html/2608.23831v2",
  "code_url": "https://async-rl-intermediate-information.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2608.23650",
  "slug": "concept-guided-exploration-building-persistent-actionable-scene-graphs",
  "title": "Concept-Guided Exploration: Building Persistent, Actionable Scene Graphs",
  "abstract": "The perception of 3D space by mobile robots is rapidly moving from flat metric grid representations to hybrid metric-semantic graphs built from human-interpretable concepts. While most approaches first build metric maps and then add semantic layers, we explore an alternative, concept-first architecture in which spatial understanding emerges from asynchronous concept agents that directly instantiate and manage semantic entities. Our robot employs two spatial concepts (room and door), implemented as autonomous processes within a cognitive distributed architecture. These concept agents cooperatively build a shared scene graph representation of indoor layouts through active exploration and incremental validation. The key architectural principle is hierarchical constraint propagation: Room instantiation provides geometric and semantic priors to guide and support door detection within wall boundaries. The resulting structure is maintained by a complementary functional principle based on prediction-matching loops. This approach is designed to yield an actionable, human-interpretable spatial representation without relying on any pre-existing global metric map, supporting scalable operation and persistent, task-relevant understanding in structured indoor environments.",
  "published": "2026-08-24",
  "updated": "2026-08-24",
  "year": "2026",
  "authors": [
   "No\u00e9 Zapata",
   "Gerardo P\u00e9rez",
   "Alejandro Torrej\u00f3n",
   "Pedro N\u00fa\u00f1ez",
   "Pablo Bustos"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "Applied Sciences",
  "venue_source": "semantic-scholar",
  "citations": 6,
  "influential_citations": 1,
  "tldr": "A novel algorithm that detects potentially hazardous situations for humans and selects appropriate robotic actions to eliminate these dangers in real time is proposed, offering a promising approach to enhancing human\u2013robot interaction in potentially hazardous environments.",
  "doi": "10.3390/app152011084",
  "oa_pdf": "https://www.mdpi.com/2076-3417/15/20/11084/pdf?version=1760609229",
  "s2_authors": [
   {
    "name": "N. Zapata",
    "id": "2218282479",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Gerardo P\u00e9rez",
    "id": "2204425713",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Lucas Bonilla",
    "id": "2293312700",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Pedro N\u00fa\u00f1ez Trujillo",
    "id": "1787200",
    "h_index": 21,
    "papers": 62
   },
   {
    "name": "P. Bachiller",
    "id": "46426211",
    "h_index": 9,
    "papers": 35
   },
   {
    "name": "Pablo Bustos",
    "id": "2293312234",
    "h_index": 2,
    "papers": 4
   }
  ],
  "comment": "",
  "topics": [
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.23650v1",
  "pdf_url": "https://arxiv.org/pdf/2608.23650v1",
  "html_url": "https://arxiv.org/html/2608.23650v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.35
 },
 {
  "id": "2608.23486",
  "slug": "geowam-visual-geometry-world-action-models-for-autonomous-driving",
  "title": "GeoWAM: Visual Geometry World Action Models for Autonomous Driving",
  "abstract": "World action models (WAMs) have recently gained increasing attention as a framework for jointly modeling scene evolution and ego actions in autonomous driving. Most existing WAMs learn scene dynamics in pixel space by combining a video-generation backbone for future-observation prediction with an action head for ego-trajectory prediction. Pixels, however, provide only an indirect representation of these dynamics: they entangle geometry and motion with appearance, texture, and illumination, forcing the model to infer three-dimensional transformations from two-dimensional observations. We argue that geometry, represented by point clouds, offers a more natural state space for driving because it explicitly captures spatial structure and the rigid and non-rigid transformations that govern scene evolution while directly aligning with the space in which driving actions are executed. Building on this insight, we introduce \\textbf{GeoWAM}, a visual geometry world action model for autonomous driving. Rather than predicting future images, GeoWAM is pretrained to forecast future scene geometry, yielding representations that jointly encode spatial structure and temporal evolution. A geometry-conditioned action head then leverages these learned geometric dynamics to predict future ego trajectories. Extensive open-loop and closed-loop evaluations show that visual geometry world modeling yields substantially stronger driving policies than image-based alternatives, establishing future-geometry prediction as an effective pretraining objective for autonomous driving.",
  "published": "2026-08-24",
  "updated": "2026-08-25",
  "year": "2026",
  "authors": [
   "Yiren Lu",
   "Xin Ye",
   "Jiaming Liu",
   "Philip Jacobson",
   "Jin Yao",
   "Yi-chung Chen",
   "Liam Merino",
   "Dhruva Dixith Kurra",
   "Min Cai",
   "Tom Lampo",
   "Yu Yin",
   "Danhua Guo",
   "Burhan Yaman"
  ],
  "author_count": 13,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "GeoWAM, a visual geometry world action model for autonomous driving, is introduced, which is pretrained to forecast future scene geometry, yielding representations that jointly encode spatial structure and temporal evolution.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yiren Lu",
    "id": "2141540331",
    "h_index": 7,
    "papers": 23
   },
   {
    "name": "Xin Ye",
    "id": "2293787205",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Jiaming Liu",
    "id": "2267506419",
    "h_index": 13,
    "papers": 38
   },
   {
    "name": "Jin Yao",
    "id": "2287524199",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Yi-Chung Chen",
    "id": "2117968025",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "L.C. Merino",
    "id": "2330635738",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Dhruva Dixith Kurra",
    "id": "2279830838",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Min Cai",
    "id": "2459198497",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Tom Lampo",
    "id": "2441894711",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Yu Yin",
    "id": "2315441075",
    "h_index": 5,
    "papers": 19
   },
   {
    "name": "Danhua Guo",
    "id": "2068110259",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Burhaneddin Yaman",
    "id": "23166935",
    "h_index": 20,
    "papers": 67
   }
  ],
  "comment": "Project page: https://yiren-lu.com/project_pages/geowam/",
  "topics": [
   "world-models",
   "spatial-3d",
   "navigation",
   "foundation-pretraining",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.23486v2",
  "pdf_url": "https://arxiv.org/pdf/2608.23486v2",
  "html_url": "https://arxiv.org/html/2608.23486v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.23478",
  "slug": "act-with-intent-distilling-behavior-intent-for-vision-language-action",
  "title": "Act with Intent: Distilling Behavior Intent for Vision-Language-Action Models",
  "abstract": "Vision-Language-Action (VLA) models can turn multimodal context into robot actions, but their action decoders are still trained largely by behavior cloning. This supervises which motor command was demonstrated while leaving implicit the local objective served by the behavior under the instruction. Future-based supervision enriches action learning with frames, latent observations, trajectories, or motion representations, but these signals capture particular realizations of what may happen rather than the shared semantic objective of the forthcoming behavior. We propose Intention Distillation (INDI), which distills behavior-level intent into the action decoder. During training, a frozen teacher VLM interprets a demonstrated segment from the current observation, instruction, coarse action summary, and corresponding execution video. From its standard inputs, the deployed VLA recovers the resulting multimodal intent representation at an intermediate decoder layer and uses it to organize action prediction together with representations of how the behavior unfolds and what it achieves. On SimplerEnv-Bridge, INDI improves GR00T-N1.7 from 64.3% to 84.7%, and on RoboCasa Kitchen it improves the controlled GR00T-N1.7 baseline from 64.1% to 70.3%, with consistent gains on $\u03c0_{0.5}$ across both benchmarks. In real-world tasks, INDI improves average success from 62.0% to 68.7%, with gains of up to 12.0 pp on longer-horizon tasks. Further analyses show that the recovered latent is used by the decoder, captures behavior objective and execution progress, and organizes downstream predictions in an objective-dependent manner. These results show that action decoders benefit from explicitly modeling the semantic objective of the behavior they generate.",
  "published": "2026-08-24",
  "updated": "2026-08-24",
  "year": "2026",
  "authors": [
   "Sangoh Lee",
   "Sangwoo Mo",
   "Wook-Shin Han"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Intention Distillation (INDI) is proposed, which distills behavior-level intent into the action decoder and organizes downstream predictions in an objective-dependent manner, and shows that action decoders benefit from explicitly modeling the semantic objective of the behavior they generate.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Sangoh Lee",
    "id": "2294147509",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Sangwoo Mo",
    "id": "2299940987",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Wook-Shin Han",
    "id": "2294367075",
    "h_index": 3,
    "papers": 14
   }
  ],
  "comment": "Project page: https://leesangoh.github.io/indi-project-page/",
  "topics": [
   "vla",
   "imitation-diffusion"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.23478v1",
  "pdf_url": "https://arxiv.org/pdf/2608.23478v1",
  "html_url": "https://arxiv.org/html/2608.23478v1",
  "code_url": "https://leesangoh.github.io/indi-project-page/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2608.23452",
  "slug": "reward-free-continual-adaptation-for-resilient-space-robots",
  "title": "Reward-Free Continual Adaptation for Resilient Space Robots",
  "abstract": "Space robots operate in extreme environments where hardware degradation can critically compromise traditional control strategies. While continual reinforcement learning offers a promising mechanism for online adaptation, it inherently requires access to a reward signal during deployment. However, precise reward computation in space is often infeasible due to the lack of external tracking systems and the overall complexity of the environment. To address the challenge of unobservable rewards, we introduce a reward-free continual learning framework that leverages latent-state world models. By pre-training a model-based agent across diverse simulations, the world model learns a robust predictor of the reward structure within its latent space. Upon deployment to an environment with severe hardware degradation, we freeze the observation encoder and reward predictor to update only the transition dynamics of the world model through unsupervised rollouts. By training the policy entirely on imagined trajectories generated by this updated world model, the agent adapts to altered dynamics without receiving new rewards. We demonstrate our approach across simulated planetary traversal, orbital navigation, and precision assembly tasks subjected to severe morphological failures.",
  "published": "2026-08-24",
  "updated": "2026-08-24",
  "year": "2026",
  "authors": [
   "Andrej Orsula",
   "Miguel Olivares-Mendez",
   "Carol Martinez"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work introduces a reward-free continual learning framework that leverages latent-state world models that adapts to altered dynamics without receiving new rewards across simulated planetary traversal, orbital navigation, and precision assembly tasks subjected to severe morphological failures.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Andrej Orsula",
    "id": "2179879892",
    "h_index": 3,
    "papers": 15
   },
   {
    "name": "M. Olivares-Mendez",
    "id": "2241386003",
    "h_index": 18,
    "papers": 64
   },
   {
    "name": "Carol Mart\u00ednez",
    "id": "2275137699",
    "h_index": 4,
    "papers": 16
   }
  ],
  "comment": "Accepted for publication at the Third Conference on AI in and for Space (SPAICE 2026) | The source code is available at https://github.com/AndrejOrsula/space_robotics_bench",
  "topics": [
   "world-models",
   "rl-control",
   "navigation",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.23452v1",
  "pdf_url": "https://arxiv.org/pdf/2608.23452v1",
  "html_url": "https://arxiv.org/html/2608.23452v1",
  "code_url": "https://github.com/AndrejOrsula/space_robotics_bench",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2608.23405",
  "slug": "momadv2-reliable-temporal-memory-for-end-to-end-autonomous-driving",
  "title": "MomADv2: Reliable Temporal Memory for End-to-End Autonomous Driving",
  "abstract": "Long-horizon planning is critical for safe autonomous driving in complex scenarios. Existing methods improve planning continuity with temporal memory, but such memory may become invalid and mislead decisions when the driving command changes. Thus, selectively leveraging useful history while suppressing command-inconsistent memory remains a key challenge. To address this issue, we propose MomADv2, a reliable state-space memory framework for long-horizon end-to-end autonomous driving. At its core, MomADv2 introduces a Selective State-Space Planning Memory Query Module, which filters historical planning queries based on temporal continuity and command consistency, selects planning modes relevant to the current command, and models the evolution of planning intentions through a selective state-space mechanism. To further alleviate local trajectory deviations and error accumulation in long-horizon planning, we design a Flow-Matching Trajectory Residual Refiner. It learns a continuous residual correction field from the refined planning output to the expert trajectory, enabling fine-grained trajectory refinement while preserving the stability of anchor-based planning. Extensive experiments on closed-loop NAVSIM and Bench2Drive, as well as open-loop nuScenes, demonstrate that MomADv2 improves long-horizon planning consistency and reduces the average collision rate by 15.6% over MomAD under 6-second planning.",
  "published": "2026-08-24",
  "updated": "2026-08-24",
  "year": "2026",
  "authors": [
   "Ziying Song",
   "Shengkai Zhang",
   "Lin Liu",
   "Peiliang Wu",
   "Lei Yang",
   "Dongyang Xu",
   "Bin Sun",
   "Li Wang",
   "Shaoqing Xu",
   "Caiyan Jia",
   "Yadan Luo"
  ],
  "author_count": 11,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "MomADv2, a reliable state-space memory framework for long-horizon end-to-end autonomous driving, introduces a Selective State-Space Planning Memory Query Module, which filters historical planning queries based on temporal continuity and command consistency, and models the evolution of planning intentions through a selective state-space mechanism.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ziying Song",
    "id": "2367119325",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Shengkai Zhang",
    "id": "2315625136",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Lin Liu",
    "id": "2232574139",
    "h_index": 11,
    "papers": 24
   },
   {
    "name": "Peiliang Wu",
    "id": "2391119240",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Lei Yang",
    "id": "2257380929",
    "h_index": 11,
    "papers": 17
   },
   {
    "name": "Dongyang Xu",
    "id": "2292214165",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Bin Sun",
    "id": "2284302666",
    "h_index": 4,
    "papers": 18
   },
   {
    "name": "Li Wang",
    "id": "2278451900",
    "h_index": 6,
    "papers": 16
   },
   {
    "name": "Shaoqing Xu",
    "id": "2278397378",
    "h_index": 10,
    "papers": 24
   },
   {
    "name": "Caiyan Jia",
    "id": "2257423721",
    "h_index": 10,
    "papers": 21
   },
   {
    "name": "Yadan Luo",
    "id": "2279402741",
    "h_index": 4,
    "papers": 11
   }
  ],
  "comment": "16 pages, 6 figures",
  "topics": [
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.23405v1",
  "pdf_url": "https://arxiv.org/pdf/2608.23405v1",
  "html_url": "https://arxiv.org/html/2608.23405v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.23354",
  "slug": "optisight-bridging-semantic-reasoning-and-geometric-control-for-embodi",
  "title": "OptiSight: Bridging Semantic Reasoning and Geometric Control for Embodied Navigation",
  "abstract": "Autonomous indoor navigation requires both semantic understanding and precise geometric control. We propose OptiSight, a hybrid framework that combines Vision-Language Model reasoning with deterministic visual servoing through a finite-state Chain-of-Thought architecture. Grounded-SAM localizes open-vocabulary targets, while camera projection geometry converts visual observations into navigation commands without requiring dense mapping. The VLM is queried only at key decision points, reducing computational overhead while geometric control handles continuous navigation. Experiments in AI Habitat demonstrate reliable zero-shot navigation across diverse indoor scenarios, including obstacle avoidance and semantic ambiguity, while operating within an 8~GB VRAM budget. The source code is available at https://github.com/avanalperen/OptiSight-Python-Multimodal-CoT-for-Visual-Reasoning.",
  "published": "2026-08-24",
  "updated": "2026-08-24",
  "year": "2026",
  "authors": [
   "Alperen Avan",
   "Jordi Sanchez-Riera"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "OptiSight is proposed, a hybrid framework that combines Vision-Language Model reasoning with deterministic visual servoing through a finite-state Chain-of-Thought architecture that demonstrates reliable zero-shot navigation across diverse indoor scenarios, including obstacle avoidance and semantic ambiguity, while operating within an 8~GB VRAM budget.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Alperen Avan",
    "id": "2459189996",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Jordi Sanchez-Riera",
    "id": "1401515040",
    "h_index": 10,
    "papers": 29
   }
  ],
  "comment": "",
  "topics": [
   "navigation",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.23354v1",
  "pdf_url": "https://arxiv.org/pdf/2608.23354v1",
  "html_url": "https://arxiv.org/html/2608.23354v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.23320",
  "slug": "ros2smolvla-enabling-small-vision-language-action-models-for-integrati",
  "title": "ROS2SmolVLA: Enabling Small Vision-Language-Action Models for Integration into Industrial-Grade Lightweight Robots",
  "abstract": "Industrial demand changes the paradigms of production. Due to smaller batch sizes and more variations in products, companies face a growing challenge to adopt more adaptive production systems. In particular, robot-based automation is usually static and fails to respond to constantly changing processes. Vision-Language-Action (VLA) Models are a promising opportunity to mitigate this challenge by generating robot actions based on the observed system state. However, current research either focuses on large models that cannot be computed on premise, creating compliance and security challenges, or use lab-grade robot hardware that obscures exploitation in real industrial settings. In this work, we adapt Hugging Face's SmolVLA for Universal Robots lightweight robots. Further, we release the open-source repository ROS2SmolVLA that implements an interface for ROS 2 to SmolVLA, and makes it applicable for industrial-grade hardware. By this, we allow a lenient adoption into lab and industrial environments. We validate the functionality of SmolVLA for a Universal Robots UR10e using a pick-and-place task and give implementation guidelines. Our findings support that SmolVLA is a well-suited option for small-sized tasks that need to be computed on premise.",
  "published": "2026-08-24",
  "updated": "2026-08-24",
  "year": "2026",
  "authors": [
   "Nils Mandischer",
   "Noah B\u00f6ckmann",
   "Ludwig Holl",
   "Lars Mikelsons"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work adapts Hugging Face's SmolVLA for Universal Robots lightweight robots, and releases the open-source repository ROS2SmolVLA that implements an interface for ROS 2 to SmolVLA, and makes it applicable for industrial-grade hardware.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Nils Mandischer",
    "id": "2315811839",
    "h_index": 2,
    "papers": 11
   },
   {
    "name": "Noah B\u00f6ckmann",
    "id": "2459199155",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Ludwig Holl",
    "id": "2459192363",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "L. Mikelsons",
    "id": "2580478",
    "h_index": 12,
    "papers": 83
   }
  ],
  "comment": "Accepted at 8th International Conference on Industry of the Future and Smart Manufacturing, Padua & Venice, 2026",
  "topics": [
   "vla"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.23320v1",
  "pdf_url": "https://arxiv.org/pdf/2608.23320v1",
  "html_url": "https://arxiv.org/html/2608.23320v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.23304",
  "slug": "design-of-a-biomimetic-joint-covering-skin-with-tissue-like-structure",
  "title": "Design of a Biomimetic Joint-Covering Skin with Tissue-Like Structure to Enhance Proprioception in a Musculoskeletal Humanoid",
  "abstract": "Proprioception in musculoskeletal humanoids is typically estimated primarily from muscle sensing, while the role of cutaneous deformation around joints remains insufficiently explored. In biological systems, mechanoreceptors distributed within soft tissue complement muscle feedback and support reliable joint state estimation. This study presents the design of a biomimetic joint-covering skin with a tissue-like layered structure that integrates pressure- and stretch-sensitive elements within the joint-covering tissue. The proposed skin is implemented on the musculoskeletal humanoid Musashi-W, and its independent proprioceptive capability as well as its integration with muscle sensing are evaluated. Experimental results show that the proposed skin alone achieves joint angle estimation with an average error of approximately 3 degrees. Furthermore, integration with muscle sensing improves estimation accuracy. Owing to its joint-covering structure, the skin may mechanically mitigate the influence of external disturbances on the muscles, and the integration of multiple modalities suggests the possibility of contributing to the identification of external stimuli that are difficult to interpret using muscle sensing alone. This work presents a design methodology for biomimetic joint-covering skin and demonstrates that such tissue-structured skin can serve as an effective approach for extending proprioceptive systems in musculoskeletal humanoids.",
  "published": "2026-08-24",
  "updated": "2026-08-26",
  "year": "2026",
  "authors": [
   "Akihiro Miki",
   "Shun Hasegawa",
   "Yoshimoto Ribayashi",
   "Kento Kawaharazuka",
   "Kei Okada"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A design methodology for biomimetic joint-covering skin with a tissue-like layered structure that integrates pressure- and stretch-sensitive elements within the joint-covering tissue is presented and it is demonstrated that such tissue-structured skin can serve as an effective approach for extending proprioceptive systems in musculoskeletal humanoids.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Akihiro Miki",
    "id": "2152445434",
    "h_index": 3,
    "papers": 34
   },
   {
    "name": "Shun Hasegawa",
    "id": "152589754",
    "h_index": 7,
    "papers": 37
   },
   {
    "name": "Yoshimoto Ribayashi",
    "id": "2198247203",
    "h_index": 2,
    "papers": 21
   },
   {
    "name": "Kento Kawaharazuka",
    "id": "8308607",
    "h_index": 17,
    "papers": 222
   },
   {
    "name": "Kei Okada",
    "id": "2248244895",
    "h_index": 5,
    "papers": 70
   }
  ],
  "comment": "Accepted at IROS 2026, website - https://poyotamu000.github.io/musashiw-joint-covering-skin/ YouTube - https://www.youtube.com/watch?v=L9xU2wMkRRg",
  "topics": [
   "humanoids"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.23304v2",
  "pdf_url": "https://arxiv.org/pdf/2608.23304v2",
  "html_url": "https://arxiv.org/html/2608.23304v2",
  "code_url": "https://poyotamu000.github.io/musashiw-joint-covering-skin/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.5
 },
 {
  "id": "2608.23258",
  "slug": "progressively-learning-heterogeneous-skills-in-a-unified-latent-space",
  "title": "Progressively Learning Heterogeneous Skills in a Unified Latent Space",
  "abstract": "We propose HetSkills, a novel framework designed to progressively learn heterogeneous skills within a unified latent space for physics-based character control. The core idea is to treat this latent space as a shared executable interface, enabling seamless integration of skills learned from diverse data sources, supervision forms, and tasks. HetSkills begins by learning a tracking skill that establishes a strong foundation in motion control and creates a shared motion decoder, which can be reused across tasks without the need for retraining or separate controllers. To prevent the text-to-motion skill from exploiting shortcut pathways instead of learning language semantics, we introduce motion intuition distillation to ground text-to-motion generation in language semantics and a task-guidance module that dynamically adjusts actions based on high-level language instructions. This enables HetSkills to preserve natural motion while continuously expanding its skill repertoire, making it highly adaptable for long-horizon tasks. Experimental results demonstrate the effectiveness in motion tracking, text-to-motion generation, motion completion, and downstream task adaptation, achieving impressive success rates even under challenging conditions.",
  "published": "2026-08-24",
  "updated": "2026-08-24",
  "year": "2026",
  "authors": [
   "Yue-Yi Zhang",
   "Ming Gong",
   "Linpu He",
   "Wei-Shi Zheng",
   "Zhilin Zhao"
  ],
  "author_count": 5,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "To prevent the text-to-motion skill from exploiting shortcut pathways instead of learning language semantics, motion intuition distillation is introduced to ground text-to-motion generation in language semantics and a task-guidance module that dynamically adjusts actions based on high-level language instructions is introduced.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yuehan Zhang",
    "id": "2455665845",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Ming Gong",
    "id": "2459191552",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Linpu He",
    "id": "2330593433",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Wei-Shi Zheng",
    "id": "2347118490",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Zhilin Zhao",
    "id": "2404622405",
    "h_index": 2,
    "papers": 6
   }
  ],
  "comment": "23 pages, 18 figures",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.23258v1",
  "pdf_url": "https://arxiv.org/pdf/2608.23258v1",
  "html_url": "https://arxiv.org/html/2608.23258v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.23224",
  "slug": "think-only-when-needed-prompt-authority-control-for-selective-slow-pat",
  "title": "Think Only When Needed: Prompt-Authority Control for Selective Slow-Path Intervention in Vision-Language-Action Manipulation",
  "abstract": "Retrieval can efficiently and effectively augment a frozen vision--language--action (VLA) policy without retraining, yet retrieved text becomes a control intervention once it enters the executed prompt. In a matched audit, raw appended text reduces mean success from 92.47\\% to 3.00\\%, while meaningful and length-matched meaningless appends both fail on all 500 states. This result identifies \\emph{prompt-form collapse}: changing the instruction form, rather than adding useful semantics, can dominate execution. We introduce TOWN-VLA (Think Only When Needed), a prompt-authority interface that separates candidate generation from permission to alter the policy input. A fixed compatibility rule authorizes a canonical compact instruction; otherwise, the interface restores the original Base prompt exactly. Across 900 audited routes, every route follows this contract: 525 routes recover Base with matching hashes, and all 375 authorized prompts preserve the task signature. On a matched $4\\times7$ LIBERO-Plus evaluation with 10{,}030 episodes per method, success rises from 69.5\\% to 73.1\\% ($+362$ episodes; 95\\% CI 1.89--5.45 points), improving on six perturbation axes and all four suites. On a physical PiPER arm with a frozen \\pizerofive{} checkpoint, success rises from 52.7\\% to 78.7\\% over 150 trials per method ($p=3.16\\times10^{-6}$). Prompt authority is enforceable for a frozen controller; oracle-free admission calibration is the next deployment target.",
  "published": "2026-08-24",
  "updated": "2026-08-24",
  "year": "2026",
  "authors": [
   "Zhiruo Zhou",
   "Zelin Li",
   "Xiwen Chen",
   "Jiazhuo Li",
   "Chenwei Wang",
   "Huiming Chen",
   "Xiaojun Zhu"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "TOWN-VLA (Think Only When Needed), a prompt-authority interface that separates candidate generation from permission to alter the policy input, is introduced, a prompt-authority interface that separates candidate generation from permission to alter the policy input.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhiruo Zhou",
    "id": "96381098",
    "h_index": 8,
    "papers": 15
   },
   {
    "name": "Zelin Li",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Xiwen Chen",
    "id": "2260602938",
    "h_index": 9,
    "papers": 44
   },
   {
    "name": "Jiazhuo Li",
    "id": "2292294220",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Chenwei Wang",
    "id": "2459228962",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Huiming Chen",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Xiaojun Zhu",
    "id": "2342023587",
    "h_index": 4,
    "papers": 14
   }
  ],
  "comment": "",
  "topics": [
   "vla"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.23224v1",
  "pdf_url": "https://arxiv.org/pdf/2608.23224v1",
  "html_url": "https://arxiv.org/html/2608.23224v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.23204",
  "slug": "guided-riemannian-optimization-guro-bridging-model-predictive-control",
  "title": "Guided Riemannian Optimization (GuRO): Bridging Model Predictive Control and Decision Transformers",
  "abstract": "Decision-making in high-dimensional, nonlinear systems remains a central challenge in robotics. While model-based methods like Model Predictive Control (MPC) offer sample efficiency and interpretability, their performance degrades when the dynamics model is inaccurate or long-horizon predictions are required. Conversely, model-free reinforcement learning (RL) learns policies directly from interaction but suffers from high sample complexity and unstable optimization. Recent advances in sequence modeling have inspired transformer-based decision-making frameworks that can unify MPC and RL, but their training typically faces significant optimization challenges due to highly non-convex loss landscapes. In this work, we propose a novel framework that integrates MPC with RL in a sequence decision-making framework and leverages a curvature-aware optimization to efficiently tackle non-convex loss landscapes. MPC provides predictions of locally optimal trajectories that guide the decision transformer, removing the need for extensive offline pretraining. To address the slow and unstable convergence of traditional optimizers, we train the policy in a Riemannian parameter space using an efficient Riemannian (curvature-aware) method, leading to faster and more robust optimization. We evaluate our framework on high-dimensional quadruped control tasks and demonstrate consistent improvements over strong baselines, including TRPO, SAC, and Online Decision Transformer, achieving higher returns and faster convergence.",
  "published": "2026-08-24",
  "updated": "2026-08-25",
  "year": "2026",
  "authors": [
   "Hossein Abdi",
   "Satya Prakash Dash",
   "Mingfei Sun"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A novel framework is proposed that integrates MPC with RL in a sequence decision-making framework and leverages a curvature-aware optimization to efficiently tackle non-convex loss landscapes and achieves higher returns and faster convergence.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hossein Abdi",
    "id": "2325949698",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Satya Dash",
    "id": "143650055",
    "h_index": 20,
    "papers": 91
   },
   {
    "name": "Mingfei Sun",
    "id": "2290173433",
    "h_index": 2,
    "papers": 15
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "rl-control",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.23204v2",
  "pdf_url": "https://arxiv.org/pdf/2608.23204v2",
  "html_url": "https://arxiv.org/html/2608.23204v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.23163",
  "slug": "spinning-quadrotor-hover-thrust-augmentation-with-passive-lifting-surf",
  "title": "Spinning Quadrotor: Hover Thrust Augmentation with Passive Lifting Surfaces",
  "abstract": "Conventional multirotor aerial vehicles actively suppress yaw rotation during hover, expending power to maintain a fixed heading despite the fact that yaw regulation is not required for force balance or altitude control. This paper challenges that paradigm by proposing a spinning quadrotor architecture that intentionally operates at a sustained yaw rate, converting power traditionally spent on yaw regulation into useful aerodynamic effects. A dynamic model of the spinning quadrotor is developed, analysis for low Re range is conducted to choose an airfoil for lifting surfaces. Preliminary hardware tests show a 22% reduction in thrust required. These findings suggest that intentional yaw rotation, rather than being suppressed, can be exploited as a design mechanism for efficient and robust multirotor flight.",
  "published": "2026-08-24",
  "updated": "2026-08-24",
  "year": "2026",
  "authors": [
   "Aniketh Parkala",
   "Harikumar Kandath"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "10.1109/ICUAS69441.2026.11598692",
  "oa_pdf": "https://doi.org/10.48550/arxiv.2608.23163",
  "s2_authors": [
   {
    "name": "Aniketh Parkala",
    "id": "2450035949",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Harikumar Kandath",
    "id": "40994401",
    "h_index": 8,
    "papers": 67
   }
  ],
  "comment": "8 pages",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.23163v1",
  "pdf_url": "https://arxiv.org/pdf/2608.23163v1",
  "html_url": "https://arxiv.org/html/2608.23163v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.23140",
  "slug": "mivifi-bridging-perspective-and-fisheye-domains-for-training-multi-vie",
  "title": "MIVIFI: Bridging Perspective and Fisheye Domains for Training Multi-View Fisheye Image Generation Models",
  "abstract": "Achieving 360\u00b0 coverage is critical for the visual perception systems of autonomous vehicles. Fisheye cameras offer a cost-effective solution by enabling full surround coverage with as few as two sensors. However, existing multi-view fisheye datasets are limited, and synthesizing rare corner cases typically requires computationally expensive 3D simulations, hindering the training. While generative models have achieved significant success in standard perspective imagery, their application to wide-angle distortion remains unexplored. In this work, we formally introduce the novel problem of multi-view fisheye image generation conditioned on volumetric semantic representations and present two distinct methods. We first propose SyntheOcc-FE, which adapts the SyntheOcc architecture to fisheye data. While effective, this method is constrained by the scarcity of fisheye datasets, which limits its generalization. To overcome these limitations, we propose our second method, MIVIFI (multi-view fisheye), which leverages cross-domain learning with Equirectangular Projections. By bridging the gap between dataset domains using KITTI-360 fisheye images alongside nuScenes multi-view standard images, our approach enables high-fidelity manipulation of scene content. This framework enables the structural modification of semantic occupancy inputs to introduce or eliminate specific actors and facilitates the rendering of diverse meteorological conditions and illumination scenarios absent in the limited fisheye datasets. Quantitative and qualitative experiments demonstrate that our methods achieve robust photorealistic multi-view fisheye image generation and highlight the specific advantages of our cross-domain strategy for handling data scarcity.",
  "published": "2026-08-24",
  "updated": "2026-08-24",
  "year": "2026",
  "authors": [
   "Matthias Neuwirth-Trapp",
   "Beg\u00fcm Altunbas",
   "Jiayi Wang",
   "Yan Xia",
   "Maarten Bieshaar",
   "Xinyu Huang",
   "Daniel Cremers"
  ],
  "author_count": 7,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work formally introduces the novel problem of multi-view fisheye image generation conditioned on volumetric semantic representations and presents two distinct methods, which achieve robust photorealistic multi-view fisheye image generation and highlight the specific advantages of the cross-domain strategy for handling data scarcity.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Matthias Neuwirth-Trapp",
    "id": "2376265720",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Beg\u00fcm Altunba\u015f",
    "id": "2210862696",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Jiaying Wang",
    "id": "2457916604",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yan Xia",
    "id": "2239199844",
    "h_index": 7,
    "papers": 25
   },
   {
    "name": "Maarten Bieshaar",
    "id": "2375064019",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Xinyu Huang",
    "id": "2160051569",
    "h_index": 9,
    "papers": 26
   },
   {
    "name": "Daniel Cremers",
    "id": "2279831537",
    "h_index": 7,
    "papers": 22
   }
  ],
  "comment": "Accepted at the IEEE International Conference on Intelligent Transportation Systems (ITSC) 2026",
  "topics": [
   "spatial-3d",
   "data-teleop",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.23140v1",
  "pdf_url": "https://arxiv.org/pdf/2608.23140v1",
  "html_url": "https://arxiv.org/html/2608.23140v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.23138",
  "slug": "pointing-vla-typed-spatial-grounding-interfaces-for-vision-language-ac",
  "title": "Pointing-VLA: Typed Spatial Grounding Interfaces for Vision-Language-Action Manipulation",
  "abstract": "Vision-language-action (VLA) models often expose spatial grounding through autoregressive text coordinates or opaque action tokens, creating brittle interfaces between multimodal reasoning and robot execution. We present Pointing-VLA, a typed hidden-state spatial readout built on Embodied-R1. Geometry-specific heads predict normalized points, object-functional grounding (OFG) heatmaps, and visual trajectories without serializing geometry as text. For the evaluated Bridge/WidowX and physical pick-place deployments, an explicit execution contract assigns PICK to source-conditioned OFG and PLACE to Pointing, providing direct stage-aligned spatial targets. Pointing-VLA achieves SOTA performance on Bridge/WidowX, averaging 72.9\\% across the evaluated four-task set without Bridge-specific finetuning under collision-enabled CuRobo execution. Pointing and OFG show complementary strengths across native and cross-dataset evaluations. The OFG/contact readout transfers to NORA-1.5, preserving or improving success while reducing recorded controller time by more than 20$\\times$; typed heads are also 6.68--6.90$\\times$ faster than Embodied-R1 text decoding on a shared external suite. When integrated as spatial guidance for a $\u03c0_{0.5}$ action policy, Pointing-VLA raises autonomous real-robot success from 52.7\\% to 80.7\\% across three visual contexts. These results establish typed spatial readouts as an efficient, inspectable interface between embodied reasoning and robot execution.",
  "published": "2026-08-24",
  "updated": "2026-08-24",
  "year": "2026",
  "authors": [
   "Xiwen Chen",
   "Zelin Li",
   "Zhiruo Zhou",
   "Huiming Chen",
   "Chenwei Wang",
   "Xiaojun Zhu"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The results establish typed spatial readouts as an efficient, inspectable interface between embodied reasoning and robot execution as an efficient, inspectable interface between embodied reasoning and robot execution.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xiwen Chen",
    "id": "2448000948",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Zelin Li",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Zhiruo Zhou",
    "id": "96381098",
    "h_index": 8,
    "papers": 15
   },
   {
    "name": "Huiming Chen",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Chenwei Wang",
    "id": "2459228962",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Xiaojun Zhu",
    "id": "2458050231",
    "h_index": 0,
    "papers": 1
   }
  ],
  "comment": "",
  "topics": [
   "vla"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.23138v1",
  "pdf_url": "https://arxiv.org/pdf/2608.23138v1",
  "html_url": "https://arxiv.org/html/2608.23138v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.23100",
  "slug": "shaping-the-evolutionary-dynamics-of-robot-morphology-via-adaptive-con",
  "title": "Shaping the Evolutionary Dynamics of Robot Morphology via Adaptive Control Learning",
  "abstract": "Robot co-design via bi-level optimization couples within-lifetime controller learning for fitness evaluation with cross-generational morphological evolution. Prior work has established that well-adapted morphology facilitates faster control learning, a property termed morphological intelligence. Yet how control learning reciprocally shapes morphological evolution remains unexplored. This paper examines both directions for a holistic account of brain-body interplay. We first show that morphological contributions to control learning decouple into two orthogonal dimensions. We formalize the convergence speed as morphological intelligence and identify the performance ceiling as a complementary quantity termed true potential. A concise functional relation is then established to jointly characterize both quantities from individual learning curves, which, when aggregated at the population level, capture evolutionary profiles. Through extensive experiments on simulated voxel-based soft robots, we reveal that premature fitness evaluation systematically underestimates true potential and biases selection towards fast learners. This restricts design space exploration, compromising both optimization efficiency and morphological diversity. Notably, the widely recognized morphological Baldwin effect emerges as an artifact of this bias rather than a general evolutionary tendency. We therefore propose AdaControl, which monitors disproportionate selection for morphological intelligence during evolution and allocates minimally sufficient control learning for unbiased fitness evaluation. With AdaControl, a simple genetic algorithm rivals state-of-the-art generative-model-based co-design methods in discovering diverse high-performing designs while cutting computation by up to 80% versus exhaustive control.",
  "published": "2026-08-24",
  "updated": "2026-08-24",
  "year": "2026",
  "authors": [
   "Junru Song",
   "Yang Yang",
   "Yaqing Xu",
   "Ying Wen",
   "Wei Peng",
   "Guozhen Li",
   "Wei'en Zhou",
   "Wen Yao"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "AdaControl is proposed, which monitors disproportionate selection for morphological intelligence during evolution and allocates minimally sufficient control learning for unbiased fitness evaluation and rivals state-of-the-art generative-model-based co-design methods in discovering diverse high-performing designs while cutting computation by up to 80% versus exhaustive control.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Junru Song",
    "id": "2221337153",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Yang Yang",
    "id": "2293750643",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Yaqing Xu",
    "id": "2457215589",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Ying Wen",
    "id": "2155576705",
    "h_index": 8,
    "papers": 17
   },
   {
    "name": "Wei Peng",
    "id": "2258148771",
    "h_index": 8,
    "papers": 25
   },
   {
    "name": "Guozhen Li",
    "id": "2306652109",
    "h_index": 8,
    "papers": 10
   },
   {
    "name": "Weien Zhou",
    "id": "2148943712",
    "h_index": 18,
    "papers": 69
   },
   {
    "name": "Wen Yao",
    "id": "2283959173",
    "h_index": 5,
    "papers": 12
   }
  ],
  "comment": "",
  "topics": [
   "navigation",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.23100v1",
  "pdf_url": "https://arxiv.org/pdf/2608.23100v1",
  "html_url": "https://arxiv.org/html/2608.23100v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.23068",
  "slug": "switched-turn-based-adaptive-source-seeking-strategy-using-estimation",
  "title": "Switched Turn-based Adaptive Source Seeking Strategy using Estimation and Information-driven Direction of Improvement",
  "abstract": "Source seeking arises in applications such as gas leak localization, radiation monitoring, and environmental surveillance, where the origin of an unknown signal field must be estimated from spatial measurements. In practice, the source location is not directly observable and must be inferred from noisy scalar measurements collected during motion.In robotic source seeking, estimation and motion are closely linked: measurements improve the source estimate, while the chosen trajectory affects the quality of future measurements.Existing loop-based geometric strategies generate feasible motion but do not explicitly use estimation uncertainty to regulate direction updates.This paper presents a loop-based source-seeking framework that combines Extended Kalman Filter (EKF) estimation with Fisher Information Matrix (FIM)-based direction selection. The source estimate is updated during motion, and the heading is changed at loop boundaries using both estimation uncertainty and predicted information gain. A measurement-based stopping condition is used to detect convergence without requiring prior knowledge of the source location.Simulation results under stationary and moving source scenarios demonstrate improved tracking performance and reduced estimation error compared to purely information-driven or estimate-driven strategies.",
  "published": "2026-08-24",
  "updated": "2026-08-24",
  "year": "2026",
  "authors": [
   "Shubhra Banerjee",
   "Satadal Ghosh"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper presents a loop-based source-seeking framework that combines Extended Kalman Filter estimation with Fisher Information Matrix-based direction selection, and the source estimate is updated during motion, and the heading is changed at loop boundaries using both estimation uncertainty and predicted information gain.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "S. Banerjee",
    "id": "2111153172",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Satadal Ghosh",
    "id": "2302138506",
    "h_index": 2,
    "papers": 11
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.23068v1",
  "pdf_url": "https://arxiv.org/pdf/2608.23068v1",
  "html_url": "https://arxiv.org/html/2608.23068v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.23055",
  "slug": "macro-action-topological-navigation-under-noisy-localization-using-rei",
  "title": "Macro-Action Topological Navigation under Noisy Localization using Reinforcement Learning",
  "abstract": "Navigating large, photorealistic 3D apartments from raw pixels is widely considered infeasible for plain reinforcement learning. We build an agent that does it anyway, estimating its own pose from the camera alone. The agent has to reach several target objects in sequence, and their positions change between episodes, so it must explore to find them. It builds on our earlier object-centric topological controller, which still read the agent's true pose and its object detections from the simulator. Here we replace that true pose with an onboard, object-centric estimate. For each object we keep a bank of ORB features that, when the object is seen again, yield a rough pose measurement, which a minimal Extended Kalman Filter (EKF) fuses with a motion model. As on a real robot, the executed motions are noisy. The estimate drifts, but the agent and the nearby objects drift together, so a locally consistent pose is enough to follow each short edge and then home in visually on the target, which lets us replace full SLAM with a much smaller model, closer to how biological navigation appears to work. In the photorealistic Habitat simulator, the agent reaches its target objects from vision alone, with a pose that only needs to be locally consistent.",
  "published": "2026-08-24",
  "updated": "2026-08-24",
  "year": "2026",
  "authors": [
   "Simon Hakenes",
   "Tobias Glasmachers"
  ],
  "author_count": 2,
  "categories": [
   "cs.LG",
   "cs.RO"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work builds on the earlier object-centric topological controller, which still read the agent's true pose and its object detections from the simulator, but replaces that true pose with an onboard, object-centric estimate.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Simon Hakenes",
    "id": "84408176",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Tobias Glasmachers",
    "id": "2295625275",
    "h_index": 1,
    "papers": 4
   }
  ],
  "comment": "15 pages, Accepted at the Artificial Intelligence Symposium (AIS) 2026",
  "topics": [
   "sim2real",
   "rl-control",
   "spatial-3d",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.23055v1",
  "pdf_url": "https://arxiv.org/pdf/2608.23055v1",
  "html_url": "https://arxiv.org/html/2608.23055v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.23040",
  "slug": "roboracer-arena-scaling-high-fidelity-autonomous-racing-in-isaac-sim",
  "title": "RoboRacer Arena: Scaling High-Fidelity Autonomous Racing in Isaac Sim",
  "abstract": "RoboRacer offers a standardized platform for research using 1:10-scale autonomous vehicles, but the variety of available tracks hinders the process of acquiring policies. Although existing occupancy-grid simulators allow for the quick addition of new maps, they fail to include physical contact, while 3D simulators require each circuit to be implemented as a separate asset, thus limiting their scalability. In order to overcome this issue, we have developed RoboRacer Arena, a system that creates 3D racing environments directly from occupancy maps. Our method starts by using a flood fill algorithm to extract the drivable corridors and to identify the track boundaries, which are then used to establish the barriers. A distance field is calculated to define the collision boundaries. The track surfaces, collision properties, and materials are assembled into a USD stage, which allows for the automated and reproducible generation of the environment in Isaac Sim. The input maps can be obtained from SLAM sessions, from rescaled Formula 1 circuits, or from natural-language descriptions. When the input is based on natural language, we use Gemma 4 31B to generate a track specification without specifying any coordinates or geometry. To guarantee consistency and reproducibility, we apply geometric screening, procedural generation, and raster-level validation. The simulation environments are initialized in a time range of 1.18 to 2.48 seconds, with the initialization time increasing linearly as the raster size increases. In 30 matched trials involving 10 tracks and 3 seeds, 21 maps were generated and all passed validation. RoboRacer Arena currently contains 130 tracks and supports the generation of tracks from natural language. In benchmark tests, the system attains 8,707 vehicle-steps per second when using 256 parallel rigid-body vehicles, excluding the time taken for rendering and policy execution.",
  "published": "2026-08-24",
  "updated": "2026-08-24",
  "year": "2026",
  "authors": [
   "Mihaela-Larisa Clement",
   "Agnes Poks",
   "Ezio Bartocci"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA 2027",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "RoboRacer Arena is a system that creates 3D racing environments directly from occupancy maps and supports the generation of tracks from natural language, and applies geometric screening, procedural generation, and raster-level validation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Mihaela-Larisa Clement",
    "id": "2351607941",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "A. Poks",
    "id": "1671245920",
    "h_index": 4,
    "papers": 18
   },
   {
    "name": "E. Bartocci",
    "id": "46289371",
    "h_index": 39,
    "papers": 242
   }
  ],
  "comment": "Submitted to ICRA 2027",
  "topics": [
   "sim2real",
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.23040v1",
  "pdf_url": "https://arxiv.org/pdf/2608.23040v1",
  "html_url": "https://arxiv.org/html/2608.23040v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.23000",
  "slug": "free-energy-gated-plasticity-for-real-time-online-motor-learning-in-ph",
  "title": "Free-Energy-Gated Plasticity for Real-Time Online Motor Learning in Physical Human-Robot Interaction",
  "abstract": "Fully online embodied learning requires synaptic adaptation to acquire new behaviors while preserving previously learned dynamics during ongoing interaction. We extend the Predictive-Coding-inspired Variational Recurrent Neural Network (PV-RNN) to continuously adapt its synaptic weights and propose Free-Energy-Gated Plasticity (FEGP), which regulates the effective learning rate according to variational free energy. In real-time physical human-robot interaction, a randomly initialized network acquired three cyclic motor patterns without offline pretraining, replay, or task-boundary signals, with all three patterns emerging in autonomous rollouts. Controlled experiments over ten randomized teaching streams and five network initializations per stream showed that FEGP substantially improved repertoire coverage and retention of previously acquired patterns after they left the recent observation window. Neither a constant learning rate matched to the gate's time-averaged effective rate nor replay of the same gain values with disrupted temporal organization reproduced these improvements. These results indicate that the temporal allocation of plasticity relative to model-environment mismatch, rather than simply its average magnitude or distribution, is critical for maintaining previously acquired behaviors during continued online learning.",
  "published": "2026-08-24",
  "updated": "2026-08-25",
  "year": "2026",
  "authors": [
   "Hiroki Sawada",
   "Jun Tani"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO",
   "cs.HC"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The Predictive-Coding-inspired Variational Recurrent Neural Network is extended to continuously adapt its synaptic weights and Free-Energy-Gated Plasticity (FEGP) is proposed, which regulates the effective learning rate according to variational free energy.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hiroki Sawada",
    "id": "2147156916",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Jun Tani",
    "id": "2264615654",
    "h_index": 3,
    "papers": 7
   }
  ],
  "comment": "",
  "topics": [
   "foundation-pretraining",
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.23000v2",
  "pdf_url": "https://arxiv.org/pdf/2608.23000v2",
  "html_url": "https://arxiv.org/html/2608.23000v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.22990",
  "slug": "instructmove-a-text-indispensable-benchmark-for-instruction-following",
  "title": "InstructMove: A Text-Indispensable Benchmark for Instruction-Following Manipulation",
  "abstract": "Vision-language-action (VLA) models have made general-purpose robot manipulation increasingly plausible by conditioning robot actions on natural-language instructions. A key test of such generality is whether policies actually follow language instructions. Yet many manipulation benchmarks leave this ability underdetermined: the intended object or destination is often visually salient or uniquely feasible, allowing policies to succeed without grounding the instruction. We argue that instruction-following evaluation should be text-indispensable: multiple actions should be visually and physically plausible, while only one should be consistent with the language instruction. We introduce InstructMove, a text-indispensable benchmark for instruction-following manipulation. InstructMove instantiates this principle in pick-and-place scenes with semantic distractors, decomposing instruction following into category identification, attribute discrimination, spatial reasoning, and compositional pick-and-place. InstructMove supports a train-eval protocol with InstructMove training data and held-out evaluation tasks, with additional diagnostics for language dependence. Experiments with representative VLA policies show that InstructMove provides a controlled testbed for diagnosing visual shortcuts and that InstructMove simulation data can improve real-world instruction-following manipulation performance. Code: https://github.com/HorizonRobotics/RoboOrchardSim",
  "published": "2026-08-24",
  "updated": "2026-08-24",
  "year": "2026",
  "authors": [
   "Mengao Zhao",
   "Ziang Li",
   "Chaodong Huang",
   "Mengchen Ma",
   "Haoyi Jiang",
   "Yiwei Jin",
   "Xinjie Wang",
   "Yun Du",
   "Xuewu Lin",
   "Taojun Ding",
   "Hongyu Xie",
   "Jackson Jiang",
   "Chunlei Yu",
   "Kaihua Zhang",
   "Lichao Huang",
   "Liu Liu",
   "Tianwei Lin",
   "Zhizhong Su"
  ],
  "author_count": 18,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "It is argued that instruction-following evaluation should be text-indispensable: multiple actions should be visually and physically plausible, while only one should be consistent with the language instruction.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Mengao Zhao",
    "id": "2373941237",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Ziang Li",
    "id": "2366602735",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Chaodong Huang",
    "id": "2373562831",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Mengchen Ma",
    "id": "2450001720",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Haoyi Jiang",
    "id": "2317648374",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Yiwei Jin",
    "id": "2362810365",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Xinjie Wang",
    "id": "2333518293",
    "h_index": 5,
    "papers": 15
   },
   {
    "name": "Yun Du",
    "id": "2372420138",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Xuewu Lin",
    "id": "2267389066",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Taojun Ding",
    "id": "2219296283",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Hongyu Xie",
    "id": "2333325881",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Jackson Jiang",
    "id": "2377922107",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Chunlei Yu",
    "id": "2378055435",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Kaihua Zhang",
    "id": "2459263730",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Lichao Huang",
    "id": "2267402336",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Liu Liu",
    "id": "2332597209",
    "h_index": 8,
    "papers": 31
   },
   {
    "name": "Tianwei Lin",
    "id": "2268009422",
    "h_index": 8,
    "papers": 15
   },
   {
    "name": "Zhizhong Su",
    "id": "2374211357",
    "h_index": 3,
    "papers": 7
   }
  ],
  "comment": "22 pages",
  "topics": [
   "vla",
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.22990v1",
  "pdf_url": "https://arxiv.org/pdf/2608.22990v1",
  "html_url": "https://arxiv.org/html/2608.22990v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.22983",
  "slug": "csymplan-certified-symbolic-planning-and-control-for-high-dof-manipula",
  "title": "CSymPlan: Certified Symbolic Planning and Control for High-DOF Manipulators",
  "abstract": "Robot manipulators are commonly engineered around a decoupled motion-generation stack: a planner computes a collision-free path and a lower-level controller tracks the resulting reference. This separation is computationally convenient, but it can produce references that are difficult to execute under actuator limits, tracking error, model mismatch, and small obstacle clearances. We present CSymPlan, a certified symbolic planning and control framework for high-DOF manipulators with two complementary implementations: an offline implementation that precomputes certified reach-avoid feedback policies for known workspaces; and an online implementation that synthesizes or updates symbolic policies at runtime from changing task and perception information using parallelization. The offline implementation reduces the manipulator dynamics to a sampled perturbed double-integrator model in operational space through feedback linearization, treats torque-realization errors, modeling inaccuracies, and measurement uncertainty as bounded disturbances, and refines the synthesized symbolic policy to the Franka FR3 through a quantization--lookup--torque realization pipeline. The online implementation uses the same abstraction and refinement interface, but replaces the precomputed policy table with a runtime pFaces request--synthesis--execution loop. In randomized simulated benchmarks and perception-driven Franka FR3 experiments, both implementations complete reach-avoid tasks with zero safety violations; whenever no certified action exists, the robot holds, replans, or stops safely instead of executing an uncertified command.",
  "published": "2026-08-24",
  "updated": "2026-08-24",
  "year": "2026",
  "authors": [
   "Aditya Narendra",
   "Ashok Kumar Saini",
   "Mahathi Anand",
   "Mahmoud Khaled",
   "Fares J. Abu-Dakka",
   "Abdalla Swikir"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "CSymPlan is presented, a certified symbolic planning and control framework for high-DOF manipulators with two complementary implementations: an offline implementation that precomputes certified reach-avoid feedback policies for known workspaces; and an online implementation that synthesizes or updates symbolic policies at runtime from changing task and perception information using parallelization.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Aditya Narendra",
    "id": "2325097237",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Ashok Kumar Saini",
    "id": "2459201311",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "M. Anand",
    "id": "2379935857",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Mahmoud Khaled",
    "id": "144466838",
    "h_index": 8,
    "papers": 20
   },
   {
    "name": "Fares J. Abu-Dakka",
    "id": "2257270913",
    "h_index": 7,
    "papers": 22
   },
   {
    "name": "Abdalla Swikir",
    "id": "27572423",
    "h_index": 8,
    "papers": 41
   }
  ],
  "comment": "",
  "topics": [
   "hardware-codesign",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.22983v1",
  "pdf_url": "https://arxiv.org/pdf/2608.22983v1",
  "html_url": "https://arxiv.org/html/2608.22983v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.22976",
  "slug": "privileged-critic-training-enables-sensor-free-thruster-fault-adaptati",
  "title": "Privileged Critic Training Enables Sensor-Free Thruster Fault Adaptation in End-to-End RL",
  "abstract": "Fault-tolerant navigation for thruster-actuated robots requires online adaptation to failures that are neither binary nor fully observable: thrusters may degrade continuously, fail dead, or jam stuck-open. Classical fault detection pipelines require dedicated sensors unavailable at deployment; oracle controllers that observe the true failure state are equally impractical. We show that privileged critic training is sufficient for sensor-free fault adaptation: giving the PPO value function access to the true degradation state dgt during training, while the actor receives only standard task observations, shapes a policy that compensates for failures at deployment without any dedicated fault sensing. We propose RAFT (Recurrent Asymmetric Fault Tolerant), a policy with recurrent memory trained with a privileged asymmetric critic. Evaluated on a floating-platform robot (8 thrusters, 1 reaction wheel) under up to four simultaneous thruster failures, RAFT achieves 70.2% success at four concurrent failures, closing 84% of the gap from a failure-naive baseline (4.8%) to an oracle policy that sees the full degradation state at deployment (82.4%). All code, checkpoints, and data are open-source.",
  "published": "2026-08-24",
  "updated": "2026-08-24",
  "year": "2026",
  "authors": [
   "Ricard Marsal I Castan",
   "Miguel A. Olivares-M\u00e9ndez"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "It is shown that privileged critic training is sufficient for sensor-free fault adaptation: giving the PPO value function access to the true degradation state dgt during training, and shapes a policy that compensates for failures at deployment without any dedicated fault sensing.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ricard M. Castan",
    "id": "2362503919",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Miguel A. Olivares-M\u00e9ndez",
    "id": "2386534297",
    "h_index": 1,
    "papers": 5
   }
  ],
  "comment": "",
  "topics": [
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.22976v1",
  "pdf_url": "https://arxiv.org/pdf/2608.22976v1",
  "html_url": "https://arxiv.org/html/2608.22976v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.22896",
  "slug": "supermap-a-spatio-temporal-slam-system-for-visual-language-navigation",
  "title": "SuperMap: A Spatio-Temporal SLAM System for Visual-Language Navigation",
  "abstract": "Robotic navigation in human environments requires a spatio-temporal semantic representation that can rec- oncile open-vocabulary perception with long-term environmental changes. While foundation models provide strong zero-shot recognition, their predictions are intermittent and view-dependent, and naively integrating them into mapping pipelines leads to identity drift and stale semantics over time. We present SuperMap, a 4D spatio-temporal mapping framework for language-guided navigation that integrates high-frequency geometric SLAM with asynchronous open-vocabulary perception. Our core contribution is a consistency-driven mapping engine that combines 3D-aware instance association/re-activation with a principled existence-and-label confidence update to maintain stable object identities and prune outdated map content under occlusions and scene changes. SuperMap produces a queryable 4D scene-graph representation that interfaces naturally with Vision-Language Models by supporting compositional queries over object semantics, relations, We demonstrate SuperMap on benchmarks and real robots, including dynamic scenes with appearance/disappearance and relocation, and provide ablations and runtime analysis. We release the full system as open-source to provide the community with a deployable baseline for open-vocabulary spatio-temporal mapping. Project website: superodometry.com/supermap.",
  "published": "2026-08-24",
  "updated": "2026-08-24",
  "year": "2026",
  "authors": [
   "Shibo Zhao",
   "Guofei Chen",
   "Honghao Zhu",
   "Zhiheng Li",
   "Changwei Yao",
   "Nader Zantout",
   "Seungchan Kim",
   "Wenshan Wang",
   "Ji Zhang",
   "Sebastian Scherer"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "This work presents SuperMap, a 4D spatio-temporal mapping framework for language-guided navigation that integrates high-frequency geometric SLAM with asynchronous open-vocabulary perception and releases the full system as open-source to provide the community with a deployable baseline for open-vocabulary spatio-temporal mapping.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Shibo Zhao",
    "id": "2278582891",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Guofei Chen",
    "id": "2296753914",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Honghao Zhu",
    "id": "2258680234",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Zhiheng Li",
    "id": "2450169091",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Changwei Yao",
    "id": "2450158024",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Nader Zantout",
    "id": "2329370999",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Seungchan Kim",
    "id": "2278742776",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Wenshan Wang",
    "id": "2296220021",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Ji Zhang",
    "id": "2350764419",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Sebastian A. Scherer",
    "id": "2265953722",
    "h_index": 3,
    "papers": 8
   }
  ],
  "comment": "",
  "topics": [
   "spatial-3d",
   "navigation",
   "foundation-pretraining",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.22896v1",
  "pdf_url": "https://arxiv.org/pdf/2608.22896v1",
  "html_url": "https://arxiv.org/html/2608.22896v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2608.22869",
  "slug": "unimem-unifying-multimodal-memory-and-control-for-vision-language-acti",
  "title": "UniMem: Unifying Multimodal Memory and Control for Vision-Language-Action Models",
  "abstract": "While Vision-Language-Action (VLA) models have leveraged internet-scale pretraining and task-focused finetuning to achieve strong performance on long-horizon tasks, they often struggle with non-Markovian tasks that require memory. Existing approaches to memory typically involve additional Vision-Language-Models (VLMs) for long-term memory management, introducing a memory bottleneck and a fractured training pipeline. Conditioning on multiple historical frames can provide the VLA with access to more descriptive features of past scenes, but can degrade performance if frames are chosen at arbitrary, fixed intervals. To address these limitations, we present UniMem, a framework that unifies high-level, multimodal memory and low-level control under one backbone. UniMem employs an event classifier for memory updates, a keyframe encoder for dense spatial memory, and a keyframe caching technique to minimize overhead during policy rollouts. We evaluate UniMem across five simulation and four hardware tasks targeting sequential and spatial memory, demonstrating that our unified, single-model system outperforms fixed-interval image sampling baselines (93.4% vs. 68.2%) in simulation and hierarchical baselines (80.0% vs. 43.5%) in hardware, while offering faster inference and a simple training pipeline for easy adoption. Project website: https://losterberg3.github.io/unimem-vla/",
  "published": "2026-08-24",
  "updated": "2026-08-24",
  "year": "2026",
  "authors": [
   "Lars Osterberg",
   "Maggie Wang",
   "Mac Schwager"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "UniMem is presented, a framework that unifies high-level, multimodal memory and low-level control under one backbone that outperforms fixed-interval image sampling baselines in simulation and hierarchical baselines in hardware, while offering faster inference and a simple training pipeline for easy adoption.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Lars W. Osterberg",
    "id": "2323996376",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "M. Wang",
    "id": "2027032530",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Mac Schwager",
    "id": "2360172520",
    "h_index": 6,
    "papers": 14
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.22869v1",
  "pdf_url": "https://arxiv.org/pdf/2608.22869v1",
  "html_url": "https://arxiv.org/html/2608.22869v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.22800",
  "slug": "triplet2track-a-hierarchical-system-with-object-centric-representation",
  "title": "Triplet2Track: A Hierarchical System with Object-Centric Representations for Reliable Long-Horizon Manipulation",
  "abstract": "Ensuring reliability in uncertain environments remains difficult for long-horizon robotic manipulation. End-to-end VLA models are data-heavy and opaque, making diagnosis and verification difficult. Hierarchical pipelines are more interpretable, but their plans are often weakly grounded in observations, weakly aligned with low-level actions, and computed without online feedback, leading to open-loop behavior and hallucinations. To address these issues, we introduce the Triplet-to-Track System (TTS), a closed-loop long-horizon imitation learning system that uses human videos to reduce reliance on robot-collected data. TTS represents high-level subgoals as instance-grounded triplets, translates them into continuous track priors for execution, and monitors task progress from observations for online replanning. Across diverse real-world long-horizon tasks, TTS achieves a 74.8\\% average success rate and supports object-level and compositional generalization.",
  "published": "2026-08-24",
  "updated": "2026-08-24",
  "year": "2026",
  "authors": [
   "Jianxiang Liu",
   "Gaojing Zhang",
   "Chuan Wen",
   "Qipeng Liu",
   "Yuxuan Zhao",
   "Ning Guo",
   "Wenzhao Lian"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The Triplet-to-Track System (TTS), a closed-loop long-horizon imitation learning system that uses human videos to reduce reliance on robot-collected data, achieves a 74.8\\% average success rate and supports object-level and compositional generalization.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jianxiang Liu",
    "id": "2266695219",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Gaojing Zhang",
    "id": "2382841296",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Chuan Wen",
    "id": "2381956964",
    "h_index": 5,
    "papers": 20
   },
   {
    "name": "Qipeng Liu",
    "id": "2282237503",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Yuxuan Zhao",
    "id": "2391839100",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Ning Guo",
    "id": "2393077879",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Wenzhao Lian",
    "id": "2323497285",
    "h_index": 3,
    "papers": 16
   }
  ],
  "comment": "8 pages, 6 figures. Accepted for presentation at the 2026 IEEE International Conference on Systems, Man, and Cybernetics (SMC 2026)",
  "topics": [
   "vla",
   "egocentric-data",
   "imitation-diffusion",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.22800v1",
  "pdf_url": "https://arxiv.org/pdf/2608.22800v1",
  "html_url": "https://arxiv.org/html/2608.22800v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.22799",
  "slug": "reproducible-vision-guided-6-dof-robotic-manipulator-with-a-mixed-step",
  "title": "Reproducible Vision-Guided 6-DoF Robotic Manipulator with a Mixed Stepper-Driver Architecture and Browser-Native Control",
  "abstract": "We present the NeuralNexus Arm, an open, low-cost 6-DOF robotic manipulator built by an undergraduate engineering team, together with the design decisions and debugging experience needed to reproduce it. The arm is driven by a single STM32H743 microcontroller on a custom printed circuit board (PCB) and combines two stepper-driver strategies on one controller: push-pull 3.3 V step/direction outputs for onboard TMC2209 drivers on the three wrist joints, and open-drain outputs for external CL57T and DM542 drivers on the three high-torque proximal joints. We describe the mechanical design, mixed-driver electronics, interrupt-driven firmware, a MATLAB/Simscape-based inverse-kinematics pipeline, a browser-native control interface using the Web Serial API, and a lightweight vision pipeline for object localisation and autonomous pick-and-place tasks. We also document non-obvious hardware and firmware failure modes encountered during the transition from a development board to the custom PCB as reproducibility guidance. All design files and firmware are released openly. The platform actuates all six axes under coordinated control at a 2 kHz update rate and executes both manual and pre-recorded motions from the browser interface.",
  "published": "2026-08-24",
  "updated": "2026-08-24",
  "year": "2026",
  "authors": [
   "Lasan Perera",
   "Deneth Priyadarshana",
   "Dulana Pitiwaduge",
   "Isitha Dinujaya",
   "Mokshan Colambage"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.CV",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Lasan Perera",
    "id": "2459197020",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Deneth Priyadarshana",
    "id": "2459195587",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Dulana Pitiwaduge",
    "id": "2459195561",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Isitha Dinujaya",
    "id": "2459198460",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Mokshan Colambage",
    "id": "2459198289",
    "h_index": 0,
    "papers": 0
   }
  ],
  "comment": "13 pages, 16 figures, 7 tables. Design files and firmware: https://github.com/Lasan-Perera/6-dof-arm-neuralnexus",
  "topics": [
   "hardware-codesign",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.22799v1",
  "pdf_url": "https://arxiv.org/pdf/2608.22799v1",
  "html_url": "https://arxiv.org/html/2608.22799v1",
  "code_url": "https://github.com/Lasan-Perera/6-dof-arm-neuralnexus",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2608.22731",
  "slug": "llm-based-selection-of-incongruent-verbal-and-nonverbal-behavior-for-v",
  "title": "LLM-Based Selection of Incongruent Verbal and Nonverbal Behavior for Virtual Humans",
  "abstract": "Nonverbal behavior generation systems for virtual agents often take an utterance as input and generate nonverbal behaviors that emphasize or illustrate the content of the verbal channel. However, human nonverbal behavior is shaped by more than the content of the speech. It is also influenced by speaker roles, interpersonal relationships, social context, and the cognitive and emotional states of the interactants. As a result, the nonverbal channel may reinforce, weaken, qualify, or even contradict the verbal channel. It may also reveal internal states that are hidden or only indirectly implied in speech, including emotional \"leakage\" that may be incidental to the immediate interaction. Modeling this richer relationship between verbal and nonverbal behavior is important for designing virtual agents that exhibit realistic, human-like behavior. It is especially critical in training contexts that require nuanced social interpretation, such as counseling simulations involving virtual patients. Drawing on Ekman's framework of verbal nonverbal relationships, we propose a taxonomy of categories in which mismatches between verbal and nonverbal behavior can occur. We then examine alternative approaches for realizing these behaviors using large language models, focusing on whether LLMs can select contextually appropriate mismatched verbal and nonverbal behaviors from a given dialogue and social interaction context. Finally, we evaluate the resulting behaviors in a human-subject study, assessing whether context-driven nonverbal behavior, when embodied in a virtual human, produces the intended effects on observers.",
  "published": "2026-08-24",
  "updated": "2026-08-24",
  "year": "2026",
  "authors": [
   "Parisa Ghanad Torshizi",
   "Stacy Marsella"
  ],
  "author_count": 2,
  "categories": [
   "cs.AI",
   "cs.HC",
   "cs.RO"
  ],
  "primary_category": "cs.AI",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes a taxonomy of categories in which mismatches between verbal and nonverbal behavior can occur, and proposes a taxonomy of categories in which context-driven nonverbal behavior, when embodied in a virtual human, produces the intended effects on observers.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "P. Torshizi",
    "id": "2256152467",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Stacy Marsella",
    "id": "2306034644",
    "h_index": 2,
    "papers": 5
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.22731v1",
  "pdf_url": "https://arxiv.org/pdf/2608.22731v1",
  "html_url": "https://arxiv.org/html/2608.22731v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.22701",
  "slug": "physics-filtering-favors-the-generalization-of-robot-learning",
  "title": "Physics Filtering Favors the Generalization of Robot Learning",
  "abstract": "Living organisms exhibit extraordinary adaptability to unseen environments through their intrinsic physical structures and lifelong feedback-driven learning. Endowing robots with comparable generalization is critical for reliable operation in the real world. While recent approaches attempt to improve generalization by scaling training data, such strategies remain impractical for robotics, where collecting real-world demonstrations at the scale of large language models is prohibitively costly and slow. Contrary to this reliance on massive datasets, we show that robots can generalize effectively under dynamics uncertainties even with limited training data by leveraging a feedback mechanism, namely PhyFilter, that corrects learning outputs with physics-filtered learning residuals. PhyFilter operates as a lightweight, model-agnostic module whose parameters can be automatically optimized through an auto-learning algorithm, eliminating manual tuning and enabling seamless integration with diverse robot policies. We validate PhyFilter across four representative robotic systems, demonstrating that it enables quadruped robots to generalize to unseen terrains, payload variations, and speed ranges; drones to flight under unseen wind disturbances; aerial manipulators to achieve centimeter-level in-air capture despite wind and mass uncertainties; and acceleration differentiators to remain robust with distribution shift. These results show that physics-filtered feedback can serve as a powerful alternative to massive data scaling.",
  "published": "2026-08-24",
  "updated": "2026-08-24",
  "year": "2026",
  "authors": [
   "Jindou Jia",
   "Shixuan Han",
   "Meng Wang",
   "Gen Li",
   "Zihan Yang",
   "Sicheng Zhou",
   "Kexin Guo",
   "Jianfei Yang",
   "Xiang Yu",
   "Wei Wang",
   "Lei Guo"
  ],
  "author_count": 11,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "It is shown that robots can generalize effectively under dynamics uncertainties even with limited training data by leveraging a feedback mechanism, namely PhyFilter, that corrects learning outputs with physics-filtered learning residuals, and shows that physics-filtered feedback can serve as a powerful alternative to massive data scaling.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jindou Jia",
    "id": "3230183",
    "h_index": 9,
    "papers": 27
   },
   {
    "name": "Shixu Han",
    "id": "117887276",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Meng Wang",
    "id": "2146060953",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Gen Li",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Zihan Yang",
    "id": "2313686511",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Sicheng Zhou",
    "id": "2108885765",
    "h_index": 7,
    "papers": 14
   },
   {
    "name": "Kexin Guo",
    "id": "2301930605",
    "h_index": 6,
    "papers": 27
   },
   {
    "name": "Jianfei Yang",
    "id": "2404007795",
    "h_index": 2,
    "papers": 27
   },
   {
    "name": "Xiang Yu",
    "id": "2403003287",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Wei Wang",
    "id": "2270568699",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Lei Guo",
    "id": "2316668864",
    "h_index": 2,
    "papers": 9
   }
  ],
  "comment": "Accepted by npj Robotics",
  "topics": [
   "humanoids",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.22701v1",
  "pdf_url": "https://arxiv.org/pdf/2608.22701v1",
  "html_url": "https://arxiv.org/html/2608.22701v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2608.22679",
  "slug": "contextrast-robust-multi-scale-contextual-contrastive-learning-for-sem",
  "title": "Contextrast++: Robust Multi-Scale Contextual Contrastive Learning for Semantic Segmentation",
  "abstract": "Semantic segmentation has rapidly advanced with deep learning; however, challenges remain in effectively capturing local and global contexts as well as addressing the long-tailed distribution problem. To tackle these issues, we present Contextrast++, a robust contrastive learning method for semantic segmentation that improves multi-scale feature integration and mitigates class imbalance issues. Our method consists of two key components: 1) contextual contrastive learning (CCL) and 2) boundary-aware negative (BANE) sampling. CCL includes three subcomponents: adaptive fusion module, pixel-to-anchor (PA) loss, and anchor-to-anchor (AA) loss. The adaptive fusion module dynamically balances local and global feature integration, resulting in a more context-aware representation. While the PA loss leverages the fused multi-scale features to improve feature representation learning, the AA loss focuses on addressing the long-tailed distribution problem by utilizing a memory bank that stores a fixed number of class-balanced representative anchors. Meanwhile, BANE sampling enhances segmentation precision by selecting hard negatives from misclassified boundary regions, which refines fine-grained details during contrastive learning. As verified in extensive experiments using public datasets, we demonstrate that Contextrast++ substantially improves semantic segmentation performance over existing contrastive learning-based state-of-the-art approaches, while introducing no additional computational overhead during inference.",
  "published": "2026-08-24",
  "updated": "2026-08-24",
  "year": "2026",
  "authors": [
   "Changki Sung",
   "Hyungtae Lim",
   "Wanhee Kim",
   "Youngwoo Seo",
   "Hyun Myung"
  ],
  "author_count": 5,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "It is demonstrated that Contextrast++ substantially improves semantic segmentation performance over existing contrastive learning-based state-of-the-art approaches, while introducing no additional computational overhead during inference.",
  "doi": "10.1109/TPAMI.2026.3721845",
  "oa_pdf": "https://doi.org/10.48550/arxiv.2608.22679",
  "s2_authors": [
   {
    "name": "Chan-Yong Sung",
    "id": "2064470465",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Hyungtae Lim",
    "id": "151025116",
    "h_index": 20,
    "papers": 50
   },
   {
    "name": "Wanhee Kim",
    "id": "2296948108",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Youngwoo Seo",
    "id": "2067714022",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Hyun Myung",
    "id": "2264289096",
    "h_index": 7,
    "papers": 14
   }
  ],
  "comment": "Accepted to IEEE Transactions on Pattern Analysis and Machine Intelligence (TPAMI), 2026",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.22679v1",
  "pdf_url": "https://arxiv.org/pdf/2608.22679v1",
  "html_url": "https://arxiv.org/html/2608.22679v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.22678",
  "slug": "raco-reliability-aware-coarse-goal-optimization-for-inspection-oriente",
  "title": "RACO: Reliability-Aware Coarse-Goal Optimization for Inspection-Oriented UAV Vision-Language Navigation",
  "abstract": "UAV vision-language navigation (UAV-VLN) is commonly evaluated as goal reaching, but inspection-oriented deployment requires the agent to stop within a valid inspection region and avoid falsely confirming visually or semantically similar distractors. This requirement exposes a key weakness in existing coarse-to-fine UAV-VLN policies: the coarse goal predicted before local refinement is often treated as reliable, although it may drift toward plausible but incorrect object regions and limit the ability of the local stage to recover. To systematically evaluate this problem, we introduce LG-UVI, an object-centric inspection evaluation setting derived from CityNav/CityRefer. LG-UVI extends standard UAV-VLN episodes with target objects, hard distractors, type-aware inspection regions, and diagnostics for inspection-region arrival and object-level confirmation. To address this inspection-oriented setting, we further propose RACO, a reliability-aware adaptive coarse-to-fine navigation framework. Instead of treating the predicted coarse goal as a fixed waypoint, RACO views it as a runtime hypothesis and uses object-level candidate anchors to check and correct coarse localization before Stage 1 and at the Stage 1-to-Stage 2 boundary. RACO also applies scale-adaptive terminal refinement to handle terminal near-miss cases using runtime-observable geometric and anchor-based evidence. Under a unified online evaluation protocol, RACO improves SR over the reproduced HETT baseline by 9.53 and 7.98 percentage points on validation-unseen and test-unseen, respectively. It also improves inspection-region arrival and reduces false verification risk, showing that coarse-goal reliability optimization is an effective complement to existing coarse-to-fine UAV-VLN policies.",
  "published": "2026-08-24",
  "updated": "2026-08-24",
  "year": "2026",
  "authors": [
   "Sen Wang",
   "Yiming Sun",
   "Jiaxuan He",
   "Pengfei Zhu"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "RACO is proposed, a reliability-aware adaptive coarse-to-fine navigation framework that improves inspection-region arrival and reduces false verification risk, showing that coarse-goal reliability optimization is an effective complement to existing coarse-to-fine UAV-VLN policies.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Senhao Wang",
    "id": "2456610654",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yiming Sun",
    "id": "2108938109",
    "h_index": 8,
    "papers": 22
   },
   {
    "name": "Jiaxuan He",
    "id": "2409719562",
    "h_index": 5,
    "papers": 24
   },
   {
    "name": "Pengfei Zhu",
    "id": "2269743082",
    "h_index": 16,
    "papers": 71
   }
  ],
  "comment": "",
  "topics": [
   "sim2real",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.22678v1",
  "pdf_url": "https://arxiv.org/pdf/2608.22678v1",
  "html_url": "https://arxiv.org/html/2608.22678v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.22675",
  "slug": "vikpath-a-vision-kansformer-framework-for-effective-obstacle-avoidance",
  "title": "VikPath: A Vision Kansformer Framework for Effective Obstacle Avoidance in Self-Supervised Pathfinding",
  "abstract": "Pathfinding is a fundamental problem in artificial intelligence and autonomous systems. Traditional heuristic-based algorithms, such as A*, rely on predefined heuristic functions to guide the search process. Although effective in structured environments, their search efficiency can degrade substantially in complex, obstacle-rich scenarios, where handcrafted heuristics may provide limited guidance. Recent studies have explored learning-based approaches to improve pathfinding efficiency; however, most existing methods rely on supervised learning and require labels generated by conventional planners or obtained through manual annotation. As a result, their performance is inherently influenced by the quality of the underlying supervision and may degrade when the labeling heuristics fail to capture complex environmental structures. Moreover, existing methods primarily optimize for path length while paying limited attention to obstacle clearance and trajectory smoothness, which can lead to paths that are difficult or unsafe to execute in real-world environments. To address these limitations, we propose $\\Design$, a self-supervised pathfinding framework that jointly considers obstacle proximity and path smoothness. At its core, our novel \\textit{Vision Kansformer} module learns representations of obstacle distributions without relying on labeled trajectories, enabling the model to better adapt to complex environments. We further introduce a sharp-turn penalty to encourage smoother and more practically executable paths. Extensive experiments demonstrate that, compared with state-of-the-art (SOTA) approaches, $\\Design$ achieves an average of 3.28\\% greater obstacle clearance and 87.07\\% lower inference latency while maintaining smooth path generation.",
  "published": "2026-08-24",
  "updated": "2026-08-24",
  "year": "2026",
  "authors": [
   "Junyao Wang",
   "Yulin Xu",
   "Mohammad Abdullah Al Faruque"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A self-supervised pathfinding framework that jointly considers obstacle proximity and path smoothness is proposed, and a sharp-turn penalty is introduced to encourage smoother and more practically executable paths.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Junyao Wang",
    "id": "2155090832",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Yulin Xu",
    "id": "2456838062",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "M. A. Faruque",
    "id": "1755024",
    "h_index": 40,
    "papers": 206
   }
  ],
  "comment": "",
  "topics": [
   "navigation",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.22675v1",
  "pdf_url": "https://arxiv.org/pdf/2608.22675v1",
  "html_url": "https://arxiv.org/html/2608.22675v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.22671",
  "slug": "exact-finite-length-theory-of-uniform-car-parking-spatial-laws-absorpt",
  "title": "Exact Finite-Length Theory of Uniform Car Parking: Spatial Laws, Absorption, and Aggregation",
  "abstract": "The uniform car-parking process is the one-dimensional random sequential adsorption of unit cars on a segment of finite length $s$: cars arrive at uniformly random positions and park wherever they fit, until no gap admits another. This paper develops the exact finite-$s$ theory. The joint density of the parked positions is resolved into jamming cells, on each of which it is a rational function, and evaluated by a subset recursion in $O(2^n n)$ operations; the marginal and gap order statistics are obtained as hyperlogarithms whose weight is fixed by the number of coordinates integrated out; and the absorption count and the aggregate quantities are treated through the integral equation descending from R\u00e9nyi.",
  "published": "2026-08-24",
  "updated": "2026-08-24",
  "year": "2026",
  "authors": [
   "Ganesh P Kumar"
  ],
  "author_count": 1,
  "categories": [
   "cs.RO",
   "cs.DS",
   "cs.SC",
   "math.PR",
   "math.RA"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ganesh P. Kumar",
    "id": "145529136",
    "h_index": 5,
    "papers": 10
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.22671v1",
  "pdf_url": "https://arxiv.org/pdf/2608.22671v1",
  "html_url": "https://arxiv.org/html/2608.22671v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.23629",
  "slug": "macro-operator-generation-and-predicate-selection-for-tamp-operator-le",
  "title": "Macro-Operator Generation and Predicate Selection for TAMP Operator Learning",
  "abstract": "Creating symbolic operators by hand is one of the main bottlenecks in deploying Task and Motion Planning systems (TAMP). Recent works show that these operators can instead be learned directly from demonstration data. Existing methods, however, typically learn each action in isolation and cannot capture the recurring multi-step structure of manipulation tasks, so the search becomes intractable on long sequential tasks. A further inefficiency arises in the symbolic state: every provided predicate is evaluated at every search node, even when it never appears in any learned operator. We present a system that addresses both problems together. Its central component is the automatic generation of macro-operators, composite actions that compress a recurring sequence of individual actions into a single planning step. Our system discovers causally linked action pairs directly from the training data, where one action produces exactly the condition that the next one requires, and turns each pair into a new operator. Alongside this, our system prunes every predicate that no learned operator references, which shrinks the symbolic state evaluated at each search node. Together, these changes shorten the effective planning horizon, and the benefit they bring grows with the length of the task. Across four TAMP domains, our method reaches up to a 4.6x planning speedup compared to the baseline method, namely Learning Operators for TAMP. More importantly, it solves a long sequential task that the baseline cannot solve. Macro-operator discovery thus not only accelerates planning but, in certain domains, determines solvability in practice.",
  "published": "2026-08-23",
  "updated": "2026-08-23",
  "year": "2026",
  "authors": [
   "Can Emir Bora",
   "Emre Ugur"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This system discovers causally linked action pairs directly from the training data, where one action produces exactly the condition that the next one requires, and turns each pair into a new operator, and not only accelerates planning but, in certain domains, determines solvability in practice.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Can Emir Bora",
    "id": "2459245953",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Emre Ugur",
    "id": "2326299482",
    "h_index": 1,
    "papers": 14
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.23629v1",
  "pdf_url": "https://arxiv.org/pdf/2608.23629v1",
  "html_url": "https://arxiv.org/html/2608.23629v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.22657",
  "slug": "physical-agentic-ai-an-architecture-for-orchestrating-a-robot-crew-wit",
  "title": "Physical Agentic AI: An Architecture for Orchestrating a Robot Crew with LLMs",
  "abstract": "Agentic AI frameworks interpret open-ended task goals and decompose them into multi-step plans. Richer information about embodiment-specific capabilities, physical preconditions, and cross-robot coordination improves grounding, but does not eliminate infeasible, mistimed, or unsafe physical actions. Physical robot crews therefore require an explicit architectural interface between semantic planning and execution, where every planned action is verified against robot capabilities, system state, and workflow constraints before actuation. This paper introduces Physical Agentic AI, a framework for skill-grounded robot agent orchestration, in which each robot exposes a typed library of executable skills while a foundation model planner decomposes a task into phases and assigns each phase to a robot-skill pair. A Robot Orchestration layer exposes the skill library, robot state, named locations, and workflow contracts to a non-actuating Mission Planner, while a deterministic Robot Orchestrator validates and authorizes one skill at a time. We evaluate on a drone-UGV search-and-dispatch mission, where every mission in every condition is executed live in Gazebo, and on a humanoid-quadruped transportation task using hardware-equivalent skill interfaces plus two physical trials on a Unitree G1 and Go2. Varying planner knowledge and runtime enforcement independently, we find that retrieval raises skill grounding from 51% to 96% yet leaves informed planners dispatching 23-29% of faulted steps. Per-dispatch enforcement reduces false dispatch to 0% with no false blocks, and a held-plan ablation confirms that the gate, not plan variation, is responsible. Live execution makes the difference physical: without enforcement all eight injected faults crossed the orchestration boundary and six produced robot motion; with enforcement all eight were refused before motion.",
  "published": "2026-08-23",
  "updated": "2026-08-23",
  "year": "2026",
  "authors": [
   "Xinyuan Liu",
   "Eren Sadikoglu",
   "Riana Chatterjee",
   "Ransalu Senanayake"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.MA"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper introduces Physical Agentic AI, a framework for skill-grounded robot agent orchestration, in which each robot exposes a typed library of executable skills while a foundation model planner decomposes a task into phases and assigns each phase to a robot-skill pair.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xinyuan Liu",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Eren Sadikoglu",
    "id": "2409932520",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Riana Chatterjee",
    "id": "2459202586",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Ransalu Senanayake",
    "id": "1388248711",
    "h_index": 21,
    "papers": 78
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "navigation",
   "foundation-pretraining"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2608.22657v1",
  "pdf_url": "https://arxiv.org/pdf/2608.22657v1",
  "html_url": "https://arxiv.org/html/2608.22657v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2608.22629",
  "slug": "enhancing-sim2real-transfer-for-torque-controlled-robots-through-real2",
  "title": "Enhancing Sim2Real Transfer for Torque-Controlled Robots through Real2Sim Dynamics Estimation and Reinforcement Learning",
  "abstract": "Transferring reinforcement learning policies from simulation to Real-World robots remains a major challenge, particularly when dealing with low-level torque control, where even small modelling inaccuracies can lead to unstable or unsafe behaviours. In this work, we propose a Real2Sim2Real pipeline that improves Sim2Real transfer for torque-controlled robotic arms by combining trajectory matching, parameter optimization via genetic algorithms, and domain randomization. Using the 7-DOF Franka Emika Panda robot, we first identify friction, inertia, and gravity compensation parameters by minimizing the error between real and simulated joint trajectories. These calibrated dynamics are then used to train a TQC-based reinforcement learning agent in simulation. The trained policy is evaluated in both Gazebo and MuJoCo environments, and finally deployed on the real robot. Our results demonstrate a significant improvement in tracking accuracy and policy robustness after parameter tuning, with smooth policy transfer from simulation to the Real-World across multiple target-reaching tasks. This work highlights the effectiveness of accurate physical modelling in enabling stable and generalizable torque-based reinforcement learning policies.",
  "published": "2026-08-23",
  "updated": "2026-08-23",
  "year": "2026",
  "authors": [
   "Davide Bargellini",
   "Alex Pasquali",
   "Andrea Govoni",
   "Riccardo Zanella",
   "Gianluca Palli"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes a Real2Sim2Real pipeline that improves Sim2Real transfer for torque-controlled robotic arms by combining trajectory matching, parameter optimization via genetic algorithms, and domain randomization, and demonstrates a significant improvement in tracking accuracy and policy robustness after parameter tuning.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Davide Bargellini",
    "id": "2340224486",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Alex Pasquali",
    "id": "2226524209",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Andrea Govoni",
    "id": "2316759316",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "R. Zanella",
    "id": "143693206",
    "h_index": 9,
    "papers": 15
   },
   {
    "name": "Gianluca Palli",
    "id": "2315505803",
    "h_index": 4,
    "papers": 36
   }
  ],
  "comment": "6 pages, 8 figures. Presented at the 2026 IEEE/ASME International Conference on Advanced Intelligent Mechatronics (AIM 2026)",
  "topics": [
   "sim2real",
   "rl-control",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.22629v1",
  "pdf_url": "https://arxiv.org/pdf/2608.22629v1",
  "html_url": "https://arxiv.org/html/2608.22629v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.22591",
  "slug": "worldtoken-time-first-sequence-modeling-for-robotic-imitation-learning",
  "title": "WorldToken: Time-First Sequence Modeling for Robotic Imitation Learning",
  "abstract": "Robot policies receive heterogeneous observations at each decision step, yet sequence models differ in how they organize these inputs over time. We introduce WorldToken, a time-first policy instantiation that fuses multiview images, proprioception, and task conditioning within each policy timestep into one world token. A causal temporal Transformer models the resulting world-token sequence, and a diffusion action head generates action chunks. On 23 RoboCasa tasks, an 85.3M-parameter policy trained from scratch apart from a frozen pretrained CLIP text encoder achieves 59.45% mean closed-loop success using 2,900 generated demonstrations per task. A complete factorial sweep over five dataset sizes, five model sizes, and two training seeds shows consistent gains from additional target-domain data and diminishing returns beyond moderate model size. Under same-checkpoint history truncation, reducing visible history to one or two policy timesteps lowers closed-loop success for all 50 RoboCasa policies. On RMBench Blocks Ranking, reducing visible history from 146 to 8 seconds lowers evaluator success from 95% to 28%, while an exploratory extended rollout sustains the reference swap sequence for over 850 seconds. These results establish the empirical feasibility of the complete WorldToken instantiation and characterize its data-scaling and temporal-context behavior under the tested recipes. They do not establish superiority over alternative sequence organizations or isolate which components of the complete implementation drive the observed performance.",
  "published": "2026-08-23",
  "updated": "2026-08-23",
  "year": "2026",
  "authors": [
   "Chunkai Yang",
   "Andong Yang",
   "Chao Gao"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "WorldToken, a time-first policy instantiation that fuses multiview images, proprioception, and task conditioning within each policy timestep into one world token is introduced and its data-scaling and temporal-context behavior under the tested recipes are characterized.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Chunkai Yang",
    "id": "2364680918",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Andong Yang",
    "id": "2198395380",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Chao Gao",
    "id": "2345800743",
    "h_index": 2,
    "papers": 10
   }
  ],
  "comment": "",
  "topics": [
   "imitation-diffusion"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.22591v1",
  "pdf_url": "https://arxiv.org/pdf/2608.22591v1",
  "html_url": "https://arxiv.org/html/2608.22591v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.22507",
  "slug": "what-is-the-effect-of-running-specific-prostheses-on-long-jumps-optimi",
  "title": "What is the effect of running-specific prostheses on long jumps? Optimization-based prediction and analysis using biomechanical models",
  "abstract": "Long jumpers with below the knee amputation (BKA) that take off from their running-specific prosthesis (RSP) improved performances significantly over the last years. The long jump biomechanics differs compared to athletes without BKA and the question arises whether the spring-like properties of the RSP facilitate achieving long jumping distances. The aim of this work is to propose a long jump model for athletes with and without BKA, to evaluate it and to apply it for comparing long jump motions with and without RSP. We establish rigid multi-body system models of one athlete with and one athlete without below the knee amputation (BKA). Long jump motions are computed by solving a specific optimal control problem (OCP) with constraints enforcing a physically correct dynamics, both for motion reconstruction or motion synthesis. With the proposed long jump model, we are able to compute realistic long jump motions. We discuss the causes of differences in measured long jumps and show directions for eliminating them. For both athletes, the synthesized solutions reveal potential for performance improvement. The jumping distance of the athlete without BKA is 64cm (6.9%) longer than the one of the athlete with BKA in the synthesized solutions.",
  "published": "2026-08-23",
  "updated": "2026-08-23",
  "year": "2026",
  "authors": [
   "Anna Lena Emonds",
   "Johannes Funken",
   "Wolfgang Potthast",
   "Katja Mombaur"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes a long jump model for athletes with and without BKA, to evaluate it and to apply it for comparing long jump motions with and without RSP, and is able to compute realistic long jump motions.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Anna Lena Emonds",
    "id": "81369527",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "J. Funken",
    "id": "6373172",
    "h_index": 14,
    "papers": 50
   },
   {
    "name": "W. Potthast",
    "id": "5364448",
    "h_index": 26,
    "papers": 193
   },
   {
    "name": "K. Mombaur",
    "id": "1755673",
    "h_index": 31,
    "papers": 173
   }
  ],
  "comment": "This work has been submitted to Scientific Reports and it is currently under review",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.22507v1",
  "pdf_url": "https://arxiv.org/pdf/2608.22507v1",
  "html_url": "https://arxiv.org/html/2608.22507v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.22496",
  "slug": "a-unified-neural-aided-alignment-and-calibration-method-for-auvs",
  "title": "A Unified Neural-Aided Alignment and Calibration Method for AUVs",
  "abstract": "Autonomous underwater vehicles (AUVs) rely on the fusion of inertial navigation systems (INS) and Doppler velocity logs (DVL) for accurate navigation. Before deployment, this fusion requires a DVL initialization pipeline consisting of two stages: alignment, which estimates the rotation between the INS and DVL frames, and calibration, which estimates the DVL error terms. Conventionally, both stages are solved with model-based algorithms that demand complex vehicle maneuvers, surface-level satellite reference measurements, and simplified error models, making initialization time-consuming, trajectory-dependent, and sensitive to sensor quality. In this work, we propose a fully neural- aided DVL initialization pipeline that replaces both stages with two complementary neural networks: ResAlignNet for alignment and DCNet for calibration. The unified pipeline operates in situ on a single nearly constant-velocity trajectory and uses the same inputs as the model-based baseline. Using real-world data recorded across five distinct sensor error-term combinations, the proposed pipeline reduces the velocity root mean squared error by an average of 68.7% over the model-based baseline, using only 25s of data for initialization.",
  "published": "2026-08-23",
  "updated": "2026-08-23",
  "year": "2026",
  "authors": [
   "Guy Damari",
   "Zeev Yampolsky",
   "Itzik Klein"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A fully neural- aided DVL initialization pipeline is proposed that replaces both stages with two complementary neural networks: ResAlignNet for alignment and DCNet for calibration, which reduces the velocity root mean squared error by an average of 68.7% over the model-based baseline.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Guy Damari",
    "id": "2352278099",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Zeev Yampolsky",
    "id": "2197780146",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Itzik Klein",
    "id": "2280333335",
    "h_index": 7,
    "papers": 20
   }
  ],
  "comment": "",
  "topics": [
   "navigation",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.22496v1",
  "pdf_url": "https://arxiv.org/pdf/2608.22496v1",
  "html_url": "https://arxiv.org/html/2608.22496v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.22449",
  "slug": "empire-explicit-manipulation-planning-as-a-learnable-intermediate-repr",
  "title": "EMPIRE: Explicit Manipulation Planning as a Learnable Intermediate Representation for Egocentric Hand-Motion Forecasting",
  "abstract": "Forecasting dexterous hand motions from egocentric observations is fundamental to intelligent interactive systems. Existing VLM-based methods typically map observations directly to future motions, overlooking the underlying manipulation process that governs hand-object interactions. Moreover, end-to-end optimization couples manipulation learning with motion synthesis, causing motion-generation gradients to interfere with the pre-learned manipulation-aware representations. To overcome these limitations, we propose EMPIRE, a two-stage framework that introduces Explicit Manipulation Planning as an Intermediate Representation for Egocentric hand-motion forecasting. Stage I: Learn to Plan. EMPIRE first learns explicit manipulation plans from multimodal context to capture the progression of hand-object interactions. Stage II: Learn to Act. A motion generator synthesizes future bimanual hand motions conditioned on frozen planner representations, preventing motion-generation gradients from affecting manipulation planning. To support our method, we further construct EMPIRE-651K, a bimanual hand-motion forecasting dataset comprising 650,910 training windows across 111 tasks, each paired with an explicit per-hand manipulation plan. Under identical training and evaluation protocols, EMPIRE achieves state-of-the-art forecasting accuracy, with an MPJPE of 84.53 mm and a finger-relative error of 38.97mm. We release the code and dataset at https://github.com/wangwen-banban/EMPIRE.",
  "published": "2026-08-23",
  "updated": "2026-08-23",
  "year": "2026",
  "authors": [
   "Wen Wang",
   "Ruibing Hou",
   "Hong Chang",
   "Shiguang Shan",
   "Xilin Chen"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "EMPIRE is a two-stage framework that introduces Explicit Manipulation Planning as an Intermediate Representation for Egocentric hand-motion forecasting and achieves state-of-the-art forecasting accuracy.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Wen Wang",
    "id": "2108907500",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Ruibing Hou",
    "id": "2288193205",
    "h_index": 6,
    "papers": 24
   },
   {
    "name": "Hong Chang",
    "id": "2331892185",
    "h_index": 5,
    "papers": 16
   },
   {
    "name": "Shiguang Shan",
    "id": "2250286882",
    "h_index": 13,
    "papers": 89
   },
   {
    "name": "Xilin Chen",
    "id": null,
    "h_index": 0,
    "papers": 0
   }
  ],
  "comment": "14 pages, 10 figures, 18 tables",
  "topics": [
   "dexterous-manipulation",
   "egocentric-data",
   "sim2real"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.22449v1",
  "pdf_url": "https://arxiv.org/pdf/2608.22449v1",
  "html_url": "https://arxiv.org/html/2608.22449v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.22419",
  "slug": "robust-bimanual-vision-language-action-models-via-embarrassingly-simpl",
  "title": "Robust Bimanual Vision-Language-Action Models via Embarrassingly Simple Modality Masking",
  "abstract": "Query-based Vision-Language-Action (VLA) models offer low-latency inference that is attractive for bimanual robotic manipulation, but we observe that they can still exhibit discontinuous actions and execution failures in complex dual-arm tasks. We hypothesize that unstable multi-view and language fusion is one contributing factor in these failures, often coinciding with attention spreading to distracting regions. To improve robustness, we introduce the Modality Masking Mechanism (M3), an embarrassingly simple, training-only strategy that requires no architectural changes or large-scale robot pretraining. M3 stochastically masks subsets of modality channels during training, exposing the policy to controlled partial observations and encouraging it to rely less on distracting cues and more on evidence that remains reliable. We evaluate M3 on ten bimanual tasks from RoboTwin 2.0 and on three long-horizon real-world tasks. Compared with the Adapter baseline, M3 improves average success by 21.7% in the Clean setting and 11.4% in Clean2Rand, where policies are trained on clean demonstrations and evaluated on randomized scenes, while also improving averaged real-world full-task success by over 30%. These results suggest that structured training-time masking is a practical way to improve the robustness of query-based VLA policies for bimanual manipulation.",
  "published": "2026-08-23",
  "updated": "2026-08-23",
  "year": "2026",
  "authors": [
   "Dongzhou Cheng",
   "Ziang Li",
   "Yixiao Zhou",
   "Haojuan Li",
   "Jinghao Zhang",
   "Lei Lei",
   "Minjing Dong",
   "Jie Gui",
   "Jiaqi Wang"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The Modality Masking Mechanism (M3), an embarrassingly simple, training-only strategy that requires no architectural changes or large-scale robot pretraining, is introduced to improve robustness of query-based VLA policies for bimanual manipulation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Dongzhou Cheng",
    "id": "2380625950",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Ziang Li",
    "id": "2288110026",
    "h_index": 5,
    "papers": 18
   },
   {
    "name": "Yixiao Zhou",
    "id": "2342456608",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Haojuan Li",
    "id": "2459231933",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Jinghao Zhang",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Lei Lei",
    "id": "2336234974",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Minjing Dong",
    "id": "2322966289",
    "h_index": 4,
    "papers": 16
   },
   {
    "name": "Jie Gui",
    "id": "2303655108",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Jiaqi Wang",
    "id": "2386931557",
    "h_index": 4,
    "papers": 13
   }
  ],
  "comment": "35 pages, 22 figures, 9 tables",
  "topics": [
   "vla",
   "foundation-pretraining",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.22419v1",
  "pdf_url": "https://arxiv.org/pdf/2608.22419v1",
  "html_url": "https://arxiv.org/html/2608.22419v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.22403",
  "slug": "ld4wam-learning-latent-dynamics-from-human-videos-for-world-action-mod",
  "title": "LD4WAM: Learning Latent Dynamics from Human Videos for World Action Models",
  "abstract": "Human video is playing an increasingly central role in training World Action Models (WAMs), owing to its diversity and low collection cost relative to teleoperated robot data. However, most WAMs learn from such video only by predicting pixel-level future frames, giving dynamics that are not directly actionable, whereas motion retargeting recovers directly actionable actions but leaves a large visual gap across embodiments. We therefore propose motion-aligned latent dynamics as an embodiment-agnostic representation to bridge video priors and low-level actions. We further present LD4WAM, which pairs a Latent Dynamics Model trained with semantic reconstruction and real motion alignment with a World Dynamics Action Model built as a mixture-of-transformers (MoT), which preserves full future-video generation and uses learnable queries to distill these latent dynamics from generated futures for action conditioning. Pretrained on our curated unified dataset of over 5{,}000 hours of human and robot data, LD4WAM performs strongly in RoboTwin simulation and on real robots equipped with both grippers and dexterous hands, while generalizing well to unseen objects and backgrounds.",
  "published": "2026-08-23",
  "updated": "2026-08-23",
  "year": "2026",
  "authors": [
   "Zhenhao Shen",
   "Jiaqi Liang",
   "Jasper Lu",
   "Feng Jiang",
   "Yuran Wang",
   "Chuanbo Wei",
   "Jiayi Liu",
   "Jianchun Yang",
   "Qize Yu",
   "Jiadi You",
   "Ce Hao",
   "Guanqi He",
   "Chen Xie",
   "Ruihai Wu"
  ],
  "author_count": 14,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "LD4WAM is presented, which pairs a Latent Dynamics Model trained with semantic reconstruction and real motion alignment with a World Dynamics Action Model built as a mixture-of-transformers (MoT), which preserves full future-video generation and uses learnable queries to distill these latent dynamics from generated futures for action conditioning.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhen Shen",
    "id": "2344981593",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "Jiaqi Liang",
    "id": "2362279070",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Jasper Lu",
    "id": "2430491413",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "F. Jiang",
    "id": "2348508993",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Yuran Wang",
    "id": "2349738743",
    "h_index": 5,
    "papers": 19
   },
   {
    "name": "Chuanbo Wei",
    "id": "2459242363",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Jiayi Liu",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Jianchun Yang",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Qize Yu",
    "id": "2405639311",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Jiadi You",
    "id": "2364354912",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Ce Hao",
    "id": "2306782351",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Guanqi He",
    "id": "2279862620",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "Chengen Xie",
    "id": "2275812370",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Ruihai Wu",
    "id": "2346811648",
    "h_index": 3,
    "papers": 7
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "dexterous-manipulation",
   "egocentric-data",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.22403v1",
  "pdf_url": "https://arxiv.org/pdf/2608.22403v1",
  "html_url": "https://arxiv.org/html/2608.22403v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.22398",
  "slug": "motiondlo-hybrid-event-and-frame-based-tracking-of-deformable-linear-o",
  "title": "MotionDLO: Hybrid Event- and Frame-Based Tracking of Deformable Linear Objects",
  "abstract": "Reliably tracking moving deformable linear objects (DLOs) while simultaneously ensuring robustness, accuracy, and temporally consistent state estimation remains a fundamental challenge in robot perception. We introduce MotionDLO, a real-time tracking framework specifically designed to overcome these limitations in temporal continuity and latency. The method exploits the high temporal resolution and sparsity of event-based cameras and combines segmentation with the Coherent Point Drift (CPD) algorithm under the principles of Motion Coherence Theory. This integration enables temporally consistent shape estimation while maintaining a low computational overhead. Existing event-based tracking methods are typically computationally efficient but exhibit reduced accuracy compared to frame-based approaches, or alternatively compromise event sparsity to achieve competitive performance. To resolve this trade-off, we propose a hybrid event- and frame-based tracking architecture that preserves the complementary strengths of both sensing modalities. The event stream ensures high-frequency motion updates, while frame-based information stabilizes spatial accuracy and object identity. We demonstrate that the proposed framework reliably associates DLO instances across video sequences, enabling robust perception for robotic manipulation tasks. Experimental results validate real-time performance at 12 ms update rates and accurate shape tracking with an point-to-curve error as measurement of accuracy of up to 0.43 mm, supporting dynamic path adaptation during manipulation. The source code and demonstration datasets are publicly available.",
  "published": "2026-08-23",
  "updated": "2026-08-23",
  "year": "2026",
  "authors": [
   "Annalena Hartmann",
   "Priyamvada Ajithkumar",
   "Patrick Br\u00fcndl",
   "J\u00f6rg Franke"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "MotionDLO, a real-time tracking framework specifically designed to overcome limitations in temporal continuity and latency, is introduced, which exploits the high temporal resolution and sparsity of event-based cameras and combines segmentation with the Coherent Point Drift algorithm under the principles of Motion Coherence Theory.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "A. Hartmann",
    "id": "2408651694",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Priyamvada Ajithkumar",
    "id": "2459201920",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "P. Br\u00fcndl",
    "id": "2224167442",
    "h_index": 7,
    "papers": 41
   },
   {
    "name": "J. Franke",
    "id": "2269174141",
    "h_index": 5,
    "papers": 31
   }
  ],
  "comment": "",
  "topics": [
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.22398v1",
  "pdf_url": "https://arxiv.org/pdf/2608.22398v1",
  "html_url": "https://arxiv.org/html/2608.22398v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.22364",
  "slug": "wam-opd-on-policy-distillation-for-world-action-models",
  "title": "WAM-OPD: On-Policy Distillation for World Action Models",
  "abstract": "World action models (WAMs) couple visual future prediction with robot action generation, but accelerated students can lose task capabilities during distillation and later encounter states that are poorly represented by offline data. We study whether on-policy distillation (OPD) can repair such a student without requiring sparse-reward reinforcement learning. We introduce WAM-OPD, a deployment-consistent post-training recipe for a video-first WAM. The student acts in the environment and therefore determines the history distribution. A frozen teacher labels those student histories with coherent video and action targets, while the student action branch is trained under its own generated video plan, as it is at deployment. Joint video and action losses update lightweight adapters in the shared backbone, together with an action flow-matching regularizer. In preliminary RoboTwin 2.0 studies on two tasks, the released one-video/one-action-step Flash-WAM improves from 0.0% to 58.3% success on HANDOVER MIC, and from 16.7% to 33.3% on PUT OBJECT CABINET. These task-specific results are an initial capability proof rather than evidence of broad or uniform generalization. They nevertheless suggest that dense teacher supervision on student-induced histories is a promising post-training interface for video-first WAMs.",
  "published": "2026-08-23",
  "updated": "2026-08-23",
  "year": "2026",
  "authors": [
   "Liuhaichen Yang",
   "Zhuang Jiang",
   "Chenchao Sheng",
   "Zezhi Tang"
  ],
  "author_count": 4,
  "categories": [
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.AI",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "WAM-OPD, a deployment-consistent post-training recipe for a video-first WAM, is introduced and it is suggested that dense teacher supervision on student-induced histories is a promising post-training interface for video-first WAMs.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Liu Yang",
    "id": "2267519122",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "Zhuang Jiang",
    "id": "2446864975",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Chenchao Sheng",
    "id": "2446444133",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Zezhi Tang",
    "id": "2279109461",
    "h_index": 4,
    "papers": 12
   }
  ],
  "comment": "",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.22364v1",
  "pdf_url": "https://arxiv.org/pdf/2608.22364v1",
  "html_url": "https://arxiv.org/html/2608.22364v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.22326",
  "slug": "gcs-bridging-restoring-connectivity-of-disconnected-convex-sets-for-gr",
  "title": "GCS-Bridging: Restoring Connectivity of Disconnected Convex Sets for Graph-of-Convex-Sets Motion Planning",
  "abstract": "Graph-of-Convex-Sets (GCS)-based trajectory optimization represents collision-free regions in configuration space as a finite collection of convex sets and directly performs collision-free trajectory planning over these sets, substantially simplifying the planning process. However, existing GCS-based trajectory planning methods generally assume sufficient connectivity among the convex regions and do not explicitly address cases in which the start and goal regions belong to different connected components of the initial GCS map. To address this limitation, we propose GCS-Bridging, which reconnects disconnected convex regions through collision-free point paths followed by convex region inflation, thereby recovering the feasibility of otherwise disconnected GCS planning problems. Extensive simulations across multiple IRIS-related algorithms and scenarios demonstrate that GCS-Bridging restores missing start-to-goal connectivity in the initial GCS map with a 99.8% success rate. In addition, a hardware experiment on a single-arm Franka platform in a real-world scenario with initially disconnected start and goal regions validates the effectiveness of the proposed method in practical motion planning. Project website: https://zhouxk1997.github.io/GCS_Bridging/",
  "published": "2026-08-23",
  "updated": "2026-08-23",
  "year": "2026",
  "authors": [
   "Xiaokai Zhou",
   "Baoshi Cao",
   "Yang Liu",
   "Kui Sun",
   "Boyu Ma",
   "Zhengpu Wang",
   "Zongwu Xie"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "GCS-Bridging is proposed, which reconnects disconnected convex regions through collision-free point paths followed by convex region inflation, thereby recovering the feasibility of otherwise disconnected GCS planning problems.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xiaokai Zhou",
    "id": "2314860282",
    "h_index": 3,
    "papers": 19
   },
   {
    "name": "Baoshi Cao",
    "id": "9242444",
    "h_index": 7,
    "papers": 40
   },
   {
    "name": "Yang Liu",
    "id": "2292155018",
    "h_index": 4,
    "papers": 25
   },
   {
    "name": "Kui Sun",
    "id": "143968432",
    "h_index": 12,
    "papers": 45
   },
   {
    "name": "Boyu Ma",
    "id": "2275779097",
    "h_index": 4,
    "papers": 31
   },
   {
    "name": "Zhengpu Wang",
    "id": "2276181151",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Zongwu Xie",
    "id": "2276055560",
    "h_index": 5,
    "papers": 38
   }
  ],
  "comment": "8 pages, 3 figures",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.22326v1",
  "pdf_url": "https://arxiv.org/pdf/2608.22326v1",
  "html_url": "https://arxiv.org/html/2608.22326v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.22301",
  "slug": "the-imitator-game-benchmarking-robot-imitative-ability-beyond-action-p",
  "title": "The Imitator Game: Benchmarking Robot Imitative Ability Beyond Action Prediction",
  "abstract": "Humans imitate at the level of intent: given a demonstration, we infer its goal and carry it out with whatever tools, objects, and layouts are at hand. Current robot policies instead learn observation-to-action mappings from visual inputs and language instructions, without explicitly inferring the demonstrated task. Learning from human video thus remains largely trajectory-level: models can replay motions in near-identical scenes, but still struggle to imitate what the demonstrator intends rather than merely what they do. We introduce The Imitator Game, a four-level benchmark (L0-L3) that progressively widens the gap between the human demonstration and the robot's own scene, isolating where trajectory replay ceases to suffice and task understanding becomes necessary. We pair it with IG-10K, the largest environment-aligned paired human-robot dataset to date and the only one instantiated across all four levels in both real and simulated settings (20,000+ paired episodes, 50+ tasks, 6 domains), and Imitator Arena, an open platform for blind A/B human evaluation. Across nine state-of-the-art models, performance is stable from L0 to L2 but collapses at L3, identifying functional substitution - achieving the same intent through a different object affordance - as the decisive barrier to intent-level imitation. Human-video-conditioned models outperform caption-conditioned ones, yet every model falls below 13% zero-shot success on unseen tasks; fine-tuning IG-10K-pretrained models with only $10$ paired human-robot demonstrations yields large gains that grow with pretraining scale. The project website and access to Imitator Arena are available at https://imitator-game.github.io.",
  "published": "2026-08-23",
  "updated": "2026-08-23",
  "year": "2026",
  "authors": [
   "Xunzhe Zhou",
   "Yiyang Cai",
   "Fengyi Wang",
   "Ran Ju",
   "Hanxiang Ren",
   "Ruizhe Liu",
   "Yu Zhang",
   "Qian Luo",
   "Feng Chen",
   "Pei Zhou",
   "Yi Ma",
   "Yanchao Yang"
  ],
  "author_count": 12,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The Imitator Game is introduced, a four-level benchmark (L0-L3) that progressively widens the gap between the human demonstration and the robot's own scene, isolating where trajectory replay ceases to suffice and task understanding becomes necessary.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xunzhe Zhou",
    "id": "2392270160",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Yiyang Cai",
    "id": "2459231108",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Fengyi Wang",
    "id": "2390936686",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Ran Ju",
    "id": "2258961033",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Hanxiang Ren",
    "id": "2152103889",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Ruizhe Liu",
    "id": "2296926167",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Yu Zhang",
    "id": "2329789543",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Q. Luo",
    "id": "2242962339",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Feng Chen",
    "id": "2377276163",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Pei Zhou",
    "id": "2313269953",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Yi Ma",
    "id": "2321486531",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Yanchao Yang",
    "id": null,
    "h_index": 0,
    "papers": 0
   }
  ],
  "comment": "",
  "topics": [
   "egocentric-data",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.22301v1",
  "pdf_url": "https://arxiv.org/pdf/2608.22301v1",
  "html_url": "https://arxiv.org/html/2608.22301v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.22296",
  "slug": "tonav-task-oriented-navigation-and-action-velocity-chunk-learning-for",
  "title": "TONAV: Task-Oriented Navigation and Action-Velocity Chunk Learning for Articulated Object Quadrupedal Mobile Manipulation",
  "abstract": "Quadruped mobile manipulation requires two tightly coupled capabilities: reaching manipulation-ready configurations and maintaining stable contact throughout articulated-object interaction. However, existing methods often terminate navigation near the target, leaving a gap between reachability and manipulation readiness, while tracking lag, motion jitter, and contact instability limit continuous interaction. To address these challenges, we present TONAV, a unified framework integrating task-oriented navigation with action-velocity chunk learning. First, we introduce a position-velocity-coupled teleoperation framework that explicitly captures motion dynamics to improve master-follower consistency and collect smooth, temporally consistent demonstrations. Next, task-oriented navigation leverages vision-language reasoning to decompose high-level instructions into executable subgoals and adaptively refine the robot base toward a manipulation-ready configuration. Finally, action-velocity chunk learning jointly models joint positions and their temporal transitions under velocity supervision, enabling smooth and stable sustained-contact manipulation. Real-world experiments across diverse articulated-object tasks demonstrate that TONAV achieves higher success rates in both task-oriented navigation and complete mobile manipulation, mitigating the navigation-manipulation gap and improving continuous-contact interaction. The project page is at https://haochen611.github.io/TONAV.",
  "published": "2026-08-23",
  "updated": "2026-08-23",
  "year": "2026",
  "authors": [
   "Haoran Lin",
   "Mingyu Yang",
   "Pengfei Qi",
   "Kehan Chen",
   "Qiang Diao",
   "Liangji Zeng",
   "Wenrui Chen",
   "Yaonan Wang",
   "Kailun Yang"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "TONAV is presented, a unified framework integrating task-oriented navigation with action-velocity chunk learning that achieves higher success rates in both task-oriented navigation and complete mobile manipulation, mitigating the navigation-manipulation gap and improving continuous-contact interaction.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Haoran Lin",
    "id": "2309308759",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Ming Yang",
    "id": "2445432044",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Pengfei Qi",
    "id": "2456462304",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Kehan Chen",
    "id": "2335489827",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Qiang Diao",
    "id": "2202018081",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Liangjing Zeng",
    "id": "2362079005",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Wenrui Chen",
    "id": "2292300370",
    "h_index": 4,
    "papers": 21
   },
   {
    "name": "Yaonan Wang",
    "id": "2309372154",
    "h_index": 3,
    "papers": 13
   },
   {
    "name": "Kailun Yang",
    "id": "8689702",
    "h_index": 40,
    "papers": 285
   }
  ],
  "comment": "The project page is at https://haochen611.github.io/TONAV",
  "topics": [
   "humanoids",
   "navigation",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.22296v1",
  "pdf_url": "https://arxiv.org/pdf/2608.22296v1",
  "html_url": "https://arxiv.org/html/2608.22296v1",
  "code_url": "https://haochen611.github.io/TONAV",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2608.22294",
  "slug": "beyond-instance-slots-semantically-rich-world-models-for-physical-inte",
  "title": "Beyond Instance Slots: Semantically Rich World Models for Physical Interaction Planning",
  "abstract": "World models for physical interaction are typically trained to predict future observations or latent features; however, a planning-oriented model must answer a fundamentally different question: whether a candidate action produces a task consistent future while preserving essential relations. Monolithic state representations obscure the underlying entities, while standard instance-level object slots merely identify what is present without specifying what role each entity plays in the task context. To bridge this gap, we present the Semantically Rich World Model (SR-WM), a task-conditioned world model structured around five functional roles: gripper, target, goal, relation, and phase. Within SR-WM, a visual entity encoder extracts soft entity hypotheses from pretrained patch features, allowing segmentation masks to serve as optional proposal priors without mandating them as required state representations or inference inputs. A role binder subsequently maps these hypotheses to task-specific roles, while an action conditioned dynamics model predicts role transitions alongside fine-grained semantics, including grasp/contact, predicate establishment, relation preservation, fixture state, and phase change. Crucially, this unified role state grounds downstream multi-candidate action generation, stage-aware reranking, and violation-aware suffix resampling. Our comprehensive evaluation protocol spans all four LIBERO simulation suites, cross-suite transfer, perception diagnostics, and action sensitivity analysis. Ultimately, this formulation transforms object-centric prediction into a semantic interface linking visual dynamics with planning-oriented decision making",
  "published": "2026-08-23",
  "updated": "2026-08-27",
  "year": "2026",
  "authors": [
   "Juntao Cheng",
   "Jingkai Wang",
   "Yijun Shen",
   "Xiansheng Chen",
   "Zhiwei Yu"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The Semantically Rich World Model is presented, a task-conditioned world model structured around five functional roles: gripper, target, goal, relation, and phase, which transforms object-centric prediction into a semantic interface linking visual dynamics with planning-oriented decision making.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Juntao Cheng",
    "id": "2346104123",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Jingkai Wang",
    "id": "2456978661",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yijun Shen",
    "id": "2459238933",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Xiansheng Chen",
    "id": "2384353683",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Zhiwei Yu",
    "id": "2336248434",
    "h_index": 3,
    "papers": 13
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "dexterous-manipulation",
   "sim2real"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.22294v2",
  "pdf_url": "https://arxiv.org/pdf/2608.22294v2",
  "html_url": "https://arxiv.org/html/2608.22294v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.22278",
  "slug": "dreammimic-learning-visuomotor-whole-body-loco-manipulation-via-world",
  "title": "DreamMimic: Learning Visuomotor Whole-Body Loco-Manipulation via World Model",
  "abstract": "Vision-based whole-body loco-manipulation on humanoid robots is challenging due to partial observability, contact-rich dynamics, and the difficulty of learning long-horizon behaviors from high-dimensional visual inputs. We present \\href{https://github.com/DreamMimic/DreamMimic}{DreamMimic}, a framework that distills privileged teacher policies into vision-based humanoid controllers via world-model-assisted distillation. Instead of using a Dreamer-style RSSM for planning, we repurpose it to learn predictive latent dynamics that serve as both a representation space and an action-conditioned multi-step supervision signal, while exposing compact predictive features to the student policy to reduce long-term drift. Beyond standard reconstruction objectives for proprioceptive and visual observations, we add auxiliary prediction heads for privileged state, contact, object state, and reward estimation. These heads provide additional supervision related to agent--object interaction and task progress, encouraging the latent representation to retain signals that are useful for contact-rich loco-manipulation. We further introduce Performance-Conditioned Guidance (PCG), a reward-driven adaptive distillation schedule that computes performance scores for both teacher and student to dynamically balance guidance and exploration. PCG prevents both premature teacher annealing and excessive teacher interference in challenging visual settings. Experiments on OMOMO and BEHAVE show improved tracking-based loco-manipulation performance over strong vision-based baselines, without exposing online privileged interaction states to the student at deployment. Qualitative simulations further examine morphology and simulator changes. These results suggest that world models can provide a useful mechanism for stabilizing visual policy distillation in contact-rich humanoid behaviors.",
  "published": "2026-08-23",
  "updated": "2026-08-23",
  "year": "2026",
  "authors": [
   "Jie Yin",
   "Xingyu Lai"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A framework that distills privileged teacher policies into vision-based humanoid controllers via world-model-assisted distillation, and introduces Performance-Conditioned Guidance (PCG), a reward-driven adaptive distillation schedule that computes performance scores for both teacher and student to dynamically balance guidance and exploration.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jie Yin",
    "id": "2453488297",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Xingyu Lai",
    "id": "2459200607",
    "h_index": 0,
    "papers": 0
   }
  ],
  "comment": "accepted to IROS2026",
  "topics": [
   "world-models",
   "humanoids",
   "tactile",
   "sim2real",
   "navigation",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.22278v1",
  "pdf_url": "https://arxiv.org/pdf/2608.22278v1",
  "html_url": "https://arxiv.org/html/2608.22278v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.22187",
  "slug": "behaviorworldgen-closing-the-loop-between-action-models-and-world-simu",
  "title": "BehaviorWorldGen: Closing the Loop between Action Models and World Simulators via Controllable Behavior-Aware Structured World Generation",
  "abstract": "Modern driving action models are increasingly improved in a self-improvement loop, where a learned world simulator imagines future observations and the resulting data is fed back to refine the action model. However, the bottleneck of this loop lies in the simulators' inability to generate behaviorally plausible responses by surrounding agents, making generated data both unrealistic in interaction and imbalanced in distribution. We introduce BehaviorWorldGen, a framework that closes the loop between action models and world simulators through controllable behavior-aware structured world generation. Its core component is BehaviorFlow, a meta-action-conditioned traffic-flow model that injects interpretable behavior controls and jointly generates multi-agent rollouts. BehaviorFlow realizes the specified agent behaviors while allowing surrounding vehicles to respond to the ego and to one another. The resulting rollouts are rendered by a world simulator into realistic multi-view observations, which are paired with corrected interaction-aware trajectories for action-model refinement. Since BehaviorWorldGen uses structured trajectories as the interface between its modules, it is compatible with diverse action models and world simulators. Experiments on world generation, scene extrapolation, and policy refinement demonstrate consistent improvements, with the largest benefits concentrated on difficult interactive scenarios.",
  "published": "2026-08-23",
  "updated": "2026-08-27",
  "year": "2026",
  "authors": [
   "Jiaqi Wang",
   "Zhuo Zhang",
   "Haining Guan",
   "Tingguang Zhou",
   "Haowen Cui",
   "ChuanYe Wang",
   "Zhongyang Zhu",
   "Yulong Zheng",
   "Xuefeng Chen",
   "Zhen Yang",
   "Tianchen Deng",
   "Feiyang Tan",
   "Xiwu Chen",
   "Hangning Zhou",
   "Bo Dai",
   "Lixia Shen",
   "Xiyang Wang",
   "Jiajun Zhu"
  ],
  "author_count": 18,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "BehaviorWorldGen is introduced, a framework that closes the loop between action models and world simulators through controllable behavior-aware structured world generation through controllable behavior-aware structured world generation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jiaqi Wang",
    "id": "2456352145",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Zhuo Zhang",
    "id": "2458968187",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Haining Guan",
    "id": "2459191974",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Ting Zhou",
    "id": "2325732557",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Haowen Cui",
    "id": "2459200892",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Zhongyang Zhu",
    "id": "2458272215",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yulong Zheng",
    "id": "2459240125",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Chuanye Wang",
    "id": "2373730585",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Xuefeng Chen",
    "id": "2399090918",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Zheng Yang",
    "id": "2458550506",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Tianchen Deng",
    "id": "2398908390",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Feiyang Tan",
    "id": "2395671410",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Hangning Zhou",
    "id": "31446037",
    "h_index": 9,
    "papers": 23
   },
   {
    "name": "Bo Dai",
    "id": "2332357874",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Li-ming Shen",
    "id": "2449429453",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Xiwu Chen",
    "id": "2295949049",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Xiyang Wang",
    "id": "2309113627",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Jiajun Zhu",
    "id": "2459230701",
    "h_index": 0,
    "papers": 0
   }
  ],
  "comment": "",
  "topics": [
   "sim2real"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.22187v2",
  "pdf_url": "https://arxiv.org/pdf/2608.22187v2",
  "html_url": "https://arxiv.org/html/2608.22187v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.22149",
  "slug": "meta-ctrl-guaranteed-plan-generation-by-decoupling-syntactic-and-seman",
  "title": "Meta-Ctrl: Guaranteed Plan Generation by Decoupling Syntactic and Semantic Constraints",
  "abstract": "LLMs generate fluent plans for robots but routinely violate the syntactic and se8mantic constraints they must satisfy to execute, and existing remedies trade formal guarantees against plan quality: soft methods (affordance scoring, grounded decoding) give no guarantee, while symbolic planners (LLM+P) discard the LM's commonsense. We propose \\textbf{Meta-Ctrl}, a constrained-decoding framework that guarantees the encoded constraints while preserving the base LM's plan quality. Meta-Ctrl introduces \\emph{meta-tokens}---a compact vocabulary of grounded actions---enforcing syntax at the token level and semantics (preconditions, goals, ordering) at the action level, an exact factorization that cuts the memory of constrained decoding from over 107TB to under 2GB. With it, a small open-weight LM becomes competitive where it otherwise sits at the bottom of the leaderboard: on WAH-NL under the LoTa-Bench protocol it reaches the highest reported subgoal success rate, exceeding GPT-4's, with consistent gains across the Embodied Agent Interface. We further demonstrate it on a real tabletop robot, where every generated plan satisfies its preconditions and goals by construction.",
  "published": "2026-08-23",
  "updated": "2026-08-25",
  "year": "2026",
  "authors": [
   "Gwen Yidou-Weng",
   "Edward Sun",
   "Tianyi Ma",
   "Metin Alp Dogan",
   "Benjie Wang",
   "Allen Peng",
   "Guy Van den Broeck",
   "Yuchen Cui"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Meta-Ctrl is proposed, a constrained-decoding framework that guarantees the encoded constraints while preserving the base LM's plan quality, and is demonstrated on a real tabletop robot, where every generated plan satisfies its preconditions and goals by construction.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Gwen Yidou-Weng",
    "id": "2393209304",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Edward Sun",
    "id": "2310336411",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Tianyi Ma",
    "id": "2375765420",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Metin Alp Dogan",
    "id": "2459191012",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Benjie Wang",
    "id": "2296747138",
    "h_index": 6,
    "papers": 16
   },
   {
    "name": "Allen Peng",
    "id": "2459190947",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Guy Van den Broeck",
    "id": "1749506",
    "h_index": 43,
    "papers": 240
   },
   {
    "name": "Yuchen Cui",
    "id": "2238151901",
    "h_index": 9,
    "papers": 24
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.22149v2",
  "pdf_url": "https://arxiv.org/pdf/2608.22149v2",
  "html_url": "https://arxiv.org/html/2608.22149v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.22100",
  "slug": "contact-rich-robotic-manipulation-in-construction-via-zero-shot-learni",
  "title": "Contact-Rich Robotic Manipulation in Construction via Zero-Shot Learning: A Diffusion Policy-Guided Adaptive Control",
  "abstract": "Construction robotics and automation offer promising means of improving productivity, alleviating workforce shortages, and reducing workers' exposure to physically demanding tasks. However, reliable contact-rich robotic assembly remains challenging under tight tolerances, fabrication inaccuracies, and uncertain contact dynamics. To address this challenge, we present a framework coupling diffusion policies trained on simulation-generated pose and force/torque data with an L1-inspired adaptive controller that corrects policy-predicted actions online to compensate for unmodeled contact dynamics. We benchmark the framework against baselines in timber joinery, pipe fitting, and sequential full-scale truss assembly. It achieves 100% success on single-task assemblies and 90-100% success across sequential truss assembly subtasks, with lower, more stable contact forces than the baselines. By enabling zero-shot sim-to-real transfer for force-aware contact-rich assembly, the framework reduces costly, labor-intensive real-world data collection for policy training and advances scalable, robust automation of multistage assembly, motivating extension to broader contact-rich manipulation tasks in construction.",
  "published": "2026-08-22",
  "updated": "2026-08-22",
  "year": "2026",
  "authors": [
   "Roman Ibrahimov",
   "Salma Mozaffari",
   "Arash Adel"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "By enabling zero-shot sim-to-real transfer for force-aware contact-rich assembly, the framework reduces costly, labor-intensive real-world data collection for policy training and advances scalable, robust automation of multistage assembly, motivating extension to broader contact-rich manipulation tasks in construction.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Roman Ibrahimov",
    "id": "2323374300",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "S. Mozaffari",
    "id": "2003014006",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Arash Adel",
    "id": "2261570183",
    "h_index": 3,
    "papers": 8
   }
  ],
  "comment": "",
  "topics": [
   "tactile",
   "sim2real",
   "imitation-diffusion",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.22100v1",
  "pdf_url": "https://arxiv.org/pdf/2608.22100v1",
  "html_url": "https://arxiv.org/html/2608.22100v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.22093",
  "slug": "endonav-semantic-to-geometric-grounding-for-language-guided-robotic-en",
  "title": "EndoNav: Semantic-to-Geometric Grounding for Language-Guided Robotic Endoscopic Examination",
  "abstract": "Minimally invasive procedures performed within confined anatomical spaces depend on continuous endoscopic visualization. Current robotic endoscope systems can stabilize or reposition an endoscope, but they do not possess relevant context to provide effective visualization assistance. We present EndoNav, an anatomy-grounded natural-language framework that translates high-level surgeon commands into autonomous endoscopic visualization behaviors within patient-specific sinonasal anatomy. Spoken surgeon commands are transcribed and interpreted by an endoscopic viewpoint agent conditioned on a patient-specific anatomical scene representation. Rather than generating robot motion directly, the viewpoint agent generates structured visualization objectives that are converted into target viewpoints and inspection trajectories, which are then executed through geometry-constrained endoscope motion planning and joint-space control. We evaluate EndoNav using a structured three-pass sinus examination across three CT-derived anatomical models. For one cadaveric specimen, autonomous visualization is compared with sinus examinations performed by two resident surgeons. EndoNav achieved mean visualization IoUs of 87.04% and 84.37% relative to the two surgeon examinations, compared with an inter-surgeon IoU of 87.44%, while recovering 92.91% and 93.20% of surgeon-observed anatomical surfaces, respectively. These results demonstrate the feasibility of grounding high-level anatomical commands into patient-specific geometric objectives and translating them into anatomically constrained robotic visualization behaviors.",
  "published": "2026-08-22",
  "updated": "2026-08-22",
  "year": "2026",
  "authors": [
   "Jecia Z. Y. Mao",
   "Hisashi Ishida",
   "Kathryn Jung",
   "Masaru Ishii",
   "Russell H. Taylor",
   "Manish Sahu"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "EndoNav is presented, an anatomy-grounded natural-language framework that translates high-level surgeon commands into autonomous endoscopic visualization behaviors within patient-specific sinonasal anatomy and demonstrates the feasibility of grounding high-level anatomical commands into patient-specific geometric objectives and translating them into anatomically constrained robotic visualization behaviors.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "J. Z. Mao",
    "id": "2381336747",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Hisashi Ishida",
    "id": "2273978167",
    "h_index": 3,
    "papers": 16
   },
   {
    "name": "Kathryn Jung",
    "id": "2459159855",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Masaru Ishii",
    "id": "2261277931",
    "h_index": 4,
    "papers": 25
   },
   {
    "name": "Russell H. Taylor",
    "id": "2274126786",
    "h_index": 4,
    "papers": 14
   },
   {
    "name": "M. Sahu",
    "id": "2142002035",
    "h_index": 6,
    "papers": 30
   }
  ],
  "comment": "",
  "topics": [
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.22093v1",
  "pdf_url": "https://arxiv.org/pdf/2608.22093v1",
  "html_url": "https://arxiv.org/html/2608.22093v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.22067",
  "slug": "dele-w0-5-inferring-action-from-future-latent-state-for-robotic-manipu",
  "title": "DELE-w0.5: Inferring Action from Future Latent State for Robotic Manipulation",
  "abstract": "World-Action Models (WAMs) build robot control on video-generation backbones, which jointly predict dense future visual trajectories and robot actions. We argue that video generation is an unnecessary intermediate objective for world-action modeling. For robotic manipulation, the goal of a world model is not to reproduce how the world looks at every intermediate moment, but to predict the state that the world will reach after an action is executed. The intermediate frames only describe the visual transition between physical states, which consumes substantial model capacity and computation, but do not directly specify the physical outcome that the robot action is intended to produce. In this paper, we propose DELE-w0.5, which infers robot actions from predicted future states without relying on video generation. Concretely, DELE-w0.5 infers the action sequence from its corresponding compact future latent state. The future latent state captures the action-relevant physical outcome of robot interaction and serves as an explicit bridge between world modeling and action generation. The core design principle of DELE-w0.5 is to model how the physical world changes under robot actions, rather than how its visual appearance evolves frame by frame. This formulation removes the high-dimensional visual redundancy introduced by dense video representations, and it therefore enables cheaper training and low-latency inference. Across 480 real-robot trials on four long-horizon manipulation tasks, our DELE-w0.5 achieves the best performance among all compared policies, attaining 62.5 overall full-task success and 81.3 macro ordered-stage progress, outperforming the strongest baseline by 47.5 and 30.7 percentage points, respectively.",
  "published": "2026-08-22",
  "updated": "2026-08-26",
  "year": "2026",
  "authors": [
   "Fenghao Lei",
   "Zhixiong Huang",
   "Long Yang",
   "Jiabao Chen",
   "Peilin Huang",
   "Han Fu",
   "Zhuo Li",
   "Xiaoxue Ren"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The core design principle of DELE-w0.5 is to model how the physical world changes under robot actions, rather than how its visual appearance evolves frame by frame, which enables cheaper training and low-latency inference.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Fenghao Lei",
    "id": "2218119181",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Zhixiong Huang",
    "id": "2218293583",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Long Yang",
    "id": "2269710210",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Jiabao Chen",
    "id": "2459228742",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Peilin Huang",
    "id": "2459189835",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Han Fu",
    "id": "1510708452",
    "h_index": 6,
    "papers": 18
   },
   {
    "name": "Zhuo Li",
    "id": "2459467263",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Xiaoxue Ren",
    "id": "2265217070",
    "h_index": 3,
    "papers": 9
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.22067v3",
  "pdf_url": "https://arxiv.org/pdf/2608.22067v3",
  "html_url": "https://arxiv.org/html/2608.22067v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.22035",
  "slug": "ludi-scriptscriptstyle-0-1-an-agentic-system-for-socially-intelligent",
  "title": "Ludi${}_{\\scriptscriptstyle 0.1}$: An Agentic System for Socially Intelligent Robots",
  "abstract": "Robot foundation models have substantially advanced perception and control, but natural human-robot collaboration requires more than executing isolated commands. A robot must recognize ambiguity, maintain context across turns, communicate its intentions, and revise ongoing behavior as the user's intent changes. We present $\\scriptstyle\\mathsf{Ludi}_{\\scriptscriptstyle 0.1}$, an agentic system for socially intelligent robots that integrates interactive speech, multimodal reasoning, memory, navigation, and learned manipulation. Its decision-making core is a fine-tuned vision-language model trained on multi-turn interaction traces spanning ambiguous requests, clarifications, corrections, interruptions, mixed social and task dialogue, and multi-step tasks. A purpose-built harness manages the model-tool interaction loop, while specialized navigation and manipulation policies execute physical skills. Ludi${}_{\\scriptscriptstyle 0.1}$ demonstrates a practical path toward fluid human-robot collaboration today while producing the multimodal interaction traces needed to develop a more deeply integrated foundation model for robots and people.",
  "published": "2026-08-22",
  "updated": "2026-08-22",
  "year": "2026",
  "authors": [
   "Wooseong Chung",
   "William Cong",
   "Jakub Dworakowski",
   "Ethan Ewer",
   "Tri Wahyu Guntara",
   "Yeonwoo Jeong",
   "Tianchong Jiang",
   "Chaewon Kim",
   "Hyunseo Kim",
   "Jinwoo Kim",
   "Jinyeon Kim",
   "Yea-Seul Kim",
   "Jack Kunde",
   "Kangwook Lee",
   "Sangheon Lee",
   "Robert Nowak",
   "Junha Roh"
  ],
  "author_count": 17,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Ludi is presented, an agentic system for socially intelligent robots that integrates interactive speech, multimodal reasoning, memory, navigation, and learned manipulation, and demonstrates a practical path toward fluid human-robot collaboration today while producing the multimodal interaction traces needed to develop a more deeply integrated foundation model for robots and people.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Wooseong Chung",
    "id": "2158149529",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "William Cong",
    "id": "2334866118",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Jakub Dworakowski",
    "id": "1419489618",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Ethan Ewer",
    "id": "2323781863",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Tri Wahyu Guntara",
    "id": "1725410972",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Yeonwoo Jeong",
    "id": "144662720",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Tianchong Jiang",
    "id": "2220962414",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Chaewon Kim",
    "id": "2453815462",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Hyun-Seop Kim",
    "id": "2455951638",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Jinwoo Kim",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Jinyeon Kim",
    "id": "2145420477",
    "h_index": 9,
    "papers": 20
   },
   {
    "name": "Yea-Seul Kim",
    "id": "2459236060",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Jackson Kunde",
    "id": "2334358539",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Kangwook Lee",
    "id": "2323790154",
    "h_index": 4,
    "papers": 15
   },
   {
    "name": "Sangheon Lee",
    "id": "2248449859",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Robert Nowak",
    "id": "2263612732",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Junha Roh",
    "id": "2011080",
    "h_index": 4,
    "papers": 8
   }
  ],
  "comment": "",
  "topics": [
   "navigation",
   "foundation-pretraining",
   "hri",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.22035v1",
  "pdf_url": "https://arxiv.org/pdf/2608.22035v1",
  "html_url": "https://arxiv.org/html/2608.22035v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.22033",
  "slug": "delta-deformable-elevation-based-local-terrain-attention-encoder-for-s",
  "title": "DELTA: Deformable Elevation-Based Local Terrain Attention Encoder for Sparse-Terrain Quadrupedal Locomotion",
  "abstract": "Stable quadrupedal locomotion on sparse terrain requires selecting state-relevant terrain evidence for precise foot placement. Model-based foothold planners provide precise foothold selection but rely heavily on explicit model assumptions. Recent attention-based map encoding (AME) studies show that end-to-end reinforcement learning (RL) can learn implicit foothold guidance. However, the computational cost of dense AME encoding grows with map resolution, limiting its scalability to fine-grained sparse terrain. We propose DELTA, a Deformable Elevation-Based Local Terrain Attention encoder. DELTA predicts state-conditioned sampling locations, forms terrain evidence tokens from adaptive local elevation patches, and attends only to a fixed-size token set. With fixed sampling and patch settings, DELTA's encoder cost is independent of map resolution. Experiments show that DELTA achieves final traversal performance comparable to AME at the standard resolution while improving learning efficiency. This fixed encoder cost enables the use of higher-resolution terrain maps, improving traversal on fine-grained sparse terrain. DELTA also demonstrates strong generalization to unseen mixed evaluation courses composed of continuous and discrete terrain elements. Beyond simulation, DELTA demonstrates successful sim-to-real transfer on RAIBO2. Analysis of the learned sampling offsets and attention weights shows that DELTA samples steppable regions and attends to terrain evidence relevant to future touchdowns without foothold labels or attention supervision.",
  "published": "2026-08-22",
  "updated": "2026-08-22",
  "year": "2026",
  "authors": [
   "Sanghyun Park",
   "Moonkyu Jung",
   "Jemin Hwangbo"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA 2027",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "DELTA, a Deformable Elevation-Based Local Terrain Attention encoder is proposed, a Deformable Elevation-Based Local Terrain Attention encoder that achieves final traversal performance comparable to AME at the standard resolution while improving learning efficiency.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Sanghyun Park",
    "id": "2457742497",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Moonkyu Jung",
    "id": "2233287834",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Jemin Hwangbo",
    "id": "1707297",
    "h_index": 25,
    "papers": 49
   }
  ],
  "comment": "8 pages, 5 figures. Submitted to the 2027 IEEE International Conference on Robotics and Automation (ICRA 2027). This work has been submitted to the IEEE for possible publication. Copyright may be transferred without notice, after which this version may no longer be accessible",
  "topics": [
   "humanoids",
   "sim2real",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.22033v1",
  "pdf_url": "https://arxiv.org/pdf/2608.22033v1",
  "html_url": "https://arxiv.org/html/2608.22033v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.22028",
  "slug": "design-of-a-human-assistance-robot-system-with-contextual-action-recog",
  "title": "Design of a Human-Assistance Robot System with Contextual Action Recognition",
  "abstract": "This paper presents a conceptual design for a proactive human assisting robot system capable of recognizing human activities and responding proactively. The system leverages contextual human activity recognition to interpret human actions across diverse contexts, while behavior trees are utilized to define dynamic and interpretable robot behaviors. We outline the system architecture, incorporating contextual human action recognition (HAR), behavior trees (BTs), and ROS, using the Spot robot platform as a representative example. We explain how HAR enables the robot to provide proactive assistance, discuss its limitations, and introduce methodologies for contextual HAR to address these limitations, thereby enhancing the robot's decision-making in complex human activity scenarios.",
  "published": "2026-08-22",
  "updated": "2026-08-22",
  "year": "2026",
  "authors": [
   "Amanuel Ergogo",
   "Teresa Zieli\u0144ska"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper explains how HAR enables the robot to provide proactive assistance, discusses its limitations, and introduces methodologies for contextual HAR to address these limitations, thereby enhancing the robot's decision-making in complex human activity scenarios.",
  "doi": "10.1007/978-3-032-08359-3_16",
  "oa_pdf": "https://doi.org/10.48550/arxiv.2608.22028",
  "s2_authors": [
   {
    "name": "Amanuel Ergogo",
    "id": "2303679710",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Teresa Zieli\u0144ska",
    "id": "2243320825",
    "h_index": 3,
    "papers": 15
   }
  ],
  "comment": "Published in Automation 2025: Recent Advances in Automation, Robotics and Measurement Techniques, Lecture Notes in Networks and Systems, vol. 1687, Springer Nature, 2025. DOI: 10.1007/978-3-032-08359-3_16",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.22028v1",
  "pdf_url": "https://arxiv.org/pdf/2608.22028v1",
  "html_url": "https://arxiv.org/html/2608.22028v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.22008",
  "slug": "stakeholder-insights-for-designing-in-home-social-robots-for-dementia",
  "title": "Stakeholder Insights for Designing In-Home Social Robots for Dementia Disorientation Detection and Caregiver-Aware Intervention",
  "abstract": "Dementia disorientation detection and intervention remain under-examined as socio-technical challenges for socially assistive robots (SARs). We conducted 14 semi-structured interviews with dementia caregivers and practitioners (DCPs) to investigate how disorientation is experienced, recognised, and managed in everyday life. The findings reveal that disorientation is recurrent and fluctuating. It often emerges through behavioural cues such as repeated questioning, inappropriate activity timing and disrupted daily routines. Caregivers described orientation as emotionally charged, and direct correction may increase distress. The DCPs were generally receptive to robotic assistance when framed as supportive rather than corrective. Based on the insights, we identify essential design implications for SARs that provide context-aware orientation support, integrate into daily routines and support caregivers through timely escalation.",
  "published": "2026-08-22",
  "updated": "2026-08-25",
  "year": "2026",
  "authors": [
   "Emmanuel Akinrintoyo",
   "Nicole Salomons"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Investigating how disorientation is experienced, recognised, and managed in everyday life reveals that disorientation is recurrent and fluctuating, and essential design implications for SARs that provide context-aware orientation support, integrate into daily routines and support caregivers through timely escalation are identified.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Emmanuel Akinrintoyo",
    "id": "2336733186",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Nicole Salomons",
    "id": "2336733181",
    "h_index": 2,
    "papers": 7
   }
  ],
  "comment": "6",
  "topics": [
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.22008v2",
  "pdf_url": "https://arxiv.org/pdf/2608.22008v2",
  "html_url": "https://arxiv.org/html/2608.22008v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.21928",
  "slug": "guardianbench-a-same-scene-instruction-contrastive-benchmark-for-laten",
  "title": "GuardianBench: A Same-Scene Instruction-Contrastive Benchmark for Latent Contextual Risk in Embodied AI",
  "abstract": "In embodied AI, safety risk can be latent: a benign instruction and a safe scene become hazardous only when composed. Prior work has advanced embodied safety by varying visual contexts or evaluating execution-time dynamics, but the complementary axis of fixing the scene and varying only the instruction remains underexplored. We introduce GuardianBench, an instruction-contrastive benchmark grounded in international safety standards that isolates this latent contextual risk through 3,024 instruction-scene examples organized as same-scene Safe/Unsafe contrastive pairs across various hazard categories. Benchmarking state-of-the-art vision-language models (VLMs) reveals instruction-insensitive verdicts: models disproportionately approve both instructions under a given scene; across the primary models, average pair accuracy is only 24.1%. Our systematic rationale audit localizes the dominant failure: models fail to bind the instruction-relevant cues that differentiate safe from unsafe compositions. As a post-training case study, Verdict Log-Odds Supervision (VLOS), a lightweight verdict-level objective, substantially improves performance on open-weight backbones. Together, our latent contextual risk task formulation, standards-grounded contrastive benchmark construction, pair-level and rationale-level failure diagnosis, and benchmark-enabled verdict calibration establish GuardianBench as a controlled evaluation suite for exposing and improving safety reasoning over instruction-scene compositions under latent contextual risk.",
  "published": "2026-08-22",
  "updated": "2026-08-22",
  "year": "2026",
  "authors": [
   "Zhesheng Zhang",
   "Jiahao Lu",
   "Wei Liu",
   "Cong Pan",
   "Jianhua Yang",
   "Yixiang Chen",
   "Hongyuan Yu",
   "Mengqi Zhang",
   "Kailin Lyu",
   "Zhumin Chen",
   "Keji He"
  ],
  "author_count": 11,
  "categories": [
   "cs.AI",
   "cs.CL",
   "cs.RO"
  ],
  "primary_category": "cs.AI",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhesheng Zhang",
    "id": "2459447525",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Jiahao Lu",
    "id": "2453824346",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Wei Liu",
    "id": "2446886703",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Cong Pan",
    "id": "2458705898",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Jianhua Yang",
    "id": "2124825842",
    "h_index": 15,
    "papers": 32
   },
   {
    "name": "Yixiang Chen",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Hongyuan Yu",
    "id": "48002920",
    "h_index": 12,
    "papers": 26
   },
   {
    "name": "Mengqi Zhang",
    "id": "48985110",
    "h_index": 14,
    "papers": 36
   },
   {
    "name": "Kailin Lyu",
    "id": "2391708903",
    "h_index": 2,
    "papers": 20
   },
   {
    "name": "Zhumin Chen",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Keji He",
    "id": "51054943",
    "h_index": 7,
    "papers": 17
   }
  ],
  "comment": "21 pages, 4 figures",
  "topics": [
   "video-generation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.21928v1",
  "pdf_url": "https://arxiv.org/pdf/2608.21928v1",
  "html_url": "https://arxiv.org/html/2608.21928v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.21899",
  "slug": "cider-continual-interactive-distillation-for-embodied-reinforcement-le",
  "title": "CIDER: Continual Interactive Distillation for Embodied Reinforcement Learning",
  "abstract": "Human-in-the-loop real-world reinforcement learning enables rapid acquisition of effective robotic manipulation policies for individual tasks, often within tens of minutes. Yet it remains unclear how to extend this paradigm to continual learning, where a single policy must acquire new skills without losing previously learned behaviors. Existing real-world continual learning methods do not explicitly constrain prior behaviors, leading to severe catastrophic forgetting. We introduce Continual Interactive Distillation for Embodied Reinforcement Learning (CIDER), a continual reinforcement learning framework that freezes the accumulated historical policy as a teacher before learning each new task and interleaves task learning with distillation-based retention. We further introduce gradient routing to separate the gradients used for acquiring new tasks from those used for preserving prior behaviors. We evaluate our method with a single shared actor on six real-world household and industrial manipulation tasks. Interactive Distillation maintains high measured success on previously learned tasks across our six-task real-robot sequence while acquiring each new task in 10 to 20 minutes, whereas every baseline forgets at least one previous task. Additional ablations reveal the key design choices that govern the tradeoff between stability and plasticity in real-world continual reinforcement learning.",
  "published": "2026-08-22",
  "updated": "2026-08-22",
  "year": "2026",
  "authors": [
   "Houlin Li",
   "Minghui Xu",
   "Guo Xu",
   "Xuan Du",
   "Xiaohan Yan",
   "Chun Wang",
   "Yuxiang Yan",
   "Shukai Yang",
   "Yongcheng Liu",
   "Wei Shan",
   "Maoqing Yao"
  ],
  "author_count": 11,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work introduces Continual Interactive Distillation for Embodied Reinforcement Learning (CIDER), a continual reinforcement learning framework that freezes the accumulated historical policy as a teacher before learning each new task and interleaves task learning with distillation-based retention.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Houlin Li",
    "id": "2444282785",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Ming Xu",
    "id": "2447751624",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Guofeng Xu",
    "id": "2115724391",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Xuan Du",
    "id": "2255020667",
    "h_index": 4,
    "papers": 17
   },
   {
    "name": "Xiao Yan",
    "id": "2275997184",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Chun Wang",
    "id": "2459242001",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Yuxiang Yan",
    "id": "2459173720",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Shukai Yang",
    "id": "47569533",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Yongcheng Liu",
    "id": "2313580839",
    "h_index": 3,
    "papers": 13
   },
   {
    "name": "Wei Shan",
    "id": "2382133576",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Maoqing Yao",
    "id": "2395512682",
    "h_index": 2,
    "papers": 4
   }
  ],
  "comment": "",
  "topics": [
   "rl-control",
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.21899v1",
  "pdf_url": "https://arxiv.org/pdf/2608.21899v1",
  "html_url": "https://arxiv.org/html/2608.21899v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.21894",
  "slug": "an-interpretable-deep-learning-framework-for-material-perception-and-c",
  "title": "An Interpretable Deep Learning Framework for Material Perception and Classification from Multisensory Tactile Data",
  "abstract": "Human tactile perception relies on complex multisensory cues. Yet the relationship between tactile signals and perceptual representations remains poorly understood, limiting the integration of touch in digital environments and human-like robotic perception. To address this gap, we developed a computational framework comprising three interconnected deep learning models that map multisensory touch data to material perception, without relying on hand-crafted features. The models represent progressively different routes from tactile signals to material class: from low-level interaction signals to perceptual attribute distributions (Model 1), from predicted attribute distributions to material classification (Model 2), and directly from tactile signals to material categories, bypassing intermediate representations (Model 3). By combining deep learning with Integrated Gradients, the framework achieved high accuracy while offering interpretability, revealing which sensory modalities most strongly drive its decisions. Our results show that deep learning can approach near-perfect material classification when unconstrained by intermediate perceptual stages, but matching human-like performance is harder once those stages are modeled explicitly. Notably, thermal cues emerged as particularly informative across all models, providing robust signals for material differentiation. The results offer a computational account of how tactile signals lead to material perception and show how interpretable deep learning can both approach human-level performance and reveal cues that robotic and haptic systems need to incorporate.",
  "published": "2026-08-22",
  "updated": "2026-08-22",
  "year": "2026",
  "authors": [
   "Li Zou",
   "Dave Hogendoorn",
   "Yasemin Vardar"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.HC"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A computational framework comprising three interconnected deep learning models that map multisensory touch data to material perception, without relying on hand-crafted features, offering a computational account of how tactile signals lead to material perception and showing how interpretable deep learning can both approach human-level performance and reveal cues that robotic and haptic systems need to incorporate.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Linjie Zou",
    "id": "2191150793",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Dave Hogendoorn",
    "id": "2459159316",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yasemin Vardar",
    "id": "2298967586",
    "h_index": 3,
    "papers": 13
   }
  ],
  "comment": "7 pages, 5 figures, journal",
  "topics": [
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.21894v1",
  "pdf_url": "https://arxiv.org/pdf/2608.21894v1",
  "html_url": "https://arxiv.org/html/2608.21894v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.21879",
  "slug": "vision-guided-morphing-quadcopter-for-multi-geometry-payload-transport",
  "title": "Vision-Guided Morphing Quadcopter for Multi-Geometry Payload Transport through Narrow Passages",
  "abstract": "Aerial payload transport using multirotor unmanned aerial vehicles is challenging because payload geometry, contact interaction, grasp stability, flight control, and narrow-passage traversal are strongly coupled during pickup and transport. Object-specific grippers often cannot adapt their footprint or grasp geometry when the payload shape or passage width changes. This paper presents a vision-guided morphing quadcopter for multi-geometry payload transport through narrow passages. The proposed platform uses four hybrid arm-leg structures that function as both landing supports and grasping members. A centrally placed actuator drives a tendon-based morphing mechanism, enabling all four arms to synchronously retract or expand for object grasping, footprint reduction, and post-transport release. Onboard vision estimates the payload geometry and passage width, while endpoint force feedback is used to confirm grasp contact during payload engagement. A phase-wise mission planner, PID-based flight stabilization, and morphology-adaptive grasp controller are implemented in a MuJoCo simulation environment. The framework is evaluated using box, cylindrical, and spherical payloads, representing flat-faced, rolling-curved, and fully curved contact conditions. Across the three cases, the simulated system completes the pickup-transport-release sequence with a maximum RMS position error of 0.31 m, a final drop-zone error below 0.18 m, a compact grasp footprint of 0.09-0.21 m2, and a footprint reduction of 75.0-89.7 percent. The results demonstrate that a single-actuator morphing quadcopter can adapt its grasp footprint for the transport of payloads with different geometries while reducing its overall footprint for narrow-passage traversal.",
  "published": "2026-08-22",
  "updated": "2026-08-22",
  "year": "2026",
  "authors": [
   "Aashish Sahu",
   "Shriram Hari",
   "R. Prasanth Kumar"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The results demonstrate that a single-actuator morphing quadcopter can adapt its grasp footprint for the transport of payloads with different geometries while reducing its overall footprint for narrow-passage traversal.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Aashish Sahu",
    "id": "2342698850",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Shri Hari",
    "id": "40944666",
    "h_index": 1,
    "papers": 9
   },
   {
    "name": "R. P. Kumar",
    "id": "2257132654",
    "h_index": 2,
    "papers": 13
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.21879v1",
  "pdf_url": "https://arxiv.org/pdf/2608.21879v1",
  "html_url": "https://arxiv.org/html/2608.21879v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.21778",
  "slug": "vision-guided-target-conditioned-control-for-autonomous-excavation",
  "title": "Vision Guided Target Conditioned Control for Autonomous Excavation",
  "abstract": "Autonomous excavation requires an intelligent control system that can convert spatial work intent into coordinated bucket motion under contact-rich soil interaction. This paper presents a target-conditioned intelligent control framework for autonomous excavation in a physics-based deformable-soil simulation workflow. An image-aligned target mask serves as a visual spatial command for the desired digging region, while a mask-conditioned Action Chunking Transformer maps multi-view RGB observations, proprioception, and the target mask to temporally extended joystick commands. To reduce target-ignoring behavior, demonstrations are organized with paired-condition supervision, where the same or closely matched scene is demonstrated with different target masks and corresponding action chunks. The framework is evaluated through both a diagnostic manipulation task and an excavation simulation benchmark with single-scoop and sequential pile-clearing protocols. In manipulation, target success is 4\\% for no-condition ACT, 63\\% for non-paired mask-conditioned ACT, and 96\\% for paired-condition mask-conditioned ACT. In sequential pile clearing, paired-condition mask-conditioned ACT removes 76.8\\% of the pile versus 27.4\\% and 15.7\\% for the two baselines, with 91.0\\% human-normalized efficiency. The results show that visual target conditioning, paired demonstration structure, and action-chunk control form a practical cyber-physical simulation pipeline for excavator automation.",
  "published": "2026-08-22",
  "updated": "2026-08-22",
  "year": "2026",
  "authors": [
   "Shuai Zhao",
   "Ji-An Pan",
   "Junwei Li",
   "Xun Tang",
   "Fansen Xi",
   "Qing Xu",
   "Keqiang Li",
   "Jianqiang Wang"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The results show that visual target conditioning, paired demonstration structure, and action-chunk control form a practical cyber-physical simulation pipeline for excavator automation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Shuai Zhao",
    "id": "2455473098",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Jisong Pan",
    "id": "2447040724",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Junwei Li",
    "id": "2459431688",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Xun Tang",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Fansen Xi",
    "id": "2459159550",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Qing Xu",
    "id": "2258344505",
    "h_index": 4,
    "papers": 32
   },
   {
    "name": "Keqiang Li",
    "id": "2288319097",
    "h_index": 6,
    "papers": 21
   },
   {
    "name": "Jianqiang Wang",
    "id": "2278550163",
    "h_index": 4,
    "papers": 9
   }
  ],
  "comment": "5 pages, 3 figures, 3 tables. Accepted at ISCSIC 2026",
  "topics": [
   "tactile",
   "sim2real",
   "imitation-diffusion"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.21778v1",
  "pdf_url": "https://arxiv.org/pdf/2608.21778v1",
  "html_url": "https://arxiv.org/html/2608.21778v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.21740",
  "slug": "counteralign-counterfactual-supervision-for-vision-language-action-mod",
  "title": "CounterAlign: Counterfactual Supervision for Vision-Language-Action Models",
  "abstract": "Vision-Language-Action (VLA) models are typically trained with behavior cloning (BC) on expert demonstrations. However, BC provides only positive supervision for expert actions, without explicit negative supervision indicating which actions are instruction-inconsistent or otherwise inappropriate. Reinforcement learning (RL) can provide such corrective signals, but often relies on externally specified rewards or curated non-expert data, both of which are costly to obtain in robotics. We show that offline RL for VLA models need not rely on curated non-expert trajectories: successful expert demonstrations alone can be transformed into dense corrective supervision through instruction relabeling. Specifically, by pairing expert actions with mismatched alternative instructions, we synthesize counterfactual instruction-observation-action tuples from the dataset and combine them with adversarial discriminator training to learn an instruction-grounded reward model for offline RL, without collecting additional rollouts or annotations. On the robustness-focused LIBERO-PRO benchmark, our method improves robustness to object position and task perturbations over a strong state-of-the-art baseline. It also outperforms competitive baselines in real-robot experiments on the TX-G2 (compatible with AGIBot G2). More broadly, our results suggest that, for data-constrained VLA learning, extracting denser supervision from each demonstration can complement collecting additional data.",
  "published": "2026-08-22",
  "updated": "2026-08-22",
  "year": "2026",
  "authors": [
   "Haru Kondoh",
   "Kei Ota",
   "Asako Kanezaki",
   "Yueh-Hua Wu"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work synthesizes counterfactual instruction-observation-action tuples from the dataset and combines them with adversarial discriminator training to learn an instruction-grounded reward model for offline RL, without collecting additional rollouts or annotations.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Haruo Kondoh",
    "id": "144703519",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Keita Ota",
    "id": "104236766",
    "h_index": 10,
    "papers": 37
   },
   {
    "name": "Asako Kanezaki",
    "id": "2247709551",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Yueh-Hua Wu",
    "id": "31609618",
    "h_index": 10,
    "papers": 13
   }
  ],
  "comment": "Project page: https://counteralign.airoa.io",
  "topics": [
   "vla",
   "imitation-diffusion",
   "rl-control",
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [
   "AgiBot"
  ],
  "abs_url": "https://arxiv.org/abs/2608.21740v1",
  "pdf_url": "https://arxiv.org/pdf/2608.21740v1",
  "html_url": "https://arxiv.org/html/2608.21740v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2608.21735",
  "slug": "safety-critical-bilateral-teleoperation-for-omnidirectional-aerial-man",
  "title": "Safety-Critical Bilateral Teleoperation for Omnidirectional Aerial Manipulation Using Force-Sensorless Haptic Feedback",
  "abstract": "This paper presents a safety-critical bilateral teleoperation framework for omnidirectional aerial manipulators that integrates visual and force-sensorless haptic wrench feedback. Unlike existing approaches that either rely on onboard force/torque sensors or use model-dependent wrench estimates, which may become unreliable under model uncertainties or induce unintended feedback during free-flight, our method implements a hierarchical safety filter based on control barrier functions to avoid such limitations. The safety filter, being the key contribution, explicitly accounts for tracking errors arising from physical interaction between the aerial manipulator and its surroundings while enforcing thrust limits, a factor overlooked despite its critical importance for flight safety. This safety filter adjusts the command from the operator to ensure safe and stable aerial manipulation and avoid motor saturation. The adjustment made by the filter is mapped to haptic feedback, which is intuitive to the operator and conveys information on physical interaction and impending motor saturation. By actual experiments with a hexarotor-based omnidirectional aerial manipulator, we demonstrate that the proposed method avoids haptic feedback during free-flight, provides directionally consistent feedback under physical interaction, and can be operated for diverse manipulative tasks. Moreover, an ablation study further shows that the saturation filter improves interaction stability by explicitly preventing motor saturation and informing the operator of corrective actions.",
  "published": "2026-08-22",
  "updated": "2026-08-22",
  "year": "2026",
  "authors": [
   "Yubin Kim",
   "Jinwoo Lee",
   "Yongjun You",
   "H. Jin Kim",
   "Jeonghyun Byun"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "It is demonstrated that the proposed method avoids haptic feedback during free-flight, provides directionally consistent feedback under physical interaction, and can be operated for diverse manipulative tasks.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yubin Kim",
    "id": "2459234917",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Jinwoo Lee",
    "id": "2383463012",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Yongjun You",
    "id": "2459159806",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "H. J. Kim",
    "id": "2297190046",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Jeonghyun Byun",
    "id": "2125337822",
    "h_index": 5,
    "papers": 19
   }
  ],
  "comment": "8 pages, 10 figures. Accepted to the 2026 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS 2026)",
  "topics": [
   "rl-control",
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.21735v1",
  "pdf_url": "https://arxiv.org/pdf/2608.21735v1",
  "html_url": "https://arxiv.org/html/2608.21735v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.21699",
  "slug": "towards-insect-like-distributed-proprioception-in-actuators-and-append",
  "title": "Towards insect-like distributed proprioception in actuators and appendages for flapping-wing insect-scale aerial robots",
  "abstract": "Modern flapping-wing insect-scale air vehicles display agility similar to that of their insect counterparts; however, these impressive maneuvers are only possible with off-board sensors like optical tracking cameras. In this manuscript, we introduce two embedded proprioceptive sensors for insect-scale aerial robots: thin film piezoelectric polymers integrated directly into a driving actuator and a pitching hinge which track stroke and pitch angle, respectively. We fabricate the aforementioned size-agnostic mechanically intelligent structures (sensor-actuator, sensor-flexure) using laminate stack fabrication methods. Chirp experiments with our sensors integrated into an insect-size flapping-wing robot show accurate tracking of stroke (RMSE = 0.44 deg) and pitch (RMSE = 2.44 deg) angles in the relevant frequency range. As the first step towards demonstrating the utility of these sensors for enabling numerous onboard autonomy applications, including closed-loop wingbeat control and sensor fusion with existing insect-scale sensor suites for more accurate proprioception and localization, we show one application for each sensor. The proprioceptive hinge enables collision detection, reducing the chance of permanent damage if the robot's wing collides with an object. The proprioceptive actuator enables asynchronous flapping, which is hypothesized to increase adaptability and efficiency in insects and robots alike. A microrobot equipped with our proprioceptive actuator allows us to test these hypotheses with potential for improving flapping aerial robot performance. We foresee proprioceptive sensors having an important role in progressing both the fields of insect-scale aerial robots and robo-physics due to the bio-inspired nature and high integration level of our sensors.",
  "published": "2026-08-22",
  "updated": "2026-08-22",
  "year": "2026",
  "authors": [
   "Alexander Hedrick",
   "Arvind Gupta",
   "Kaushik Jayaram"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Alexander Hedrick",
    "id": "2257000287",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Arvind Gupta",
    "id": "2438151571",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Kaushik Jayaram",
    "id": "19851278",
    "h_index": 18,
    "papers": 63
   }
  ],
  "comment": "8 pages, 6 figures, this work has been submitted to the IEEE for possible publication. Copyright may be transferred without notice, after which this version may no longer be accessible",
  "topics": [
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.21699v1",
  "pdf_url": "https://arxiv.org/pdf/2608.21699v1",
  "html_url": "https://arxiv.org/html/2608.21699v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.26182",
  "slug": "why-did-my-robot-just-change-personality-prompting-guidelines-for-a-gr",
  "title": "Why did My Robot Just Change Personality? Prompting Guidelines for a Grounded Robot Persona in LLM-Based HRI",
  "abstract": "Large language models (LLMs) are increasingly used for verbal interaction in social robots, yet prompt design in human-robot interaction (HRI) remains underspecified. As a result, robots may present hallucinated capabilities, unclear behavioural boundaries, and misleading personas. This paper develops a framework for prompt design in LLM-based robots and introduces a structured prompt template comprising eight functional components through which robot behaviour can be specified, bounded, and adapted. The framework is grounded in a review of prior LLM-based HRI work and complemented by survey and discussion data from HRI experts gathered at the Robo-Identity workshop at IEEE RO-MAN 2025 (N=27). The qualitative findings highlight limited legibility of robot personality, the need for user adaptation, and strong ethical concerns about safety, deception, and governance. Based on these findings, we present prompting guidelines accompanied by proof-of-concept template as a structured design and reporting aid for HRI research. We argue that prompt design should be treated as a socio-technical problem rather than a minor implementation detail, requiring explicit capability boundaries, transparent behavioural assumptions, and context-sensitive safeguards to support reliable and interpretable HRI.",
  "published": "2026-08-21",
  "updated": "2026-08-21",
  "year": "2026",
  "authors": [
   "Ashita Ashok",
   "Franziska Babel",
   "Patrick Holthaus",
   "Rucha Khot",
   "Karla Bransky",
   "Fethiye Irmak Dogan",
   "Karsten Berns",
   "Silvia Rossi",
   "Minha Lee",
   "Guy Laban"
  ],
  "author_count": 10,
  "categories": [
   "cs.AI",
   "cs.HC",
   "cs.RO"
  ],
  "primary_category": "cs.AI",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "Accepted for publication at the 35th IEEE International Conference on Robot and Human Interactive Communication (RO-MAN 2026)",
  "topics": [
   "hri",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.26182v1",
  "pdf_url": "https://arxiv.org/pdf/2608.26182v1",
  "html_url": "https://arxiv.org/html/2608.26182v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.21685",
  "slug": "in-situ-reconstruction-of-the-international-space-station-using-3d-gau",
  "title": "In-Situ Reconstruction of the International Space Station Using 3D Gaussian Splatting and Astrobee",
  "abstract": "This article presents a novel 3D reconstruction and mapping of the interior of the International Space Station (ISS) using 3D Gaussian Splatting (3DGS). Using existing grayscale images from the Astrobee free-flying robot dataset, we construct a full 3D splat of the ISS' Kib\u014d or Japanese Experiment Module (JEM). 3DGS has in recent years shown promise in providing novel view synthesis of scenes captured from many images or videos, this article applies this approach to human spaceflight systems. We compare our 3DGS architecture to existing methods such as Nerfacto and TensoRF and show that reconstruction improves the state-of-the-art in both scene quality and rendering speed. We show that with as little as 500 in-situ images, a high-fidelity map can be constructed using Astrobee's Navigation Camera (NavCam) during free-flight in the JEM. These reconstructions could enable free-flyers to rapidly create and update interior maps for intra-vehicular habitats like the ISS.",
  "published": "2026-08-21",
  "updated": "2026-08-21",
  "year": "2026",
  "authors": [
   "Hudson Kim",
   "Ryan Soussan",
   "Brian Coltin",
   "Jordan Kam"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This article presents a novel 3D reconstruction and mapping of the interior of the International Space Station (ISS) using 3D Gaussian Splatting (3DGS), and shows that reconstruction improves the state-of-the-art in both scene quality and rendering speed.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hudson Kim",
    "id": "2274426284",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Ryan Soussan",
    "id": "92450179",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "B. Coltin",
    "id": "3149558",
    "h_index": 22,
    "papers": 68
   },
   {
    "name": "Jordan Kam",
    "id": "2343442916",
    "h_index": 1,
    "papers": 9
   }
  ],
  "comment": "",
  "topics": [
   "spatial-3d",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.21685v1",
  "pdf_url": "https://arxiv.org/pdf/2608.21685v1",
  "html_url": "https://arxiv.org/html/2608.21685v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.21676",
  "slug": "lifelong-robot-recomposition-via-persistent-categorical-modeling-for-u",
  "title": "Lifelong Robot Recomposition via Persistent Categorical Modeling for Unified Task-Driven Co-Design, Verification, and Planning",
  "abstract": "Robotic systems are traditionally designed and deployed in static configurations, with assumptions made at design-time becoming immutable constraints during runtime. This design-then-deploy paradigm produces performant systems under narrow operating conditions, but renders robots brittle when qualities of themselves, their tasks, or their environments unexpectedly change. We address this challenge with a compositional framework that formalizes robotic systems as abstract circuits within a strict symmetric monoidal category, in which design and runtime composition of hardware, software, and behavior are synthesized simultaneously via an SMT-based solver, with monoidal functors projecting the system into lifecycle-specific views and free symbolic variables simultaneously solving for parameters and entire component specifications within larger compositions. This persistent model also supports queries a long-lived system needs beyond plan existence across its entire lifecycle, including mapping Pareto fronts over candidate compositions, diagnosing why a composition has become infeasible, finding its minimal restoration, and reconfiguring with limited change to the deployed system. We evaluate against official implementations of optimal numeric, stream-based, and SMT-based planners all measured onboard a deployed robot and demonstrate the approach end-to-end in a search-and-rescue scenario in which the robot recognizes when it has become unfit and synthesizes and assumes new holistic configurations to restore operation. We release our solver and supporting software open-source.",
  "published": "2026-08-21",
  "updated": "2026-08-21",
  "year": "2026",
  "authors": [
   "Steven Swanbeck",
   "Mitch Pryor"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A compositional framework that formalizes robotic systems as abstract circuits within a strict symmetric monoidal category, in which design and runtime composition of hardware, software, and behavior are synthesized simultaneously via an SMT-based solver.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Steven Swanbeck",
    "id": "2100012176",
    "h_index": 3,
    "papers": 13
   },
   {
    "name": "Mitchell W. Pryor",
    "id": "2374046982",
    "h_index": 2,
    "papers": 8
   }
  ],
  "comment": "",
  "topics": [
   "hardware-codesign",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.21676v1",
  "pdf_url": "https://arxiv.org/pdf/2608.21676v1",
  "html_url": "https://arxiv.org/html/2608.21676v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.21631",
  "slug": "openscvx-an-open-source-modular-and-extensible-nonlinear-trajectory-pl",
  "title": "OpenSCvx: An Open-Source Modular and Extensible Nonlinear Trajectory Planning Package",
  "abstract": "Trajectory optimization computes dynamically feasible motions that enable autonomous systems to accomplish complex tasks while satisfying operational and environmental constraints. This tutorial presents OpenSCvx, an open-source Python framework that bridges the gap between high-level problem specification and efficient numerical optimization. Rather than requiring users to derive solver-specific mathematical formulations, OpenSCvx provides a symbolic modeling interface that automatically constructs and solves trajectory optimization problems from modular descriptions of objectives, dynamics, and constraints. Beyond simplifying problem formulation, OpenSCvx supports (i) continuous-time constraint modeling, (ii) temporal and logical specifications, (iii) automatic vectorization for scalable and batched optimization, and (iv) a modular architecture that enables new algorithms, models, and solver backends to be incorporated with minimal effort. These capabilities allow researchers and practitioners to rapidly prototype, solve, and extend state-of-the-art trajectory optimization methods.",
  "published": "2026-08-21",
  "updated": "2026-08-21",
  "year": "2026",
  "authors": [
   "Christopher R. Hayner",
   "Griffin J. Norris",
   "Fabio Spada",
   "Samet Uzun",
   "Avi Mittal",
   "Behcet Ac\u0131kmese",
   "Karen Leung"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "math.OC"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This tutorial presents OpenSCvx, an open-source Python framework that bridges the gap between high-level problem specification and efficient numerical optimization and provides a symbolic modeling interface that automatically constructs and solves trajectory optimization problems from modular descriptions of objectives, dynamics, and constraints.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Christopher R. Hayner",
    "id": "2125226143",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Griffin J. Norris",
    "id": "2139961560",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Fabio Spada",
    "id": "2339777304",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Samet Uzun",
    "id": "150309993",
    "h_index": 5,
    "papers": 16
   },
   {
    "name": "Avi Mittal",
    "id": "2194369902",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Beh\u00e7et A\u00e7ikmese",
    "id": "2991808",
    "h_index": 43,
    "papers": 278
   },
   {
    "name": "Karen Leung",
    "id": "144579384",
    "h_index": 17,
    "papers": 35
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.21631v1",
  "pdf_url": "https://arxiv.org/pdf/2608.21631v1",
  "html_url": "https://arxiv.org/html/2608.21631v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.21628",
  "slug": "exploreai-agentic-exploration-knowledge-bases-for-reproducible-observa",
  "title": "ExploreAI: Agentic Exploration Knowledge Bases for Reproducible Observable-Regression Testing of Black-Box VR and 3D Applications",
  "abstract": "Black-box VR and 3D applications are difficult to regression test because observable failures depend on where a tester moves, what objects are visible, and which views are captured. Manual exploratory testing can find such failures, but its evidence is time-consuming to reproduce; systematic sweeps are reproducible, but they lack semantic guidance and spend exploration budget on low-value viewpoints. We observe that an LLM can make the high-level decisions a human tester makes during exploration: interpreting a task, choosing which objects to inspect, grouping related objects, recording what it saw, and deciding when missing evidence should trigger another attempt. Based on this observation, we present ExploreAI, an LLM-driven agentic framework that offloads repeated perception, navigation, multi-view capture execution, and logging to specialized modules while using the LLM for planning, evidence recording, capture-policy decisions, and verification decisions. ExploreAI constructs an Exploration Knowledge Base (EKB): a structured, per-object record of one exploration run. For each object the agent finds, the EKB stores the scan evidence that exposed it, the selected target, the navigation path, the multi-view capture, and the self-verification result. The EKB is a reusable testing artifact that supports reproducible observable-regression checking across versions of a VR or 3D application. Across six indoor and outdoor scenes in Unity, AI2-THOR, and BeamNG, ExploreAI constructs high-completeness EKBs under both complete and target exploration, and an LLM-module ablation shows where semantic planning, capture policy, evidence recording, and self-verification contribute. Reproduction pilots further show that EKB-guided traces help both humans and LLM-based reproducers reproduce exact object-view evidence more effectively than conditions without EKB context.",
  "published": "2026-08-21",
  "updated": "2026-08-21",
  "year": "2026",
  "authors": [
   "Jiajie Wang",
   "Kebin Peng",
   "Wei Wang",
   "Xiaoyin Wang",
   "Sen He",
   "Xue Qin"
  ],
  "author_count": 6,
  "categories": [
   "cs.SE",
   "cs.RO"
  ],
  "primary_category": "cs.SE",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jiajie Wang",
    "id": "2459232787",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Kebin Peng",
    "id": "2327285420",
    "h_index": 3,
    "papers": 14
   },
   {
    "name": "Wei Wang",
    "id": "2297548375",
    "h_index": 0,
    "papers": 6
   },
   {
    "name": "Xiaoyin Wang",
    "id": "2297649109",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Sen He",
    "id": "2334532873",
    "h_index": 3,
    "papers": 19
   },
   {
    "name": "Xue Qin",
    "id": "2313036617",
    "h_index": 2,
    "papers": 4
   }
  ],
  "comment": "11 pages, 3 figures, 8 tables",
  "topics": [
   "navigation",
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.21628v1",
  "pdf_url": "https://arxiv.org/pdf/2608.21628v1",
  "html_url": "https://arxiv.org/html/2608.21628v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.21620",
  "slug": "why-personalization-matters-cross-subject-challenges-in-emg-imu-based",
  "title": "Why Personalization Matters: Cross-Subject Challenges in EMG-IMU-based HRI Activity Recognition",
  "abstract": "This paper investigates wearable-based recognition of human activities and gestures to support Human-Robot Interaction (HRI) in object-handover and assembly-like scenarios. Electromyography (EMG) and Inertial Measurement Unit (IMU) signals were collected using a Myo armband, culminating in a novel dataset introduced as MAGIC-HRI (Multimodal Activity, Gesture and Intention Collection) with a large taxonomy of 53 movement classes, including Brazilian Sign Language (LIBRAS) numbers (0-9), hand gestures, object/tool handover actions (pick up/give/hold), tool-manipulation tasks, and generic assembly/idle motions, collected from 11 participants with 10 samples per class (530 samples per participant). Signals are segmented by detecting muscle activation via an EMG energy envelope, then processed using sliding windows; time- and frequency-domain features are extracted. Multiple classical classifiers are tuned via cross-validated grid search, with Random Forest as the strongest baseline. A Leave-One-Subject-Out (LOSO) protocol reveals a large generalization gap, indicating substantial subject dependence. A personalized adaptation experiment suggests that injecting a small number of samples from a new user can markedly improve recognition. Overall, the study contributes a broad, HRI-driven multimodal dataset, a rigorous evaluation emphasizing generalization, and practical evidence that personalization is likely required for robust deployment in practical HRI.",
  "published": "2026-08-21",
  "updated": "2026-08-21",
  "year": "2026",
  "authors": [
   "Ruan Rithelle Chagas de Faria Carminati",
   "Giovanni Braglia",
   "Luigi Biagiotti",
   "Ronnier Frates Rohrich",
   "Andre Schneider de Oliveira",
   "Mikael Nedel Hartmann",
   "Andr\u00e9 Eugenio Lazzaretti"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A broad, HRI-driven multimodal dataset, a rigorous evaluation emphasizing generalization, and practical evidence that personalization is likely required for robust deployment in practical HRI are contributed.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "R. Carminati",
    "id": "46595794",
    "h_index": 5,
    "papers": 17
   },
   {
    "name": "Giovanni Braglia",
    "id": "2139973165",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "L. Biagiotti",
    "id": "3034713",
    "h_index": 19,
    "papers": 96
   },
   {
    "name": "R. Rohrich",
    "id": "1435351740",
    "h_index": 3,
    "papers": 27
   },
   {
    "name": "A. S. D. Oliveira",
    "id": "2302161342",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "M. Hartmann",
    "id": "1698692002",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "A. Lazzaretti",
    "id": "3225435",
    "h_index": 18,
    "papers": 135
   }
  ],
  "comment": "Data collection was approved by the Federal University of Technology-Paran\u00e1 Ethics Committee (CAAE 91430125.0.0000.0177). The MAGIC-HRI (Multimodal Activity, Gesture, and Intention Collection for HRI) dataset is available at [https://github.com/ruancarminati/MAGIC-HRI-V01.git](https://github.com/ruancarminati/MAGIC-HRI-V01.git). This paper will be presented at IEEE RO-MAN 2026",
  "topics": [
   "data-teleop",
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.21620v1",
  "pdf_url": "https://arxiv.org/pdf/2608.21620v1",
  "html_url": "https://arxiv.org/html/2608.21620v1",
  "code_url": "https://github.com/ruancarminati/MAGIC-HRI-V01.git",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2608.21592",
  "slug": "force-torque-based-kinematic-adaptation-for-robotic-manipulation-tasks",
  "title": "Force/Torque-Based Kinematic Adaptation for Robotic Manipulation Tasks",
  "abstract": "Contact-rich robotic manipulation requires an accurate model of the kinematic relationship between a robot's joints and the task features it senses. This relationship is rarely known exactly: it changes with each tool the robot picks up and shifts, sometimes almost instantaneously, as contact modes change --- especially for multi-fingered hands that make and break contact at points that are not exactly prescribed, as in full-hand grasping. This paper develops an adaptive scheme that estimates that relationship online, using only joint-angle sensing and a wrist-mounted force/torque sensor, with no exteroceptive measurement of the tool tip. We derive a provably stable kinematic update law that identifies the kinematics of an unknown tool from force/torque feedback alone, and prove stability of both the rigid case and the case with a compliance controller as an inner loop. We show that identification is confined to the directions the motion excites --- so that, for example, a tool's length is unobservable under a rigid insertion push, while a compliant loop's passive yielding partially excites it; and that with a second-order admittance the compliant certificate holds unconditionally in continuous time. We also pose the combined control and estimation problem as a Quadratic Program (QP): the formulation yields the prediction term of the update law exactly but, instructively, cannot reproduce the tracking adaptation term. We validate the scheme in simulation on a peg-in-hole insertion. This work is the first step in a research program aimed at factoring manipulation learning into a task policy which can be learned in isolation of the robot, for instance by reinforcement learning, and an adaptive kinematic component that adapts online to the particular robot, hand, or tool in use.",
  "published": "2026-08-21",
  "updated": "2026-08-21",
  "year": "2026",
  "authors": [
   "Carl Glen Henshaw"
  ],
  "author_count": 1,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "An adaptive scheme that estimates the kinematic relationship between a robot's joints and the task features it senses online, using only joint-angle sensing and a wrist-mounted force/torque sensor, with no exteroceptive measurement of the tool tip is developed.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Carl Glen Henshaw",
    "id": "2270693088",
    "h_index": 2,
    "papers": 7
   }
  ],
  "comment": "29 pages including appendices, 3 figures",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.21592v1",
  "pdf_url": "https://arxiv.org/pdf/2608.21592v1",
  "html_url": "https://arxiv.org/html/2608.21592v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.21572",
  "slug": "betting-for-sim-to-real-performance-certificates",
  "title": "Betting for Sim-to-Real Performance Certificates",
  "abstract": "Consider a typical test of a robot system: one observes a sequence of outcomes concerning some aspect of interest (crash or no crash, tracking error, time to completion), and reports a mean (crash risk, average error, mean time to completion) and, more importantly, an interval guaranteed to contain that mean at a prescribed confidence, referred to as a performance certificate. Given expensive real-world trials, the sample size is therefore small, and the certificate is often loose. Now consider the same procedure, except that before each real outcome is revealed, the operator ``peeks'' at a large bank of simulated results, and places a bet on where the real outcome will land. As the real outcomes settle the bets, the operator gains or loses wealth. One's ``trust'' over simulators also shifts within the portfolio. This paper develops that idea into a sim-to-real betting certificate framework with three contributions: (i) An algorithm that links a scalable bank of simulators to effective bets, and the accumulated betting wealth to the certificate. (ii) A proof that the returned certificate is anytime valid, covering the true mean with the prescribed probability, using any simulator bank. (iii) The guaranteed wealth-regret bounds yield configuration principles for the proposed algorithm and simulator bank design to deliver tight certificates. Experiments across synthetic distributions and real-world robot tests, covering both replayed standardized testing outcomes and online runtime evaluation, show the proposed method narrows the certificate by $51.6\\%\\pm16\\%$ against classic and state-of-the-art baselines, and by $32.26\\%\\pm8\\%$ in the extremely limited-sample regime ($\\leq30$ samples).",
  "published": "2026-08-21",
  "updated": "2026-08-21",
  "year": "2026",
  "authors": [
   "Yujia Chen",
   "Bowen Weng"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Experiments across synthetic distributions and real-world robot tests, covering both replayed standardized testing outcomes and online runtime evaluation, show the proposed method narrows the certificate by $51.6 against classic and state-of-the-art baselines, and by $32.26 in the extremely limited-sample regime.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yujia Chen",
    "id": "2326733965",
    "h_index": 1,
    "papers": 9
   },
   {
    "name": "Bowen Weng",
    "id": "51497959",
    "h_index": 11,
    "papers": 41
   }
  ],
  "comment": "",
  "topics": [
   "sim2real",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.21572v1",
  "pdf_url": "https://arxiv.org/pdf/2608.21572v1",
  "html_url": "https://arxiv.org/html/2608.21572v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.21554",
  "slug": "model-based-reinforcement-learning-for-heterogeneous-multi-robot-task",
  "title": "Model-Based Reinforcement Learning for Heterogeneous Multi-Robot Task Assignment Under Distribution Shifts",
  "abstract": "Heterogeneous multi-robot service systems must assign requests to compatible robots, construct feasible schedules, and adapt as new tasks arrive online. Historical data can help anticipate future demand, but relying too heavily on inaccurate predictions can degrade performance under distribution shifts. We develop a prediction-aware adaptive rollout framework for heterogeneous multi-robot task assignment with scheduled and real-time requests. The problem is formulated as a finite-horizon stochastic dynamic program incorporating robot-task compatibility, ordered service requirements, routing constraints, service windows, and end-of-horizon return requirements. The proposed policy evaluates current assignments using sampled future request scenarios while restricting immediate commitments to requests already observed. To enable online use, the framework combines pruned candidate controls, wait actions, and an interaction-aware base policy for efficient future-cost estimation. Robustness to forecast error is provided by adaptively reweighting predicted requests based on recent prediction mismatch and selectively re-optimizing assigned but unstarted requests. We also introduce a historical-data-driven procedure for selecting the heterogeneous fleet composition before deployment. In a case study using real nursing-task requests from hospital inpatient floors, the proposed approach achieves near-complete service and reduces serviced-request wait times relative to reactive, token-passing, prediction-positioning, and myopic greedy baselines, with the largest improvements in tail-delay metrics.",
  "published": "2026-08-21",
  "updated": "2026-08-21",
  "year": "2026",
  "authors": [
   "Daniel Garces",
   "Sara Castro",
   "Adrian Haimovich",
   "Byron Crowe",
   "Stephanie Gil"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.LG",
   "cs.MA"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A prediction-aware adaptive rollout framework for heterogeneous multi-robot task assignment with scheduled and real-time requests that achieves near-complete service and reduces serviced-request wait times relative to reactive, token-passing, prediction-positioning, and myopic greedy baselines.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Daniel Garces",
    "id": "2192609360",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Sara Castro",
    "id": "2459159113",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Adrian D Haimovich",
    "id": "2350803973",
    "h_index": 2,
    "papers": 24
   },
   {
    "name": "Byron Crowe",
    "id": "2459158910",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Stephanie Gil",
    "id": "2265381326",
    "h_index": 2,
    "papers": 9
   }
  ],
  "comment": "34 pages, 14 figures, 4 tables",
  "topics": [
   "rl-control",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.21554v1",
  "pdf_url": "https://arxiv.org/pdf/2608.21554v1",
  "html_url": "https://arxiv.org/html/2608.21554v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.21550",
  "slug": "golem-modular-humanoid-autonomy-towards-electric-vehicle-battery-disas",
  "title": "GOLEM: Modular Humanoid Autonomy Towards Electric Vehicle Battery Disassembly",
  "abstract": "Disassembling end-of-life electric vehicle (EV) battery packs is dull and dangerous work, performed almost entirely by humans. We present GOLEM (Generalized Open Library of Embodied Modules), an end-to-end, open-source system architecture for EV battery disassembly with the Unitree H1-2 humanoid robot in which walking, manipulation, dynamic stability, navigation, and spatial memory are independent modules with abstract interfaces, so that methods are easily developed, interchanged, and compared. GOLEM is deployed as a Docker-based ROS 2 abstraction in which MuJoCo and IsaacLab digital twins expose interfaces matching the physical robot. GOLEM's composability and per-module customization enable development and demonstration of humanoid EV battery disassembly, from simulation to reality. GOLEM provides fair comparison between humanoid modules, enabling evaluation as a capability ladder, in which one module is characterized at a time and added as a rung: LiDAR-inertial navigation places the robot within 13.0cm of a 6m goal; a learned standing controller recovers from external disturbances that sampling-based lower-body MPC does not; and grasping loosened fasteners from a real Hyundai Ioniq 5 pack degrades from 97% tethered to 87% free-standing to 37% under navigation-induced pose variance. Source code is available at the project page https://golem-humanoid.github.io",
  "published": "2026-08-21",
  "updated": "2026-08-21",
  "year": "2026",
  "authors": [
   "Max Conway",
   "William Xie",
   "Allen Devaraj",
   "Yutong Zhang",
   "Niraj Pudasaini",
   "Mateo Feit",
   "Adam Abid",
   "Zachary Allen",
   "Chen Liu",
   "Xuan Tan",
   "Jensen Lavering",
   "Jason Chen",
   "Lyle Antieau",
   "Anthony Von Pischke",
   "Alessandro Roncone",
   "Zachary Sunberg",
   "Nikolaus Correll"
  ],
  "author_count": 17,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Max Conway",
    "id": "114932147",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "W. Xie",
    "id": "2291098851",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "A. Devaraj",
    "id": "2447160065",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yutong Zhang",
    "id": "2361648003",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Niraj Pudasaini",
    "id": "2092477521",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Mateo Feit",
    "id": "2459157047",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Adam Abid",
    "id": "2459157857",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Zachery Allen",
    "id": "82508369",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Chen Liu",
    "id": "2287759845",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "X. Tan",
    "id": "2112782090",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Jensen Lavering",
    "id": "2290918920",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Jason Chen",
    "id": "2459236946",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Lyle Antieau",
    "id": "2424644728",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Anthony Von Pischke",
    "id": "2459157572",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Alessandro Roncone",
    "id": "3077150",
    "h_index": 5,
    "papers": 31
   },
   {
    "name": "Zachary Sunberg",
    "id": "2303653497",
    "h_index": 2,
    "papers": 15
   },
   {
    "name": "N. Correll",
    "id": "2886493",
    "h_index": 36,
    "papers": 174
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "humanoids",
   "navigation"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2608.21550v1",
  "pdf_url": "https://arxiv.org/pdf/2608.21550v1",
  "html_url": "https://arxiv.org/html/2608.21550v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2608.21533",
  "slug": "model-free-adaptive-parameter-tuning-for-efficient-multi-robot-warehou",
  "title": "Model-Free Adaptive Parameter Tuning for Efficient Multi-Robot Warehouse Operations",
  "abstract": "Robotic Fulfillment Centers (FCs) store inventory on shelves (pods) arranged in dense blocks. Retrieving a target pod that is buried deep in a block requires moving obstructing pods out of the way (i.e., digout). Multi-robot planners use parameterized cost functions to control digout behavior, producing a spectrum of strategies: at one extreme, obstructing pods are sent to other blocks (using more robots in travel lanes); at the other, pods are shuffled within the block (avoiding lane congestion but increasing extraction time). Each point on this spectrum has different downstream consequences for floor congestion and throughput. The optimal operating point depends on the specific facility configuration and shifts with operational conditions such as varying station demand and congestion patterns, making offline tuning impractical. We present an adaptive parameter tuning framework based on Extremum Seeking Control (ESC) that continuously adjusts planner parameters in response to measured throughput. ESC performs model-free optimization by perturbing parameters with sinusoidal dither signals and correlating perturbations with performance changes to estimate gradients, making it robust to the multi-minute delayed effects and credit assignment challenges inherent in large FC operations. Simulation studies demonstrate that the adaptive policy improves upon fixed policies across several conditions. We observe an improvement in throughput by an average of 5.0% across map and robot fleet size variations, and by 8.4% under dynamic operating conditions. This work eliminates manual parameter provisioning and enables real-time adaptation, providing a self-tuning paradigm for FC storage operations.",
  "published": "2026-08-21",
  "updated": "2026-08-21",
  "year": "2026",
  "authors": [
   "Pratap Tokekar",
   "Mouhacine Benosman",
   "Rahul Chandan",
   "Alexandre Ormiga Galvao Barbosa",
   "Michael Caldara",
   "Joseph W. Durham"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work eliminates manual parameter provisioning and enables real-time adaptation, providing a self-tuning paradigm for FC storage operations, and presents an adaptive parameter tuning framework based on Extremum Seeking Control that continuously adjusts planner parameters in response to measured throughput.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Pratap Tokekar",
    "id": "2390456",
    "h_index": 31,
    "papers": 192
   },
   {
    "name": "M. Benosman",
    "id": "1730046",
    "h_index": 26,
    "papers": 158
   },
   {
    "name": "Rahul Chandan",
    "id": "2378708280",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "A. Barbosa",
    "id": "2384323099",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Michael Caldara",
    "id": "2349811857",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Joseph W. Durham",
    "id": "2140658",
    "h_index": 13,
    "papers": 29
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.21533v1",
  "pdf_url": "https://arxiv.org/pdf/2608.21533v1",
  "html_url": "https://arxiv.org/html/2608.21533v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.21358",
  "slug": "mining-beyond-earth-with-space-robots-exploration-sampling-and-extract",
  "title": "Mining beyond Earth with Space Robots: Exploration, Sampling, and Extraction",
  "abstract": "Space resource acquisition and utilization, commonly referred to as Space Mining, represent critical pathways for enabling sustained human exploration and unlocking commercial opportunities in space. These resources mainly include helium-3, water, mineral resources on the Moon and Mars, and abundant mineral deposits on asteroids. Due to the harsh conditions of space, communication delays, and high launch costs, the development of autonomous robotic systems is critical to achieving efficient, cost-effective space mining. This paper provides a comprehensive overview of space mining robotics and associated technologies. First, we review the background of space mining, including international policies, commercial entities, and recent advancements. We define a systematic six-stage architecture for space mining: Exploration is initiated by (1) remote sensing for target identification and (2) precise in situ robotic detection; Sampling progresses from (3) single-robot small-scale sampling to (4) multi-robot large-scale excavation; and Extraction integrates (5) autonomous resource extraction and (6) final integration into in situ construction or terrestrial transport. Additionally, we review and curate existing resources for space mining research, including real-world mission data, terrestrial analog datasets, and high-fidelity simulation environments. Finally, we identify critical open challenges in autonomous space mining and delineate a strategic research roadmap to bridge current technological gaps, fostering the transition toward a sustainable off-world economy. To track ongoing developments in space mining, we maintain an updated project page: https://github.com/OpenSpace-Lab/Space-Mining-with-Robotics-List.",
  "published": "2026-08-21",
  "updated": "2026-08-21",
  "year": "2026",
  "authors": [
   "Dong Li",
   "Dujun Nie",
   "Xiaotong Zhang",
   "Ruilin Wang",
   "Yuchen Li",
   "Chang Ge",
   "Chao Xiong",
   "Kaichang Di",
   "Andreas N\u00fcchter",
   "Levente Kov\u00e1cs",
   "Qingquan Li",
   "Shirong Ge",
   "Fei-Yue Wang",
   "Long Chen"
  ],
  "author_count": 14,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Dong Li",
    "id": "2331566278",
    "h_index": 0,
    "papers": 6
   },
   {
    "name": "Dujun Nie",
    "id": "2309004226",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Xiaotong Zhang",
    "id": "2165301352",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "Ruili Wang",
    "id": "2144415547",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Yuchen Li",
    "id": "2128125807",
    "h_index": 13,
    "papers": 35
   },
   {
    "name": "Changshui Ge",
    "id": "104662942",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Chaolin Xiong",
    "id": "2407896945",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "K. Di",
    "id": "144833973",
    "h_index": 25,
    "papers": 162
   },
   {
    "name": "Andreas N\u00fcchter",
    "id": "2248820505",
    "h_index": 5,
    "papers": 32
   },
   {
    "name": "Levente Kov\u00e1cs",
    "id": "2326224505",
    "h_index": 2,
    "papers": 14
   },
   {
    "name": "Qingquan Li",
    "id": "2243224925",
    "h_index": 44,
    "papers": 246
   },
   {
    "name": "Shirong Ge",
    "id": "2278862462",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Fei-Yue Wang",
    "id": "2148956730",
    "h_index": 31,
    "papers": 253
   },
   {
    "name": "Long Chen",
    "id": "2280479289",
    "h_index": 9,
    "papers": 27
   }
  ],
  "comment": "",
  "topics": [
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.21358v1",
  "pdf_url": "https://arxiv.org/pdf/2608.21358v1",
  "html_url": "https://arxiv.org/html/2608.21358v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.21355",
  "slug": "vitacphys-physical-property-aware-grasping-from-human-visual-tactile-d",
  "title": "ViTacPhys: Physical Property-Aware Grasping from Human Visual-Tactile Demonstrations",
  "abstract": "Recent vision-based action models have demonstrated strong capabilities in complex manipulation, but they rarely leverage explicit object physical properties to adapt their policies. We introduce ViTacPhys, a visual-tactile framework and data acquisition system that estimates object mass and friction-coefficient classes, together with continuous stiffness, from human manipulation demonstrations. Trained on data from 60 rigid and deformable objects, ViTacPhys combines temporal visual-tactile modeling, cross-attention multimodal fusion, and a semantic prior derived from a vision-language model. On seen objects, it achieves 97.2% mass classification accuracy, 98.8% friction-coefficient classification accuracy, and a stiffness mean absolute percentage error (MAPE) of 5.51%. On held-out objects from known categories, it achieves 87.5% mass accuracy, 97.5% friction-coefficient accuracy, and a stiffness MAPE of 9.08%. We transfer ViTacPhys from the human domain to the robot domain using limited robot teleoperation data, robot-style video augmentation, and human demonstrations with matched actions, and deploy it as an online module for adaptive grasping. The resulting physical-property-conditioned policy achieves total grasping success rates of 95.0% on in-distribution objects and 83.4% on out-of-distribution objects. For out-of-distribution objects successfully grasped by both methods, its force profiles are more consistent with human teleoperation than those produced by ACT. These results demonstrate the feasibility of explicitly estimating and conditioning on object physical properties for real-world adaptive grasping.",
  "published": "2026-08-21",
  "updated": "2026-08-21",
  "year": "2026",
  "authors": [
   "Yiwen Liu",
   "Yujun Zhu",
   "Kui Jia",
   "Zhao Liao",
   "Yangwei You",
   "Shuaijun Wang"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "ViTacPhys, a visual-tactile framework and data acquisition system that estimates object mass and friction-coefficient classes, together with continuous stiffness, from human manipulation demonstrations, demonstrates the feasibility of explicitly estimating and conditioning on object physical properties for real-world adaptive grasping.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yiwen Liu",
    "id": "2181653051",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Yujun Zhu",
    "id": "2109369565",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Kui Jia",
    "id": "2370507",
    "h_index": 63,
    "papers": 250
   },
   {
    "name": "Zhao Liao",
    "id": "2459116004",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yangwei You",
    "id": "2383111339",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Shuaijun Wang",
    "id": "2242199635",
    "h_index": 2,
    "papers": 3
   }
  ],
  "comment": "11 pages, 7 figures. Project page: https://vitacphys.github.io/ViTacPhys/",
  "topics": [
   "dexterous-manipulation",
   "egocentric-data",
   "tactile",
   "data-teleop",
   "video-generation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.21355v1",
  "pdf_url": "https://arxiv.org/pdf/2608.21355v1",
  "html_url": "https://arxiv.org/html/2608.21355v1",
  "code_url": "https://vitacphys.github.io/ViTacPhys/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2608.21332",
  "slug": "anatomy-informed-neural-networks-encoding-anatomic-priors-in-loss-and",
  "title": "Anatomy-Informed Neural Networks: Encoding Anatomic Priors in Loss and Architecture, with an SE(3) Formulation of Guidewire-Induced Aortoiliac Deformation",
  "abstract": "Deep-learning models of anatomy can be numerically plausible yet anatomically impossible, and they generalize poorly when data are scarce. We introduce Anatomy-Informed Neural Networks (AINN), in which soft anatomic priors enter as penalty terms in the loss (e.g., a branching penalty that treats a renal transplant artery off the iliac instead of the aorta as unexpected rather than impossible), in direct analogy to a physics-informed neural network, and hard anatomic priors (e.g., continuity of the vessel) are built into the architecture and state representation, making such invalid predictions impossible by construction wherever the prior admits architectural enforcement. We develop it on a clinical test case with limited data: how the aortoiliac tree deforms when a stiff wire is introduced endoluminally. This is important to contemporary aortic surgery and will matter to autonomous endovascular navigation. We lift the vessel centerline and the wire path from R^3 to curves of frames in the Lie group SE(3), and couple a Cosserat-rod wire to a tortuosity-modulated, anatomically anchored vessel through a unilateral lumen-contact inequality. The prediction is a constrained minimizer of the coupled elastic energy, with contact forces as its Lagrange multipliers. Supervision is a Wasserstein-2 optimal-transport loss between the predicted projection through the C-arm geometry and the observed angiogram, so a 2D angiogram can train a 3D prediction. The kinematics, loss and projection are verified against known ground truth; the mechanics solver only against its own optimality conditions, and predicted displacement is not yet mesh-converged. Here, no network is trained. Future work will transfer this in silico model to real CT scans and test whether it improves predictive accuracy and reduces the training data required.",
  "published": "2026-08-21",
  "updated": "2026-08-21",
  "year": "2026",
  "authors": [
   "David P. Stonko"
  ],
  "author_count": 1,
  "categories": [
   "cs.AI",
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.AI",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "An Anatomy-Informed Neural Networks (AINN) is introduced, in which soft anatomic priors enter as penalty terms in the loss, in direct analogy to a physics-informed neural network, and hard anatomic priors are built into the architecture and state representation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "D. Stonko",
    "id": "6068092",
    "h_index": 17,
    "papers": 116
   }
  ],
  "comment": "42 pages, 10 figures, 4 tables",
  "topics": [
   "navigation",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.21332v1",
  "pdf_url": "https://arxiv.org/pdf/2608.21332v1",
  "html_url": "https://arxiv.org/html/2608.21332v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.21330",
  "slug": "nesam-neuro-symbolic-kinodynamics-with-soil-adaptation-for-off-road-mo",
  "title": "NeSAM: Neuro-Symbolic Kinodynamics with Soil Adaptation for Off-Road Mobility",
  "abstract": "Accurate prediction of off-road vehicle motion over deformable terrain remains challenging because sinkage, slip, and traction vary with local soil conditions. Existing learning-based kinodynamic models directly approximate vehicle-terrain interactions from data but do not explicitly represent soil mechanics and offer limited physical interpretability. To address these limitations, we present NeSAM, a neuro-symbolic framework that combines differentiable Bekker-Wong terramechanics with learned terrain representations and a Transformer-based residual dynamics model for long-horizon, six degree-of-freedom kinodynamic prediction. The terramechanics component models soil-dependent interaction forces, while the residual model corrects discrepancies between the analytical prediction and the observed vehicle dynamics. NeSAM further estimates physically meaningful soil parameters from terrain observations and updates them online using an extended Kalman filter. We evaluate NeSAM in Verti-Bench, a simulator built on the Chrono multiphysics engine, and validate its performance on a physical Verti-4-Wheeler platform. NeSAM improves prediction accuracy by up to 30% in simulation and 29% on real-world data relative to the strongest compared baselines. When integrated with a close-loop navigation controller, NeSAM further improves traversal success rate through online soil adaptation while reduces Hausdorff distance to the reference trajectory by 69.4%, indicating improved trajectory tracking accuracy.",
  "published": "2026-08-21",
  "updated": "2026-08-21",
  "year": "2026",
  "authors": [
   "Chenhui Pan",
   "Tong Xu",
   "Francesco Cancelliere",
   "Xuesu Xiao"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Chenhui Pan",
    "id": "2084643982",
    "h_index": 8,
    "papers": 17
   },
   {
    "name": "Tong Xu",
    "id": "2320264790",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Francesco Cancelliere",
    "id": "2375384741",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Xuesu Xiao",
    "id": "2320187442",
    "h_index": 6,
    "papers": 19
   }
  ],
  "comment": "",
  "topics": [
   "sim2real",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.21330v1",
  "pdf_url": "https://arxiv.org/pdf/2608.21330v1",
  "html_url": "https://arxiv.org/html/2608.21330v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.21319",
  "slug": "unified-branch-and-bound-search-for-the-steiner-traveling-salesman-pro",
  "title": "Unified Branch-and-Bound Search for the Steiner Traveling Salesman Problem on Graphs of Convex Sets",
  "abstract": "We formalize the Steiner Traveling Salesman Problem (Steiner-TSP) on Graphs of Convex Sets (GCS), which seeks a minimum-cost closed trajectory through required convex sets while allowing optional transit vertices and revisits. To explore the resulting infinite solution space, we propose a unified branch-and-bound search over rooted walk prefixes. Additive lower-bound-graph costs bound committed prefixes, while a cut-separated connected-flow relaxation lower-bounds the residual cost of visiting every remaining target and returning to the root. Under a uniform positive-cost assumption, best-first traversal terminates after finitely many expansions on every feasible instance without an initial incumbent, whereas depth-first traversal does so once a finite incumbent is available. For a user-specified factor $\u03b5\\geq1$, a global lower bound certifies that either strategy's incumbent cost is at most $\u03b5$ times the global optimum. We further demonstrate joint sensing-mode, visitation-order, and continuous-trajectory selection for a mobile-manipulator inspection task, including action precedences expressed in linear temporal logic over finite traces (LTL$_f$). Both traversal strategies find feasible solutions on all benchmark instances within 30s with mean certified optimality gaps of 28.1% and 29.7%, respectively, whereas two recent baselines succeed on only about half of the instances",
  "published": "2026-08-21",
  "updated": "2026-08-21",
  "year": "2026",
  "authors": [
   "Jingtao Tang",
   "Hang Ma"
  ],
  "author_count": 2,
  "categories": [
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.AI",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The Steiner Traveling Salesman Problem is formalized on Graphs of Convex Sets (GCS), which seeks a minimum-cost closed trajectory through required convex sets while allowing optional transit vertices and revisits, and a unified branch-and-bound search over rooted walk prefixes is proposed.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "J. Tang",
    "id": "2118351389",
    "h_index": 7,
    "papers": 16
   },
   {
    "name": "Hang Ma",
    "id": "2275194451",
    "h_index": 5,
    "papers": 11
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.21319v1",
  "pdf_url": "https://arxiv.org/pdf/2608.21319v1",
  "html_url": "https://arxiv.org/html/2608.21319v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.21290",
  "slug": "vt-muse-multimodal-unified-sequential-visuotactile-representation-lear",
  "title": "VT-MUSE: Multimodal Unified Sequential Visuotactile Representation Learning for Manipulation",
  "abstract": "We propose VT-MUSE, a Multimodal Unified SEquential representation learning framework for visuotactilemanipulation. Existing approaches often encode visual and tactile observations independently before fusion, limiting their ability to capture fine-grained cross-modal dependencies. Moreover, most methods focus on observations at the current time step and overlook the temporal evolution of contact. VT-MUSE addresses both limitations through a two-stage representation learning framework. In Stage I, modality specific encoders are jointly adapted via cross-modal temporal alignment and masked-view consistency. In Stage II, a conditional variational latent model processes masked visual sequences together with full tactile histories. Auxiliary decoders reconstruct the masked recent visual observations and predict tactile depth changes, encouraging the latent representation to retain both global visual context and local contact dynamics. The learned representation is subsequently integrated into a lightweight Transformer policy through gated cross-attention. On the simulation benchmark, VT-MUSE outperforms the strongest baseline evaluated on all tasks by 11 percentage points and also achieves substantial improvements in real-world experiments.",
  "published": "2026-08-21",
  "updated": "2026-08-21",
  "year": "2026",
  "authors": [
   "Congsheng Xu",
   "Qiaochu Yang",
   "Fangyuan Shi",
   "Yifan Han",
   "Baijun Chen",
   "Yiming Wang",
   "Haonan Zhao",
   "Daolin Ma",
   "Xiaokang Yang",
   "Hesheng Wang"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "VT-MUSE is proposed, a Multimodal Unified SEquential representation learning framework for visuotactilemanipulation that outperforms the strongest baseline evaluated on all tasks and also achieves substantial improvements in real-world experiments.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Congsheng Xu",
    "id": "2276446441",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Qiaochu Yang",
    "id": "2338023932",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Fangyuan Shi",
    "id": "2459116667",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yifan Han",
    "id": "2323619357",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Baijun Chen",
    "id": "2370950715",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Yiming Wang",
    "id": "2459150699",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Haonan Zhao",
    "id": "2371305933",
    "h_index": 3,
    "papers": 14
   },
   {
    "name": "Daolin Ma",
    "id": "2384395304",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Xiaokang Yang",
    "id": "2362515842",
    "h_index": 6,
    "papers": 22
   },
   {
    "name": "Hesheng Wang",
    "id": "2459239418",
    "h_index": 0,
    "papers": 0
   }
  ],
  "comment": "",
  "topics": [
   "tactile",
   "sim2real"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.21290v1",
  "pdf_url": "https://arxiv.org/pdf/2608.21290v1",
  "html_url": "https://arxiv.org/html/2608.21290v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.21276",
  "slug": "the-coastline-as-a-structural-constraint-harnessing-scene-geometry-for",
  "title": "The Coastline as a Structural Constraint: Harnessing Scene Geometry for Autonomous Surface Vessel Localization",
  "abstract": "Coastal environments contain rich, largely unexploited geometric structure capable of providing globally referenced localization cues. In this work, we present two complementary localization frameworks that exploit shoreline and water-surface geometry for GPS-denied autonomous surface vessel localization. The first framework leverages LiDAR observations of the water surface to estimate roll, pitch, and heave (vertical motion), while recovering global position and heading through direct registration of shoreline observations against a satellite-derived coastline map. The second framework relies solely on passive imagery to detect the shoreline and horizon through semantic segmentation. Using the proposed coastal scene geometry, shoreline distance is inferred from monocular imagery. Shoreline observations are accumulated into short-duration local submaps, registered against the same satellite-derived coastline map, and fused within a hierarchical factor graph. Evaluated across three real-world coastal datasets, the LiDAR pipeline consistently improves trajectory accuracy over standard baselines, while the monocular architecture maintains bounded long-term drift. In addition, we establish that modern zero-shot foundation models can reliably extract shoreline observations across diverse coastal environments. Together, these results demonstrate that coastal geometry provides a powerful and dependable source of globally referenced information for GPS-denied maritime localization.",
  "published": "2026-08-21",
  "updated": "2026-08-21",
  "year": "2026",
  "authors": [
   "Derek R. Benham",
   "Joshua G. Mangelson"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Derek Benham",
    "id": "2197580631",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Joshua G. Mangelson",
    "id": "2307282760",
    "h_index": 1,
    "papers": 8
   }
  ],
  "comment": "22 pages, 13 figures, 7 tables",
  "topics": [
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.21276v1",
  "pdf_url": "https://arxiv.org/pdf/2608.21276v1",
  "html_url": "https://arxiv.org/html/2608.21276v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.21247",
  "slug": "just-noticeable-difference-modeling-for-token-compression-in-vision-la",
  "title": "Just Noticeable Difference Modeling for Token Compression in Vision-Language-Action Models",
  "abstract": "Token compression has become a key technique for reducing the inference cost of large foundation models, with approaches such as token pruning and KV-cache reuse widely adopted in vision-language models and recently explored for embodied agents. In embodied agents, tokens not only support perception and semantic understanding but also directly affect latency-sensitive closed-loop robot action prediction. Existing schemes typically guide compression using redundancy or importance cues, such as visual similarity, attention scores, and saliency. However, these cues only indirectly measure the key factor for safe compression: how much a token can change before causing an unacceptable deviation in downstream actions. This receiver-dependent tolerance is closely related to the principle of just noticeable difference (JND). Classical JND characterizes signal tolerance in the human visual system, while machine-oriented JND extends this concept to downstream machine responses. Building on this progression, we introduce Action-JND, which extends JND modeling to embodied perception by defining noticeability through the language-conditioned action response of a vision-language-action (VLA) policy in closed-loop control. A token change is considered admissible only when the induced action deviation remains within a tolerated margin. To realize this concept, we develop a lightweight token-wise JND estimator in deep visual-feature space to predict the maximum tolerable perturbation while preserving policy responses. The resulting action-tolerance score serves as a plug-and-play criterion for VLA compression paradigms, including stale-KV reuse and token pruning, prioritizing action-tolerant tokens for compression. Experiments on the LIBERO benchmark with OpenVLA and OpenVLA-OFT demonstrate that Action-JND consistently improves compression reliability, especially under aggressive compression ratios.",
  "published": "2026-08-21",
  "updated": "2026-08-21",
  "year": "2026",
  "authors": [
   "Zhuoyuan Li",
   "Rui Zhao",
   "Jin Wang",
   "Hanwei Zhu",
   "Cong Zhang",
   "Giuseppe Valenzise",
   "Weisi Lin",
   "Kin-Man Lam"
  ],
  "author_count": 8,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Action-JND is introduced, which extends JND modeling to embodied perception by defining noticeability through the language-conditioned action response of a vision-language-action (VLA) policy in closed-loop control, and develops a lightweight token-wise JND estimator in deep visual-feature space to predict the maximum tolerable perturbation while preserving policy responses.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhuoyuan Li",
    "id": "2294671187",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Rui Zhao",
    "id": "2397559157",
    "h_index": 1,
    "papers": 9
   },
   {
    "name": "Jin Wang",
    "id": "2459173397",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Hanwei Zhu",
    "id": "2376903753",
    "h_index": 2,
    "papers": 12
   },
   {
    "name": "Cong Zhang",
    "id": "2447990283",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Giuseppe Valenzise",
    "id": "2266212073",
    "h_index": 4,
    "papers": 25
   },
   {
    "name": "Weisi Lin",
    "id": "2334525697",
    "h_index": 4,
    "papers": 17
   },
   {
    "name": "K. Lam",
    "id": "144847940",
    "h_index": 50,
    "papers": 341
   }
  ],
  "comment": "15 pages, 5 figures",
  "topics": [
   "vla",
   "foundation-pretraining",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.21247v1",
  "pdf_url": "https://arxiv.org/pdf/2608.21247v1",
  "html_url": "https://arxiv.org/html/2608.21247v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.21204",
  "slug": "beyond-imitation-self-improving-robot-policies-via-off-policy-q-planni",
  "title": "Beyond Imitation: Self-Improving Robot Policies via Off-Policy Q-Planning",
  "abstract": "Behaviour Cloning (BC) has driven remarkable progress in robot manipulation, yet it is fundamentally limited by its inability to self-improve: a policy that fails cannot learn from that failure without additional human demonstrations. Reinforcement Learning fine-tuning offers a path to self-improvement but has proven difficult to scale to the multi-billion-parameter models underpinning modern robot policies. We propose Q-Planning, which equips a large visuomotor BC policy with a small off-policy Q-function. Because a Q-function estimates value rather than imitates actions, it can be trained on the same successful demonstrations as the BC policy and later absorb both successful and failed deployment rollouts, an asymmetry BC does not have. We exploit this asymmetry to enable value-guided action selection at inference (a single-step Q-weighted average over BC draws) and online self-improvement that fine-tunes only the Q-function, leaving the BC weights untouched. On LIBERO and bimanual RoboTwin, ten iterations of self-improvement lift every benchmark score we tested (LIBERO-10 93% to 99%, RoboTwin 83.8% to 91.4%) and shorten successful episodes on the near-ceiling suites (LIBERO-Object, LIBERO-Goal). On two contact-rich bimanual real-robot tasks, the same loop (BC frozen, no human intervention) improves purely from its own deployment rollouts: stack-cups 40% to 90% and insert-wallet 25% to 80% in five iterations, whereas SFT on successful rollouts alone stalls at 55% and 30%. Under an identical online budget Q-Planning is the only method, among Best-of-N, filtered SFT, IBRL, DSRL, and DAWR, that improves stably from failures without training an auxiliary actor.",
  "published": "2026-08-21",
  "updated": "2026-08-21",
  "year": "2026",
  "authors": [
   "Varun Giridhar",
   "Anant Khandelwal",
   "Jeremy A. Collins",
   "Ignat Georgiev",
   "Animesh Garg"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Q-Planning is proposed, which equips a large visuomotor BC policy with a small off-policy Q-function and exploits this asymmetry to enable value-guided action selection at inference and online self-improvement that fine-tunes only the Q-function, leaving the BC weights untouched.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Varun Giridhar",
    "id": "2309248227",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Anant Khandelwal",
    "id": "2261493723",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Jeremy A. Collins",
    "id": "2332458424",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Ignat Georgiev",
    "id": "2070720942",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Animesh Garg",
    "id": "2282540862",
    "h_index": 7,
    "papers": 19
   }
  ],
  "comment": "Project page with videos: https://varungiridhar.github.io/qplanning/",
  "topics": [
   "egocentric-data",
   "tactile",
   "imitation-diffusion",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.21204v1",
  "pdf_url": "https://arxiv.org/pdf/2608.21204v1",
  "html_url": "https://arxiv.org/html/2608.21204v1",
  "code_url": "https://varungiridhar.github.io/qplanning/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2608.21175",
  "slug": "srl-mpc-shape-aware-reinforcement-learned-model-predictive-control",
  "title": "SRL-MPC: Shape-Aware Reinforcement Learned Model Predictive Control",
  "abstract": "Safe and efficient shape-aware navigation in heterogeneous crowds and robot fleets remains challenging. Traditional approaches often assume homogeneous robots, sparse workspaces, simplified geometry, offline computation, or handcrafted parameters to make the problem tractable, which limits their deployment in dense crowd scenarios. Toward this end, we propose Shape-Aware Reinforcement Learned Model Predictive Control (SRL-MPC), a method for safe, efficient, and adaptive navigation in crowds with heterogeneous shapes without geometry simplification. To encode shape-aware safety, we formulate high-order control barrier function (HOCBF) constraints from geometric separation features (GSFs) based on support function transformation. A reinforcement learning (RL) framework then learns a neural policy that reads GSFs and outputs real-time MPC parameter updates, enabling the MPC solver to adapt to neighboring crowd geometries. The key advantage of SRL-MPC is that it preserves the safety structure and generalizability of MPC while integrating the adaptability and intelligence of RL. Experiments in randomized crowd scenarios with arbitrary shaped robot fleets demonstrate the effectiveness, scalability, and robustness of SRL-MPC. The results show that SRL-MPC substantially outperforms representative baselines in safety and adaptability. Project website: https://hanruihua.github.io/srl_mpc_project/",
  "published": "2026-08-21",
  "updated": "2026-08-21",
  "year": "2026",
  "authors": [
   "Ruihua Han",
   "Rui Gao",
   "Zhe Liu",
   "Xinyi Wang",
   "Chang Chen",
   "Shuai Wang",
   "Qi Hao",
   "Jia Pan",
   "Hengshuang Zhao"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Shape-Aware Reinforcement Learned Model Predictive Control is proposed, a method for safe, efficient, and adaptive navigation in crowds with heterogeneous shapes without geometry simplification that preserves the safety structure and generalizability of MPC while integrating the adaptability and intelligence of RL.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ruihua Han",
    "id": "2332361459",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Rui Gao",
    "id": "2267606474",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Zhe Liu",
    "id": "2404004658",
    "h_index": 2,
    "papers": 14
   },
   {
    "name": "Xinyi Wang",
    "id": "2405674653",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Chang Chen",
    "id": "2459173346",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Shuai Wang",
    "id": "2224173586",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Qi Hao",
    "id": "2265493005",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Jia Pan",
    "id": "2258308608",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Hengshuang Zhao",
    "id": "2310758544",
    "h_index": 9,
    "papers": 27
   }
  ],
  "comment": "",
  "topics": [
   "rl-control",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.21175v1",
  "pdf_url": "https://arxiv.org/pdf/2608.21175v1",
  "html_url": "https://arxiv.org/html/2608.21175v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.21083",
  "slug": "teaching-is-a-process-the-toss-framework-for-modeling-human-teaching-d",
  "title": "Teaching is a Process: The TOSS Framework for Modeling Human Teaching Decisions in Human-Interactive Robot Learning",
  "abstract": "Successful Human-Robot Teaching assumes alignment between robot processing needs and human teaching intent. To better understand this alignment, this work seeks to uncover the underlying logic that humans intuitively apply when teaching. Through an exploratory, bottom-up study with N=34, participants observing two distinct robot Reinforcement Learning (RL) scenarios, we analyze 204 intuitive teaching responses across early, middle, and late learning phases. Results reveal that teaching decisions consist of a nuanced, interconnected network of Triggers (situational catalysts), Objectives (subjective teaching targets), Signals (communicative acts), and Strategies (high-level governance) in which teachers spontaneously adopt diverse roles, acting as coaches, engineers, or designers and prioritize different objectives. Based on these results, we introduce the TOSS Framework, which conceptualizes Human-Robot teaching as a procedural loop between robot behavior and human teaching actions, in which human teaching decisions are modeled as Trigger-Signal responses modulated by teaching Objectives and Strategies. It provides future research with an openly accessible dataset and a theoretical foundation for a) understanding teaching decisions and b) simulating realistic oracles as well as c) designing human-centered teaching settings and novel robot learning algorithms that go beyond the constraints of current robot learning settings.",
  "published": "2026-08-21",
  "updated": "2026-08-21",
  "year": "2026",
  "authors": [
   "Bernhard Hilpert",
   "Kim Baraka",
   "Joost Broekens"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.HC"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "The TOSS Framework is introduced, which conceptualizes Human-Robot teaching as a procedural loop between robot behavior and human teaching actions, in which human teaching decisions are modeled as Trigger-Signal responses modulated by teaching Objectives and Strategies.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Bernhard Hilpert",
    "id": "1455857216",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Kim Baraka",
    "id": "2266517202",
    "h_index": 5,
    "papers": 30
   },
   {
    "name": "J. Broekens",
    "id": "2189563094",
    "h_index": 2,
    "papers": 10
   }
  ],
  "comment": "",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.21083v1",
  "pdf_url": "https://arxiv.org/pdf/2608.21083v1",
  "html_url": "https://arxiv.org/html/2608.21083v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2608.21056",
  "slug": "ff-mpcc-high-speed-agile-formation-flight-with-model-predictive-contou",
  "title": "FF-MPCC: High-speed Agile Formation Flight with Model Predictive Contouring Control",
  "abstract": "Flying in a prescribed formation in an agile manner remains a challenging problem in the field of UAVs, particularly when following highly-demanding trajectories that require flight at platform limits. We address this problem by proposing a novel decentralized approach to formation flight along a given path that integrates formation maintenance into the MPCC framework, allowing UAVs to adapt their progression along complex paths while respecting individual dynamic constraints and maintaining the desired formation. To this end, we introduce a novel reparametrization and synchronization method for dynamic formation geometries together with a decentralized approach to determine the desired positions for the individual UAVs. The proposed approach allows the formation to coordinate high-speed path following without compromising formation integrity. The proposed approach is validated through extensive simulation and real-world experiments involving scenarios with varying complexity of paths and changes of required formation shape on the fly. In comparison to time-parameterized trajectory tracking, we demonstrate improved formation maintenance by 65% in high-speed flight with velocities up to 21 m/s, while achieving comparable times required to reach the goal.",
  "published": "2026-08-21",
  "updated": "2026-08-21",
  "year": "2026",
  "authors": [
   "Aditya Dandwate",
   "Vit Kratky",
   "Parakh M. Gupta",
   "Martin Saska",
   "Robert Penicka"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work introduces a novel reparametrization and synchronization method for dynamic formation geometries together with a decentralized approach to determine the desired positions for the individual UAVs, allowing UAVs to adapt their progression along complex paths while respecting individual dynamic constraints and maintaining the desired formation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Aditya Dandwate",
    "id": "2372797262",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "V. Kr\u00e1tk\u00fd",
    "id": "31821861",
    "h_index": 14,
    "papers": 30
   },
   {
    "name": "Parakh M. Gupta",
    "id": "151168582",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "M. Saska",
    "id": "1719925",
    "h_index": 42,
    "papers": 280
   },
   {
    "name": "Robert P\u011bni\u010dka",
    "id": "3255874",
    "h_index": 25,
    "papers": 61
   }
  ],
  "comment": "",
  "topics": [
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.21056v1",
  "pdf_url": "https://arxiv.org/pdf/2608.21056v1",
  "html_url": "https://arxiv.org/html/2608.21056v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.21035",
  "slug": "taper-probabilistic-recovery-of-sparse-task-precedence-graphs-from-a-h",
  "title": "TaPeR: Probabilistic Recovery of Sparse Task Precedence Graphs from a Handful of Demonstrations",
  "abstract": "Long-horizon manipulation tasks are often only partially ordered. For example, when assembling an electronic device, the battery and circuit board may be installed in either order, but both must be in place before the enclosure is closed. Recovering such dependencies enables robots to flexibly reorder subtasks while preserving task validity. Existing approaches typically infer task structure from human demonstrations using both temporal and symbolic supervision. However, symbolic predicates require explicit grounding, which is difficult to obtain in realistic settings. In this work, we present an approach for extracting task dependency structures from demonstrations using only simple kinematic graphs and distributions over relative object poses. From these representations, our method estimates pairwise task-step-dependency probabilities and uses them to initialize the edge weights of a precedence graph. We then introduce a filtering pipeline that converts this graph of probability estimates into the final task dependency graph. We evaluate our approach on an existing benchmark and on a new dataset comprising longer tasks with more complex dependencies. We find that our method recovers more accurate task structures from fewer demonstrations than the baselines. Finally, we demonstrate that the inferred graphs can be used to generate multiple valid robotic execution orders for the same task.",
  "published": "2026-08-21",
  "updated": "2026-08-21",
  "year": "2026",
  "authors": [
   "Adrian R\u00f6fer",
   "Karla Stepanova",
   "Abhinav Valada"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work presents an approach for extracting task dependency structures from demonstrations using only simple kinematic graphs and distributions over relative object poses, and finds that it recovers more accurate task structures from fewer demonstrations than the baselines.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Adrian R\u00f6fer",
    "id": "1753652042",
    "h_index": 5,
    "papers": 19
   },
   {
    "name": "Karla St\u00e9p\u00e1nov\u00e1",
    "id": "2276789452",
    "h_index": 2,
    "papers": 11
   },
   {
    "name": "A. Valada",
    "id": "2131132945",
    "h_index": 17,
    "papers": 87
   }
  ],
  "comment": "8 pages, 5 figures, 3 tables, under review",
  "topics": [
   "egocentric-data"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.21035v1",
  "pdf_url": "https://arxiv.org/pdf/2608.21035v1",
  "html_url": "https://arxiv.org/html/2608.21035v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.21032",
  "slug": "roadside-cooperative-autonomous-driving-from-data-platform-to-vision-l",
  "title": "Roadside-Cooperative Autonomous Driving: From Data Platform to Vision-Language End-to-End Reasoning",
  "abstract": "Vehicle-to-Everything (V2X) cooperation enables beyond-line-of-sight perception, mitigating occlusions in single-vehicle sensing. However, existing V2X benchmarks provide limited support for closed-loop evaluation and language-grounded supervision, hindering the development of vision-language models (VLMs) for end-to-end cooperative driving. To address these limitations, we introduce V2XBench, a simulation platform featuring synchronized ego--roadside sensing and closed-loop evaluation, together with Chat-V2XBench, a progressively structured VQA dataset for cooperative reasoning. Building upon this benchmark infrastructure, we propose AURORA, an end-to-end cooperative driving framework. Equipped with a dual-view perception architecture, AURORA mitigates spatial and semantic discrepancies across ego and roadside viewpoints through a query-level Cross-View Query Alignment and Fusion (CQAF) module. Leveraging the resulting unified tokens, a LoRA-adapted VLM bridges semantic reasoning and generative trajectory planning. Extensive closed-loop evaluations on V2XBench demonstrate that AURORA achieves state-of-the-art performance in heavily occluded scenarios, with a Route Completion rate of 98.21% and a Driving Score of 76.02, while requiring low roadside communication bandwidth. Ultimately, this work pioneers an extensible V2X--VLM paradigm, paving the way for next-generation cooperative autonomous driving.",
  "published": "2026-08-21",
  "updated": "2026-08-21",
  "year": "2026",
  "authors": [
   "Yitao Xu",
   "Tong Wu",
   "Yiyan Wu",
   "Guoji Xu",
   "Yanbo Jiang",
   "Jiahao Wang",
   "Zehong Ke",
   "Junkai Jiang",
   "Fang Zhang",
   "Jianqiang Wang"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes AURORA, an end-to-end cooperative driving framework equipped with a dual-view perception architecture that pioneers an extensible V2X--VLM paradigm, paving the way for next-generation cooperative autonomous driving.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yitao Xu",
    "id": "2455640621",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Tong Wu",
    "id": "2449168703",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yiyang Wu",
    "id": "2455951948",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Guoji Xu",
    "id": "2294301335",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Yanbo Jiang",
    "id": "2267906542",
    "h_index": 4,
    "papers": 17
   },
   {
    "name": "Jiahao Wang",
    "id": "2349393345",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Zehong Ke",
    "id": "2267749889",
    "h_index": 3,
    "papers": 15
   },
   {
    "name": "Junkai Jiang",
    "id": "2218453614",
    "h_index": 7,
    "papers": 20
   },
   {
    "name": "Fang Zhang",
    "id": "2355029588",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Jianqiang Wang",
    "id": "2288181854",
    "h_index": 3,
    "papers": 10
   }
  ],
  "comment": "15 pages, 5 figures, under review",
  "topics": [
   "navigation",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.21032v1",
  "pdf_url": "https://arxiv.org/pdf/2608.21032v1",
  "html_url": "https://arxiv.org/html/2608.21032v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.21031",
  "slug": "physcap-grounding-code-as-policy-agent-with-physics-informed-explorati",
  "title": "PhysCaP: Grounding Code-as-Policy Agent with Physics-Informed Exploration",
  "abstract": "We present PhysCaP, a Physics-Informed Code-as-Policy agent for active perception in robotic manipulation. While vision-language-action policies excel at imitating demonstrations, they rely on passive observation and fail to infer latent physical properties critical for manipulation. PhysCaP augments code-as-policy frameworks with a physics-informed exploration layer that enables explicit information-seeking through interaction. It introduces training-free physical property extraction modules that estimate object mass and stiffness from robot proprioception without additional sensors. To balance exploration costs and the efficiency of information obtained, PhysCaP employs a dual-agent design: a Planner that decides when to explore and when to stop, and a Prioritizer that filters implausible interactions and ranks the remainder using a heuristic priority score, enabling efficient, targeted exploration. We evaluate PhysCaP on real-world tabletop manipulation tasks (searching for hidden objects, detecting empty cans, and finding ripe avocados) and a simulated task in LIBERO. The results show that existing passive and naive interactive baselines either fail when physical properties are hidden or over-explore, whereas PhysCaP achieves comparable performance with fewer interactions and reduced execution time. Ablation studies further validate the effectiveness of the proposed physical property extraction modules. Project page: https://physcap.github.io",
  "published": "2026-08-21",
  "updated": "2026-08-21",
  "year": "2026",
  "authors": [
   "Chen-Yu Lin",
   "Jing-Wen Chen",
   "Hsueh-En Chang",
   "Hung-An Chen",
   "Sheng-Hsun Chang",
   "Chi-Pin Huang",
   "Fu-En Yang",
   "Min-Hung Chen",
   "Yi-Ting Chen",
   "Yu-Chiang Frank Wang",
   "Shao-Hua Sun"
  ],
  "author_count": 11,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "PhysCaP augments code-as-policy frameworks with a physics-informed exploration layer that enables explicit information-seeking through interaction, and introduces training-free physical property extraction modules that estimate object mass and stiffness from robot proprioception without additional sensors.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Chen-Yu Lin",
    "id": "2459152921",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Jing-Wen Chen",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Hsueh-En Chang",
    "id": "2459141391",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Hung-An Chen",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Sheng-Hsun Chang",
    "id": "2459173112",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Chi-Pin Huang",
    "id": "2268723514",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "Fu-En Yang",
    "id": "41015732",
    "h_index": 15,
    "papers": 39
   },
   {
    "name": "Min-Hung Chen",
    "id": "2239064269",
    "h_index": 7,
    "papers": 21
   },
   {
    "name": "Yi-Ting Chen",
    "id": "2459145322",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Y. Wang",
    "id": "2332289950",
    "h_index": 7,
    "papers": 16
   },
   {
    "name": "Shao-Hua Sun",
    "id": "2386779729",
    "h_index": 2,
    "papers": 3
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.21031v1",
  "pdf_url": "https://arxiv.org/pdf/2608.21031v1",
  "html_url": "https://arxiv.org/html/2608.21031v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.20962",
  "slug": "hybrid-roller-jamming-gripper-for-object-acquisition-and-retention-und",
  "title": "Hybrid Roller-Jamming Gripper for Object Acquisition and Retention Under Pose Uncertainty",
  "abstract": "In household manipulation, pose uncertainty often results in off-centre or partial initial contact, making reliable object acquisition difficult. Roller-based grippers can actively draw objects inward but often provide limited post-capture stability, whereas granular-jamming grippers require sufficient contact before jamming to achieve strong retention. This paper presents a hybrid roller-jamming gripper that integrates active object intake and post-capture retention within a single gripper. The proposed gripper uses inward roller rotation to increase contact and draw the object toward the gripper centre, followed by vacuum-induced granular jamming to stiffen the rollers and stabilise the grasp. The paper also presents a simplified geometric analysis of the gripper and a bench-level characterisation of the prototype's force capability. The gripper prototype was mounted on a 7-DoF robotic arm and evaluated using eight test objects. Furthermore, controlled planar position and orientation offsets were applied, with each condition repeated three times. The main evaluation comprised 840 grasp trials, including 216 planar-offset trials and 624 orientation-offset trials. Overall, the gripper succeeded in 812/840 trials: 215/216 planar-offset trials and 597/624 orientation-offset trials. The ablation evaluation comprised 162 trials on three objects. The roller-only and jamming-only conditions achieved 54/81 and 24/81 successes, respectively, showing their different contributions. These results provide initial mechanism-level evidence that hybrid roller-jamming is a promising strategy for improving acquisition and retention after imperfect first contact.",
  "published": "2026-08-21",
  "updated": "2026-08-21",
  "year": "2026",
  "authors": [
   "Yijie Ren",
   "Guillaume Gourmelen",
   "Hiroyasu Iwata"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper presents a hybrid roller-jamming gripper that integrates active object intake and post-capture retention within a single gripper, followed by vacuum-induced granular jamming to stiffen the rollers and stabilise the grasp.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yijie Ren",
    "id": "2459151689",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Guillaume Gourmelen",
    "id": "1393847726",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "H. Iwata",
    "id": "2076505",
    "h_index": 23,
    "papers": 379
   }
  ],
  "comment": "11 pages, 16 figures, under review",
  "topics": [
   "dexterous-manipulation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.20962v1",
  "pdf_url": "https://arxiv.org/pdf/2608.20962v1",
  "html_url": "https://arxiv.org/html/2608.20962v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.20948",
  "slug": "neural-primitive-an-efficient-end-to-end-local-planner-with-primitive",
  "title": "Neural-Primitive: An Efficient End-to-end Local Planner with Primitive-based Imitation Learning for Autonomous Flight",
  "abstract": "Autonomous flight in unknown cluttered environments is hindered by the computation-quality-memory trilemma of onboard trajectory generation. In this paper, we propose an efficient end-to-end local planner via imitation learning. A lightweight offline-primitive-based dataset collection framework is designed to produce safe and high-quality trajectory primitives in non-convex environments. A compact neural network directly maps sensory inputs to polynomial coefficients that inherently encode higher-order dynamical information. The learned policy generates smooth, empirically collision-free and dynamically feasible trajectories in real time without back-end solving. It achieves ultra-fast computation (below 1ms on a standard desktop and average 3.68ms during onboard flight), while maintaining low onboard memory requirements (less than 1.5MiB). Extensive simulation benchmarks demonstrate superiority in both planning latency and target-reaching progress quality. Zero-shot deployment in real-world experiments further validates the robust sim-to-real transfer capability of the proposed method.",
  "published": "2026-08-21",
  "updated": "2026-08-21",
  "year": "2026",
  "authors": [
   "Zhitao Liu",
   "Guangtong Xu",
   "Zihan Wang",
   "Jialiang Hou",
   "Chao Xu",
   "Fei Gao"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A lightweight offline-primitive-based dataset collection framework is designed to produce safe and high-quality trajectory primitives in non-convex environments and demonstrates superiority in both planning latency and target-reaching progress quality.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhitao Liu",
    "id": "2349395992",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Guangtong Xu",
    "id": "2260603451",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Zihan Wang",
    "id": "2279762287",
    "h_index": 4,
    "papers": 16
   },
   {
    "name": "Jialiang Hou",
    "id": "2157186184",
    "h_index": 3,
    "papers": 13
   },
   {
    "name": "Chao Xu",
    "id": "2004428678",
    "h_index": 32,
    "papers": 75
   },
   {
    "name": "Fei Gao",
    "id": "2349210755",
    "h_index": 3,
    "papers": 6
   }
  ],
  "comment": "Accepted by IEEE Transactions on Industrial Informatics",
  "topics": [
   "sim2real",
   "imitation-diffusion"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.20948v1",
  "pdf_url": "https://arxiv.org/pdf/2608.20948v1",
  "html_url": "https://arxiv.org/html/2608.20948v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.20946",
  "slug": "fast-coordinated-bimanual-motion-planning-with-hard-constraints",
  "title": "Fast Coordinated Bimanual Motion Planning With Hard Constraints",
  "abstract": "Bimanual manipulation enables complex tasks but introduces added complexity from the high number of degrees of freedom involved. When handling rigid objects, the relative transformation between the two end effectors must remain fixed throughout the motion, manifesting as a nonlinear equality constraint that confines the feasible configuration space to a measure-zero manifold and challenges conventional motion planners. We propose a fast bimanual motion planning pipeline that enforces this hard transformation constraint continuously along the entire path, using a leader-follower parameterization: the leader's configuration is treated as a free variable, while the follower's is determined via inverse kinematics to satisfy the constraint. We extensively evaluate the method in simulation across diverse environments, constraints and bimanual platforms, achieving 19.4x faster planning than prior work while guaranteeing continuous constraint satisfaction. Real-world experiments on a bimanual Kinova Gen3 setup, involving tray transport and elongated-object manipulation, validate direct transfer of planned trajectories to physical hardware.",
  "published": "2026-08-21",
  "updated": "2026-08-21",
  "year": "2026",
  "authors": [
   "Borna Paro",
   "Luka Petrovi\u0107",
   "Ivan Markovi\u0107"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes a fast bimanual motion planning pipeline that enforces this hard transformation constraint continuously along the entire path, using a leader-follower parameterization: the leader's configuration is treated as a free variable, while the follower's is determined via inverse kinematics to satisfy the constraint.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Borna Paro",
    "id": "2324127037",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Luka Petrovi\u0107",
    "id": "39876287",
    "h_index": 9,
    "papers": 27
   },
   {
    "name": "Ivan Markovic",
    "id": "2324127540",
    "h_index": 1,
    "papers": 5
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.20946v1",
  "pdf_url": "https://arxiv.org/pdf/2608.20946v1",
  "html_url": "https://arxiv.org/html/2608.20946v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.20936",
  "slug": "graph-operator-world-models-for-morphology-parameter-generalization-in",
  "title": "Graph-Operator World Models for Morphology-Parameter Generalization in Continuous Control",
  "abstract": "World models for continuous control are commonly trained for a fixed physical system and can degrade when known morphology parameters such as link lengths, masses, damping, and actuation change. Existing approaches often provide these parameters as conditioning information, but leave unspecified which part of the learned transition should remain reusable and which part should change with morphology. We propose Graph-Operator World Models (GraphOp-WM), a structured world model for generalization across unseen morphology parameters within related articulated robot families. GraphOp-WM represents bodies and their kinematic relations as an attributed graph and factorizes each transition into a morphology-independent local dynamics basis and a morphology-conditioned structured operator. The operator combines node-local modulation, kinematic-tree coupling, and a low-rank global correction, while architectural information separation, basis normalization, and paired-morphology supervision encourage static morphology dependence to be carried by the operator pathway. Graph-level readout and edge-wise action representations provide a compatible interface for reward, value, and TD-MPC-style planning. We further define controlled MuJoCo parameter splits covering interpolation, extrapolation, and held-out compositions of link geometry, mass, damping, and actuation parameters in Hopper, Walker2d, and HalfCheetah.",
  "published": "2026-08-21",
  "updated": "2026-08-21",
  "year": "2026",
  "authors": [
   "Xu Yang",
   "Yiqin Yang",
   "Qianchuan Zhao"
  ],
  "author_count": 3,
  "categories": [
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.AI",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Graph-Operator World Models (GraphOp-WM), a structured world model for generalization across unseen morphology parameters within related articulated robot families, is proposed and controlled MuJoCo parameter splits covering interpolation, extrapolation, and held-out compositions of link geometry, mass, damping, and actuation parameters in Hopper, Walker2d, and HalfCheetah are defined.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xu Yang",
    "id": "2297289824",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Yiqin Yang",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Qianchuan Zhao",
    "id": "2273362379",
    "h_index": 8,
    "papers": 19
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.20936v1",
  "pdf_url": "https://arxiv.org/pdf/2608.20936v1",
  "html_url": "https://arxiv.org/html/2608.20936v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.20909",
  "slug": "decoupling-policy-extraction-for-offline-reinforcement-learning",
  "title": "Decoupling Policy Extraction for Offline Reinforcement Learning",
  "abstract": "Offline RL methods commonly jointly train the actor and critic, where the critic is used to guide the actor toward higher-value actions. This coupled learning process is well motivated in online RL, where an improved actor collects new data that can further update the actor and the critic. However, training data remains fixed in offline RL, making actor-side policy improvement unable to generate new data to validate or correct the critic. Moreover, retaining this coupled paradigm leads to two related challenges. Firstly, actor updates can drift toward high-valued but potentially out-of-distribution (OOD) actions and amplify critic overestimation. Secondly, conservative value estimation or behavior-cloning regularization creates a difficult trade-off between suppressing OOD actions and selecting high-value actions within the data-supported region. Motivated by this observation, we revisit the conventional offline RL paradigm and propose decoupling policy improvement from actor training. Specifically, we train the actor solely to model the behavior distribution and perform policy improvement at inference time by reranking multiple actor-generated proposals with a separately learned critic. We refer to this paradigm as the decoupled policy extraction paradigm. Under such paradigm, the actor provides behavior-supported action candidates, while the critic performs value-based selection within this candidate set. Extensive experiments show that the decoupled policy extraction paradigm outperforms both behavior cloning and jointly learned offline RL methods, while remaining effective even with a naive Q-learning critic.",
  "published": "2026-08-21",
  "updated": "2026-08-21",
  "year": "2026",
  "authors": [
   "Xuyao Lin",
   "Yixiang Shan",
   "Jinru Duan",
   "Tao Yang",
   "Xinyu Zhao",
   "Runyu Lei",
   "Yiming Zhao",
   "Jiaxin Fan",
   "Zongbao Feng",
   "Peng Jia"
  ],
  "author_count": 10,
  "categories": [
   "cs.LG",
   "cs.RO"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work revisits the conventional offline RL paradigm and proposes decoupling policy improvement from actor training, and trains the actor solely to model the behavior distribution and perform policy improvement at inference time by reranking multiple actor-generated proposals with a separately learned critic.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xuyao Lin",
    "id": "2360266789",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Yixiang Shan",
    "id": "2282534736",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Jinru Duan",
    "id": "2459104201",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Tao Yang",
    "id": "2364544280",
    "h_index": 2,
    "papers": 11
   },
   {
    "name": "Xinyu Zhao",
    "id": "2320679543",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Runyu Lei",
    "id": "2459081039",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yiming Zhao",
    "id": "2345755411",
    "h_index": 5,
    "papers": 20
   },
   {
    "name": "Jiaxing Fan",
    "id": "2449163614",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Zongbao Feng",
    "id": "2149059932",
    "h_index": 21,
    "papers": 44
   },
   {
    "name": "Peng Jia",
    "id": "2333234654",
    "h_index": 3,
    "papers": 13
   }
  ],
  "comment": "",
  "topics": [
   "imitation-diffusion",
   "rl-control",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.20909v1",
  "pdf_url": "https://arxiv.org/pdf/2608.20909v1",
  "html_url": "https://arxiv.org/html/2608.20909v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.20906",
  "slug": "a-safety-driven-architectural-framework-for-fail-operational-drone-swa",
  "title": "A Safety-Driven Architectural Framework for Fail-Operational Drone Swarms in Critical Missions",
  "abstract": "The certification of Unmanned Aerial Vehicle (UAV) swarms for safety-critical operations requires verifiable design assurance. Airworthiness standards demand deterministic reliability, whereas multi-agent coordination algorithms execute non-deterministic models. This paper proposes a mixed-criticality architectural framework that applies SAE ARP4754B methods to swarm reconfiguration. First, a hardware-isolated Safety Monitor functions as a Run-Time Assurance (RTA) gateway, decoupling the flight-critical core from the non-deterministic Swarm Manager. Second, the monitor enforces formal safety contracts based on agent Health Vectors derived systematically from a Functional Hazard Assessment (FHA). Third, the framework propagates these Health Vectors to the collective planner to trigger fail-operational task reallocation, enabling intelligent swarm behaviors without compromising flight-critical isolation. Markov reliability modeling demonstrates that the $10^{-7}$ failures per flight hour Hazardous target is theoretically achievable for our SAIL IV scenario, provided the Safety Monitor meets $C_{monitor}>0.9991$, consistent with DAL B CMD/MON implementations.",
  "published": "2026-08-21",
  "updated": "2026-08-21",
  "year": "2026",
  "authors": [
   "Luiz Giacomossi",
   "Zafer Yigit",
   "Marwan Shakarna",
   "Shoaib Saleemi",
   "Ivan Tomasic",
   "Baran \u00c7ur\u00fckl\u00fc",
   "H\u00e5kan Forsberg"
  ],
  "author_count": 7,
  "categories": [
   "eess.SY",
   "cs.MA",
   "cs.RO"
  ],
  "primary_category": "eess.SY",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A mixed-criticality architectural framework that applies SAE ARP4754B methods to swarm reconfiguration, enabling intelligent swarm behaviors without compromising flight-critical isolation is proposed.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Luiz Giacomossi",
    "id": "2141678528",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Zafer Yigit",
    "id": "2375065733",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Marwan Shakarna",
    "id": "2459082323",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Shoaib Saleemi",
    "id": "2459082841",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "I. Tomasic",
    "id": "2416514",
    "h_index": 14,
    "papers": 124
   },
   {
    "name": "Baran \u00c7ur\u00fckl\u00fc",
    "id": "2459082845",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "H\u00e5kan Forsberg",
    "id": "2257210293",
    "h_index": 2,
    "papers": 4
   }
  ],
  "comment": "10 pages, 7 figures. Accepted for presentation at the 45th AIAA/IEEE Digital Avionics Systems Conference (DASC), Orlando, FL, USA, 2026. \\c{opyright} 2026 IEEE. Personal use of this material is permitted. Permission from IEEE must be obtained for all other uses",
  "topics": [
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.20906v1",
  "pdf_url": "https://arxiv.org/pdf/2608.20906v1",
  "html_url": "https://arxiv.org/html/2608.20906v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.20904",
  "slug": "scalable-distributed-simulation-based-testing-for-automated-driving-sy",
  "title": "Scalable Distributed Simulation-Based Testing for Automated Driving Systems",
  "abstract": "Virtual scenario-based testing is a key enabler for validating automated driving systems (ADS) and intelligent transport systems (ITS). However, executing large-scale test suites involving possibly thousands of scenarios remains labor-intensive and difficult to scale. This paper presents an end-to-end, DevOps-driven framework that automates build, deployment, and distributed execution of CARLA-based scenario tests of an ADS on a lightweight Kubernetes cluster. ROS 2 applications are packaged as standardized Kubernetes Helm charts generated from repository specifications, while entire simulation environments are composed declaratively via dynamic Helmfile manifests. The paper describes how a distributed testing workflow can be implemented in Argo Workflows to provision environments, aggregate and batch OpenSCENARIO test cases from configurable sources, execute scenarios in parallel across cluster nodes, and collect logs and resource metrics. In an evaluation on a multi-node K3s cluster running 200 scenarios, the best configuration speeds up end-to-end workflow time by more than a factor of eight compared to a sequential baseline. The results demonstrate significant gains in end-to-end execution time and quantify trade-offs between parallelism, orchestration overhead, and cluster stability. The framework is further demonstrated in a real-world ADS test application with connections to scenario sources and downstream evaluation modules. This demonstrates that the approach provides a strong foundation not only for scalable simulation testing, but also for generating traceable evidence that can support safety arguments.",
  "published": "2026-08-21",
  "updated": "2026-08-21",
  "year": "2026",
  "authors": [
   "Christian Geller",
   "Benedikt Haas",
   "Lutz Eckstein"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.DB",
   "cs.SE"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "An end-to-end, DevOps-driven framework that automates build, deployment, and distributed execution of CARLA-based scenario tests of an ADS on a lightweight Kubernetes cluster and demonstrates that the approach provides a strong foundation not only for scalable simulation testing, but also for generating traceable evidence that can support safety arguments.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Christian Geller",
    "id": "1739249898",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "B. Haas",
    "id": "48476631",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Lutz Eckstein",
    "id": "2288293894",
    "h_index": 2,
    "papers": 7
   }
  ],
  "comment": "14 pages; Accepted to be published as part of the 17. Uni-DAS e.V. Workshop \"Fahrerassistenz und automatisiertes Fahren\", September 29-30, 2026",
  "topics": [
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.20904v1",
  "pdf_url": "https://arxiv.org/pdf/2608.20904v1",
  "html_url": "https://arxiv.org/html/2608.20904v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.20891",
  "slug": "imu-free-body-frame-state-estimation-with-sparse-scene-flow-for-quadco",
  "title": "IMU-Free Body-Frame State Estimation with Sparse Scene Flow for Quadcopters",
  "abstract": "We present a vision-only state estimation system for X-configuration quadcopters equipped with a canonical stereo camera pair and no inertial sensors. The system operates entirely in the body frame, requiring only synchronised stereo images and motor thrust commands. A continuous-discrete extended Kalman filter on a composite manifold state $\\langle SE(3), \\mathbb{R}^3, \\ldots \\rangle$ maintains estimates of body-frame pose, velocity, angular velocity, gravity, and disturbances, using stationary scene points as implicit inertial references. Feature points are detected (FAST, Shi-Tomasi), tracked temporally (SSD, Lucas-Kanade) and matched across cameras (NCC), with search regions predicted from filter-derived pose and point uncertainty. Chi-squared gating on the normalised innovation admits only stationary points to the filter. The system also produces a sparse 3D point cloud carrying per-point position, velocity and joint covariance. These come from a 4-view (two stereo pairs at two timestamps) full bundle adjustment that jointly estimates position and velocity from stereo disparity and temporal parallax, with the filter-derived relative pose as a prior. Feature points in the EKF do not enter the solver; their information is reflected through the pose prior. Point cloud density is spatially adaptive: an external focus point directs allocation, producing dense coverage in the region of attention and sparse coverage elsewhere. The output is a body-frame state estimate, a calibrated pose change, and a sparse scene flow. It is intended as a measurement source for a downstream world model anchored in the current body frame, without dependence on GPS, IMU, or any world-frame infrastructure, though the architecture accommodates their future integration.",
  "published": "2026-08-21",
  "updated": "2026-08-21",
  "year": "2026",
  "authors": [
   "Daniel Gr\u00f8nhaug",
   "Sofie Markeset",
   "Mathias Kolberg"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A vision-only state estimation system for X-configuration quadcopters equipped with a canonical stereo camera pair and no inertial sensors, intended as a measurement source for a downstream world model anchored in the current body frame, without dependence on GPS, IMU, or any world-frame infrastructure, though the architecture accommodates their future integration.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Daniel Gr\u00f8nhaug",
    "id": "2407211787",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Sofie Markeset",
    "id": "2459081236",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Mathias Kolberg",
    "id": "2459080781",
    "h_index": 0,
    "papers": 1
   }
  ],
  "comment": "56 pages, 5 figures, 2 tables. Evaluated on the VID dataset (arXiv:2103.11152)",
  "topics": [
   "world-models",
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.20891v1",
  "pdf_url": "https://arxiv.org/pdf/2608.20891v1",
  "html_url": "https://arxiv.org/html/2608.20891v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.20890",
  "slug": "a-collaborative-multi-modality-interaction-for-vla-based-end-to-end-au",
  "title": "A Collaborative Multi-Modality Interaction for VLA-based End-to-End Autonomous Driving",
  "abstract": "Vision-Language-Action (VLA) models have emerged as a powerful paradigm for end-to-end autonomous driving by jointly integrating perception, reasoning, and decision making within a unified multimodal framework. However, most existing VLA models formulate end-to-end autonomous driving as a visual question answering task, leading to unreliable and less interpretable decision reasoning. In addition, they fail to establish effective multi-modal interaction across heterogeneous sensors, thereby limiting robust scene perception and reliable driving reasoning in long-tail driving scenarios. To this end, we propose a robust VLA-based end-to-end autonomous driving system that combines multi-modality interaction with multi-trajectory planning and optimization, enabling more reliable, interpretable, and safer driving decisions. Our method comprises three core components: (1) Affinity-Guided Optimal Transport for main-auxiliary modality two-way interaction; (2) Distribution-Consistent Modality Transfer for heterogeneous modality distribution transfer and cross-modal interaction; (3) Multi-modal Multi-Trajectory Planning along with Perception-Oriented Trajectory Refinement for better driving decisions to long-tail driving scenarios. Experimental results in open-loop and closed-loop datasets demonstrate improvements in safety long-horizon driving reasoning and road scene perception over existing driving systems, highlighting the ability of our mutli-modality interaction and multi-trajectory planning and optimization for scalable VLA-based systems.",
  "published": "2026-08-21",
  "updated": "2026-08-21",
  "year": "2026",
  "authors": [
   "Jingtao Sun",
   "Xiaohai He",
   "Yike Zhang",
   "Dong Huang",
   "Yaonan Wang",
   "Ajmal Mian",
   "Mike Zheng Shou"
  ],
  "author_count": 7,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Improvements in safety long-horizon driving reasoning and road scene perception over existing driving systems are demonstrated over existing driving systems, highlighting the ability of the mutli-modality interaction and multi-trajectory planning and optimization for scalable VLA-based systems.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jingtao Sun",
    "id": "2188827333",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Xiaohai He",
    "id": "2447255032",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yike Zhang",
    "id": "2407638784",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Dongsheng Huang",
    "id": "2453843323",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yaonan Wang",
    "id": "2278782811",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Ajmal Saeed Mian",
    "id": "1747500",
    "h_index": 67,
    "papers": 327
   },
   {
    "name": "M. Shou",
    "id": "2047358650",
    "h_index": 49,
    "papers": 278
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.20890v1",
  "pdf_url": "https://arxiv.org/pdf/2608.20890v1",
  "html_url": "https://arxiv.org/html/2608.20890v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.20874",
  "slug": "multi-modal-traffic-sign-detection-with-semantic-attributes-for-autono",
  "title": "Multi-Modal Traffic Sign Detection with Semantic Attributes for Autonomous Driving",
  "abstract": "Reliable traffic sign detection is a prerequisite for the global deployment of autonomous driving systems, where regulatory compliance and road safety depend on perceiving signs correctly across regions, ranges, and weather conditions. Despite recent progress, vision-based methods continue to face three fundamental limitations: poor cross-regional generalization due to high diversity across countries, degraded performance on small-object detection at long ranges (traffic signs occupy as little as $10{\\times}10$ pixels at 200m), and fragile temporal tracking under the strongly non-linear perspective distortion that occurs as a vehicle approaches a sign. In this paper, we address the problem of robust, long-range, region-agnostic traffic sign perception by combining camera and Light Detection and Ranging (LiDAR) sensing. We present a multi-modal detection framework whose Intensity-Aware Deformable Fusion module aligns retro-reflective LiDAR cues with camera features, anchoring detection on geometric invariants rather than region-specific visual appearance. We further introduce a dual motion-model tracker that explicitly accounts for non-linear perspective transformations during vehicle approach, substantially improving temporal consistency over linear motion assumptions. Additionally, we develop a semantic attribute classification pipeline that estimates occlusion level, readability, sign embeddedness, and road relevance, providing actionable context to downstream planning. Extensive evaluation on our dataset, spanning 60+ countries and 2,500+ hours of driving data, shows that the proposed pipeline achieves an Object Miss Ratio (OMR) of 0.49% across 221,068 evaluation sequences, demonstrating globally generalizable traffic sign perception in commercial-grade autonomous driving systems.",
  "published": "2026-08-21",
  "updated": "2026-08-21",
  "year": "2026",
  "authors": [
   "Meda Lazar",
   "Sourab Sridhar",
   "Shashwata Gupta",
   "Alexandra Tripcea",
   "Varun Ravi",
   "Senthil Yogamani"
  ],
  "author_count": 6,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A dual motion-model tracker that explicitly accounts for non-linear perspective transformations during vehicle approach is introduced, substantially improving temporal consistency over linear motion assumptions, and a semantic attribute classification pipeline that estimates occlusion level, readability, sign embeddedness, and road relevance is developed, providing actionable context to downstream planning.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Meda Lazar",
    "id": "2459083452",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Sourab Sridhar",
    "id": "2459083207",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Shashwata Gupta",
    "id": "2459230954",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Alexandra Tripcea",
    "id": "2459083430",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "V. Ravi",
    "id": "2452881424",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "S. Yogamani",
    "id": "2601522",
    "h_index": 41,
    "papers": 133
   }
  ],
  "comment": "",
  "topics": [
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.20874v1",
  "pdf_url": "https://arxiv.org/pdf/2608.20874v1",
  "html_url": "https://arxiv.org/html/2608.20874v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.20852",
  "slug": "demonstration-guided-humanoid-stand-up-on-an-emulated-deformable-surfa",
  "title": "Demonstration-Guided Humanoid Stand-Up on an Emulated Deformable Surface",
  "abstract": "This paper presents a reference-guided reinforcement learning framework to generate stand-up motion for a 29-DOF Unitree G1 humanoid on deformable soft ground, using a human demonstration recorded on hard ground. The terrain compliance is modelled using solref and solimp parameters from MuJoCo's rigid body soft-contact model. The rewards consists of (i) reference motion tracking through residual joint-position control and (ii) explicit recovery objectives such as pelvis height, torso uprightness, and the final posture. First, the policy is trained with the specified rewards considering hard ground. Next, the terrain stiffness is lowered by updating solref and the nominal surface penetration zone is expanded using solimp. Subsequent training enables the policy to adapt to the delayed support force generation due to significant surface penetration during contact-intensive phases while preserving the original demonstration pattern. The learned policy successfully completes the fallen-to-standing task in simulation, reaching the targeted pelvis height and uprightness, with a maximum contact penetration of approximately 40 mm during the process. The proposed method is demonstrated on two stand-up sequences and successfully achieves the final recovery objective on both hard and soft ground. Ablation studies show that reference tracking alone is insufficient for successful stand-up, and that explicit recovery rewards are essential.",
  "published": "2026-08-21",
  "updated": "2026-08-21",
  "year": "2026",
  "authors": [
   "Aniruddh Kushwah",
   "Vyankatesh Ashtekar",
   "Ashish Dutta"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A reference-guided reinforcement learning framework to generate stand-up motion for a 29-DOF Unitree G1 humanoid on deformable soft ground, using a human demonstration recorded on hard ground and successfully achieves the final recovery objective on both hard and soft ground is presented.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Aniruddh Kushwah",
    "id": "2459083041",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Vyankatesh Ashtekar",
    "id": "2226291635",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Ashish Dutta",
    "id": "2265002637",
    "h_index": 3,
    "papers": 17
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "egocentric-data",
   "rl-control"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2608.20852v1",
  "pdf_url": "https://arxiv.org/pdf/2608.20852v1",
  "html_url": "https://arxiv.org/html/2608.20852v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2608.20823",
  "slug": "natural-sit-to-stand-motion-synthesis-for-humanoids-via-guided-assista",
  "title": "Natural Sit-to-Stand Motion Synthesis For Humanoids via Guided Assistance Curricula and Staged Rewards",
  "abstract": "A humanoid has infinitely many ways to stand up from sitting while maintaining balance, making sit-to-stand (STS) a challenging control problem. We synthesise natural humanoid STS motion from scratch using reinforcement learning, without demonstrations or reference trajectories. A single Proximal Policy Optimisation policy learns smooth, human-like rising driven by three complementary components. (i) A coupled force/chair-height curriculum is used. A vertical pelvis-assist force aids early trajectory exploration and decays over training. Taller chairs are unlocked with decaying assisting force. This ensures that the policy masters a viable STS trajectory at each chair height before being exposed to harder ones, avoiding the premature distribution shift that otherwise collapses generalisation. (ii) Motion robustness is achieved by randomly sampling from a large number of inverse kinematics-generated initial and target poses spanning over eight chair heights. (iii) A set of rewards is defined inspired from biomechanics and optimal control studies. They shape the robot's angular momentum for seat-off, and enable support-region transition via centre of pressure attraction function to ensure smooth low-effort actuation. On a deterministic force-free evaluator, the policy attains more than 97% balanced-standing success across eight chair heights. The policy generalises smooth motion across chair heights and enables the robot to rise from substantially deep-seated postures as compared to the state of the art.",
  "published": "2026-08-21",
  "updated": "2026-08-21",
  "year": "2026",
  "authors": [
   "Meet Pal Singh",
   "Vyankatesh Ashtekar",
   "Ashish Dutta"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work synthesise natural humanoid STS motion from scratch using reinforcement learning, without demonstrations or reference trajectories, and generalises smooth motion across chair heights and enables the robot to rise from substantially deep-seated postures as compared to the state of the art.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Meeta Singh",
    "id": "2454924704",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Vyankatesh Ashtekar",
    "id": "2226291635",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Ashish Dutta",
    "id": "2265002637",
    "h_index": 3,
    "papers": 17
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "rl-control",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.20823v1",
  "pdf_url": "https://arxiv.org/pdf/2608.20823v1",
  "html_url": "https://arxiv.org/html/2608.20823v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.20817",
  "slug": "ghosttac-manipulating-tactile-sensors-without-physical-contact",
  "title": "GhostTac: Manipulating Tactile Sensors without Physical Contact",
  "abstract": "Tactile sensors are integral components of modern robotic systems, enabling robots to perceive and interact with the physical environment through tactile feedback. Despite their importance, the physical-layer security of tactile sensors has received little attention in prior work. In this paper, we present GhostTac, to the best of our knowledge, the first contactless attack that manipulates tactile sensing via electromagnetic interference (EMI). We identify that EMI exploits the nonlinear rectification and limited bandwidth amplification effects, allowing carefully crafted EMI signals to be converted into a persistent DC offset that bypasses on-board filtering and induces stable measurement deviations. Building on this mechanism, GhostTac enables fine-grained and controllable manipulation of sensor outputs by reshaping the spatial distribution and manipulating the magnitude at the targeted location. Such interference can induce unintended and harmful robot behaviors, such as causing a domestic robot to exert excessive force, resulting in physical damage or human injury. We evaluate GhostTac on 10 sensor modules and 2 dexterous hands, covering 15 tactile sensors of different types, and demonstrate consistent attack effectiveness across all tested devices. We further present three case studies on tactile grasping, slip detection, and material classification to illustrate practical impacts in real robotic tasks. We envision that our findings shed light on a new physical attack vector against tactile sensing in robotic systems.",
  "published": "2026-08-21",
  "updated": "2026-08-24",
  "year": "2026",
  "authors": [
   "Kun Wang",
   "Xuancun Lu",
   "Ruochen Zhou",
   "Kai Wang",
   "Tongjun Ye",
   "Yihao Shao",
   "Chen Yan",
   "Xiaoyu Ji",
   "Wenyuan Xu"
  ],
  "author_count": 9,
  "categories": [
   "cs.CR",
   "cs.RO"
  ],
  "primary_category": "cs.CR",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper presents GhostTac, to the best of its knowledge, the first contactless attack that manipulates tactile sensing via electromagnetic interference (EMI), and identifies that EMI exploits the nonlinear rectification and limited bandwidth amplification effects.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Kunrong Wang",
    "id": "2457349158",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Xuancun Lu",
    "id": "2226431069",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Ruochen Zhou",
    "id": "121462075",
    "h_index": 2,
    "papers": 15
   },
   {
    "name": "Kai Wang",
    "id": "2160215423",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Tongjun Ye",
    "id": "2219259838",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Yihao Shao",
    "id": "2455437249",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Chen Yan",
    "id": "2116547768",
    "h_index": 7,
    "papers": 25
   },
   {
    "name": "Xiaoyu Ji",
    "id": "2312342565",
    "h_index": 2,
    "papers": 16
   },
   {
    "name": "Wenyuan Xu",
    "id": "2312701488",
    "h_index": 6,
    "papers": 26
   }
  ],
  "comment": "Accepted at ACM CCS 2026",
  "topics": [
   "dexterous-manipulation",
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.20817v2",
  "pdf_url": "https://arxiv.org/pdf/2608.20817v2",
  "html_url": "https://arxiv.org/html/2608.20817v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.20784",
  "slug": "rethinking-demonstration-unlearning-in-imitation-learning-for-robotics",
  "title": "Rethinking Demonstration Unlearning in Imitation Learning for Robotics",
  "abstract": "Imitation learning for robotics depends on human demonstrations, some of which people may later ask to remove. Retraining without them is the natural reference, but its cost grows with policy and dataset scale, motivating cheaper operators that edit a trained policy. Metrics inherited from machine unlearning, such as forgetting loss or a single membership attack, do not establish what an edit removed from a policy acting in closed loop. We therefore introduce a retrain-calibrated audit that reads demonstration unlearning along two axes: behavior, whether the edited policy acts like one retrained without the removed demonstrations, and evidence, whether an auditor can still detect it was trained on them. The behavior axis measures action divergence to that retrain at matched states, calibrated by a floor built from independent retrains, so a policy at the floor is as close to a retrain as retrains are to each other. The evidence axis applies a per-demonstration membership attack against a retrain null, reporting both its rank and its absolute member-loss level, since rank alone accepts operators that inflate member losses past the null. A conformal test then combines both axes into one hypothesis of joint retrain consistency, against a fleet of independent retrains large enough to reject at conventional significance. Across five preregistered conditions on three real-robot policy classes and two simulation suites, the axes dissociate in both directions on one checkpoint, as an edit may repair task behavior while leaving evidence unchanged, or reduce evidence while moving behavior away from retraining. On the ACT arm, a redirect edit restores blind-scored robot success to 18 of 20 trials.",
  "published": "2026-08-21",
  "updated": "2026-08-21",
  "year": "2026",
  "authors": [
   "Jiazhuo Li",
   "Yu Zhang",
   "Yiming Fei",
   "Kangkang Dong",
   "Xiaojun Zhu",
   "Houde Liu",
   "Jinze Tao"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A retrain-calibrated audit is introduced that reads demonstration unlearning along two axes: behavior, whether the edited policy acts like one retrained without the removed demonstrations, and evidence, whether an auditor can still detect it was trained on them.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jiazhuo Li",
    "id": "2292294220",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Yu Zhang",
    "id": "2329789543",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Yiming Fei",
    "id": "2321595615",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Kangkang Dong",
    "id": "51053421",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Xiaojun Zhu",
    "id": "2342023587",
    "h_index": 4,
    "papers": 14
   },
   {
    "name": "Houde Liu",
    "id": "2293440971",
    "h_index": 3,
    "papers": 28
   },
   {
    "name": "Jinze Tao",
    "id": "2396436519",
    "h_index": 1,
    "papers": 6
   }
  ],
  "comment": "21 pages, 7 figures, 14 tables",
  "topics": [
   "egocentric-data",
   "imitation-diffusion"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.20784v1",
  "pdf_url": "https://arxiv.org/pdf/2608.20784v1",
  "html_url": "https://arxiv.org/html/2608.20784v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.20735",
  "slug": "foretime-vla-causal-future-token-distillation-from-a-world-action-mode",
  "title": "ForeTime-VLA: Causal Future-Token Distillation from a World Action Model for Conveyor-Belt Manipulation",
  "abstract": "Manipulating moving objects requires a policy to anticipate contact events, yet vision-language-action (VLA) policies are commonly fine-tuned from the current observation alone. World action models (WAMs) learn predictive dynamics, but running a video-scale teacher or explicitly imagining future frames at deployment is costly. We introduce ForeTime-VLA, a dense pi0.5 policy that distills a future-aware, action-equivalent representation from a frozen Fast-WAM-derived teacher while remaining causal at inference. Offline, current and future video latents are compressed into a whitened 64-D target. Online, an eight-frame history encoder predicts this target together with manipulation phase and normalized time-to-transition. Four future tokens and one phase token condition the VLM prefix, while the predicted future and transition horizon condition the action expert. Training retains the original flow-matching action target and adds cosine, relational geometry, phase, time-to-transition, and action-equivalence objectives. On a deduplicated conveyor-belt dataset, we compare 40k-step checkpoints on 768 matched windows per split. Test MAE decreases from 0.134119 to 0.130593 (2.63%; paired-bootstrap 95% CI: 0.82-4.48% improvement), and test L2 decreases by 3.02%, at a 2.46-2.93% latency cost. In quantitative real-robot evaluation, ForeTime-VLA achieves 81.1% stationary and 58.9% slow-moving grasp success, exceeding the next-best reference by 12.2 and 22.2 percentage points, respectively. Across three belt speeds, it completes 44/90 grasps versus 23/90 for pi0.5, including 11/30 versus 2/30 at fast speed. The agreement between offline orientation gains and reduced real-robot contact-pose failures supports causal future-token distillation as an effective way to improve dynamic manipulation without deploying the world-model teacher.",
  "published": "2026-08-21",
  "updated": "2026-08-24",
  "year": "2026",
  "authors": [
   "Siyuan Ma",
   "Yutian Zhang",
   "Boshi Zhang",
   "Qinglian Wu",
   "Jiaqi Zhai",
   "Dong Wei",
   "Xiaojin Huang"
  ],
  "author_count": 7,
  "categories": [
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.AI",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The agreement between offline orientation gains and reduced real-robot contact-pose failures supports causal future-token distillation as an effective way to improve dynamic manipulation without deploying the world-model teacher.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Siyuan Ma",
    "id": "2458696249",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yutian Zhang",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Boshi Zhang",
    "id": "2458694154",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Qinglian Wu",
    "id": "2458705991",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Jiaqi Zhai",
    "id": "2458627634",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Dongjun Wei",
    "id": "2457983330",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Xiaojin Huang",
    "id": "2450293815",
    "h_index": 0,
    "papers": 2
   }
  ],
  "comment": "8 pages, 5 figures. Introduces ForeTime-VLA, a causal future-token distillation method for conveyor-belt manipulation from a frozen world action model teacher",
  "topics": [
   "world-models",
   "vla",
   "dexterous-manipulation"
  ],
  "orgs": [
   "Physical Intelligence"
  ],
  "abs_url": "https://arxiv.org/abs/2608.20735v2",
  "pdf_url": "https://arxiv.org/pdf/2608.20735v2",
  "html_url": "https://arxiv.org/html/2608.20735v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2608.20655",
  "slug": "nonlinear-model-predictive-control-for-trajectory-tracking-of-differen",
  "title": "Nonlinear Model Predictive Control for Trajectory Tracking of Differentially Flat Fixed-Wing Aerial Systems",
  "abstract": "Planning and control of fixed-wing Unmanned Aerial Vehicles (UAVs) are challenging due to nonlinear dynamics, aerodynamic limits, and environmental disturbances. Differential flatness offers a principled way to generate fast, feasible trajectories, but its use has largely been confined to model-free controllers, which lack predictive capabilities and demand tuning. In this paper, we propose a unified framework that integrates differential flatness-based trajectory generation with Nonlinear Model Predictive Control (NMPC), combining computationally efficient planning with predictive, constraint-aware control. To further improve robustness, we introduce a wind-aware sampling strategy embedded within the NMPC framework, enabling the generation of dynamically feasible reference trajectories that proactively account for wind disturbances while strictly enforcing aerodynamic and control input constraints. We validate the proposed framework through extensive simulations and real-world flight experiments, demonstrating improved tracking accuracy and robustness for complex trajectories, particularly when using the proposed wind-aware sampling strategy under strong wind conditions.",
  "published": "2026-08-21",
  "updated": "2026-08-21",
  "year": "2026",
  "authors": [
   "Nishanth Bobbili",
   "Pratyaksh Rao",
   "Luca Morando",
   "Luca Masci",
   "Giuseppe Loianno"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A unified framework that integrates differential flatness-based trajectory generation with Nonlinear Model Predictive Control (NMPC) is proposed, combining computationally efficient planning with predictive, constraint-aware control and validated through extensive simulations and real-world flight experiments.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Nishanth Bobbili",
    "id": "2266838509",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "P. Rao",
    "id": "2257453052",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Luca Morando",
    "id": "2282968376",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Luca Masci",
    "id": "2343635589",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Giuseppe Loianno",
    "id": "1759737",
    "h_index": 39,
    "papers": 169
   }
  ],
  "comment": "",
  "topics": [
   "rl-control",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.20655v1",
  "pdf_url": "https://arxiv.org/pdf/2608.20655v1",
  "html_url": "https://arxiv.org/html/2608.20655v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.20626",
  "slug": "pneumatic-units-for-logic-based-sequential-excitation-pulse-in-wearabl",
  "title": "Pneumatic Units for Logic-based Sequential Excitation (PULSE) in Wearable Haptic Devices",
  "abstract": "Soft, wearable robotic devices can deliver haptic feedback to support a wide range of tasks, such as extended reality, training various skills, and rehabilitation. Pneumatic actuation can deliver complex haptic feedback, is lightweight and compliant, and can be incorporated into textiles, making it promising for wearable applications. These soft pneumatic devices, however, typically require a valve and input for each pneumatic actuator, making it challenging to develop fully portable devices for at-home use. In this work we present a pneumatic unit for logic-based sequential excitation (PULSE). The PULSE is a flat, textile-based pneumatic actuator with embedded fluidic logic. By combining these actuators into a fluidic ring oscillator, we decreased the typical amount of required pneumatic inputs for a haptic forearm sleeve by 60%, with the ability to scale. We built the ring oscillator by optimizing design variables to reach desired periods of oscillation. We demonstrated a set of tactile stroking cues with periods ranging from 1.16 to 1.56 s and forces ranging from 1.07 to 2.04 N. We assessed the sleeve's ability to render differentiable, pleasant, and continuous haptic cues in a user study. The forearm sleeve containing PULSEs successfully delivered four directional cues and guided users to target wrist angles with fast reaction times, low overshoot amounts, and a 93.3% average accuracy of correct initial directions.",
  "published": "2026-08-20",
  "updated": "2026-08-20",
  "year": "2026",
  "authors": [
   "Jessica Healey",
   "Anoush Sepehri",
   "Michael T. Tolley",
   "Tania K. Morimoto"
  ],
  "author_count": 4,
  "categories": [
   "cs.HC",
   "cs.RO"
  ],
  "primary_category": "cs.HC",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A pneumatic unit for logic-based sequential excitation (PULSE) is presented, a flat, textile-based pneumatic actuator with embedded fluidic logic and the sleeve's ability to render differentiable, pleasant, and continuous haptic cues in a user study is assessed.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jessica Healey",
    "id": "2278790274",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Anoush Sepehri",
    "id": "2301244975",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "M. Tolley",
    "id": "48724645",
    "h_index": 41,
    "papers": 146
   },
   {
    "name": "Tania K. Morimoto",
    "id": "2300488454",
    "h_index": 6,
    "papers": 29
   }
  ],
  "comment": "9 pages, 6 figures",
  "topics": [
   "tactile",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.20626v1",
  "pdf_url": "https://arxiv.org/pdf/2608.20626v1",
  "html_url": "https://arxiv.org/html/2608.20626v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.20556",
  "slug": "logic-vla-a-temporal-logic-conditioned-vision-language-action-model",
  "title": "Logic-VLA: A Temporal Logic Conditioned Vision-Language-Action Model",
  "abstract": "Vision-language-action (VLA) models can follow natural-language (NL) task instructions, but such instructions may not precisely specify safety-critical or spatiotemporal requirements on the resulting behavior. We introduce Logic-VLA, a formal-requirement-aware VLA that conditions on Signal Temporal Logic (STL) specifications supplied at inference time. Logic-VLA uses a syntax-graph-based STL encoder pre-trained to capture temporal logic semantics. Policy adaptation proceeds in two stages: STL-conditioned supervised fine-tuning on satisfying demonstrations is followed by trajectory-level preference optimization over matched satisfying-violating rollout pairs using a flow-matching surrogate for Identity Preference Optimization. This formulation improves formal requirement satisfaction while preserving the nominal NL task. We evaluate Logic-VLA in closed-loop quadcopter navigation simulation across randomized photorealistic environments and test generalization to STL formulas unseen during training. Across the evaluation benchmarks, Logic-VLA improves STL satisfaction rate over an STL-blind base policy by 24.8 to 40.7 percentage points (pp) while reducing nominal NL task success by at most 1.8 pp, showing that a single VLA can adapt its behavior to varying formal requirements without requiring a separate policy for each specification.",
  "published": "2026-08-20",
  "updated": "2026-08-20",
  "year": "2026",
  "authors": [
   "Celina Shiyu Wang",
   "Yiqi Zhao",
   "Junjie Ye",
   "Yue Wang",
   "Jyotirmoy V. Deshmukh"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.LO",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "Logic-VLA is introduced, a formal-requirement-aware VLA that conditions on Signal Temporal Logic (STL) specifications supplied at inference time, showing that a single VLA can adapt its behavior to varying formal requirements without requiring a separate policy for each specification.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Celina Shiyu Wang",
    "id": "2445511242",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yiqi Zhao",
    "id": "6825484",
    "h_index": 15,
    "papers": 71
   },
   {
    "name": "Junjie Ye",
    "id": "2267498972",
    "h_index": 6,
    "papers": 18
   },
   {
    "name": "Yue Wang",
    "id": "2257326367",
    "h_index": 12,
    "papers": 20
   },
   {
    "name": "J. Deshmukh",
    "id": "2256346716",
    "h_index": 2,
    "papers": 17
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.20556v1",
  "pdf_url": "https://arxiv.org/pdf/2608.20556v1",
  "html_url": "https://arxiv.org/html/2608.20556v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2608.20546",
  "slug": "koala-gripper-co-designing-robotic-grippers-and-data-capture-devices-f",
  "title": "Koala Gripper: Co-designing Robotic Grippers and Data-Capture Devices for Scaling Dexterous Manipulation Learning",
  "abstract": "As the demand for larger manipulation datasets grows, handheld robotic gripper data collection and the associated gripper designs become more vital. Current data collection device designs trend towards matching the morphologies of existing robotic grippers, sacrificing ergonomics and manipulation performance. In this paper, we propose a co-design framework that guides the simultaneous development of both data collection and robotic execution devices by weaving both platform constraints into the design process. Through this workflow, we present the Koala Gripper system, a data capture device and robotic gripper platform that improves dexterity and grasp capability compared to parallel jaw grippers while preserving scalability and ease-of-use. The design introduces a novel force-optimized finger/trigger linkage mechanism with directional reflected mass characteristics, a unique monolithic dual-thumb, and user-centered ergonomic design. The design's actuated robotic fingers are backdrivable, with effective mass on the order of tens of grams. We show that these grippers are capable of secure grasps over a wide range of objects, forceful tool use, and precise singulation. We further validate the platform by deploying it with an end-to-end data collection and policy execution pipeline that highlights its capabilities through learning from demonstration. More information available at http://koalagripper.rai-inst.com",
  "published": "2026-08-20",
  "updated": "2026-08-20",
  "year": "2026",
  "authors": [
   "Amar Hajj-Ahmad",
   "Zubin Kremer Guha",
   "Tim Fofonoff",
   "Zhi Ern Teoh",
   "Ciar\u00e1n T. O'Neill",
   "Ben Thacher",
   "Igor Fala",
   "Vidullan Surendran",
   "Murphy Wonsick",
   "Peter Whitney",
   "David Watkins"
  ],
  "author_count": 11,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A co-design framework is proposed that guides the simultaneous development of both data collection and robotic execution devices by weaving both platform constraints into the design process, and presents the Koala Gripper system, a data capture device and robotic gripper platform that improves dexterity and grasp capability compared to parallel jaw grippers while preserving scalability and ease-of-use.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Amar Hajj-Ahmad",
    "id": "2036926747",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Zubin Kremer Guha",
    "id": "2098024799",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Tim Fofonoff",
    "id": "2459082321",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Zhi Ern Teoh",
    "id": "2285958",
    "h_index": 8,
    "papers": 13
   },
   {
    "name": "Ciar\u00e1n T. O\u2019Neill",
    "id": "1466555051",
    "h_index": 12,
    "papers": 15
   },
   {
    "name": "Ben Thacher",
    "id": "2459082277",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Igor Fala",
    "id": "2459082938",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Vidullan Surendran",
    "id": "120571366",
    "h_index": 4,
    "papers": 18
   },
   {
    "name": "Murphy Wonsick",
    "id": "8545008",
    "h_index": 8,
    "papers": 15
   },
   {
    "name": "Peter D. Whitney",
    "id": "143754758",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "David Watkins",
    "id": "2068456081",
    "h_index": 5,
    "papers": 10
   }
  ],
  "comment": "Paper website: http://koalagripper.rai-inst.com Paper video: http://www.youtube.com/watch?v=ZoygFCWAVhg",
  "topics": [
   "dexterous-manipulation",
   "data-teleop",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.20546v1",
  "pdf_url": "https://arxiv.org/pdf/2608.20546v1",
  "html_url": "https://arxiv.org/html/2608.20546v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.20478",
  "slug": "endolift-language-disambiguated-latent-conditioned-rectified-flow-for",
  "title": "EndoLIFT: Language-Disambiguated Latent-Conditioned Rectified Flow for Bidirectional Endoscopic Control",
  "abstract": "Routine gastrointestinal endoscopy is intrinsically bidirectional: the instrument is advanced to reach target anatomy and later withdrawn or retroflexed for inspection, while an external cue may require earlier reversal. When the requested phase changes before the visual scene does, nearly identical observations can require opposite axial actions. We identify and formalize this ambiguity in bidirectional endoscopic control as intent aliasing. We propose EndoLIFT (Endoscopic Language-Instruction Flow with Trajectory Latents), a vision-language-action policy that combines explicit language-based intent conditioning with a latent-conditioned rectified-flow action expert. The policy receives RGB, a language instruction, and the previous-action state; a 32-D variational trajectory latent stochastically conditions continuous action-chunk generation. Controlled same-observation instruction swaps establish that language selects the axial mode, independently of whether the trajectory latent is present. Relative to the matched model without latent conditioning, EndoLIFT improves navigation-direction accuracy by 11.1 percentage points and reduces wrong-direction advance by 83\\%. An architecture-controlled 1-bit mode-flag reference exhibits weaker canonical-anchor switching, while EndoLIFT retains 82.8\\% intent-following accuracy across 44 held-out linguistic variants. In closed-loop evaluation, EndoLIFT improves overall success by 30 percentage points over EndoLIFT w/o VTL on both the seen colon phantom and the unseen lung and stomach phantoms, and completes 10/10 ex-vivo porcine-trachea trials. These results separate language-based intent selection from the trajectory latent's contribution to directional correctness and robust retraction.",
  "published": "2026-08-20",
  "updated": "2026-08-20",
  "year": "2026",
  "authors": [
   "Chi Kit Ng",
   "Yidong Zhang",
   "Lui Siu Hing",
   "Jinsong Lin",
   "Tianchun Wu",
   "Ho Yin Chim",
   "Zhiqing Tang",
   "Tao Yang",
   "Huxin Gao",
   "Trevor Yeung",
   "Raymond Shing-Yan Tang",
   "Hongliang Ren"
  ],
  "author_count": 12,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "EndoLIFT (Endoscopic Language-Instruction Flow with Trajectory Latents) is proposed, a vision-language-action policy that combines explicit language-based intent conditioning with a latent-conditioned rectified-flow action expert and separate language-based intent selection from the trajectory latent's contribution to directional correctness and robust retraction.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Chikit Ng",
    "id": "2307447792",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Yidong Zhang",
    "id": "2447667111",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Lui Siu Hing",
    "id": "2459080050",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Jinsong Lin",
    "id": "2445481374",
    "h_index": 0,
    "papers": 5
   },
   {
    "name": "Tian-Hang Wu",
    "id": "2457960414",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "H. Chim",
    "id": "2340789789",
    "h_index": 2,
    "papers": 13
   },
   {
    "name": "Zhiqing Tang",
    "id": "2455447929",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Tao Yang",
    "id": "2445488647",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Huxin Gao",
    "id": "1653073415",
    "h_index": 13,
    "papers": 50
   },
   {
    "name": "Trevor M. Yeung",
    "id": "2277829260",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Raymond Shing-Yan Tang",
    "id": "2280535467",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Hongliang Ren",
    "id": "2260612957",
    "h_index": 13,
    "papers": 54
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.20478v1",
  "pdf_url": "https://arxiv.org/pdf/2608.20478v1",
  "html_url": "https://arxiv.org/html/2608.20478v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.20467",
  "slug": "learning-based-measurement-robust-control-barrier-functions-for-obstac",
  "title": "Learning-Based Measurement-Robust Control Barrier Functions for Obstacle Avoidance under State Estimation Error",
  "abstract": "Safety filters are an effective tool for enforcing constraints in safety-critical systems, but most existing methods assume perfect state information, which is rarely available in practice. Recent work has begun to close this gap by developing filtering mechanisms that are robust to state estimation error, but these methods can still exhibit safety violations or overly conservative behavior as estimation error grows. Focusing on obstacle avoidance, we develop two new control barrier function (CBF) formulations: drift-measurement-robust (DMR)-CBFs and neural measurement-robust (NMR)-CBFs. The DMR-CBF augments the standard CBF condition with an inner optimization over the worst-case uncertainty in the drift dynamics, improving robustness to estimation error. This DMR-CBF then supervises a pretraining phase for the NMR-CBF, which replaces the inner optimization with a learned term. The NMR-CBF is subsequently finetuned through differentiable trajectory rollouts, yielding a filter that achieves empirical safety comparable to the DMR-CBF while reducing both conservativeness and computational cost. We provide theoretical analysis of the DMR-CBF along with numerical results on a planar double integrator and a 12D quadrotor, where both proposed approaches prevent collisions while other robust methods either fail or are overly conservative. Finally, we deployed the NMR-CBF on a Unitree Go2, enabling successful navigation of an obstacle field under odometry errors that caused a standard CBF to collide.",
  "published": "2026-08-20",
  "updated": "2026-08-20",
  "year": "2026",
  "authors": [
   "Nicholas Rober",
   "Yixuan Jia",
   "Jonathan P. How"
  ],
  "author_count": 3,
  "categories": [
   "eess.SY",
   "cs.RO"
  ],
  "primary_category": "eess.SY",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work develops two new control barrier function (CBF) formulations: drift-measurement-robust (DMR)-CBFs and neural measurement-robust (NMR)-CBFs and provides theoretical analysis of the DMR-CBF along with numerical results on a planar double integrator and a 12D quadrotor.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Nicholas Rober",
    "id": "2264297285",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Y. Jia",
    "id": "2322458151",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Jonathan P. How",
    "id": "2261739177",
    "h_index": 5,
    "papers": 18
   }
  ],
  "comment": "8 Pages, 6 figures",
  "topics": [
   "rl-control",
   "navigation",
   "foundation-pretraining",
   "safety-eval"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2608.20467v1",
  "pdf_url": "https://arxiv.org/pdf/2608.20467v1",
  "html_url": "https://arxiv.org/html/2608.20467v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2608.20433",
  "slug": "humanoid-musical-robots-as-experimental-interfaces-for-music-evoked-em",
  "title": "Humanoid Musical Robots as Experimental Interfaces for Music-Evoked Emotion",
  "abstract": "Advances in technology have led to increasingly sophisticated musical humanoid robots. However, their use has largely been limited to performance and related research in human-robot interaction. In this position paper, we propose a novel perspective: musical humanoid robots as experimental interfaces for investigating music-evoked emotions. We argue that current research is constrained by paradigms relying on pre-recorded auditory stimuli, which fail to capture the multimodal, embodied, and interactive nature of real-world musical experience. Building on existing theories of music cognition and emotion, we identify mechanisms that require controlled manipulation of both acoustic and non-acoustic variables. We show that humanoid robots are well-suited as they enable parametric control of performance variables, reproducibility across trials, and the decoupling and recombination of auditory, visual, and interactive components. We illustrate the technical feasibility of this perspective through a case study of the WAseda Saxophonist Robot 5 (WAS-5), demonstrating reproducible control of acoustic and interaction variables that are prerequisites for future music-emotion experiments. Our work positions musical humanoid robots as a methodological platform that enables future controlled investigations of music-evoked emotions.",
  "published": "2026-08-20",
  "updated": "2026-08-20",
  "year": "2026",
  "authors": [
   "Vincent K. M. Cheung",
   "Jia-Yeu Lin"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO",
   "cs.HC",
   "cs.MM",
   "cs.SD"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "It is shown that humanoid robots are well-suited as they enable parametric control of performance variables, reproducibility across trials, and the decoupling and recombination of auditory, visual, and interactive components.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Vincent K. M. Cheung",
    "id": "2275064497",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Jia-Yeu Lin",
    "id": "9570197",
    "h_index": 5,
    "papers": 20
   }
  ],
  "comment": "Opinion paper accepted for presentation at the Sound and Music Computing (SMC) Conference 2026 (5-7 November in Zagreb, Croatia)",
  "topics": [
   "humanoids",
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.20433v1",
  "pdf_url": "https://arxiv.org/pdf/2608.20433v1",
  "html_url": "https://arxiv.org/html/2608.20433v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.20284",
  "slug": "towards-surgical-world-action-modeling-a-preliminary-joint-visual-traj",
  "title": "Towards Surgical World-Action Modeling: A Preliminary Joint Visual-Trajectory Forecasting for Surgical Motion Planning",
  "abstract": "Reliable surgical planning requires models to anticipate not only how instruments will move, but also how the operative visual state will evolve together with such motion. Existing approaches typically treat future scene generation and instrument trajectory prediction as two separate tasks. Scene-only models cannot directly evaluate the accuracy of future instrument motion at the trajectory level, while trajectory-only models fail to capture the visual consequences of instrument movement, leaving the consistency between predicted trajectories and future scene evolution unaddressed. Jointly forecasting both provides a more complete account of surgical action-scene dynamics by enabling explicit trajectory-level evaluation while simultaneously modeling the corresponding visual evolution. To bridge this gap, we present a preliminary joint visual-trajectory world-action model that simultaneously forecasts future visual states and instrument trajectories from historical surgical observations. Specifically, we encode historical video frames and tool trajectories into latent representations, which are processed by a temporal-spatial encoder and subsequently decoded through separate visual-state and trajectory prediction heads. Based on this preliminary architecture, a chunked autoregressive rollout is repeatedly applied to predict fifteen future steps. The chunked strategy consistently outperforms direct one-shot prediction across all evaluated horizons, improving first-segment PSNR from 18.86 to 23.11 dB and reducing ADE from 45.77 to 22.22 pixels. These results demonstrate the initial feasibility of joint visual-motion forecasting. However, we observe progressive visual degradation and accumulated trajectory errors over longer prediction horizons, which remain important challenges for future surgical world-action modeling.",
  "published": "2026-08-20",
  "updated": "2026-08-20",
  "year": "2026",
  "authors": [
   "Weiliang Huang",
   "Huanrong Liu",
   "Bob Zhang",
   "Qi Dou",
   "Zhen Chen",
   "Yun Gu",
   "Guy Rosman",
   "Qingbiao Li"
  ],
  "author_count": 8,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A preliminary joint visual-trajectory world-action model is presented that simultaneously forecasts future visual states and instrument trajectories from historical surgical observations and demonstrates the initial feasibility of joint visual-motion forecasting.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Weiliang Huang",
    "id": "2353661364",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Huanrong Liu",
    "id": "2269890777",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Bobo Zhang",
    "id": "2456270150",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Qi Dou",
    "id": "2287929188",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Zhen Chen",
    "id": "2458694587",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Yun Gu",
    "id": "2240758345",
    "h_index": 1,
    "papers": 10
   },
   {
    "name": "Guy Rosman",
    "id": "2116952",
    "h_index": 27,
    "papers": 113
   },
   {
    "name": "Qingbiao Li",
    "id": "2108053899",
    "h_index": 12,
    "papers": 33
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.20284v1",
  "pdf_url": "https://arxiv.org/pdf/2608.20284v1",
  "html_url": "https://arxiv.org/html/2608.20284v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.20275",
  "slug": "dart-s-reachability-audited-active-suspension-preconditioning-for-off",
  "title": "DART-S: Reachability-Audited Active-Suspension Preconditioning for Off-Road Vehicle Jumps",
  "abstract": "Airborne torque reaction cannot recover takeoff errors beyond the wheel angular-momentum budget. DART-S applies ramp-face suspension preconditioning to change pitch, pitch rate, and wheel spin before liftoff, thereby shifting the queried state and altering the remaining authority budget. To predict how each suspension action reshapes this state-budget pair, DART-S employs a local calibration map. A support-aware selector combines the predicted shift with local outcome evidence and an interval-reachability screen; an exact-pair audit reports residual authority. Across 600 new runs in 72 independent BeamNG sessions, every positive, negative, and boundary query follows its prespecified branch. At the confirmed 40\u00b0/13 m/s boundary, DART-S attains 24/24 post-touchdown attitude-criterion successes versus 0/24 for DART (session-level Holm-adjusted p=0.0234). At 11.5 m/s, a 0.35 s timing action attains 23/24 versus 0/24 for the static preset (p=0.0156). The 200 rad/s command guard keeps drivetrain hard-limit exceedance at zero across all 600 runs. The source code will be available at https://github.com/MeridianCAS/DART-S",
  "published": "2026-08-20",
  "updated": "2026-08-20",
  "year": "2026",
  "authors": [
   "Yu Hu",
   "Fangzhou Zhao",
   "Liang Chen",
   "Chen Min",
   "Wei Li",
   "Mingyuan Sang",
   "Jiajia Ma",
   "Shican Chen",
   "Di Pang",
   "Baolei Chen"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "DART-S applies ramp-face suspension preconditioning to change pitch, pitch rate, and wheel spin before liftoff, thereby shifting the queried state and altering the remaining authority budget.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yu Hu",
    "id": "2274058079",
    "h_index": 4,
    "papers": 17
   },
   {
    "name": "Fangzhou Zhao",
    "id": "2314064471",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Liang Chen",
    "id": "2311506031",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Chen Min",
    "id": "2061285173",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Wei Li",
    "id": "2278779611",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Mingyuan Sang",
    "id": "2422377105",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Jiajia Ma",
    "id": "2458707435",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Shican Chen",
    "id": "2455130827",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Di Pang",
    "id": "2458624456",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Baolei Chen",
    "id": "2455130857",
    "h_index": 0,
    "papers": 1
   }
  ],
  "comment": "9 pages, 6 figures",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.20275v1",
  "pdf_url": "https://arxiv.org/pdf/2608.20275v1",
  "html_url": "https://arxiv.org/html/2608.20275v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.20251",
  "slug": "video2doortraversal-push-door-traversal-via-simulated-door-twins",
  "title": "Video2DoorTraversal: Push Door Traversal via Simulated Door Twins",
  "abstract": "Door opening and traversal is a long-horizon loco-manipulation task that requires precise handle interaction and coordinated base-arm control. We present Video2DoorTraversal, a single-video real-to-sim-to-real framework for wheel-legged mobile manipulators. Given one RGB video of a real door, DoorTwin reconstructs an instance-aligned, articulated, and simulation-ready door twin with realistic geometry and appearance. A simulation-in-the-loop agent converts the recovered articulation into a parameterized skill program and iteratively refines failed rollouts to generate physically executable demonstrations. These demonstrations are used to train ArticuACT, a dual-depth policy that predicts coordinated base, arm, and gripper commands using robot-centric camera conditioning and interaction-aware supervision. With all perception and policy inference running onboard, the system achieves a 96.57% average success rate across five real doors and an 80.95% zero-shot success rate on structurally similar unseen doors, while completing the full approach, opening, and traversal sequence in approximately 13s on average. Project Page: https://video2doortraversal.github.io/.",
  "published": "2026-08-20",
  "updated": "2026-08-20",
  "year": "2026",
  "authors": [
   "Xincheng Tang",
   "Yiji Chen",
   "Youhan Xie",
   "Wanyu Li",
   "Zhengjie Shu",
   "Lai Jiang",
   "Wenkang Hu",
   "Yitong Li",
   "Jinchuang Zhang",
   "Xibin Song",
   "Ruigang Yang"
  ],
  "author_count": 11,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Video2DoorTraversal, a single-video real-to-sim-to-real framework for wheel-legged mobile manipulators and ArticuACT, a dual-depth policy that predicts coordinated base, arm, and gripper commands using robot-centric camera conditioning and interaction-aware supervision are presented.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xinchen Tang",
    "id": "2339261250",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Yijia Chen",
    "id": "2457936020",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Youhan Xie",
    "id": "2445236311",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Wanyu Li",
    "id": "2338507078",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Zhengjie Shu",
    "id": "2372589987",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Lai Jiang",
    "id": "2348880772",
    "h_index": 4,
    "papers": 14
   },
   {
    "name": "Wenkang Hu",
    "id": "2239034488",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Yitong Li",
    "id": "2304609281",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Jinchuan Zhang",
    "id": "2362749014",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Xibin Song",
    "id": "2243008885",
    "h_index": 5,
    "papers": 17
   },
   {
    "name": "Ruigang Yang",
    "id": "2271528832",
    "h_index": 0,
    "papers": 5
   }
  ],
  "comment": "8 pages, 6 figures",
  "topics": [
   "humanoids",
   "sim2real"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.20251v1",
  "pdf_url": "https://arxiv.org/pdf/2608.20251v1",
  "html_url": "https://arxiv.org/html/2608.20251v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.20114",
  "slug": "decowam-decoupled-whole-body-world-action-model-for-legged-mobile-mani",
  "title": "DECOWAM: Decoupled Whole-Body World-Action Model for Legged Mobile Manipulation",
  "abstract": "Mobile manipulation requires a robot to predict how locomotion and arm motion jointly alter future observations and control. Existing world-action models, developed largely for fixed-base platforms, do not explicitly distinguish camera ego-motion from base and arm actions. Here we introduce DECOWAM, a whole-body world-action model that separates these factors through dedicated conditional interfaces. DECOWAM freezes an adapted FastWAM backbone and trains residual adapters, an action-equivalent future bottleneck distilled from privileged observations, adversarially separated base and arm latents, and base-velocity conditioning for video prediction. We further introduce ARMDOG, a real-robot dataset that synchronizes video, whole-body state and action, and language. On a fixed replay protocol, DECOWAM improved both future-video and action prediction over FastWAM, reducing action MSE by 21.7% with 25.95M trainable adaptation parameters. Across 79 closed-loop trials per method, it achieved the highest observed whole-body coordination and base-displacement robustness among the compared systems, while task completion remained comparable to the strongest baseline. These results show that embodiment-aware factorization can support parameter-efficient joint visual prediction and whole-body control under moving viewpoints.",
  "published": "2026-08-20",
  "updated": "2026-08-21",
  "year": "2026",
  "authors": [
   "Siyuan Ma",
   "Boshi Zhang",
   "Yutian Zhang",
   "Qinglian Wu",
   "Jiaqi Zhai",
   "Dong Wei",
   "Qiaojun Yu"
  ],
  "author_count": 7,
  "categories": [
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.AI",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "DECOWAM is introduced, a whole-body world-action model that separates camera ego-motion from base and arm actions through dedicated conditional interfaces and shows that embodiment-aware factorization can support parameter-efficient joint visual prediction and whole-body control under moving viewpoints.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Siyuan Ma",
    "id": "2458696249",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Boshi Zhang",
    "id": "2458694154",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Yutian Zhang",
    "id": "2453953893",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Qinglian Wu",
    "id": "2458705991",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Jiaqi Zhai",
    "id": "2458627634",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Dongjun Wei",
    "id": "2457983330",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Qiaojun Yu",
    "id": "2153795608",
    "h_index": 10,
    "papers": 22
   }
  ],
  "comment": "8 pages, 5 figures. Introduces DECOWAM, a decoupled whole-body world-action model for legged mobile manipulation, and the ARMDOG real-robot dataset",
  "topics": [
   "world-models",
   "humanoids",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.20114v2",
  "pdf_url": "https://arxiv.org/pdf/2608.20114v2",
  "html_url": "https://arxiv.org/html/2608.20114v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.20111",
  "slug": "planning-oriented-end-to-end-autonomous-driving-architectures-evaluati",
  "title": "Planning-Oriented End-to-End Autonomous Driving: Architectures, Evaluation, and Emerging Paradigms",
  "abstract": "End-to-end autonomous driving has evolved from camera-to-control regression toward planning-oriented systems that use structured representations, trajectory-level outputs, and increasingly realistic evaluation protocols. This survey reviews this transition across behavior cloning, conditional imitation learning, privileged distillation, BEV and vectorized planning, unified perception-prediction-planning architectures, world-model-based planners, and vision-language-action systems. We argue that the key distinction in modern end-to-end driving is not whether intermediate representations are used, but whether they are learned, supervised, and evaluated to support safe, feasible, and route-compliant planning. To organize the literature, we synthesize existing methods along four axes: input representation, planning output, supervision signal, and evaluation protocol. We further examine the benchmark shift from open-loop trajectory matching to closed-loop simulation, non-reactive real-log evaluation, long-tail testing, and human-preference-aware metrics. Our analysis highlights that architectural progress is difficult to interpret without benchmark-consistent evaluation, and that displacement-based open-loop metrics alone provide limited evidence for safe and human-aligned driving. We conclude with open challenges in uncertainty-aware planning, learner-expert mismatch, runtime safety assurance, language-action grounding, world-model validation, and reproducible benchmarking.",
  "published": "2026-08-20",
  "updated": "2026-08-20",
  "year": "2026",
  "authors": [
   "Yanchen Guan",
   "Xingcheng Liu",
   "Bin Rao",
   "Chengyue Wang",
   "Guofa Li",
   "Yunjian Li",
   "Lishengsa Yue",
   "Zhiyong Cui",
   "Chengzhong Xu",
   "Zhenning Li"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO",
   "cs.ET"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This survey reviews this transition across behavior cloning, conditional imitation learning, privileged distillation, BEV and vectorized planning, unified perception-prediction-planning architectures, world-model-based planners, and vision-language-action systems, finding open challenges in uncertainty-aware planning, learner-expert mismatch, runtime safety assurance, language-action grounding, world-model validation, and reproducible benchmarking.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yanchen Guan",
    "id": "2283730555",
    "h_index": 6,
    "papers": 22
   },
   {
    "name": "Xingcheng Liu",
    "id": "2373678154",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Bin Rao",
    "id": "2335956675",
    "h_index": 3,
    "papers": 17
   },
   {
    "name": "Chengyue Wang",
    "id": "2272610917",
    "h_index": 14,
    "papers": 47
   },
   {
    "name": "Guofa Li",
    "id": "2272344125",
    "h_index": 7,
    "papers": 12
   },
   {
    "name": "Yunjian Li",
    "id": "2211238884",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Lishengsa Yue",
    "id": "2281352151",
    "h_index": 6,
    "papers": 18
   },
   {
    "name": "Zhiyong Cui",
    "id": "2288423696",
    "h_index": 7,
    "papers": 14
   },
   {
    "name": "Cheng-zhong Xu",
    "id": "2272208211",
    "h_index": 15,
    "papers": 33
   },
   {
    "name": "Zhenning Li",
    "id": "2273365115",
    "h_index": 15,
    "papers": 53
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "vla",
   "sim2real",
   "imitation-diffusion",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.20111v1",
  "pdf_url": "https://arxiv.org/pdf/2608.20111v1",
  "html_url": "https://arxiv.org/html/2608.20111v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.20087",
  "slug": "towards-professional-tennis-styles-for-humanoid-robots-with-adaptive-m",
  "title": "Towards Professional Tennis Styles for Humanoid Robots with Adaptive Motion Planning and Tracking",
  "abstract": "Humanoid robots have recently demonstrated promising capabilities in real-world ball sports. However, achieving professional motion styles while maintaining strong task performance remains challenging. In this work, we propose AdaPT, an Adaptive Motion Planning and Tracking framework that learns professional tennis serving and rally styles directly from broadcast videos. This hierarchical design is motivated by the key insight that the planner generates stylistic kinematic motions, while the tracker executes them with minimal interference with planning. Despite its effectiveness in simulation, a substantial sim-to-real gap emerges: tracking performance inevitably degrades on real robots, and this degradation is partially overlooked by autoregressive planning and further compounded by noisy perception. To address these issues, our adaptation mechanism improves tracking robustness by learning to track randomized execution speeds, while conditioning the planner on a learned motion-speed adapter to mitigate compounding errors. Real-world experiments on the Unitree G1 demonstrate the effectiveness of our adaptation mechanism in bridging the sim-to-real gap. We further deploy AdaPT policies on the full-size Dobot Atom humanoid robot (1.7m) and demonstrate in-the-wild serving without motion capture. Beyond these results, our real-world experiments reveal both algorithmic and engineering insights for future humanoid ball-sports systems. Videos and code are available on our \\href{https://humanoidtennis.github.io/AdaPT/}{project website}.",
  "published": "2026-08-20",
  "updated": "2026-08-20",
  "year": "2026",
  "authors": [
   "Tao Huang",
   "Ruofei Liu",
   "Xuchen Tang",
   "Xinyin Zhang",
   "Junli Ren",
   "Huayi Wang",
   "Feiyu Jia",
   "Yukai Qi",
   "Kangning Yin",
   "Weishuai Zeng",
   "Lipeng Chen",
   "Xi Li",
   "Ting Wu",
   "Kailin Li",
   "Ruoli Dai",
   "Jingbo Wang",
   "Lei Han",
   "Jiangmiao Pang"
  ],
  "author_count": 18,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes AdaPT, an Adaptive Motion Planning and Tracking framework that learns professional tennis serving and rally styles directly from broadcast videos, and improves tracking robustness by learning to track randomized execution speeds, while conditioning the planner on a learned motion-speed adapter to mitigate compounding errors.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Tao Huang",
    "id": "2311459920",
    "h_index": 8,
    "papers": 11
   },
   {
    "name": "Ruofei Liu",
    "id": "2454707266",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Xuchen Tang",
    "id": "2458695610",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Xinyi Zhang",
    "id": "2457529687",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Junli Ren",
    "id": "2331744206",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "Huayi Wang",
    "id": "2315948077",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Feiyu Jia",
    "id": "2345925355",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Yukai Qi",
    "id": "2458659438",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Kangning Yin",
    "id": "2289841621",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Weishuai Zeng",
    "id": "2315923491",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Lipeng Chen",
    "id": "2244132623",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Xi Li",
    "id": "2458703388",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Ting Wu",
    "id": "2221276582",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Kailin Li",
    "id": "2023790905",
    "h_index": 9,
    "papers": 15
   },
   {
    "name": "Ruoli Dai",
    "id": "2347118701",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Jingbo Wang",
    "id": "2363513787",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "Lei Han",
    "id": "2362318590",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Jiangmiao Pang",
    "id": "2405891293",
    "h_index": 2,
    "papers": 17
   }
  ],
  "comment": "14 pages",
  "topics": [
   "humanoids",
   "egocentric-data",
   "sim2real",
   "safety-eval"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2608.20087v1",
  "pdf_url": "https://arxiv.org/pdf/2608.20087v1",
  "html_url": "https://arxiv.org/html/2608.20087v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2608.20084",
  "slug": "evidence-gated-task-and-motion-planning-with-vision-language-models",
  "title": "Evidence-Gated Task and Motion Planning with Vision-Language Models",
  "abstract": "Robots executing long-horizon manipulation tasks from natural-language instructions must reason about both semantic task structure and geometric feasibility. However, under partial observability, the availability of goal-relevant objects may be uncertain. In such cases, approaches that combine Vision-Language Models (VLMs) with Task and Motion Planning (TAMP) may generate subgoals that rely on the VLM's prior knowledge without observational support, leading to execution failures or unintended outcomes. We propose Evidence Acquisition and Feasibility Gating (EAFG), a framework that acquires visual evidence through VLM-generated exploratory subgoals and TAMP-based execution. EAFG then applies a feasibility gate to decide whether to proceed with task planning, acquire further evidence, or halt. Our experiments show that, in cooking tasks with ambiguous object use, EAFG improves recipe completion by discovering task-relevant objects before planning. For instructions requiring an absent object, EAFG promotes appropriate halt decisions and reduces repeated attempts to manipulate that object.",
  "published": "2026-08-20",
  "updated": "2026-08-20",
  "year": "2026",
  "authors": [
   "Tsunehiko Tanaka",
   "Matthew Stephenson",
   "Alistair Macvicar",
   "Edgar Simo-Serra"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Evidence Acquisition and Feasibility Gating (EAFG) is proposed, a framework that acquires visual evidence through VLM-generated exploratory subgoals and TAMP-based execution and applies a feasibility gate to decide whether to proceed with task planning, acquire further evidence, or halt.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Tsunehiko Tanaka",
    "id": "2283097692",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Matthew Stephenson",
    "id": "52149123",
    "h_index": 14,
    "papers": 65
   },
   {
    "name": "Alistair Macvicar",
    "id": "2341641905",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Edgar Simo-Serra",
    "id": "2282965779",
    "h_index": 3,
    "papers": 7
   }
  ],
  "comment": "",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.20084v1",
  "pdf_url": "https://arxiv.org/pdf/2608.20084v1",
  "html_url": "https://arxiv.org/html/2608.20084v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.20043",
  "slug": "wave-based-bilateral-teleoperation-between-nonlinear-manipulators-with",
  "title": "Wave-Based Bilateral Teleoperation between Nonlinear Manipulators with Direct Contact Force Feedback",
  "abstract": "We study bilateral teleoperation between nonlinear, multi-DOF robotic manipulators in the presence of constant communication delays. Unlike classical wave-transformation architectures that transmit a coordinating force, we consider the case where the environmental force is reflected to the master side to enhance teleoperation transparency. Since direct contact force feedback might destabilize the closed-loop system, we first develop a passivity-shortage characterization for the Euler--Lagrange remote system using a linear matrix inequality (LMI) approach. An upper strictly passive communication law is then employed to compensate for the computed passivity shortage so that the closed-loop stability under delays as well as position and force synchronization are preserved under appropriate conditions. Simulations with nonlinear 2-DOF robotic manipulators in different settings illustrate our approach.",
  "published": "2026-08-20",
  "updated": "2026-08-20",
  "year": "2026",
  "authors": [
   "G. Q. Bao Tran",
   "Takanori Miyoshi",
   "Ho Duc Tho"
  ],
  "author_count": 3,
  "categories": [
   "eess.SY",
   "cs.RO",
   "math.DS",
   "math.OC"
  ],
  "primary_category": "eess.SY",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work develops a passivity-shortage characterization for the Euler--Lagrange remote system using a linear matrix inequality (LMI) approach and employs an upper strictly passive communication law to compensate for the computed passivity shortage.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "G. B. Tran",
    "id": "2088078486",
    "h_index": 5,
    "papers": 24
   },
   {
    "name": "Takanori Miyoshi",
    "id": "2376450205",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Ho Duc Tho",
    "id": "23548736",
    "h_index": 7,
    "papers": 20
   }
  ],
  "comment": "65th IEEE Conference on Decision and Control (CDC), Honolulu, HI, USA, Dec. 2026",
  "topics": [
   "tactile",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.20043v1",
  "pdf_url": "https://arxiv.org/pdf/2608.20043v1",
  "html_url": "https://arxiv.org/html/2608.20043v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.19977",
  "slug": "learning-highly-dynamic-skills-transition-for-quadruped-jumping-throug",
  "title": "Learning Highly Dynamic Skills Transition for Quadruped Jumping Through Constrained Space",
  "abstract": "Although legged animals are capable of performing explosive motions while traversing confined spaces, replicating this behavior in quadrupedal robots has been a longstanding challenge. Here, we propose a hierarchical reinforcement learning pipeline that empowers the robots to perform aggressive locomotion through constrained obstacles--a narrow gate. The imitation learning technique is used to train the low-level policy, which mimics the behaviors of real animals and forms a set of diverse skills. The high-level controller, having an awareness of the capability of low-level skills and acquiring the gate information via vision-based detection, determines the suitable maneuvers with collision-free trajectories to traverse it dynamically. Notably, we also verify that this framework can be extended to other highly dynamic tasks. This is one of the first works that perform autonomous and agile aerial gate traversal tasks on ground-walking robots, extending the lifelike agility of legged robots to match that of their biological counterparts.",
  "published": "2026-08-20",
  "updated": "2026-08-20",
  "year": "2026",
  "authors": [
   "Zeren Luo",
   "Jiahui Zhang",
   "Yimin Han",
   "Ji Ma",
   "Minghao Lu",
   "Ioannis Havoutis",
   "Peng Lu"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "This work proposes a hierarchical reinforcement learning pipeline that empowers the robots to perform aggressive locomotion through constrained obstacles--a narrow gate, extending the lifelike agility of legged robots to match that of their biological counterparts.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zeren Luo",
    "id": "2265130531",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Jiahui Zhang",
    "id": "2336923892",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Yimin Han",
    "id": "2368345854",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Ji Ma",
    "id": "2336920494",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "M. Lu",
    "id": "2149494949",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Ioannis Havoutis",
    "id": "2281743264",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Peng Lu",
    "id": "2291231307",
    "h_index": 6,
    "papers": 22
   }
  ],
  "comment": "15 pages, 12 figures",
  "topics": [
   "humanoids",
   "imitation-diffusion",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.19977v1",
  "pdf_url": "https://arxiv.org/pdf/2608.19977v1",
  "html_url": "https://arxiv.org/html/2608.19977v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2608.19968",
  "slug": "pvra-a-pointwise-key-point-voting-framework-for-robotic-assembly",
  "title": "PVRA: A Pointwise Key-point Voting Framework for Robotic Assembly",
  "abstract": "Modern computer vision has enabled partial autonomy in robotic assembly manipulation. However, performing autonomous manipulation of a progressive assembly demands a more specific set of skills, in addition to perceiving the objects. Through a comparative analysis of research in the associated domains, we deduce that object-centric perception must advance towards learning assembly dependencies to predict meaningful actionable outputs for autonomous assembly manipulation. Subsequently, we present a 3D keypoint-based modular learning framework to learn assembly dependencies to infer actionable outputs given a RGB-D input of an assembly scene. We train and evaluate our trained network on an assembly pose estimation dataset and compare it against object-centric baselines with an augmented set of metrics for progressive assemblies.",
  "published": "2026-08-20",
  "updated": "2026-08-20",
  "year": "2026",
  "authors": [
   "Kulunu Samarawickrama",
   "Roel Pieters"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A 3D keypoint-based modular learning framework to learn assembly dependencies to infer actionable outputs given a RGB-D input of an assembly scene and compares it against object-centric baselines with an augmented set of metrics for progressive assemblies.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Kulunu Samarawickrama",
    "id": "2096953680",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Roel Pieters",
    "id": "2154221501",
    "h_index": 4,
    "papers": 10
   }
  ],
  "comment": "14 pages, 3 figures. Accepted for presentation at the European Conference on Robotics (ECoR) 2026",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.19968v1",
  "pdf_url": "https://arxiv.org/pdf/2608.19968v1",
  "html_url": "https://arxiv.org/html/2608.19968v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.19955",
  "slug": "mild-tractable-terrain-modeling-for-learning-improved-bipedal-locomoti",
  "title": "MILD: Tractable Terrain Modeling for Learning Improved Bipedal Locomotion on Deformable Surfaces",
  "abstract": "Enabling robots to walk on yielding terrain is vital for applications ranging from disaster response to planetary exploration. While bipedal robots hold immense potential, their locomotion on deformable surfaces remains limited as current simulators fail to capture the spatiotemporal heterogeneity of such yielding substrates. We present MILD, featuring a physics-grounded discrete-element contact solver that accurately simulates spatially varying foot-terrain interactions. Complementing this model, we train a terrain-aware locomotion controller via deep reinforcement learning with latent modulation and proprioceptive estimation. Quantitative comparisons against state-of-the-art methods show our approach generates more diverse and realistic contact scenarios during training, resulting in controllers that exhibit natural adaptation on real deformable surfaces. Through hardware experiments, we demonstrate the system's capability for online terrain identification and adaptation across a wide range of surface stiffness.",
  "published": "2026-08-20",
  "updated": "2026-08-20",
  "year": "2026",
  "authors": [
   "Zeren Luo",
   "Jiahui Zhang",
   "Zhe Xu",
   "Wanyue Li",
   "Xinqi Li",
   "Xuechao Chen",
   "Zhangguo Yu",
   "Annan Tang",
   "Peng Lu"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "RA-L",
  "venue_source": "semantic-scholar",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "MILD is presented, featuring a physics-grounded discrete-element contact solver that accurately simulates spatially varying foot-terrain interactions and train a terrain-aware locomotion controller via deep reinforcement learning with latent modulation and proprioceptive estimation.",
  "doi": "10.1109/LRA.2025.3645520",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zeren Luo",
    "id": "2265130531",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Jiahui Zhang",
    "id": "2336923892",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Zhe Xu",
    "id": "2300244077",
    "h_index": 4,
    "papers": 18
   },
   {
    "name": "Wanyue Li",
    "id": "2221132466",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Xinqi Li",
    "id": "2291314080",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Xuechao Chen",
    "id": "1849860",
    "h_index": 18,
    "papers": 227
   },
   {
    "name": "Zhangguo Yu",
    "id": "2291855583",
    "h_index": 4,
    "papers": 38
   },
   {
    "name": "Annan Tang",
    "id": "2273992257",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Peng Lu",
    "id": "2291231307",
    "h_index": 6,
    "papers": 22
   }
  ],
  "comment": "8 pages, 9 figures",
  "topics": [
   "humanoids",
   "sim2real",
   "rl-control",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.19955v1",
  "pdf_url": "https://arxiv.org/pdf/2608.19955v1",
  "html_url": "https://arxiv.org/html/2608.19955v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.8
 },
 {
  "id": "2608.19826",
  "slug": "calming-robot-pitches-exploring-the-influence-of-robot-voice-pitch-on",
  "title": "Calming Robot Pitches? Exploring the Influence of Robot Voice Pitch on Children's Stress Levels",
  "abstract": "This study examined whether variations in robot speech pitch influence children's stress levels during a robot-guided game. Although lower-pitched voices have been shown to facilitate stress regulation in human communication, it remains unclear whether this effect generalizes to synthetic voices in child-robot interactions. Twenty-seven Dutch children aged 8-12 years were randomly assigned to interact with a Zenbo Junior II robot using either a lower-pitched or a higher-pitched voice. The interaction consisted of an introduction followed by a timed LEGO-building game. Stress levels, measured with an adapted version of CAM-S, increased during the game, confirming the stress-inducing nature of the task. No differences emerged between pitch conditions. These findings suggest that the benefits of lower pitch in reducing stress may not directly translate to child-robot interactions. Possible explanations include children's developing sensitivity to emotional tone, mismatches between the robot's voice and appearance, or the use of fixed pitch changes that sound unnatural, since real speech varies dynamically across multiple dimensions. Future research examining combinations of prosodic cues (beyond pitch alone) could provide further insights and help inform robot voice design for effective stress regulation support for children.",
  "published": "2026-08-20",
  "updated": "2026-08-20",
  "year": "2026",
  "authors": [
   "Nina G. M. van Roij",
   "Emilia I. Barakova",
   "Briana Isaila",
   "Aoju Chen"
  ],
  "author_count": 4,
  "categories": [
   "cs.HC",
   "cs.RO"
  ],
  "primary_category": "cs.HC",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "It is suggested that the benefits of lower pitch in reducing stress may not directly translate to child-robot interactions, and combinations of prosodic cues could provide further insights and help inform robot voice design for effective stress regulation support for children.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Nina G. M. van Roij",
    "id": "2458627222",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Emilia I. Barakova",
    "id": "2266520925",
    "h_index": 5,
    "papers": 30
   },
   {
    "name": "Briana Isaila",
    "id": "2458627169",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Aoju Chen",
    "id": "2393380168",
    "h_index": 2,
    "papers": 2
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.19826v1",
  "pdf_url": "https://arxiv.org/pdf/2608.19826v1",
  "html_url": "https://arxiv.org/html/2608.19826v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.19794",
  "slug": "towards-general-embodied-intelligence-integrating-large-language-model",
  "title": "Towards general embodied intelligence: integrating large language models, knowledge bases, and reasoning capabilities to build the next generation of AI agents",
  "abstract": "The convergence of large language models (LLMs), structured knowledge bases (KBs), and reasoning ability (RA) presents a promising trajectory toward general embodied intelligence (GEI). This paper reviews the evolution of LLM-centered intelligent systems, emphasising their integration with knowledge representation, logical reasoning, and physical embodiment. We analyse LLM architectures, pre-training methods, and inference mechanisms, along with their interaction with external knowledge sources and structured reasoning frameworks. Furthermore, we examine embodied intelligence (EI) paradigms wherein agents learn and act in physical environments. To synthesise these dimensions, we present a conceptual framework that illustrates the synergy among LLMs, KBs, RA, and embodiment, serving as a guiding model for perception, reasoning, and action rather than an implemented engineering architecture. To advance toward GEI, we identify five key challenges: efficient LLM deployment, closed-loop knowledge integration, hybrid symbolic-neural reasoning, perception-action grounding, and continual learning. This survey provides a comprehensive roadmap for developing adaptive, multimodal agents capable of operating in complex, dynamic settings.",
  "published": "2026-08-20",
  "updated": "2026-08-20",
  "year": "2026",
  "authors": [
   "Fujiang Yuan",
   "Xia Huang",
   "Lusheng Wang",
   "Jun Ding",
   "Zhen Tian",
   "Yuxin Wang",
   "Shaojie Gu",
   "Yuki Funabora",
   "Yanhong Peng",
   "Zebing Mao"
  ],
  "author_count": 10,
  "categories": [
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.AI",
  "venue": "",
  "venue_source": "",
  "citations": 5,
  "influential_citations": 0,
  "tldr": "This survey provides a comprehensive roadmap for developing adaptive, multimodal agents capable of operating in complex, dynamic settings, and identifies five key challenges: efficient LLM deployment, closed-loop knowledge integration, hybrid symbolic-neural reasoning, perception-action grounding, and continual learning.",
  "doi": "10.1504/IJHM.2026.154223",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Fujiang Yuan",
    "id": "2352365048",
    "h_index": 8,
    "papers": 26
   },
   {
    "name": "Xia Huang",
    "id": "2352297033",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Lusheng Wang",
    "id": "2440936367",
    "h_index": 11,
    "papers": 38
   },
   {
    "name": "Jun Ding",
    "id": "2371108445",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Zhen Tian",
    "id": "2379617087",
    "h_index": 2,
    "papers": 11
   },
   {
    "name": "Yuxin Wang",
    "id": "2335589760",
    "h_index": 4,
    "papers": 30
   },
   {
    "name": "Shaojie Gu",
    "id": "2352893014",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Yuki Funabora",
    "id": "2377060",
    "h_index": 8,
    "papers": 154
   },
   {
    "name": "Yanhong Peng",
    "id": "2379608589",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Zebing Mao",
    "id": "2301161913",
    "h_index": 3,
    "papers": 4
   }
  ],
  "comment": "",
  "topics": [
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.19794v1",
  "pdf_url": "https://arxiv.org/pdf/2608.19794v1",
  "html_url": "https://arxiv.org/html/2608.19794v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.78
 },
 {
  "id": "2608.19776",
  "slug": "cotograsp-contact-topology-conditioned-dexterous-grasp-synthesis-via-c",
  "title": "CoToGrasp: Contact-Topology-Conditioned Dexterous Grasp Synthesis via Canonical Workspace Learning",
  "abstract": "Current dexterous grasp planners primarily optimize for physical stability, focusing on whether an object can be grasped rather than how it should be grasped to support downstream functional tasks. However, conditioning grasp synthesis on specific human grasp taxonomies typically requires prohibitively expensive, object-annotated datasets. To address these limitations, we propose CoToGrasp, a novel generative framework that synthesizes diverse, stable grasps strictly conditioned on specific contact topologies. To bypass the data collection bottleneck, CoToGrasp is trained entirely in an object-agnostic manner. We introduce a feature-based canonical workspace that projects local object features into a unified gripper-centric domain, effectively decoupling the semantic functional intent from the arbitrary object geometry. By learning the intrinsic contact manifold of the gripper within this workspace, our model achieves zero-shot generalization to unseen objects at inference. Extensive evaluations on the large-scale DexGraspNet dataset demonstrate that CoToGrasp achieves state-of-the-art performance, outperforming existing taxonomy-guided planners. Finally, we demonstrate the physical viability and kinematic feasibility of our synthesized contact topologies on a physical robot platform. Code is available on our project website at https://cea-list.github.io/cotograspweb/ .",
  "published": "2026-08-20",
  "updated": "2026-08-21",
  "year": "2026",
  "authors": [
   "Julien Merand",
   "Boris Meden",
   "Liming Chen",
   "Mathieu Grossard"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "CoToGrasp is a novel generative framework that synthesizes diverse, stable grasps strictly conditioned on specific contact topologies, and introduces a feature-based canonical workspace that projects local object features into a unified gripper-centric domain, effectively decoupling the semantic functional intent from the arbitrary object geometry.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Julien M\u00e9rand",
    "id": "2381850286",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Boris Meden",
    "id": "2748324",
    "h_index": 4,
    "papers": 16
   },
   {
    "name": "Liming Chen",
    "id": "2349738633",
    "h_index": 2,
    "papers": 11
   },
   {
    "name": "Mathieu Grossard",
    "id": "2395951347",
    "h_index": 1,
    "papers": 5
   }
  ],
  "comment": "Project website at https://cea-list.github.io/cotograspweb/",
  "topics": [
   "dexterous-manipulation",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.19776v2",
  "pdf_url": "https://arxiv.org/pdf/2608.19776v2",
  "html_url": "https://arxiv.org/html/2608.19776v2",
  "code_url": "https://cea-list.github.io/cotograspweb/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.3
 },
 {
  "id": "2608.19759",
  "slug": "goag-generative-and-object-agnostic-grasp-planner-for-dexterous-roboti",
  "title": "GOAG: Generative and Object-Agnostic Grasp Planner for Dexterous Robotic Manipulation",
  "abstract": "Multifingered grasping is a crucial robotic skill, but current deep-learning grasp planners often struggle to generalize to new objects because they are trained on limited, object-specific datasets. We introduce a fundamentally different approach, grounded in the observation that the gripper and the object share identical surface geometry at their mutual contact points. We propose GOAG: Generative and Object-Agnostic Grasp Planner for Dexterous Robotic Manipulation, a novel deep generative model that learns a compact latent representation of a specific gripper's contact surface distribution, enabling the efficient sampling of valid grasp configurations without relying on object-specific training data. We show that by introducing object features only at inference time, our model can effectively retrieve admissible contact areas that are compatible with the gripper's capabilities. We validate our approach through extensive experiments on established grasp protocols in both simulated and real-world scenarios, demonstrating its effectiveness with different grippers from the literature. Our method delivers state-of-the-art results on the objects from the MultiDex dataset, achieving an average success rate of 86.93%. It offers significantly faster processing when generating numerous grasps, while matching the performance of leading approaches specifically trained on this dataset. Unlike these methods, our approach does not rely on object-specific training data, highlighting the advantages of object-agnostic learning. It effectively addresses the generalization challenges faced by traditional data-driven grasp planners. Code and videos are available on our project website https://cea-list.github.io/goagweb/ .",
  "published": "2026-08-20",
  "updated": "2026-08-20",
  "year": "2026",
  "authors": [
   "Julien Merand",
   "Boris Meden",
   "Mathieu Grossard",
   "Liming Chen"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "GOAG is proposed: Generative and Object-Agnostic Grasp Planner for Dexterous Robotic Manipulation, a novel deep generative model that learns a compact latent representation of a specific gripper's contact surface distribution, enabling the efficient sampling of valid grasp configurations without relying on object-specific training data.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Julien M\u00e9rand",
    "id": "2381850286",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Boris Meden",
    "id": "2748324",
    "h_index": 4,
    "papers": 16
   },
   {
    "name": "Mathieu Grossard",
    "id": "2395951347",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Liming Chen",
    "id": "2349738633",
    "h_index": 2,
    "papers": 11
   }
  ],
  "comment": "Project website: https://cea-list.github.io/goagweb/",
  "topics": [
   "dexterous-manipulation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.19759v1",
  "pdf_url": "https://arxiv.org/pdf/2608.19759v1",
  "html_url": "https://arxiv.org/html/2608.19759v1",
  "code_url": "https://cea-list.github.io/goagweb/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.3
 },
 {
  "id": "2608.19740",
  "slug": "keeping-the-franka-emika-panda-alive-a-ros-2-stack-with-a-reliable-pos",
  "title": "Keeping the Franka Emika Panda alive: a ROS 2 stack with a reliable position interface",
  "abstract": "This paper presents an open-source software stack that restores ROS 2 support for the Franka Emika Panda robot while resolving the long-standing unreliability of its external position control interface. We first analyze the root causes of unstable position control and show that the observed vibrations and protective stops arise from the timing of the external control loop and sampling jitter, rather than from limitations of the robot itself. Building on this analysis, we introduce an asynchronous hardware interface that decouples real-time communication from the ROS 2 control loop, a rate-matching mechanism for slower command sources, and a position-domain reference generation strategy that produces reliable, smooth position commands. Experimental validation shows that the proposed architecture reliably tracks velocity references by reducing motion artifacts introduced by the official implementation, and the stack is validated across motion planning, compliance control, position-controlled manipulation, and haptic teleoperation on two independent Panda platforms. By restoring a modern, reliable, and open ROS 2 ecosystem for the Panda, this work lowers the barrier to developing safe, responsive, and reproducible human-robot collaboration applications that integrate planning, perception, interaction, and shared autonomy. Code and videos are available on our website at https://sites.google.com/view/fer-ros2/.",
  "published": "2026-08-20",
  "updated": "2026-08-20",
  "year": "2026",
  "authors": [
   "Antonio Langella",
   "Davide Risi",
   "Vincenzo Petrone",
   "Enrico Ferrentino",
   "Pasquale Chiacchio"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "An open-source software stack that restores ROS 2 support for the Franka Emika Panda robot while resolving the long-standing unreliability of its external position control interface and introduces an asynchronous hardware interface that decouples real-time communication from the ROS 2 control loop.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "A. Langella",
    "id": "40928025",
    "h_index": 28,
    "papers": 170
   },
   {
    "name": "D. Risi",
    "id": "2082202118",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "V. Petrone",
    "id": "47838431",
    "h_index": 11,
    "papers": 31
   },
   {
    "name": "Enrico Ferrentino",
    "id": "26974091",
    "h_index": 9,
    "papers": 35
   },
   {
    "name": "Pasquale Chiacchio",
    "id": "2261555049",
    "h_index": 3,
    "papers": 15
   }
  ],
  "comment": "12 pages, 10 figures, submitted to ICINCO 2026",
  "topics": [
   "data-teleop",
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.19740v1",
  "pdf_url": "https://arxiv.org/pdf/2608.19740v1",
  "html_url": "https://arxiv.org/html/2608.19740v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.19729",
  "slug": "safebranch-branch-pair-safety-alignment-for-embodied-agents",
  "title": "SafeBranch: Branch-Pair Safety Alignment for Embodied Agents",
  "abstract": "Vision-language-model-based embodied agents can complete instructed tasks but often violate safety constraints in the process, a problem recently framed as interactive safety. Training such agents to act safely is difficult, since safety and task success are distinct objectives, and safety arises only at a small number of safety-critical steps within a trajectory. Standard supervision is insufficient: imitating safe trajectories teaches behavior without explaining why it is safe, and contrasting arbitrary safe and unsafe trajectories mixes the safety signal with unrelated differences. We propose SafeBranch, a framework that aligns an embodied actor on safety through branch pairs constructed from the actor's own unsafe rollouts via environment rollback. SafeBranch rolls each unsafe rollout back to the safety-critical step that caused the violation, queries the actor for a safe alternative, and pairs the original action with the alternative so that the two branches differ only at that step. The trained actor acts safely at deployment with no critic in the loop. On IS-Bench, SafetyALFRED, and out-of-distribution variants with unseen tasks and objects, it handles safety reliably without sacrificing task success, achieving roughly ten times more safe successes than the untrained baseline on the unseen-object variant.",
  "published": "2026-08-20",
  "updated": "2026-08-20",
  "year": "2026",
  "authors": [
   "Hyunse Lee",
   "Jiwoo Jeong",
   "Haneul Lee",
   "Kyochul Jang",
   "Youngjae Yu",
   "Woojin Lee"
  ],
  "author_count": 6,
  "categories": [
   "cs.AI",
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.AI",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "SafeBranch is proposed, a framework that aligns an embodied actor on safety through branch pairs constructed from the actor's own unsafe rollouts via environment rollback, achieving roughly ten times more safe successes than the untrained baseline on the unseen-object variant.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hyunse Lee",
    "id": "2383107403",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Jiwoo Jeong",
    "id": "2335165657",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "H. Lee",
    "id": "2452058437",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Kyochul Jang",
    "id": "2304479297",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Youngjae Yu",
    "id": "2392301985",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Woojin Lee",
    "id": "2296232551",
    "h_index": 3,
    "papers": 15
   }
  ],
  "comment": "25 pages, 12 figures",
  "topics": [
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.19729v1",
  "pdf_url": "https://arxiv.org/pdf/2608.19729v1",
  "html_url": "https://arxiv.org/html/2608.19729v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.19684",
  "slug": "learning-hierarchical-skill-policies-with-offline-quality-diversity-re",
  "title": "Learning Hierarchical Skill Policies with Offline Quality-Diversity Reinforcement Learning",
  "abstract": "Recent studies investigate how to leverage pre-collected datasets to improve the policy performance and sample efficiency of RL. One promising approach to achieve this goal is to employ a two-stage strategy: In the first stage, diverse skills are extracted as a low-level policy from a given dataset, and a high-level policy is trained to solve a specific task in the second stage. Typically, extraction of the low-level policy is performed based on unsupervised learning such as trajectory VAE. However, a limitation of this approach is that the quality of the low-level policy highly depends on the quality of the dataset. To address this issue, we introduce QDOS (Quality-Diversity Offline Skill learning), a unified pipeline for robust offline-to-online learning. Our approach incorporates an Advantage-Weighted Quality-Diversity pretraining objective, which weights the skill extraction and diversity objectives by the estimated advantage of each trajectory segment. This approach allows the model to extract diverse and high-value skills. By providing robust and task-relevant skill representations, QDOS significantly improves the quality of the embedded skill space used by the low-level policy. We further integrate this with a dual dataset reuse strategy, where offline data is used both for skill pretraining and for populating the online replay buffer via pseudo-labeling. Experiments demonstrate that QDOS significantly outperforms strong baselines in structured manipulation tasks and unstructured locomotion tasks, confirming its ability to accelerate exploration and improve final returns in challenging sparse-reward domains.",
  "published": "2026-08-20",
  "updated": "2026-08-20",
  "year": "2026",
  "authors": [
   "Tanachai Anakewat",
   "Takayuki Osa",
   "Tatsuya Harada"
  ],
  "author_count": 3,
  "categories": [
   "cs.AI",
   "cs.LG",
   "cs.RO"
  ],
  "primary_category": "cs.AI",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "QDOS (Quality-Diversity Offline Skill learning), a unified pipeline for robust offline-to-online learning, incorporates an Advantage-Weighted Quality-Diversity pretraining objective, which weights the skill extraction and diversity objectives by the estimated advantage of each trajectory segment.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Tanachai Anakewat",
    "id": "2279830217",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Takayuki Osa",
    "id": "2253748734",
    "h_index": 5,
    "papers": 21
   },
   {
    "name": "Tatsuya Harada",
    "id": "2253762492",
    "h_index": 4,
    "papers": 20
   }
  ],
  "comment": "IROS 2026",
  "topics": [
   "humanoids",
   "rl-control",
   "navigation",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.19684v1",
  "pdf_url": "https://arxiv.org/pdf/2608.19684v1",
  "html_url": "https://arxiv.org/html/2608.19684v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.19671",
  "slug": "sage-ergodic-control-for-autonomous-and-adaptive-inspection-of-subsea",
  "title": "SAGE: Ergodic Control for Autonomous and Adaptive Inspection of Subsea Infrastructure",
  "abstract": "Subsea Christmas Trees (XTs) are underwater structures that use valves for directing oil flow, needing constant inspection. But not every valve carries the same risk at the same time: a valve with a suspected leak needs to be revisited far more often than one with a clean history, and that risk picture changes during the mission as new leaks are found. To handle this, we present SAGE (Semantic and Adaptive Generative Ergodicity), an ergodic-control architecture that allocates vehicle time in proportion to a live, sensor-derived risk distribution rather than a scripted route. We study a two-XT scenario, with five valves in total, and compare a fixed-loop A* tour against SAGE. Both methods can be tuned to spend similar total time near a high-risk valve, but only ergodic control also checks it more often: in simulation, a dominant-risk valve was revisited every 5.8 s under ergodic control against a fixed 8.1 s for every valve under A*, regardless of risk, so a leak can go unnoticed for barely two-thirds as long. Because the tracked distribution is recomputed rather than planned once, a newly detected leak shifts vehicle behavior on the next control cycle with no explicit re-planning step and no operator in the loop, which a fixed tour cannot do without a discrete re-route. We derive the ergodic control law behind this behavior and report simulation results on the five-valve scenario.",
  "published": "2026-08-20",
  "updated": "2026-08-20",
  "year": "2026",
  "authors": [
   "Markus Buchholz",
   "Ignacio Carlucho",
   "Yvan R. Petillot"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Markus Buchholz",
    "id": "2322676042",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Ignacio Carlucho",
    "id": "2282540912",
    "h_index": 4,
    "papers": 27
   },
   {
    "name": "Yvan R. P\u00e9tillot",
    "id": "2283839920",
    "h_index": 5,
    "papers": 29
   }
  ],
  "comment": "This work has been accepted to the IEEE IROS 2026 AQ2UASIM workshop",
  "topics": [
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.19671v1",
  "pdf_url": "https://arxiv.org/pdf/2608.19671v1",
  "html_url": "https://arxiv.org/html/2608.19671v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.19661",
  "slug": "world-model-grounded-llm-planning-for-auv-and-asv-navigation-near-offs",
  "title": "World-Model-Grounded LLM Planning for AUV and ASV Navigation Near Offshore Wind Farms",
  "abstract": "Large language models can turn a natural-language mission into a sequence of robot actions, but they do not have a sense of physics: they cannot judge how long a command should run, or whether it will make the robot drift into an obstacle. We proposed the use of a world model to expand the capabilities of Large Language model-based planners. Our method has three components: a physics-grounded neural world model, a three-phase gradient-based trajectory optimizer, and a Model Predictive Controller (MPC)-style closed-loop replanner with a trust-region guard. The language model decides what to do, and the world model decides how long, whether that means driving eight thrusters through 6 DOF or two differential thrusters through 3 DOF. We evaluate two marine vehicle classes operating near offshore wind infrastructure: a 6-DOF Autonomous Underwater Vehicle (AUV) and a 3-DOF differential-drive Autonomous Surface Vehicle (ASV). In five benchmark missions per platform, both vehicles reach every goal with zero predicted collisions, and both transfer to GazeboSim under ocean current, waves, and thruster dynamics, remaining collision-free and cutting GazeboSim goal-distance error versus the ungrounded baseline by 70-82% (ASV) and roughly 93% (AUV), after a residual fine-tuning pass that separately reduces surrogate rollout Root Mean Square Error (RMSE) by 60% (AUV) and 69% (ASV). For the ASV we further demonstrate a Vision language model (VLM)-assisted semantic-mapping pipeline that extracts obstacles and environmental context from satellite imagery, nautical charts, and forecast Application Programming Interface (API) instead of onboard sensors, reaching 96% navigability accuracy as a drop-in replacement for hand-specified obstacle geometry.",
  "published": "2026-08-20",
  "updated": "2026-08-20",
  "year": "2026",
  "authors": [
   "Markus Buchholz",
   "Ignacio Carlucho",
   "Yvan R. Petillot"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A Vision language model (VLM)-assisted semantic-mapping pipeline is demonstrated that extracts obstacles and environmental context from satellite imagery, nautical charts, and forecast Application Programming Interface (API) instead of onboard sensors, reaching 96% navigability accuracy as a drop-in replacement for hand-specified obstacle geometry.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Markus Buchholz",
    "id": "2322676042",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Ignacio Carlucho",
    "id": "2282540912",
    "h_index": 4,
    "papers": 27
   },
   {
    "name": "Yvan R. P\u00e9tillot",
    "id": "2283839920",
    "h_index": 5,
    "papers": 29
   }
  ],
  "comment": "This work has been accepted to the IEEE IROS 2026 AQ2UASIM workshop",
  "topics": [
   "world-models",
   "rl-control",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.19661v1",
  "pdf_url": "https://arxiv.org/pdf/2608.19661v1",
  "html_url": "https://arxiv.org/html/2608.19661v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.19635",
  "slug": "magnetically-self-sealed-mr-haptic-actuator-with-pwm-based-excitation",
  "title": "Magnetically Self-Sealed MR Haptic Actuator With PWM-Based Excitation and High-Fidelity Torque Control",
  "abstract": "Accurate and stable torque rendering is essential for safe and perceptive human--machine interaction. Magnetorheological fluid (MRF)-based actuators offer a compact and rapidly controllable solution for haptic feedback, but their practical implementation requires reliable fluid sealing, low-hysteresis excitation, accurate torque control, and stable long-duration operation. This article presents an integrated MRF haptic system featuring a compact magnetically self-sealed rotary actuator, low-hysteresis PWM operation, high-fidelity model-based torque rendering, and stable performance during long-time operation. Magnetostatic simulation guides the arrangement of magnetic and nonmagnetic materials to focus flux in the multidisk torque and permanent-magnet sealing regions, enabling a maximum 600 N$\\cdot$mm/A output. Experiments show that higher PWM frequencies reduce hysteresis and improve repeatability. At 10 kHz, the response is represented by a nonlinear model that varies with the direction and speed of torque change. The real-time controller combines feedforward, hysteresis compensation, PI feedback, and sliding-mode correction. Compared with PID, it reduces square-wave overshoot, undershoot, and steady-state RMSE by 77.4\\%, 61.9\\%, and 68.3\\%, respectively. It tracks sinusoidal and biomechanics-model-based references, and a 1.5-h test shows only a 2.5 $^\\circ$C rise near the coil with no clear tracking loss. This high-fidelity torque rendering will fundamentally transform human--robot collaboration by making interactions safer, more efficient, and more intuitive.",
  "published": "2026-08-20",
  "updated": "2026-08-20",
  "year": "2026",
  "authors": [
   "Dong Qiang",
   "Tian Yuan",
   "Song Yang",
   "Kequan Xia",
   "Thomas Reddyhoff",
   "Yikun Zhang",
   "Cheng Cheng",
   "Min Yu"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Dong Qiang",
    "id": "2064237710",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Tian Yuan",
    "id": "2454919613",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Song Yang",
    "id": "2364694887",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Kequan Xia",
    "id": "2333333089",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "T. Reddyhoff",
    "id": "2310346038",
    "h_index": 3,
    "papers": 19
   },
   {
    "name": "Yikun Zhang",
    "id": "2458699690",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Cheng Cheng",
    "id": "2116393639",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Min Yu",
    "id": "2319846514",
    "h_index": 3,
    "papers": 8
   }
  ],
  "comment": "Submitted to IEEE/ASME Transactions on Mechatronics. 16 pages, 9 figures, including supplementary material",
  "topics": [
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.19635v1",
  "pdf_url": "https://arxiv.org/pdf/2608.19635v1",
  "html_url": "https://arxiv.org/html/2608.19635v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.19613",
  "slug": "what-matters-for-latent-actions-in-robot-learning",
  "title": "What Matters for Latent Actions in Robot Learning",
  "abstract": "Latent Action Models (LAMs) have emerged as a promising paradigm for enabling robot learning to leverage large-scale unlabeled videos through latent actions that serve as compact surrogates for physical actions. Despite rapid progress, research on LAM remains highly fragmented, with existing methods evaluating different design choices in isolation under inconsistent experimental settings, making it difficult to identify the factors that truly determine downstream robotic manipulation performance. In this work, we present the first comprehensive empirical study of latent action learning for robotic manipulation. We unify representative LAM methods within a common autoencoding framework and systematically investigate 41 LAM design choices across three dimensions, including latent action modeling paradigms, learning objectives and regularization methods, and latent action integration strategies. We further examine four proxy metrics for evaluating latent action quality and assess their ability to reliably predict downstream robotic manipulation performance. Extensive experiments on three widely used benchmarks provide strong empirical evidence that fine-tuning vision-language model (VLM) backbones with latent actions provides a stronger initialization for downstream policy learning, with further validation on real-world robot manipulation tasks.",
  "published": "2026-08-20",
  "updated": "2026-08-20",
  "year": "2026",
  "authors": [
   "Xizhou Bu",
   "Qingda Hu",
   "Lei Zhou",
   "Lingfeng Zhang",
   "Yingbo Tang",
   "Zihao Liu",
   "Xinyi Tao",
   "Zhiqiang Ma",
   "Qingqiu Huang",
   "Chufeng Tang",
   "Hongbo Wang",
   "Jing Zhang",
   "Jiayi Ma",
   "Hangjun Ye",
   "Wei Li",
   "Xiaoshuai Hao"
  ],
  "author_count": 16,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work unify representative LAM methods within a common autoencoding framework and systematically investigate 41 LAM design choices across three dimensions, including latent action modeling paradigms, learning objectives and regularization methods, and latent action integration strategies.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xizhou Bu",
    "id": "2275199622",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Qingda Hu",
    "id": "2319805850",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Lei Zhou",
    "id": "2454198297",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Lingfeng Zhang",
    "id": "2327199957",
    "h_index": 5,
    "papers": 16
   },
   {
    "name": "Yingbo Tang",
    "id": "2349203748",
    "h_index": 7,
    "papers": 17
   },
   {
    "name": "Zihao Liu",
    "id": "2281025215",
    "h_index": 2,
    "papers": 15
   },
   {
    "name": "Xinyi Tao",
    "id": "2458575441",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Zhiqiang Ma",
    "id": "49975866",
    "h_index": 20,
    "papers": 62
   },
   {
    "name": "Qingqiu Huang",
    "id": "2342351978",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Chufeng Tang",
    "id": "2340583",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Hongbo Wang",
    "id": "2191361147",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Jing Zhang",
    "id": "2455886898",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Jiayi Ma",
    "id": "2261475380",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Hangjun Ye",
    "id": "2384401186",
    "h_index": 7,
    "papers": 34
   },
   {
    "name": "Wei Li",
    "id": "2372550386",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Xiaoshuai Hao",
    "id": "2386788371",
    "h_index": 5,
    "papers": 21
   }
  ],
  "comment": "Project page: https://carldegio.github.io/latent_action.github.io",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.19613v1",
  "pdf_url": "https://arxiv.org/pdf/2608.19613v1",
  "html_url": "https://arxiv.org/html/2608.19613v1",
  "code_url": "https://carldegio.github.io/latent_action.github.io",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2608.19593",
  "slug": "the-verification-gap-in-networked-physical-ai-a-post-semantic-communic",
  "title": "The Verification Gap in Networked Physical AI: A Post-Semantic Communication Framework",
  "abstract": "A task-effective proposal is not yet a justified physical action. In networked Physical AI, a proposal may be understood while valid, timely, proposal-bound evidence or the authority required to finalize an action remains unavailable. We call this mismatch the verification gap and introduce a Post-Semantic Communication Framework for the systems interface between proposal formation and physical execution. The framework begins with application-declared evidence requirements, represents qualifying observations as evidence records, validates supporting and conflicting records through one path, and separates evidence sufficiency from authorized finalization and a downstream runtime gate. It further distinguishes evidence transfer, which can enlarge the record set reachable by a finalizer, from evidence coordination, which can suppress transmission around records already held at the finalization endpoint. Finite-state framework checks verify that the evaluator implements the declared distinctions consistently. Under the declared model, the controlled communication study exposes a finalizer-dependent asymmetry: sender-finalized Feedback uses evidence transfer to expand evidence reachability throughout the feasible plotted region, whereas receiver-finalized Feedback uses coordination to suppress redundant payload until loss, latency, freshness, and deadline costs shift selection to One-way. Finally, an episode-level reporting schema defines common denominators for future measured Physical-AI studies.",
  "published": "2026-08-20",
  "updated": "2026-08-20",
  "year": "2026",
  "authors": [
   "Shunsuke Saruwatari"
  ],
  "author_count": 1,
  "categories": [
   "cs.RO",
   "cs.IT",
   "cs.NI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A Post-Semantic Communication Framework for the systems interface between proposal formation and physical execution is introduced and evidence transfer is distinguished from evidence coordination, which can suppress transmission around records already held at the finalization endpoint.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "S. Saruwatari",
    "id": "1788325",
    "h_index": 14,
    "papers": 159
   }
  ],
  "comment": "9 pages, 3 figures, 3 tables",
  "topics": [
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.19593v1",
  "pdf_url": "https://arxiv.org/pdf/2608.19593v1",
  "html_url": "https://arxiv.org/html/2608.19593v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.19589",
  "slug": "orthoskillvla-continual-skill-learning-via-gradient-informed-skill-sub",
  "title": "OrthoSkillVLA: Continual Skill Learning via Gradient-Informed Skill Subspace Adaptation",
  "abstract": "Pretrained Vision-Language-Action models provide a strong foundation for robot learning, but sequentially adapting them to diverse skills can perturb the representations and velocity mappings used by previous skills, leading to catastrophic forgetting. Architecture-based approaches improve retention by isolating skills but lead to increased inference footprint. Recent subspace-constrained methods restrict parameter updates in an orthogonal subspace to minimize interference but impose a unified constraint on the entire model. We analyze the distinct roles of internal VLA components and identify two VLA-specific challenges. First, the VLM maintains broad semantic representations, making it vulnerable to capacity exhaustion, whereas the ActionHead refines semantics into localized velocity patterns that are highly sensitive to perturbations. Second, the final velocity decoder serves as a readout layer. Freezing it forms an output-stage expressivity bottleneck, while updating it risks overwriting previous velocity mappings. To this end, we propose OrthoSkillVLA, a parameter-efficient framework for continual skill learning in pretrained VLA models without demonstration replay. Given the representation heterogeneity, we impose separate subspace constraints on the VLM and ActionHead, preserving reusable semantic capacity while protecting localized velocity patterns. For the output layer, we introduce a lightweight feature-aware MoE decoder, where each skill is allocated a compact expert and a training-free router selects the expert according to feature-space affinity. Extensive simulated and real-world evaluations, together with ablations, demonstrate that OrthoSkillVLA better preserves prior skills while acquiring new ones.",
  "published": "2026-08-20",
  "updated": "2026-08-20",
  "year": "2026",
  "authors": [
   "Jiaqi Wang",
   "Zhou Fang",
   "Qiongfeng Shi",
   "Yi Zhou"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes OrthoSkillVLA, a parameter-efficient framework for continual skill learning in pretrained VLA models without demonstration replay, and introduces a lightweight feature-aware MoE decoder that better preserves prior skills while acquiring new ones.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jiaqi Wang",
    "id": "2329131600",
    "h_index": 4,
    "papers": 20
   },
   {
    "name": "Zhou Fang",
    "id": "2299924394",
    "h_index": 2,
    "papers": 12
   },
   {
    "name": "Q. Shi",
    "id": "2424619362",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Yi Zhou",
    "id": "2326251703",
    "h_index": 3,
    "papers": 7
   }
  ],
  "comment": "Accepted by PRCV 2026",
  "topics": [
   "vla",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.19589v1",
  "pdf_url": "https://arxiv.org/pdf/2608.19589v1",
  "html_url": "https://arxiv.org/html/2608.19589v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.19574",
  "slug": "hitac-wam-a-hierarchical-tactile-world-action-model-for-contact-rich-r",
  "title": "HiTac-WAM: A Hierarchical Tactile World Action Model for Contact-Rich Robot Manipulation",
  "abstract": "World action models jointly predict future visual observations and actions, whereas existing tactile-aware variants typically represent future touch as an image or latent stream without modeling the physical dependencies that organize tactile states hierarchically. We present HiTac-WAM, a hierarchical tactile world action model that forecasts a sequence of future tactile states for each candidate action chunk before execution. The forecast factorizes into contact state, a 3D deformation field, and slip risk, organized as a directed hierarchy in which each downstream stage is conditioned on stop-gradient signals from preceding stages. A directed attention mask allows tactile queries to attend to the video-action context of each candidate while preventing video and action queries from attending to tactile tokens. For planning, HiTac-WAM ranks candidate action chunks using tactile forecasts and task-progress estimates. For execution, the selected tactile forecast is retained as a reference; persistent discrepancies between predicted and observed tactile states trigger corrective replanning. HiTac-WAM achieves a mean contact F1 of 0.921; under matched training budgets, the directed hierarchy reduces 3D displacement L2 error by 17.6% relative to the deformation-only predictor and improves slip AUPRC by 60.4% relative to the slip-only predictor. Across chip grasping, blackboard erasing, and USB insertion, selection guided by the hierarchical forecasts increases the average real-robot success rate from 31.1% to 61.1%, while the full system attains 72.2%.",
  "published": "2026-08-20",
  "updated": "2026-08-20",
  "year": "2026",
  "authors": [
   "Chao Xue",
   "Chaofan Zhang",
   "Wenxuan Ma",
   "Guocai Yao",
   "Shaowei Cui",
   "Shuo Wang"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "HiTac-WAM is presented, a hierarchical tactile world action model that forecasts a sequence of future tactile states for each candidate action chunk before execution, organized as a directed hierarchy in which each downstream stage is conditioned on stop-gradient signals from preceding stages.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Chao Xue",
    "id": "2454126195",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Chaofan Zhang",
    "id": "2256775583",
    "h_index": 7,
    "papers": 19
   },
   {
    "name": "Wenxuan Ma",
    "id": "2276183477",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Guocai Yao",
    "id": "2376597393",
    "h_index": 6,
    "papers": 23
   },
   {
    "name": "Shaowei Cui",
    "id": "1853836031",
    "h_index": 17,
    "papers": 57
   },
   {
    "name": "Shuo Wang",
    "id": "2360886366",
    "h_index": 3,
    "papers": 13
   }
  ],
  "comment": "8 pages, 7 figures, and 3 tables",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.19574v1",
  "pdf_url": "https://arxiv.org/pdf/2608.19574v1",
  "html_url": "https://arxiv.org/html/2608.19574v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.19537",
  "slug": "multimodal-trajectory-planning-for-surface-vehicles-using-turning-circ",
  "title": "Multimodal Trajectory Planning for Surface Vehicles using Turning Circle-based Control Barrier Functions",
  "abstract": "This paper presents a guide path-free multimodal trajectory planning framework for autonomous surface vehicles operating in dynamic environments. The proposed method integrates model predictive control (MPC) with a turning circle-based control barrier function (TC-CBF). Unlike conventional Euclidean distance-based CBFs (ED-CBFs), which evaluate safety solely based on proximity, the TC-CBF accounts for the nonholonomic motion and finite turning capability of a surface vehicle. Its geometric formulation identifies feasible avoidance regions according to the vehicle's turning circles and generates distinct left- and right-turning avoidance modes. These modes allow the optimization solver to explore and select topologically different trajectories without relying on globally planned guide paths, as required by many conventional multimodal planning approaches. By embedding the avoidance direction directly into the safety constraint, the proposed framework alleviates the local-minimum and deadlock problems of single-mode MPC while maintaining computational efficiency. Extensive simulations involving multiple moving vessels demonstrate that the proposed method achieves higher success rates, fewer safety violations, and smaller residual violations than single-mode baselines across all tested traffic densities.",
  "published": "2026-08-20",
  "updated": "2026-08-20",
  "year": "2026",
  "authors": [
   "Changyu Lee"
  ],
  "author_count": 1,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper presents a guide path-free multimodal trajectory planning framework for autonomous surface vehicles operating in dynamic environments that alleviates the local-minimum and deadlock problems of single-mode MPC while maintaining computational efficiency.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Changyu Lee",
    "id": "49010303",
    "h_index": 7,
    "papers": 21
   }
  ],
  "comment": "This work has been submitted to an Elsevier journal for possible publication",
  "topics": [
   "rl-control",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.19537v1",
  "pdf_url": "https://arxiv.org/pdf/2608.19537v1",
  "html_url": "https://arxiv.org/html/2608.19537v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.19536",
  "slug": "cvsd-reg-cross-modal-visual-semantic-prior-distillation-for-robust-lid",
  "title": "CVSD-Reg: Cross-Modal Visual Semantic Prior Distillation for Robust LiDAR Registration",
  "abstract": "Learning-based global point cloud registration has achieved remarkable progress, yet its reliance on geometric representations makes existing methods sensitive to variations in point density, scan pattern, viewpoint, and sensor characteristics. We propose CVSD-Reg, a robust global LiDAR registration framework that distills visual semantic priors from a vision foundation model into LiDAR representations. In Stage 1, a Point Transformer V3 student learns from a frozen DINOv2 teacher through contrastive distillation and spherical-manifold alignment, which preserves the hyperspherical geometry of the teacher embedding space. Self-supervised InfoNCE consistency and soft $\\mathrm{SE}(3)$ invariance further encourage viewpoint-robust descriptors. In Stage 2, the distilled representation is adapted to registration through correspondence learning, density-aware point-dropout augmentation, and end-to-end pose optimization. With a single checkpoint, CVSD-Reg generalizes to both single-sensor and zero-shot cross-sensor scenarios without sensor-specific adaptation and remains entirely camera-free at inference. On KITTI, nuScenes, and HeLiPR, CVSD-Reg achieves strict success rate (SR@0.5\\,m/$1^\\circ$) of 97.7$\\%$, 99.0$\\%$, and 99.3$\\%$, respectively, including 97.3$\\%$ on sparse 16-beam Velodyne scans. It outperforms state-of-the-art geometric registration methods by up to 44.0 percentage points without requiring camera inputs or post-hoc ICP refinement.",
  "published": "2026-08-20",
  "updated": "2026-08-20",
  "year": "2026",
  "authors": [
   "Eunsoo Im",
   "Junghun Suh",
   "Gyeonggwan Lee",
   "Seunghwan Hong"
  ],
  "author_count": 4,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "CVSD-Reg is proposed, a robust global LiDAR registration framework that distills visual semantic priors from a vision foundation model into LiDAR representations and generalizes to both single-sensor and zero-shot cross-sensor scenarios without sensor-specific adaptation and remains entirely camera-free at inference.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Eunsoo Im",
    "id": "2315933569",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Junghun Suh",
    "id": "121602022",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Gyeonggwan Lee",
    "id": "2374988535",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Seunghwan Hong",
    "id": "2137273",
    "h_index": 4,
    "papers": 23
   }
  ],
  "comment": "",
  "topics": [
   "spatial-3d",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.19536v1",
  "pdf_url": "https://arxiv.org/pdf/2608.19536v1",
  "html_url": "https://arxiv.org/html/2608.19536v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.19522",
  "slug": "lf-gicp-parameter-free-degeneracy-aware-lidar-odometry-via-a-voxel-nor",
  "title": "LF-GICP: Parameter-Free Degeneracy-Aware LiDAR Odometry via a Voxel-Normal Localizability Field",
  "abstract": "Scan-to-map LiDAR odometry drifts unboundedly along the unobservable axes of geometrically degenerate environments like tunnels and corridors, and existing degeneracy handling requires environment-specific parameter tuning. This paper presents a parameter-free approach. We show that in voxelized GICP the Gauss--Newton (GN) Hessian masks translational degeneracy, because covariance regularization keeps the translation block artificially well-conditioned. We bypass this with a regularization-free voxel-normal localizability field and two of its statistics: a normalized fraction $f_0$ detecting directional anisotropy, and an absolute per-voxel mass $\u03bb_0$ distinguishing information absence (tunnels) from dilution (dense open scenes). A temporal-median gate combines both to trigger Fisher-information correspondence weighting. Calibrated once by fixed rules on two short sequences and then frozen, LF-GICP achieves the lowest KITTI relative translation error ($0.865\\%$) under an identical evaluation protocol against re-run baselines, outperforms them on GEODE tunnels and MulRan, leads the HeLiPR mean, and generalizes across four sensor types without re-tuning. We further demonstrate empirically that straight, uniform tunnels remain unobservable along their axis for LiDAR-only registration.",
  "published": "2026-08-20",
  "updated": "2026-08-20",
  "year": "2026",
  "authors": [
   "Eunsoo Im"
  ],
  "author_count": 1,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "LF-GICP achieves the lowest KITTI relative translation error under an identical evaluation protocol against re-run baselines, outperforms them on GEODE tunnels and MulRan, leads the HeLiPR mean, and generalizes across four sensor types without re-tuning.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Eunsoo Im",
    "id": "2315933569",
    "h_index": 2,
    "papers": 3
   }
  ],
  "comment": "",
  "topics": [
   "sim2real"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.19522v1",
  "pdf_url": "https://arxiv.org/pdf/2608.19522v1",
  "html_url": "https://arxiv.org/html/2608.19522v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.21444",
  "slug": "agentic-ai-for-safety-critical-multi-drone-systems-challenges-and-oppo",
  "title": "Agentic AI for Safety-critical Multi-drone Systems: Challenges and Opportunities",
  "abstract": "Multi-drone systems are increasingly positioned for safety-critical missions such as search and rescue (SAR) and critical infrastructure monitoring. Yet, real-world adoption remains constrained not only by autonomy performance, but by the difficulty of integrating agentic behavior into professional work: operators must understand, trust, and govern automation under uncertainty, time pressure, and accountability. This position paper synthesizes the ambitions and lessons from two ongoing efforts: NAMUR, which explores LLM-supported robot control in SAR and firefighting contexts, and PERSIST, which explores persistent drone operations for monitoring and security at critical infrastructure sites. We argue that agentic AI should be approached as a socio-technical design problem, where interfaces, oversight mechanisms, and evaluation practices are as critical as algorithms. We outline a human-centered, participatory, and iterative research approach aimed at uncovering stakeholder needs, shaping agent capabilities through successive prototypes, and producing transferable proof-of-concept systems and evaluation strategies for other safety-critical contexts.",
  "published": "2026-08-19",
  "updated": "2026-08-19",
  "year": "2026",
  "authors": [
   "Timothy Merritt",
   "Alejandro Jarabo-Pe\u00f1as",
   "Juan Bravo-Arrabal",
   "Maria-Theresa Bahodi",
   "Anders Lyhne Christensen"
  ],
  "author_count": 5,
  "categories": [
   "cs.AI",
   "cs.ET",
   "cs.HC",
   "cs.RO"
  ],
  "primary_category": "cs.AI",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "It is argued that agentic AI should be approached as a socio-technical design problem, where interfaces, oversight mechanisms, and evaluation practices are as critical as algorithms.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "T. Merritt",
    "id": "2309012325",
    "h_index": 3,
    "papers": 16
   },
   {
    "name": "Alejandro Jarabo-Pe\u00f1as",
    "id": "2114849337",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Juan Bravo-Arrabal",
    "id": "2047110218",
    "h_index": 6,
    "papers": 16
   },
   {
    "name": "Maria-Theresa Bahodi",
    "id": "2321477929",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Anders Lyhne Christensen",
    "id": "2404738595",
    "h_index": 2,
    "papers": 9
   }
  ],
  "comment": "9 pages, 4 figures, presented at the AgentCraft Workshop at IUI2026 Conference, https://agentcraft-iui.github.io/2026/ - CEUR-WS.org proceedings forthcoming",
  "topics": [
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.21444v1",
  "pdf_url": "https://arxiv.org/pdf/2608.21444v1",
  "html_url": "https://arxiv.org/html/2608.21444v1",
  "code_url": "https://agentcraft-iui.github.io/2026/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2608.19492",
  "slug": "beyond-multimodal-alignment-certifying-physical-language-through-respo",
  "title": "Beyond Multimodal Alignment: Certifying Physical Language through Response Substitution and Ordered Execution",
  "abstract": "World models increasingly treat compact multimodal representations as interfaces between perception and physical interaction, yet existing probes do not establish whether different sensors carry the same executable meaning or whether that meaning survives a new action composition. We introduce an operational capability hierarchy and the Disjoint-Bridge Operator-Substitution Certificate (DBOSC), which asks whether independently trained modality compilers enter a frozen response chart interchangeably on evidence outside their training panels. On Cluster Haptic, audio and acceleration representations of the same unseen surface are 4.5x closer in response space than wrong-surface pairings, with the gap holding for all 19 held-out surfaces; unsealing withheld responses confirms that every branch predicts the physics better than the population chart. We then test ordered execution in a controlled elastoplastic system with complementary modality blind spots. At the pre-registered budget, the prerequisite refuses the stack because the frozen executor cannot advance even an exact chart coordinate through a held-out program. At a converged budget, the same rank-three chart executes those programs (oracle NMSE 0.18), fusion improves on both modalities, and 14 of 16 registered checks pass; the two failures arise because a diagonal restriction of the fused information matrix performs as well as the full one. Clearing the gate is a property of the executor, not the chart: an executor emitting whole programs instead of shared per-step dynamics is 38x worse than an entity-blind predictor on the same chart. A matching non-identifiability result explains why compression and fusion alone cannot determine an unseen composition law. These results separate attribute access, response substitution, fusion closure, and ordered execution into distinct, separately testable achievements.",
  "published": "2026-08-19",
  "updated": "2026-08-19",
  "year": "2026",
  "authors": [
   "Kaizhen Tan",
   "Xin Xu",
   "Siru Tao",
   "Yixiao Li",
   "Hanzhe Hong",
   "Yang Feng",
   "Heqing Du"
  ],
  "author_count": 7,
  "categories": [
   "cs.LG",
   "cs.RO"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "An operational capability hierarchy and the Disjoint-Bridge Operator-Substitution Certificate are introduced and a matching non-identifiability result explains why compression and fusion alone cannot determine an unseen composition law are explained.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Kaizhen Tan",
    "id": "2380624774",
    "h_index": 1,
    "papers": 14
   },
   {
    "name": "Xin Xu",
    "id": "2454075825",
    "h_index": 0,
    "papers": 5
   },
   {
    "name": "Siru Tao",
    "id": "2362506658",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Yixiao Li",
    "id": "2455182273",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Han Hong",
    "id": "2093217886",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Yang Feng",
    "id": "2261199971",
    "h_index": 14,
    "papers": 37
   },
   {
    "name": "Heqing Du",
    "id": "2430741201",
    "h_index": 0,
    "papers": 4
   }
  ],
  "comment": "",
  "topics": [
   "world-models"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.19492v1",
  "pdf_url": "https://arxiv.org/pdf/2608.19492v1",
  "html_url": "https://arxiv.org/html/2608.19492v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.19490",
  "slug": "fine-tuning-vlas-with-self-demonstrated-generative-control-for-multi-t",
  "title": "Fine-Tuning VLAs with Self-Demonstrated Generative Control for Multi-Task Manipulation",
  "abstract": "State-of-the-art vision-language-action (VLA) models such as $\u03c0_{0.5}$ exhibit strong semantic understanding, instruction following and task behavior. However, when deployed on new robots, even minor mismatches in hardware configuration relative to pretraining can cause severe performance drops. Finetuning the VLA on in-domain expert data from the new embodiment improves performance on the expert task but leads to a loss in its original instruction following and behavioral priors. In this paper, we propose a self-supervised method that generates online interaction rollouts from the zero-shot VLA as additional training data for finetuning. Our experiments show this finetuning scheme yields strong multi-task policies that, on the target robot, (1) inherit prior tasks distilled from the zero-shot model, (2) enable generalist instruction following, while (3) learning new skills from expert data with improved sample efficiency. We demonstrate the success of our approach across test sets probing generalization on a real ALOHA robot and a new simulation benchmark in RoboTwin. Video results are available at https://self-supervised-control.pages.dev/",
  "published": "2026-08-19",
  "updated": "2026-08-19",
  "year": "2026",
  "authors": [
   "Prachi Garg",
   "Steve Xing",
   "Prahit Yaugand",
   "Saurabh Gupta",
   "Derek Hoiem"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper proposes a self-supervised method that generates online interaction rollouts from the zero-shot VLA as additional training data for finetuning and demonstrates the success of this approach across test sets probing generalization on a real ALOHA robot and a new simulation benchmark in RoboTwin.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Prachi Garg",
    "id": "2135373709",
    "h_index": 5,
    "papers": 26
   },
   {
    "name": "Steve Xing",
    "id": "2458571060",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Prahit Yaugand",
    "id": "2458571476",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Saurabh Gupta",
    "id": "47924870",
    "h_index": 19,
    "papers": 38
   },
   {
    "name": "Derek Hoiem",
    "id": "2261388484",
    "h_index": 5,
    "papers": 19
   }
  ],
  "comment": "Project Page: https://self-supervised-control.pages.dev/",
  "topics": [
   "vla",
   "sim2real",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.19490v1",
  "pdf_url": "https://arxiv.org/pdf/2608.19490v1",
  "html_url": "https://arxiv.org/html/2608.19490v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.19453",
  "slug": "when-automata-meet-streams-temporal-logic-compilation-for-stream-based",
  "title": "When Automata Meet Streams: Temporal Logic Compilation for Stream-Based Robotics Task and Motion Planning",
  "abstract": "Stream-based robotics Task and Motion Planning (TAMP) integrates discrete symbolic planning with dynamically generated continuous geometric parameters, such as poses, grasps, and trajectories. However, stream-based planners typically reason only about goal reachability, whereas long-horizon tasks also demand adherence to temporal specifications, such as safety-critical ordering, invariance, and liveness constraints. No methods currently exist to enforce such temporal constraints for stream-based solvers because streams generate an expanding geometric object set via iterative stream refinement loops during planning, rendering existing temporal-logic compilation techniques incompatible. We therefore present Synchronous Action Monitoring with Token Destruction (SAM-TD), a compilation method that enforces arbitrary Linear Temporal Logic over finite traces ($\\textrm{LTL}_f$) specifications in stream-based TAMP. SAM-TD translates arbitrary $\\textrm{LTL}_f$ constraints into automata and embeds regressed automaton guards into action schemas, which are pre-specified before planning begins. By doing so, SAM-TD can handle objects generated by streams during planning, thus circumventing the need to enumerate a fixed object set or modify the underlying planner. During search, SAM-TD synchronously updates automaton states and uses a validity token shared across all automata to prune constraint-violating branches. We show that SAM-TD supports dynamically generated stream objects from iterative stream refinements during plan search. Experimental results provide the first ever demonstration of stream-based TAMP under $\\textrm{LTL}_f$ constraints in three robotics PDDLStream environments. Furthermore, on standard discrete PDDL benchmarks, SAM-TD is competitive with state-of-the-art temporal-constraint compilation methods.",
  "published": "2026-08-19",
  "updated": "2026-08-19",
  "year": "2026",
  "authors": [
   "Sayem Nazmuz Zaman",
   "Cyrus Neary"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO",
   "cs.FL"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Experimental results provide the first ever demonstration of stream-based TAMP under $\\textrm{LTL}_f$ constraints in three robotics PDDLStream environments and show that SAM-TD is competitive with state-of-the-art temporal-constraint compilation methods.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Sayem Nazmuz Zaman",
    "id": "2238209284",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Cyrus Neary",
    "id": "2370936413",
    "h_index": 3,
    "papers": 6
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.19453v1",
  "pdf_url": "https://arxiv.org/pdf/2608.19453v1",
  "html_url": "https://arxiv.org/html/2608.19453v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.19443",
  "slug": "hybrid-feedback-sampling-for-sample-efficient-model-predictive-control",
  "title": "Hybrid Feedback Sampling for Sample-Efficient Model Predictive Control",
  "abstract": "Thanks to its parallelizability and flexibility, sampling-based Model Predictive Control (MPC) has become widely popular for controlling real-world robotic systems. However, for high-dimensional and open-loop unstable dynamical systems, the required number of samples to improve the control sequence will grow exponentially with the horizon, leading to poor sample efficiency and numerical instability. This paper investigates the instability of shooting methods in sampling-based MPC and shows that the optimal sampling proposal distribution can be realized by sampling with an optimized feedback policy. We refer to this algorithm as Feedback Sampling MPC (FS-MPC). FS-MPC involves a hybrid sampling design which balances local and global search based on the system stability and the available computation budget. Our theoretical analysis shows that our hybrid sampling approach achieves faster convergence than standard MPPI and better optimality than standard feedback sampling. Empirically, in diverse contact-rich control tasks like humanoid loco-manipulation and dexterous manipulation, we show that FS-MPC successfully tackles dynamically unstable tasks where standard sample-based approaches struggle, and strictly outperforms feedback policies alone. Finally, we validate our method on humanoid robot locomotion and manipulation tasks in the real world.",
  "published": "2026-08-19",
  "updated": "2026-08-19",
  "year": "2026",
  "authors": [
   "Chaoyi Pan",
   "Zeji Yi",
   "John Zhang",
   "Zachary Manchester",
   "Guannan Qu",
   "Guanya Shi"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper investigates the instability of shooting methods in sampling-based MPC and shows that the optimal sampling proposal distribution can be realized by sampling with an optimized feedback policy, and develops a hybrid sampling design which balances local and global search based on the system stability and the available computation budget.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Chaoyi Pan",
    "id": "2279668676",
    "h_index": 9,
    "papers": 14
   },
   {
    "name": "Zeji Yi",
    "id": "2279549366",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "John Z. Zhang",
    "id": "2268802798",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Zachary Manchester",
    "id": "7987394",
    "h_index": 23,
    "papers": 69
   },
   {
    "name": "Guannan Qu",
    "id": "2279549911",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Guanya Shi",
    "id": "2396503175",
    "h_index": 3,
    "papers": 6
   }
  ],
  "comment": "15 pages, 6 figures",
  "topics": [
   "dexterous-manipulation",
   "humanoids",
   "tactile",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.19443v1",
  "pdf_url": "https://arxiv.org/pdf/2608.19443v1",
  "html_url": "https://arxiv.org/html/2608.19443v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.19425",
  "slug": "scape-scenario-conditioned-simulation-augmented-policy-evaluation",
  "title": "SCAPE: Scenario-Conditioned Simulation-Augmented Policy Evaluation",
  "abstract": "Reliable performance evaluation is a central bottleneck for deploying robot-learning policies in real-world conditions. Real-world testing is faithful but costly and difficult to scale, whereas simulation-based testing scales easily but is inevitably biased by the sim-to-real gap. Existing simulation-augmented methods combine limited real-world rollouts with abundant simulation proxies, but focus on performance averaged over initial conditions and deployment settings. Such population-level averages obscure scenario-specific variation and provide limited guidance about when and where a policy can be safely deployed. We propose SCAPE, a scenario-conditioned simulation-augmented policy evaluation framework that predicts scenario-conditioned real-world policy performance using limited paired sim-and-real samples and large-scale simulation rollouts. SCAPE corrects sim-to-real bias in simulation labels before training the prediction model and calibrates prediction uncertainty through conformal prediction. We validate SCAPE on autonomous driving and quadruped velocity tracking. In sim-to-sim studies, SCAPE reduces scenario-level prediction error by 4.9%/34.7% (driving) and 14.5%/27.7% (quadruped) relative to scene-conditioned neural and aggregate statistical baselines on average. We further evaluate a velocity-tracking policy deployed on a physical Unitree Go2. SCAPE also improves testing sample efficiency, produces narrower calibrated prediction intervals, generalizes better to out-of-distribution scenarios, and enables fine-grained deployment strategies.",
  "published": "2026-08-19",
  "updated": "2026-08-19",
  "year": "2026",
  "authors": [
   "Dijie Zhu",
   "Seunghun Oh",
   "Ruopeng Huang",
   "Zhiyu Huang",
   "Jiaqi Ma",
   "Chen Tang"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "SCAPE is proposed, a scenario-conditioned simulation-augmented policy evaluation framework that predicts scenario-conditioned real-world policy performance using limited paired sim-and-real samples and large-scale simulation rollouts and improves testing sample efficiency.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Dijie Zhu",
    "id": "2458621212",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Seunghun Oh",
    "id": "2316526529",
    "h_index": 1,
    "papers": 9
   },
   {
    "name": "Ruopeng Huang",
    "id": "2140387138",
    "h_index": 13,
    "papers": 33
   },
   {
    "name": "Zhiyu Huang",
    "id": "2333367109",
    "h_index": 6,
    "papers": 16
   },
   {
    "name": "Jiaqi Ma",
    "id": "2367691803",
    "h_index": 6,
    "papers": 23
   },
   {
    "name": "Chen Tang",
    "id": "1491105028",
    "h_index": 16,
    "papers": 38
   }
  ],
  "comment": "22 pages",
  "topics": [
   "humanoids",
   "sim2real",
   "navigation",
   "safety-eval"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2608.19425v1",
  "pdf_url": "https://arxiv.org/pdf/2608.19425v1",
  "html_url": "https://arxiv.org/html/2608.19425v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2608.19375",
  "slug": "learning-the-right-abstraction-neural-reduced-dynamics-for-complex-rob",
  "title": "Learning the Right Abstraction: Neural Reduced Dynamics for Complex Robot Control",
  "abstract": "High-fidelity embodied AI simulators provide realistic evaluation of complex robotic systems, but their computational cost limits their direct use for large-scale reinforcement learning campaigns. We advocate the use of less accurate but more expeditious simulations, which might draw on data-driven, e.g., neural dynamics, models. This contribution argues that the practical value of a neural dynamics model for complex robot control lies in learning the \\emph{right abstraction}: a reduced state that preserves the control-relevant physics of the high-fidelity system while enabling high-throughput policy learning. We develop a neural reduced dynamics (NRD) framework that separates the state the model propagates from what can be supplied as an input or recovered analytically, trains policies entirely inside the frozen learned model, and validates them back in the high-fidelity simulator. Two case studies instantiate it across three control tasks: terrain-aware HMMWV trajectory tracking on rigid, bumpy and deformable Continuum Representation Model (CRM) terrain; and goal reaching for a stock tracked vehicle and its front-mounted articulated arm. Every policy transfers back to the high-fidelity simulator. A single policy trained inside the terrain-conditioned dynamics model, and given no terrain input of its own, attains lower median and mean tracking error than both single-terrain specialists on all three terrains, including zero-shot bumpy terrain. Quantitatively, the tracked vehicle reaches 100 of 100 goals and the arm 97 of 100, with zero contacts or joint-limit violations. The NRD models advance roughly four orders of magnitude faster in simulated time than the high-fidelity simulator scenes they replace, making iterative on-policy learning practical and supporting neural reduced dynamics as a bridge between accurate but expensive physics simulation and scalable robot learning.",
  "published": "2026-08-19",
  "updated": "2026-08-19",
  "year": "2026",
  "authors": [
   "Harry Zhang",
   "Dan Negrut"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A neural reduced dynamics framework is developed that separates the state the model propagates from what can be supplied as an input or recovered analytically, trains policies entirely inside the frozen learned model, and validates them back in the high-fidelity simulator.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Harry Zhang",
    "id": "2143441449",
    "h_index": 4,
    "papers": 24
   },
   {
    "name": "Dan Negrut",
    "id": "2297681684",
    "h_index": 3,
    "papers": 9
   }
  ],
  "comment": "",
  "topics": [
   "sim2real",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.19375v1",
  "pdf_url": "https://arxiv.org/pdf/2608.19375v1",
  "html_url": "https://arxiv.org/html/2608.19375v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.19372",
  "slug": "the-missing-touch-spatially-distributed-tactile-feedback-brings-teleop",
  "title": "The Missing Touch: Spatially Distributed Tactile Feedback Brings Teleoperation Closer to Human Dexterity",
  "abstract": "A fundamental challenge in robotic teleoperation is enabling an operator to control a remote robot as effortlessly and intuitively as their own hands. Despite the growing use of teleoperation to collect demonstration data for training autonomous robot policies, teleoperated robot performance still falls significantly short of human dexterity, even for basic tasks. Here, we present evidence that a key factor contributing to this performance gap is the absence of spatially distributed tactile feedback. Using a two-degree-of-freedom (DoF) bilateral force-feedback telemanipulator paired with a 32-DoF tactile fingertip display, we show that operator performance improves significantly when localized deformations on the remote manipulator are faithfully reproduced on the operator's fingertip. In a series of teleoperation tasks, reproducing distributed contact information not only accelerated task performance but also brought teleoperated movements closer to natural human behavior by minimizing corrective actions and task completion steps, thereby reducing the deviation between teleoperated and natural trajectories by 29$\\unicode{x2013}$79%. Furthermore, we found that increasing the resolution of the tactile feedback$\\unicode{x2014}$by refining how finely the measured displacements were quantized for reproduction$\\unicode{x2014}$compressed the state-space distribution of teleoperated motions, which has been associated with improved training outcomes for autonomous robot policies. Together, these results suggest that spatially distributed tactile feedback is essential for closing the gap between human and teleoperated dexterity and training the next generation of autonomous robots.",
  "published": "2026-08-19",
  "updated": "2026-08-19",
  "year": "2026",
  "authors": [
   "Rohan Kota",
   "Gregory Reardon",
   "J. Edward Colgate"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "It is suggested that spatially distributed tactile feedback is essential for closing the gap between human and teleoperated dexterity and training the next generation of autonomous robots.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Rohan Kota",
    "id": "2390607152",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Gregory Reardon",
    "id": "2376448245",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "J. Colgate",
    "id": "153362742",
    "h_index": 9,
    "papers": 16
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.19372v1",
  "pdf_url": "https://arxiv.org/pdf/2608.19372v1",
  "html_url": "https://arxiv.org/html/2608.19372v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.19281",
  "slug": "approve-visual-end-user-in-the-loop-robot-programming-with-llms",
  "title": "APPROVE: Visual End-User-in-the-Loop Robot Programming with LLMs",
  "abstract": "Programming robots remains challenging for non-experts, as traditional methods require expert knowledge and even block-based interfaces often lack flexibility. Recent work has explored Large Language Models (LLMs) to automatically generate robot programs from natural language, but these systems remain limited by a lack of transparency, missing mechanisms to ensure alignment with user intent, and little support for reuse. We present APPROVE (AI-Powered Programming for Robots with Visual End-User Feedback), an LLM-based multi-modal end-user programming framework that integrates natural language input with a block-based interface and an explicit user confirmation step. Generated programs are visualized using a block-based interface in Blockly, allowing users to confirm, modify, or reject them before execution. Confirmed functions are stored in a library for reuse, gradually building a set of reliable program components. Our approach contributes a human-centered design for LLM-based robot programming that emphasizes user trust, intent alignment, and reusability.",
  "published": "2026-08-19",
  "updated": "2026-08-19",
  "year": "2026",
  "authors": [
   "Bijan Kavousian",
   "Miray \u00d6zakkas",
   "Josefine Monnet",
   "Oliver Petrovic",
   "Christian Brecher"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.HC"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "APPROVE (AI-Powered Programming for Robots with Visual End-User Feedback), an LLM-based multi-modal end-user programming framework that integrates natural language input with a block-based interface and an explicit user confirmation step that contributes a human-centered design for LLM-based robot programming.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Bijan Kavousian",
    "id": "2358247970",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Miray \u00d6zakkas",
    "id": "2353192686",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Josefine Monnet",
    "id": "2332847660",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Oliver Petrovic",
    "id": "2266765531",
    "h_index": 3,
    "papers": 15
   },
   {
    "name": "Christian Brecher",
    "id": "2287571543",
    "h_index": 3,
    "papers": 20
   }
  ],
  "comment": "Accepted for publication in Procedia CIRP, Proceedings of the 20th CIRP Conference on Intelligent Computation in Manufacturing Engineering (ICME 2026)",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.19281v1",
  "pdf_url": "https://arxiv.org/pdf/2608.19281v1",
  "html_url": "https://arxiv.org/html/2608.19281v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.19280",
  "slug": "multi-tool-robotics-enables-in-situ-sample-manipulation-for-time-resol",
  "title": "Multi-Tool Robotics Enables In-Situ Sample Manipulation for Time-Resolved Synchrotron Measurements",
  "abstract": "The high photon flux at synchrotron beamlines allows for the measurement of fast dynamical processes. However, beamline radiation-safety protocols prohibit human intervention during X-ray experiments, limiting the ability to perform versatile real-time sample manipulations during continuous data acquisition. Here we present a robotic platform at an X-ray scattering beamline to enable real-time sample handling and processing in the experimental hutch, revealing previously inaccessible transient in-situ dynamics in perovskite thin films. This modular multi-tool robotic architecture enables in-hutch sample manipulation beyond human-access constraints, establishing a foundation for automated and autonomous synchrotron experimentation.",
  "published": "2026-08-19",
  "updated": "2026-08-19",
  "year": "2026",
  "authors": [
   "Aditya Bondada",
   "Elizabeth M. Wall",
   "Eric Yuan Xiao",
   "Quinn C. Burlingame",
   "Yueh-Lin Loo",
   "Esther H. R. Tsai",
   "Ruipeng Li"
  ],
  "author_count": 7,
  "categories": [
   "cond-mat.mtrl-sci",
   "cs.RO"
  ],
  "primary_category": "cond-mat.mtrl-sci",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Aditya Bondada",
    "id": "2307917944",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Elizabeth M. Wall",
    "id": "2380029679",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Eric Yuan Xiao",
    "id": "2458572243",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Q. Burlingame",
    "id": "14447199",
    "h_index": 18,
    "papers": 37
   },
   {
    "name": "Yueh\u2010Lin Loo",
    "id": "2229696074",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Esther H. R. Tsai",
    "id": "3653945",
    "h_index": 19,
    "papers": 48
   },
   {
    "name": "Ruipeng Li",
    "id": "35241937",
    "h_index": 61,
    "papers": 350
   }
  ],
  "comment": "",
  "topics": [
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.19280v1",
  "pdf_url": "https://arxiv.org/pdf/2608.19280v1",
  "html_url": "https://arxiv.org/html/2608.19280v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.19188",
  "slug": "partialbigrasp-inferring-hidden-local-geometry-for-bimanual-grasping-f",
  "title": "PartialBiGrasp: Inferring Hidden Local Geometry for Bimanual Grasping from Partial Views",
  "abstract": "Dual-arm robotic grasping is essential for manipulating large, heavy, and geometrically complex objects that cannot be reliably handled using a single manipulator. These large objects often contain only sparse graspable regions determined by local geometric properties such as thickness, edge structure, and gripper clearance. Prior bimanual grasping methods assume access to a full point cloud of the object which inherently contains this geometric information, but may not be accessible in real scenarios. This work proposes PartialBiGrasp, a dual-arm grasp generation framework that operates directly on partial point cloud observations. Our model learns geometric features implicitly through convolutional occupancy networks, enabling local reasoning about graspability, collision-free contact regions, and object thickness. We leverage this understanding to generate force-closure compliant grasp pairs, which are further refined using a sampling-based optimization to correct for ambiguity caused by incomplete geometry. We evaluate our approach using analytical force-closure metrics, large-scale simulation experiments, and real-world robot evaluations on noisy partial point clouds of novel objects, demonstrating robust and physically stable dual-arm grasp generation.",
  "published": "2026-08-19",
  "updated": "2026-08-19",
  "year": "2026",
  "authors": [
   "Ayush Kaura",
   "Vignesh Vembar",
   "Md Faizal Karim",
   "Keshab Patra",
   "K Madhava Krishna"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes PartialBiGrasp, a dual-arm grasp generation framework that operates directly on partial point cloud observations that learns geometric features implicitly through convolutional occupancy networks, enabling local reasoning about graspability, collision-free contact regions, and object thickness.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ayush Kaura",
    "id": "2381107062",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Vignesh Vembar",
    "id": "2381987150",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Md Faizal Karim",
    "id": "2295669480",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Keshab Patra",
    "id": "2281504009",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "K. M. Krishna",
    "id": "2256988413",
    "h_index": 2,
    "papers": 5
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.19188v1",
  "pdf_url": "https://arxiv.org/pdf/2608.19188v1",
  "html_url": "https://arxiv.org/html/2608.19188v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.19182",
  "slug": "adept-accelerating-dexterity-via-pre-training-and-post-training-using",
  "title": "ADEPT: Accelerating Dexterity via Pre-Training and Post-Training using Reinforcement Learning",
  "abstract": "We introduce Accelerating Dexterity via Pre-Training (ADEPT), a large-scale reinforcement learning (RL) framework for learning sim-to-real transferable dexterity across high degree-of-freedom (DoF) robot embodiments that can solve long-horizon tasks directly from raw visuo-tactile perception. ADEPT pretrains a dexterous policy on a generic object reposing task, then post-trains downstream policies with this pretrained behavior as a prior. ADEPT enables learning new behaviors that are otherwise difficult to discover from scratch on multi-fingered robots and avoids learning the same set of skills over again for every new downstream task. The pretrained policy zero-shots the reposing phase of downstream tasks, but na\u00efve RL fine-tuning rapidly degrades this capability during transfer. We address this with a stable post-training recipe combining behavior-cloning distillation, critic warm-up, and conservative on-policy updates. To safely exploit the full kinematic dexterity, we introduce a joint-space Geometric Fabric that mediates between the RL policy and the robot. We distill post-trained teachers into perceptive students that zero-shot sim-to-real transfer on two embodiments: a 23 DoF Kuka-Allegro with two RGB cameras, and a 29 DoF Flexiv-Sharpa with two RGB cameras and five vision-based tactile sensors, and can solve long-horizon tasks from challenging initial states with dexterity at human-level speed.",
  "published": "2026-08-19",
  "updated": "2026-08-19",
  "year": "2026",
  "authors": [
   "Jayjun Lee",
   "Jessica Yin",
   "Asif Rana",
   "Nicholas Blauch",
   "Sam Mady",
   "Mohak Bhardwaj",
   "Nima Fazeli",
   "Nathan Ratliff",
   "Karl Van Wyk",
   "Ankur Handa"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jayjun Lee",
    "id": "2322455757",
    "h_index": 4,
    "papers": 14
   },
   {
    "name": "Jessica Yin",
    "id": "1390925803",
    "h_index": 7,
    "papers": 15
   },
   {
    "name": "Asif Rana",
    "id": "2359456723",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Nicholas M. Blauch",
    "id": "32080534",
    "h_index": 7,
    "papers": 23
   },
   {
    "name": "S. Mady",
    "id": "2363452219",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "M. Bhardwaj",
    "id": "22275235",
    "h_index": 10,
    "papers": 17
   },
   {
    "name": "Nima Fazeli",
    "id": "2718552",
    "h_index": 19,
    "papers": 83
   },
   {
    "name": "Nathan D. Ratliff",
    "id": "2240527931",
    "h_index": 8,
    "papers": 11
   },
   {
    "name": "Karl Van Wyk",
    "id": "2423933",
    "h_index": 20,
    "papers": 53
   },
   {
    "name": "Ankur Handa",
    "id": "2328010136",
    "h_index": 6,
    "papers": 8
   }
  ],
  "comment": "Project page: https://adept-dexterity.github.io/",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "sim2real",
   "rl-control",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.19182v1",
  "pdf_url": "https://arxiv.org/pdf/2608.19182v1",
  "html_url": "https://arxiv.org/html/2608.19182v1",
  "code_url": "https://adept-dexterity.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2608.19085",
  "slug": "da-wam-decision-aligned-future-latents-for-driving-world-models",
  "title": "DA-WAM: Decision-Aligned Future Latents for Driving World Models",
  "abstract": "Anticipating how scenes evolve under ego actions is fundamental to safe autonomous driving, yet the full potential of world models for decision-making remains unrealized. The critical challenge lies in ensuring that future modeling is not merely predictive, but decision-informative: the predicted future must directly shape which trajectory is selected. Existing approaches decouple future representation learning from planning optimization, or share predicted states across trajectory candidates, thereby diluting the action-specific consequences that ought to guide selection. To bridge this gap, we propose DA-WAM, a framework that unifies predictive representation learning, action-conditioned future modeling, and trajectory scoring under a single decision-making objective. DA-WAM maintains predictive supervision throughout planner optimization via an online encoder and a stable momentum target, allowing future representations to co-evolve with the driving task. An action-conditioned predictor generates a distinct future latent state per trajectory candidate, which is then evaluated by a future-latent-conditioned factorized scorer. For the expert-matched trajectory, the predicted future latent is supervised by the observed future representation, while safety-critical hard negatives provide additional supervision near planning boundaries. Extensive experiments on NAVSIM-v1 and NAVSIM-v2 demonstrate state-of-the-art performance, while ablations and diagnostic analyses validate the key components.",
  "published": "2026-08-19",
  "updated": "2026-08-20",
  "year": "2026",
  "authors": [
   "Ruiguo Zhong",
   "Benshan Ma",
   "Xiaolong Chen",
   "Lang Zhang",
   "Mingyue Feng",
   "Yaonong Wang",
   "Pei Liu",
   "Jun Ma"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "DA-WAM is proposed, a framework that unifies predictive representation learning, action-conditioned future modeling, and trajectory scoring under a single decision-making objective and demonstrates state-of-the-art performance on NAVSIM-v1 and NAVSIM-v2.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ruiguo Zhong",
    "id": "2333359555",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Benshan Ma",
    "id": "2374987893",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Xiaolong Chen",
    "id": "2349941196",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Lang Zhang",
    "id": "2374280219",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Mingyue Feng",
    "id": "2310397283",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Yaonong Wang",
    "id": "2378861215",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Pei Liu",
    "id": "2359687414",
    "h_index": 3,
    "papers": 19
   },
   {
    "name": "Jun Ma",
    "id": "2349362645",
    "h_index": 4,
    "papers": 11
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.19085v2",
  "pdf_url": "https://arxiv.org/pdf/2608.19085v2",
  "html_url": "https://arxiv.org/html/2608.19085v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.19059",
  "slug": "lt-mem-volatility-aware-spatio-temporal-memory-for-lifelong-scene-unde",
  "title": "LT-Mem: Volatility-Aware Spatio-Temporal Memory for Lifelong Scene Understanding",
  "abstract": "Long-term robot operation in evolving environments requires object-level understanding that persists across repeated revisits. Existing systems either overwrite history to maintain an up-to-date map or store semantic snapshots without consistent cross-session object identity, resulting in temporal amnesia: the systematic loss of object history that prevents answering queries such as \"Where has the green chair been across all sessions?\" We propose LT-Mem, a volatility-aware memory evolution framework that unifies spatially aligned instance-level 3D perception with volatility-conditioned temporal reasoning. First, a multi-session SLAM backbone provides spatially aligned per-object observations across sessions. Second, a reasoning layer governs how object memory evolves: deterministic evidence scoring preserves cross-session identity, and a volatility-aware policy selects among overwrite, hold, and multi-hypothesis actions based on each object's dynamics. Third, the resulting Tri-Memory structure (Live, Delta, Meta) preserves both current states and event histories, enabling longitudinal object-centric reasoning. We further introduce LT-VQA, a dataset and evaluation suite comprising multi-session recordings, persistent identity annotations, and temporal QA pairs. Experiments show that LT-Mem consistently outperforms baselines across all metrics while consuming an order of magnitude fewer tokens, and ablations confirm that gains are driven by the structured memory architecture rather than LLM capacity.",
  "published": "2026-08-19",
  "updated": "2026-08-19",
  "year": "2026",
  "authors": [
   "Yumin Lee",
   "Hyoseok Ju",
   "Giseop Kim"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "LT-Mem is proposed, a volatility-aware memory evolution framework that unifies spatially aligned instance-level 3D perception with volatility-conditioned temporal reasoning and introduces LT-VQA, a dataset and evaluation suite comprising multi-session recordings, persistent identity annotations, and temporal QA pairs.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yu-Han Lee",
    "id": "2145405676",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Hyoseok Ju",
    "id": "2416215942",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Giseop Kim",
    "id": "2243365255",
    "h_index": 3,
    "papers": 13
   }
  ],
  "comment": "8 pages, 8 figures, 6 tables. Accepted to the 2026 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS 2026)",
  "topics": [
   "spatial-3d",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.19059v1",
  "pdf_url": "https://arxiv.org/pdf/2608.19059v1",
  "html_url": "https://arxiv.org/html/2608.19059v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.19004",
  "slug": "autonomous-agricultural-tractor-integrated-weed-detection-and-lidar-na",
  "title": "Autonomous Agricultural Tractor: Integrated Weed Detection and LiDAR Navigation for Precision Paddy Farming",
  "abstract": "Site-specific weed management in paddy farming offers substantial reductions in herbicide use over conventional broadcast spraying, but field deployment has been limited by three persistent challenges: robust crop-row navigation under canopy where GNSS degrades, real-time visual discrimination between rice and morphologically diverse weeds, and the asymmetric cost of misclassifying rice as weed, which is irreversible. This paper presents AgriNav, an integrated autonomous tractor system built around four ROS-coupled modules: a custom PyTorch reimplementation of WeedDet for rice detection, a parallel lightweight 1.68M-parameter CNN-FPN variant with asymmetric class weighting, an inverted-logic discrimination module that protects the rice class through a hardcoded confidence-gate veto, and a 6-state constant-velocity-turn-rate Extended Kalman Filter fusing GNSS, IMU, and wheel odometry with three-level outage bridging. Our primary system-level contribution is a four-mechanism LiDAR-camera fusion bridge that uses the navigation LiDAR for region-of-interest constraint, world-coordinate projection, ground-plane filtering, and bidirectional confidence fusion at zero additional hardware cost. Simulation experiments demonstrate continuous position tracking through a 20-second GNSS outage, crop row detection confidence above 0.9 throughout operation, and rice-detection confidences from 0.32 to 0.95 across paddy, aerial, and post-flood imagery. The LiDAR ROI constraint reduces detection inference region by an estimated 30 to 50 percent.",
  "published": "2026-08-19",
  "updated": "2026-08-19",
  "year": "2026",
  "authors": [
   "Benjamin Merryman-Smith",
   "Tony Nguyen",
   "Bilal Dogutas",
   "Krish Shah",
   "Anthony Raphael",
   "Sudip Dhakal"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Benjamin Merryman-Smith",
    "id": "2458372505",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Tony Nguyen",
    "id": "2216180687",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Bilal Dogutas",
    "id": "2458372417",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Krish Shah",
    "id": "2345103180",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Anthony Raphael",
    "id": "2458372500",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Sudip Dhakal",
    "id": "2122691154",
    "h_index": 5,
    "papers": 9
   }
  ],
  "comment": "",
  "topics": [
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.19004v1",
  "pdf_url": "https://arxiv.org/pdf/2608.19004v1",
  "html_url": "https://arxiv.org/html/2608.19004v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.18948",
  "slug": "roboedit-turning-human-manipulation-videos-into-scalable-robot-experie",
  "title": "RoboEdit: Turning Human Manipulation Videos into Scalable Robot Experience",
  "abstract": "Collecting robot hand-object interaction data is costly and embodiment-specific, yet abundant human-object videos remain unusable for robot training. We present RoboEdit, a human-to-robot video editing suite that transforms human manipulation videos into action-consistent, physically plausible robot videos with aligned 3D hand states. To enable scalable supervision, we introduce RoboEdit-ADC, an automatic pipeline that reconstructs and retargets 3D interactions from RGB videos across embodiments. This pipeline generates RoboEdit-14M, a large-scale dataset of 174K aligned video pairs (14M frames) spanning seven robot embodiments, diverse scenes, and interaction types. The core editing engine, RoboEdit-Trans, employs cross-embodiment adaptation modules to preserve temporal coherence while adapting appearance and motion. It further integrates a 3D Robot-State Decoder to recover per-frame hand states for structured motion supervision. Experiments show that RoboEdit achieves state-of-the-art editing quality and supports downstream robot control policies in real-world manipulation tasks. Ultimately, the RoboEdit suite unlocks the vast potential of unlabeled human videos, providing scalable, high-fidelity visual and 3D motion supervision for generalizable robot learning.",
  "published": "2026-08-19",
  "updated": "2026-08-19",
  "year": "2026",
  "authors": [
   "Yaowei Guo",
   "Zeng Tao",
   "Yuxin Jiang",
   "Yunuo Chen",
   "Zhiyang Dou",
   "Yuxiang Ma",
   "Yin Yang",
   "Demetri Terzopoulos",
   "Ying Jiang",
   "Chenfanfu Jiang"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "RoboEdit is presented, a human-to-robot video editing suite that transforms human manipulation videos into action-consistent, physically plausible robot videos with aligned 3D hand states, providing scalable, high-fidelity visual and 3D motion supervision for generalizable robot learning.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yaowei Guo",
    "id": "2198053384",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Zeng Tao",
    "id": "2261831274",
    "h_index": 8,
    "papers": 25
   },
   {
    "name": "Yuxin Jiang",
    "id": "2361654632",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Yunuo Chen",
    "id": "2344397292",
    "h_index": 4,
    "papers": 16
   },
   {
    "name": "Zhiyang Dou",
    "id": "2292386296",
    "h_index": 8,
    "papers": 24
   },
   {
    "name": "Yuxiang Ma",
    "id": "2144365142",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Yin Yang",
    "id": "2381261597",
    "h_index": 1,
    "papers": 10
   },
   {
    "name": "D. Terzopoulos",
    "id": "70287739",
    "h_index": 18,
    "papers": 87
   },
   {
    "name": "Ying Jiang",
    "id": "2282079101",
    "h_index": 7,
    "papers": 28
   },
   {
    "name": "Chenfanfu Jiang",
    "id": "2267866705",
    "h_index": 16,
    "papers": 77
   }
  ],
  "comment": "14 pages, 13 figures. Supplementary material included",
  "topics": [
   "egocentric-data",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.18948v1",
  "pdf_url": "https://arxiv.org/pdf/2608.18948v1",
  "html_url": "https://arxiv.org/html/2608.18948v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.18840",
  "slug": "beyond-placement-and-articulation-usage-driven-code-scenes-for-embodie",
  "title": "Beyond Placement and Articulation: Usage-Driven Code Scenes for Embodied Interaction",
  "abstract": "Indoor scene synthesis provides essential environments for embodied AI, robotic manipulation, and simulation-based policy learning. Recent code-based scene generation methods produce editable and extensible environments, yet they remain focused on visual construction and object-level articulation, leaving the functional usage of scenes largely unmodeled. To address this problem, we present RoomWright, an agentic usage-driven framework for generating 3D scenes represented entirely as code for embodied interaction. RoomWright performs usage-driven object reasoning, which treats each anchor as a task centre and admits task-required objects and their affordances. A code agent further enables multi-part interaction by compiling each interaction into a trigger, condition, effect rule that updates structured object states, capturing causal dependencies across objects. Moreover, since manipuland orientation is ambiguous and hard to recover from pixels, RoomWright alleviates this via annotation-informed usage-guided orientation. Extensive experiments demonstrate the effectiveness of our method. The resulting scenes are executable, editable, and simulation-ready, providing interactive environments for embodied AI and policy learning.",
  "published": "2026-08-19",
  "updated": "2026-08-19",
  "year": "2026",
  "authors": [
   "Zijian Xiao",
   "Zipeng Ye",
   "Jinkun Hao",
   "Xiong Yang",
   "Yuchen Xie",
   "Ran Yi"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "RoomWright is presented, an agentic usage-driven framework for generating 3D scenes represented entirely as code for embodied interaction, providing interactive environments for embodied AI and policy learning.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zijian Xiao",
    "id": "2456286154",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Zipeng Ye",
    "id": "7818093",
    "h_index": 11,
    "papers": 21
   },
   {
    "name": "Jinkun Hao",
    "id": "2295705469",
    "h_index": 3,
    "papers": 13
   },
   {
    "name": "Xiong Yang",
    "id": "2449150930",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yu-Han Xie",
    "id": "2382004631",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Ran Yi",
    "id": "2294877670",
    "h_index": 6,
    "papers": 18
   }
  ],
  "comment": "",
  "topics": [
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.18840v1",
  "pdf_url": "https://arxiv.org/pdf/2608.18840v1",
  "html_url": "https://arxiv.org/html/2608.18840v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.18787",
  "slug": "dream2reward-transition-alignment-reward-models-from-positive-demonstr",
  "title": "Dream2Reward: Transition-Alignment Reward Models from Positive Demonstrations for Robotic Manipulation",
  "abstract": "Learning robotic policies requires dense rewards that remain informative when behavior departs from successful demonstrations. Progress-based rewards estimate how far an observation has advanced along a nominal successful trajectory, but may remain high after an incorrect transition. We introduce Dream2Reward, which learns a language-conditioned successful latent transition field from positive demonstrations. Given the visual history up to a transition start, the model predicts the latent displacement associated with successful execution and scores the observed displacement through signed directional and symmetric magnitude agreement. This transition-level comparison penalizes wrong-direction, overshooting, and stagnant motion even when the resulting observation appears to show progress. Dream2Reward requires no failure annotations, progress labels, or synthetic negatives, and produces a dense causal reward. Across mechanism diagnostics and shared-trajectory evaluations, it provides stronger success-failure separation and more informative feedback on low-quality behavior than progress-based alternatives. Across online and offline policy learning, the same frozen reward model reduces reward hacking and supports stronger downstream performance, including in real-robot manipulation. These results show that comparing realized motion with predicted successful change provides an effective way to convert positive demonstrations into dense rewards for robot learning.",
  "published": "2026-08-19",
  "updated": "2026-08-19",
  "year": "2026",
  "authors": [
   "Haoyu Zhang",
   "Zecui Zeng",
   "Bin Wang",
   "Lusong Li",
   "Liang Lin",
   "Long Cheng"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work introduces Dream2Reward, which learns a language-conditioned successful latent transition field from positive demonstrations that provides stronger success-failure separation and more informative feedback on low-quality behavior than progress-based alternatives.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Haoyu Zhang",
    "id": "2276656133",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Zecui Zeng",
    "id": "2304012939",
    "h_index": 4,
    "papers": 14
   },
   {
    "name": "Bin Wang",
    "id": "2293648083",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Lusong Li",
    "id": "2268580598",
    "h_index": 6,
    "papers": 19
   },
   {
    "name": "Liang Lin",
    "id": "2316519420",
    "h_index": 7,
    "papers": 18
   },
   {
    "name": "Long Cheng",
    "id": "2287293379",
    "h_index": 3,
    "papers": 7
   }
  ],
  "comment": "12 pages, 7 figures",
  "topics": [
   "rl-control",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.18787v1",
  "pdf_url": "https://arxiv.org/pdf/2608.18787v1",
  "html_url": "https://arxiv.org/html/2608.18787v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.18778",
  "slug": "engineering-psychological-safety-in-autonomous-vehicles-a-systems-theo",
  "title": "Engineering Psychological Safety in Autonomous Vehicles: A Systems-Theoretic Framework for Psychological Safety in Autonomous Vehicles and its Validation in Real-World Scenarios",
  "abstract": "Despite rapid technological advances, the societal acceptability of autonomous vehicles (AVs) remains limited by psychological barriers that extend beyond traditional concerns of physical safety. While factors such as trust and perceived safety are known to influence user acceptance, there is a lack of formalized mechanisms and engineering methods to systematically identify, assess, and mitigate psychological risks arising from human-AV interactions. To address this gap, this work proposes and validates a systems-theoretic framework for the assessment of psychological safety in autonomous vehicles. First, a comprehensive psychological safety risk model is defined, extending the Systems-Theoretic Accident Model and Processes (STAMP) to incorporate key psychological constructs such as trust, perceived control, predictability, and perceived support. Based on this model, a hazard analysis method (AV-PsySafe) is developed to systematically identify psychological hazards, unsafe control actions, and loss scenarios, while introducing a Psychological Safety Integrity Level (PsySIL) to support risk prioritization. Second, the applicability and relevance of the framework are evaluated through its deployment in realistic autonomous vehicle scenarios. A structured validation approach is implemented, including a methodological guide, standardized analysis templates, and the collection of analyst feedback. The results demonstrate that the framework can be consistently applied by practitioners, producing meaningful insights into psychological risks. Overall, this work establishes both the theoretical foundations and practical feasibility of a unified approach to co-assessing psychological and physical safety in autonomous systems, contributing to more human-centred and trustworthy AV development.",
  "published": "2026-08-19",
  "updated": "2026-08-19",
  "year": "2026",
  "authors": [
   "Yandika Sirgabsou",
   "Benjamin Hardin",
   "Fran\u00e7ois Leblanc",
   "Efi Raili",
   "David Jackson",
   "Pericle Salvini",
   "Lars Kunze",
   "Marina Jirotka"
  ],
  "author_count": 8,
  "categories": [
   "cs.HC",
   "cs.RO"
  ],
  "primary_category": "cs.HC",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yandika Sirgabsou",
    "id": "2156457686",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Benjamin Hardin",
    "id": "2275727888",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Fran\u00e7ois Leblanc",
    "id": "2143936907",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Efi Raili",
    "id": "2330081854",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "David Jackson",
    "id": "2065117486",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Pericle Salvini",
    "id": "153723511",
    "h_index": 14,
    "papers": 58
   },
   {
    "name": "L. Kunze",
    "id": "121364177",
    "h_index": 14,
    "papers": 67
   },
   {
    "name": "M. Jirotka",
    "id": "2270317002",
    "h_index": 6,
    "papers": 21
   }
  ],
  "comment": "",
  "topics": [
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.18778v1",
  "pdf_url": "https://arxiv.org/pdf/2608.18778v1",
  "html_url": "https://arxiv.org/html/2608.18778v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.18770",
  "slug": "to-go-far-go-together-diverse-preferences-induce-a-curriculum-for-rewa",
  "title": "To Go Far, Go Together: Diverse Preferences Induce a Curriculum for Reward Optimization",
  "abstract": "Learning a reward model from human feedback and optimizing a policy against it is one approach to aligning AI systems with individual users. From a fairness perspective, existing work improves such alignment by developing data-efficient and accurate reward models that capture minority preferences despite scarce data. We push this line of inquiry one step further and argue that data-efficient and accurate per-user reward models are not sufficient: users whose reward models are difficult to \\textit{optimize} at the policy level can become a new underserved group. We start from the observation that one user's reward model can be easy to optimize from the initial policy while another's is not. We argue that, given a sufficiently diverse user population, a curriculum naturally emerges between easy- and hard-to-optimize reward models. Building on this insight, we propose CurriPO, which grows a tree-structured curriculum to accommodate diverse user-specific objectives, covering the population in a single traversal. Specifically, CurriPO automatically constructs a curriculum over diverse user reward models, allowing it to branch from the existing curriculum and reuse reward models previously incorporated into the curriculum. To the best of our knowledge, this is the first work to explicitly exploit multi-user structure to address optimization in AI alignment. Extensive experiments on personalized continuous control in a simulated environment show that CurriPO achieves $1.2$--$2.1\\times$ the population satisfaction of the strongest baseline while substantially reducing training time. Additional analysis attributes much of this improvement to the users left underserved by conventional optimization.",
  "published": "2026-08-19",
  "updated": "2026-08-19",
  "year": "2026",
  "authors": [
   "Taehyung Kim",
   "Jongeun Choi"
  ],
  "author_count": 2,
  "categories": [
   "cs.LG",
   "cs.RO"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work argues that, given a sufficiently diverse user population, a curriculum naturally emerges between easy- and hard-to-optimize reward models, and proposes CurriPO, which grows a tree-structured curriculum to accommodate diverse user-specific objectives, covering the population in a single traversal.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Taehyung Kim",
    "id": "2292429402",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Jongeun Choi",
    "id": "2238150779",
    "h_index": 5,
    "papers": 15
   }
  ],
  "comment": "14 pages, 7 figures",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.18770v1",
  "pdf_url": "https://arxiv.org/pdf/2608.18770v1",
  "html_url": "https://arxiv.org/html/2608.18770v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.18701",
  "slug": "softvtbench-a-deformation-aware-visuo-tactile-dataset-and-benchmark-fo",
  "title": "SoftVTBench: A Deformation-Aware Visuo-Tactile Dataset and Benchmark for Deformable-Object Manipulation",
  "abstract": "Physical interaction quality is central to deformable-object manipulation, yet most benchmarks evaluate task success alone. A policy may complete the task while allowing slip or causing excessive compression. A primary bottleneck is the absence of visuo-tactile datasets that pair policy-visible contact observations with independent physical ground truth over complete tasks. We introduce SoftVTBench, a visuo-tactile dataset for physical-interaction-aware deformable-object manipulation. It contains 4,000 expert demonstrations and more than 50 assets, including volumetric deformable objects and visually matched rigid twins. At 20 Hz, each episode synchronizes multi-view RGB, dual-finger tactile RGB and marker motion, proprioception, language, and binary and continuous gripper actions, alongside evaluator-only finite-element (FEM) states. Building upon this dataset, we establish a closed-loop benchmark that uses fixed object-specific calibration to define the Deformation-aware Success Rate (DSR), which counts a rollout as successful only when it completes the task and keeps peak normalized deformation within tolerance. Across Diffusion Policy, $\u03c0_{0.5}$, and FastWAM, all 12 in-distribution configurations contain successful rollouts that violate the deformation tolerance, accounting for 0.7--24% of each configuration's successes. Under distribution shift, visuo-tactile variants achieve higher task success in all six policy--suite comparisons and higher DSR in five, whereas their in-distribution benefits are mixed. These results show that making touch available does not by itself ensure effective multimodal fusion. SoftVTBench therefore provides a common visuo-tactile resource for studying not only whether a policy succeeds, but how it physically interacts with deformable objects and when touch improves that interaction.",
  "published": "2026-08-19",
  "updated": "2026-08-19",
  "year": "2026",
  "authors": [
   "Bowen Jing",
   "Mingxin Wang",
   "Ruiyang Hao",
   "Chenchen Ge",
   "Hanwen Shen",
   "Junjie He",
   "Yang Cui",
   "Yiming Hou",
   "Weitao Zhou",
   "Jiawei Wang",
   "Minglei Li",
   "Dandan Zhang",
   "Ding Zhao",
   "Houde Liu",
   "Xiaofan Li",
   "Si Liu",
   "Ping Luo",
   "Haibao Yu"
  ],
  "author_count": 18,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Results show that making touch available does not by itself ensure effective multimodal fusion, and SoftVTBench provides a common visuo-tactile resource for studying not only whether a policy succeeds, but how it physically interacts with deformable objects and when touch improves that interaction.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Bowen Jing",
    "id": "2333354814",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Mingxin Wang",
    "id": "2115447018",
    "h_index": 1,
    "papers": 13
   },
   {
    "name": "Ruiyang Hao",
    "id": "2000239184",
    "h_index": 9,
    "papers": 14
   },
   {
    "name": "Chenchen Ge",
    "id": "2448002689",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Hanwen Shen",
    "id": "2448346170",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Junjie He",
    "id": "2448058828",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Yang Cui",
    "id": "2448110836",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yiming Hou",
    "id": "2448350747",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Weitao Zhou",
    "id": "2148943699",
    "h_index": 7,
    "papers": 29
   },
   {
    "name": "Jiawei Wang",
    "id": "2448016120",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Minglei Li",
    "id": "2448263547",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Dandan Zhang",
    "id": "2354220119",
    "h_index": 3,
    "papers": 22
   },
   {
    "name": "Ding Zhao",
    "id": "47783130",
    "h_index": 30,
    "papers": 81
   },
   {
    "name": "Houde Liu",
    "id": "2293440971",
    "h_index": 3,
    "papers": 26
   },
   {
    "name": "Xiaofan Li",
    "id": "2394589651",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Si Liu",
    "id": "2291108679",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Ping Luo",
    "id": "2257349788",
    "h_index": 8,
    "papers": 22
   },
   {
    "name": "Haibao Yu",
    "id": "2162290793",
    "h_index": 12,
    "papers": 26
   }
  ],
  "comment": "",
  "topics": [
   "tactile",
   "imitation-diffusion"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.18701v1",
  "pdf_url": "https://arxiv.org/pdf/2608.18701v1",
  "html_url": "https://arxiv.org/html/2608.18701v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.18672",
  "slug": "orienteering-problem-with-uncertain-time-varying-rewards-framework-and",
  "title": "Orienteering Problem with Uncertain Time-Varying Rewards: Framework and Benchmark for Everyday Service Robotics",
  "abstract": "We present the orienteering problem with uncertain time-varying rewards (OP-UTVR), a novel variant of the orienteering problem (OP). While most existing OP formulations assume rewards to be known in advance, practical applications involve uncertain and time-varying rewards, as with shifting customer demand for delivery agents. OP-UTVR relaxes this assumption by allowing agents to estimate reward dynamics from observations and forecast future rewards. This enables informed routing decisions despite stochastic reward changes and inevitable prediction errors. We address this problem using three planners that differ in planning horizon and online adaptivity, and derive theoretical bounds on their performance under reward stochasticity. We further introduce a mobile service robot benchmark for OP-UTVR, where a robot navigates among pedestrians in indoor environments. Experiments reveal trade-offs between planning horizon and adaptivity, and demonstrate the effectiveness of long-horizon planning with online adaptation.",
  "published": "2026-08-19",
  "updated": "2026-08-19",
  "year": "2026",
  "authors": [
   "Masafumi Endo",
   "Kohei Honda",
   "Yuu Jinnai",
   "Ryo Yonetani"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work presents the orienteering problem with uncertain time-varying rewards (OP-UTVR), a novel variant of the orienteering problem (OP), and introduces a mobile service robot benchmark for OP-UTVR, where a robot navigates among pedestrians in indoor environments.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Masafumi Endo",
    "id": "2052789578",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Kohei Honda",
    "id": "2302804783",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Yuu Jinnai",
    "id": "3387240",
    "h_index": 14,
    "papers": 42
   },
   {
    "name": "Ryo Yonetani",
    "id": "2323508335",
    "h_index": 2,
    "papers": 5
   }
  ],
  "comment": "11 pages, 6 figures",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.18672v1",
  "pdf_url": "https://arxiv.org/pdf/2608.18672v1",
  "html_url": "https://arxiv.org/html/2608.18672v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.18662",
  "slug": "dynamic-spectraformer-for-ultra-high-definition-underwater-image-enhan",
  "title": "Dynamic SpectraFormer for Ultra-High-Definition Underwater Image Enhancement",
  "abstract": "Underwater images suffer from color distortion, haze, and poor visibility due to light refraction and absorption in water. These challenges significantly impact the utilization of Autonomous Underwater Vehicles (AUVs) or marine robots. Typically, color and brightness distortions manifest at lower frequencies, while edge and texture distortions are prevalent at higher frequencies. Traditional methods struggle to concurrently rectify these mixed distortions as they primarily concentrate on the spatial domain. To address these issues, we introduce the Dynamic SpectraFormer, which enhances underwater images through a frequency domain transformer. The Dynamic SpectraFormer introduces an ultra-high-resolution sparse spectrum attention module, which could capture the long-term dependency without losing the universal approximating power. Additionally, we have developed a dynamic spectrum weight generation layer that serves as an adaptive spectrum band selector, accentuating critical frequency bands and suppressing less relevant ones. Consequently, this method significantly improves underwater image quality by addressing both high- and low-frequency distortions. Our extensive ablation studies and comparative evaluations consolidate the Dynamic SpectraFormer's efficacy across multiple underwater image enhancement benchmarks. The source code is available at https://github.com/arifence2024/DynamicSpectraFormer.git.",
  "published": "2026-08-19",
  "updated": "2026-08-19",
  "year": "2026",
  "authors": [
   "Zhiqiang Hu",
   "Tao Yu",
   "Shouren Huang",
   "Masatoshi Ishikawa"
  ],
  "author_count": 4,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "IROS",
  "venue_source": "semantic-scholar",
  "citations": 3,
  "influential_citations": 0,
  "tldr": "",
  "doi": "10.1109/IROS58592.2024.10802529",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhiqiang Hu",
    "id": "2239440449",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Tao Yu",
    "id": "2256865604",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Shouren Huang",
    "id": "2205923626",
    "h_index": 10,
    "papers": 55
   },
   {
    "name": "Masatoshi Ishikawa",
    "id": "2275353838",
    "h_index": 1,
    "papers": 15
   }
  ],
  "comment": "8 pages, 7 figures. Published in the 2024 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS 2024)",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.18662v1",
  "pdf_url": "https://arxiv.org/pdf/2608.18662v1",
  "html_url": "https://arxiv.org/html/2608.18662v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.1
 },
 {
  "id": "2608.18647",
  "slug": "progressive-experience-fusion-for-multi-task-world-model-control-in-en",
  "title": "Progressive Experience Fusion for Multi-Task World Model Control in Endovascular Navigation",
  "abstract": "Autonomous endovascular navigation could support the delivery of mechanical thrombectomy to underserved areas, but controllers must navigate long, multi-stage paths across varying vascular anatomies. This study investigates Progressive Experience Fusion (PEF) to train a multi-task TD-MPC2 controller. We additionally evaluate a heuristic that changes the Model Predictive Path Integral planning horizon using residual action-sequence dispersion, and fine-tuning in a patient-specific simulation. Across five subtasks in ten known training anatomies with held-out targets, PEF achieved a mean success rate of 74%, compared with 37% for Soft Actor-Critic (p < 0.001) and 65% for base TD-MPC2 (p = 0.053). A PEF controller with adaptive-horizon planning trained on 30 vasculatures achieved a mean success rate of 90% in ten held-out vasculatures. The PEF agent successfully transferred to an unseen in vitro stroke patient vasculature under fluoroscopy, achieving a mean path ratio improvement from 63% to 80% with fine-tuning (p < 0.001), following 40x103 fine-tuning steps (corresponding to approximately 107 min of clinical inter-hospital transfer time). This work represents a proof of concept for multi-vasculature training and patient-specific adaptation, while further validation is required before clinical deployment.",
  "published": "2026-08-19",
  "updated": "2026-08-19",
  "year": "2026",
  "authors": [
   "Harry Robertshaw",
   "Maxence Boels",
   "Nikola Fischer",
   "Sebastien Ourselin",
   "Christos Bergeles",
   "Alejandro Granados",
   "Thomas C Booth"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Progressive Experience Fusion is investigated to train a multi-task TD-MPC2 controller to train a multi-vasculature training and patient-specific adaptation, while further validation is required before clinical deployment.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Harry Robertshaw",
    "id": "2306899054",
    "h_index": 6,
    "papers": 21
   },
   {
    "name": "Maxence Boels",
    "id": "1517921244",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "N. Fischer",
    "id": "2124702922",
    "h_index": 5,
    "papers": 26
   },
   {
    "name": "S\u00e9bastien Ourselin",
    "id": "2244394305",
    "h_index": 9,
    "papers": 56
   },
   {
    "name": "Christos Bergeles",
    "id": "2241433151",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Alejandro Granados",
    "id": "2306899115",
    "h_index": 5,
    "papers": 17
   },
   {
    "name": "Thomas C. Booth",
    "id": "2223750483",
    "h_index": 5,
    "papers": 22
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "rl-control",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.18647v1",
  "pdf_url": "https://arxiv.org/pdf/2608.18647v1",
  "html_url": "https://arxiv.org/html/2608.18647v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.18632",
  "slug": "evaluation-of-monocular-slam-systems-on-high-altitude-nadir-uav-footag",
  "title": "Evaluation of Monocular SLAM Systems on High-Altitude Nadir UAV Footage",
  "abstract": "Aerial nadir video combines weak geometric constraints with severe perceptual aliasing, making it a difficult regime for monocular SLAM. We benchmark five monocular SLAM systems on local UAV flights, synthetic city-scale imagery, and long-range aerial sequences. To isolate visual performance, we provide no inertial or GNSS aiding. Performance varies strongly with environment and trajectory scale: MASt3R-SLAM achieves the lowest mean horizontal MAE on the five DJI flights (0.53% of reference path length), whereas no system consistently preserves global trajectory shape on the long GES and ALTO sequences. Overall, DROID-SLAM performs best, averaging 2.88% of reference path length across completed runs. Vertical position remains poor, and large-area trajectories remain highly distorted despite loop-closure capability. Current monocular SLAM methods are by themselves therefore insufficient for reliable visual-only aerial navigation.",
  "published": "2026-08-19",
  "updated": "2026-08-24",
  "year": "2026",
  "authors": [
   "Ga\u0161per Spagnolo",
   "Matej Dobrevski",
   "Danijel Sko\u010daj"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Five monocular SLAM systems are benchmarked on local UAV flights, synthetic city-scale imagery, and long-range aerial sequences to isolate visual performance, and current methods are found insufficient for reliable visual-only aerial navigation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ga\u0161per Spagnolo",
    "id": "2394082695",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Matej Dobrevski",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "D. Sko\u010daj",
    "id": "2238343",
    "h_index": 32,
    "papers": 125
   }
  ],
  "comment": "6 pages, accepted to ERK 2026",
  "topics": [
   "spatial-3d",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.18632v2",
  "pdf_url": "https://arxiv.org/pdf/2608.18632v2",
  "html_url": "https://arxiv.org/html/2608.18632v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.18625",
  "slug": "payload-swing-estimation-and-damping-without-payload-parameters-for-mu",
  "title": "Payload Swing Estimation and Damping Without Payload Parameters for Multirotor UAVs",
  "abstract": "Cable-suspended payload transport by multirotor UAVs is flexible but generates periodic swing disturbance that degrades tracking and risks instability. Existing anti-swing methods require additional sensors or precise identification of cable length and payload mass, limiting field deployment. We propose a swing-estimation and damping method using only the onboard IMU and throttle command, requiring no payload parameters. An extended Kalman filter extracts the periodic disturbance with the unknown pendulum frequency as an estimated state, and an active damping controller adds a correction angle to the attitude loop to dissipate pendulum energy. Flight experiments confirm robust damping across a tested range of cable-length and mass variations.",
  "published": "2026-08-19",
  "updated": "2026-08-19",
  "year": "2026",
  "authors": [
   "K. Taki",
   "K. Umemoto"
  ],
  "author_count": 2,
  "categories": [
   "eess.SY",
   "cs.RO"
  ],
  "primary_category": "eess.SY",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "K. Taki",
    "id": "2267048671",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "K. Umemoto",
    "id": "152545806",
    "h_index": 0,
    "papers": 1
   }
  ],
  "comment": "",
  "topics": [
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.18625v1",
  "pdf_url": "https://arxiv.org/pdf/2608.18625v1",
  "html_url": "https://arxiv.org/html/2608.18625v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.18618",
  "slug": "labdex-a-hierarchical-benchmark-for-dexterous-manipulation-in-laborato",
  "title": "LabDex: A Hierarchical Benchmark for Dexterous Manipulation in Laboratories",
  "abstract": "Autonomous laboratories hold great promise for accelerating scientific discovery. To achieve this vision, robots are supposed to dexterously manipulate diverse labware and instruments and execute long-horizon, state-dependent experimental procedures. Yet existing benchmarks do not jointly capture dexterous hand use, real-world laboratory interactions, and multi-stage experimental procedures, limiting systematic training and evaluation. To bridge this gap, we introduce LabDex, a large-scale real-world dataset and benchmark for dexterous manipulation in chemistry laboratories, organized around a hierarchical task taxonomy spanning atomic skills, compositional tasks, and long-horizon experiments. First, LabDex is cross-platform and, for the first time, unifies real-world and simulation platforms under a common framework, providing standardized task definitions, demonstrations, and evaluation protocols. Second, LabDex is large-scale and systematically organizes chemistry laboratory operations into three interconnected levels: Atomic Skills, which characterize fundamental dexterous manipulation capabilities; Compositional Skills; and Long-Horizon Laboratory Workflows. This hierarchical design not only supports the evaluation of end-task performance, but also enables the analysis of how fundamental dexterous skills compose and influence more complex laboratory operations. We conduct cross-level evaluations of representative robot learning methods in both real-world and simulation environments. The experimental results validate the effectiveness of the LabDex task design and demonstration data, and show that the benchmark supports the training and systematic evaluation of existing robotic policies across laboratory dexterous manipulation tasks at different levels, providing a foundation for further research and development of autonomous laboratory robots.",
  "published": "2026-08-19",
  "updated": "2026-08-19",
  "year": "2026",
  "authors": [
   "Zhipeng Tang",
   "Sihang Chen",
   "Sha Zhang",
   "Peihao Yang",
   "Yan Liu",
   "Wentao Zhao",
   "Xinrui Liu",
   "Rui Huang",
   "Wensheng Du",
   "Yuting Huang",
   "Jiajun Deng",
   "Lidian Wang",
   "Yuan Zhang",
   "Yanyong Zhang"
  ],
  "author_count": 14,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The experimental results validate the effectiveness of the LabDex task design and demonstration data, and show that the benchmark supports the training and systematic evaluation of existing robotic policies across laboratory dexterous manipulation tasks at different levels, providing a foundation for further research and development of autonomous laboratory robots.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhipeng Tang",
    "id": "2357714106",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Sihan Chen",
    "id": "2447731695",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Sha Zhang",
    "id": "2283448371",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "P. Yang",
    "id": "2431641766",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yan Liu",
    "id": "2458653754",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Wentao Zhao",
    "id": "2308142092",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Xinrui Liu",
    "id": "2446315609",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Rui Huang",
    "id": "2300852411",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Wensheng Du",
    "id": "2458367325",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Yuting Huang",
    "id": "2356577580",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Jiajun Deng",
    "id": "2287933949",
    "h_index": 8,
    "papers": 22
   },
   {
    "name": "Lidian Wang",
    "id": "2327109570",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Yuan Zhang",
    "id": "2458631245",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Yanyong Zhang",
    "id": "2341792088",
    "h_index": 2,
    "papers": 7
   }
  ],
  "comment": "13 pages, 3 figures",
  "topics": [
   "dexterous-manipulation",
   "sim2real"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.18618v1",
  "pdf_url": "https://arxiv.org/pdf/2608.18618v1",
  "html_url": "https://arxiv.org/html/2608.18618v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.18569",
  "slug": "the-role-of-grid-cells-in-reducing-spatial-aliasing-in-hippocampal-pla",
  "title": "The Role of Grid Cells in Reducing Spatial Aliasing in Hippocampal Place Representations",
  "abstract": "Spatial aliasing occurs when two or more distinct locations produce highly similar place-cell representations, primarily due to environmental symmetry or repetitive structures. This issue is most pronounced when place representations are constructed solely from boundary vector cell (BVC) inputs, because symmetric or repetitive structures can yield indistinguishable sensory patterns across multiple locations in an environment. This work introduces grid cell signals to mitigate spatial aliasing in such settings. Because grid cells contribute periodic, internally generated spatial signals that vary independently of environmental geometry, they play a key role in disambiguating perceptually identical locations. We integrate multiple modules of analytically constructed grid cells with BVC-driven place cells and show that this leads to a 94--99% reduction in spatial aliasing relative to a BVC-only baseline across three environments: an open environment without obstacles; an environment with a cross-shaped central obstacle creating high visual symmetry; and a maze environment. The greatest improvement occurs in the environment with the highest visual symmetry. These results indicate that grid cells provide information complementary to boundary-based inputs, yielding more reliable place representations in geometrically ambiguous environments.",
  "published": "2026-08-19",
  "updated": "2026-08-19",
  "year": "2026",
  "authors": [
   "Alexander Johnson",
   "Obadah Ghizawi",
   "Ali A. Minai"
  ],
  "author_count": 3,
  "categories": [
   "cs.NE",
   "cs.AI",
   "cs.RO",
   "eess.SY",
   "q-bio.NC"
  ],
  "primary_category": "cs.NE",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Grid cell signals are introduced to mitigate spatial aliasing in geometrically ambiguous environments and indicate that grid cells provide information complementary to boundary-based inputs, yielding more reliable place representations in geometrically ambiguous environments.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Alexander B. Johnson",
    "id": "2308846297",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Obadah Ghizawi",
    "id": "2458351890",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "A. Minai",
    "id": "9112544",
    "h_index": 28,
    "papers": 186
   }
  ],
  "comment": "IEEE World Congress on Computational Intelligence, Masstricht, Netherlands, June 2026",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.18569v1",
  "pdf_url": "https://arxiv.org/pdf/2608.18569v1",
  "html_url": "https://arxiv.org/html/2608.18569v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.18552",
  "slug": "real-time-control-constrained-ddp-for-underactuated-balancing-of-legge",
  "title": "Real-Time Control-Constrained DDP for Underactuated Balancing of Legged Robots",
  "abstract": "This paper presents a real-time control-constrained Differential Dynamic Programming (DDP) framework for underactuated legged robots. To address the limitation of classical DDP in handling control constraints, we propose an Accelerated Projected Gradient (APG)-based control-constrained DDP (ABC-DDP), which efficiently computes constrained solutions and identifies active sets without repeated Karush-Kuhn-Tucker (KKT) inversions. A virtual constraint is introduced to integrate control constraints within a feasibility-driven multiple-shooting framework, enabling stable optimization even from dynamically infeasible initializations. The proposed method supports real-time model predictive control (MPC) with short horizons under strong underactuation. Simulation results demonstrate static two-leg standing under external disturbances, along with diverse dynamic motions including slow catwalk, upright walking, and high-speed running within a unified MPC framework. To the best of our knowledge, this is the first demonstration of static two-leg standing of a quadruped robot achieved using real-time finite-horizon MPC.",
  "published": "2026-08-19",
  "updated": "2026-08-19",
  "year": "2026",
  "authors": [
   "SeongWon Nam",
   "Hyunyong Lee",
   "Hansol Kang",
   "Jiman Park",
   "Yeongwoo Son",
   "Bumsu Yi",
   "Jaeyoung Oh",
   "Hyouk Ryeol Choi"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO",
   "math.OC"
  ],
  "primary_category": "cs.RO",
  "venue": "RA-L",
  "venue_source": "semantic-scholar",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper presents a real-time control-constrained Differential Dynamic Programming (DDP) framework for underactuated legged robots and proposes an Accelerated Projected Gradient (APG)-based control-constrained DDP (ABC-DDP), which efficiently computes constrained solutions and identifies active sets without repeated Karush-Kuhn-Tucker inversions.",
  "doi": "10.1109/LRA.2026.3723262",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "S. Nam",
    "id": "2334802484",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Hyunyong Lee",
    "id": "2363594691",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Hansol Kang",
    "id": "2269705155",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Jim Park",
    "id": "2404030007",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Y. Son",
    "id": "2334786884",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "B. Yi",
    "id": "2334770280",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "J. Oh",
    "id": "2335153317",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "H. Choi",
    "id": "6609846",
    "h_index": 46,
    "papers": 371
   }
  ],
  "comment": "Accepted for publication in IEEE Robotics and Automation Letters (RA-L)",
  "topics": [
   "humanoids",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.18552v1",
  "pdf_url": "https://arxiv.org/pdf/2608.18552v1",
  "html_url": "https://arxiv.org/html/2608.18552v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.18507",
  "slug": "an-experimental-study-of-downwash-effects-on-a-continuum-manipulator-i",
  "title": "An Experimental Study of Downwash Effects on a Continuum Manipulator Integrated with a Multirotor UAV",
  "abstract": "Continuum arm aerial manipulation systems leverage soft-manipulator compliance and dexterity for tasks in confined or hazardous environments, but propeller downwash can degrade performance, particularly near walls and the ground. This effect remains uncharacterized for continuum manipulators. This letter experimentally studies downwash-induced kinematic deviations of a tendon-driven continuum manipulator integrated with a multirotor platform. Under still-air conditions, the CM is compared with a constant-curvature (CC) model. Downwash- induced end-effector pose deviations are then quantified relative to the mean still-air experimental baseline at four propeller throttle levels in free space, and at maximum throttle near a wall, and near the ground. Vertical position and yaw show the largest deviations and are amplified by ground effect. A CC-guided Gaussian process regression (GPR) residual model is learned from experimental data that improves forward pose prediction RMSE (position by 89-95%, orientation by 47-79%), and support compensation-oriented, downwash-aware modeling of continuum arm aerial manipulation systems.",
  "published": "2026-08-19",
  "updated": "2026-08-19",
  "year": "2026",
  "authors": [
   "Anuraj Uthayasooriyan",
   "Krishna Manaswi Digumarti",
   "ernando Vanegas",
   "Felipe Gonzalez"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "RA-L",
  "venue_source": "semantic-scholar",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "10.1109/LRA.2026.3709580",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Anuraj Uthayasooriyan",
    "id": "2188178455",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Krishna Manaswi Digumarti",
    "id": "2394981727",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "F. Vanegas",
    "id": "144815206",
    "h_index": 15,
    "papers": 45
   },
   {
    "name": "Felipe Gonzalez",
    "id": "2156826978",
    "h_index": 12,
    "papers": 28
   }
  ],
  "comment": "Accepted to publish in Robotics and Automation Letters",
  "topics": [
   "dexterous-manipulation",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.18507v1",
  "pdf_url": "https://arxiv.org/pdf/2608.18507v1",
  "html_url": "https://arxiv.org/html/2608.18507v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.18488",
  "slug": "designing-social-robots-for-social-cognition-training-with-autistic-ad",
  "title": "Designing Social Robots for Social-Cognition Training with Autistic Adults",
  "abstract": "Social robots have been widely explored as tools for autism intervention, yet this literature has focused predominantly on children and has rarely involved autistic adults as active contributors to design. This creates a mismatch between existing systems and the social-cognitive challenges autistic adults actually face in everyday life, including navigating ambiguous interpersonal contexts, managing conversational timing, and interpreting implied emotional meaning. To address this gap, we conducted an online focus group and co-design session with five autistic adults to explore what a social robot for social-cognition training should do, how it should interact, and under what conditions it would be genuinely useful. The 90-minute session combined open discussion with structured co-design activities on a shared digital whiteboard, and the resulting verbal and visual data were analysed using reflexive thematic analysis. The analysis yielded seven themes that define core design requirements: the robot should function as a scaffold rather than a substitute, prioritise authenticity over comfort, provide personalised and user-controlled feedback, accommodate emotional self-awareness gaps, respect privacy and contextual boundaries, support rehearsal for real-world social situations, and remain configurable in identity, form, and expression. Together, the findings suggest that autistic adults envision the robot not as a companion or live social assistant, but as a private, configurable rehearsal partner designed to support independence over time.",
  "published": "2026-08-19",
  "updated": "2026-08-19",
  "year": "2026",
  "authors": [
   "Yuval Zohar",
   "Mordi Benhamou",
   "Guy Laban"
  ],
  "author_count": 3,
  "categories": [
   "cs.HC",
   "cs.RO"
  ],
  "primary_category": "cs.HC",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yuval Zohar",
    "id": "2458351849",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Mordi Benhamou",
    "id": "2458351809",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Guy Laban",
    "id": "1490528902",
    "h_index": 15,
    "papers": 35
   }
  ],
  "comment": "Accepted for publication at the 35th IEEE International Conference on Robot and Human Interactive Communication (RO-MAN 2026)",
  "topics": [
   "hardware-codesign",
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.18488v1",
  "pdf_url": "https://arxiv.org/pdf/2608.18488v1",
  "html_url": "https://arxiv.org/html/2608.18488v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.18470",
  "slug": "devgru-depth-guided-visual-navigation-using-a-collision-aware-recurren",
  "title": "DevGRU: Depth-guided Visual Navigation using a Collision-aware Recurrent Model",
  "abstract": "Existing visual navigation models often aim to develop foundation models that can generalize robot navigation across diverse platforms. However, many of these models are prone to collisions when deployed in complex indoor environments, particularly in structured layouts and narrow passages. To address this problem, we propose a depth image- and point-goal-conditioned navigation system, DevGRU. The proposed system employs an action predictor (AP) that generates collision-aware future trajectories, enabling effective avoidance of immediate obstacles. In conjunction with a collision predictor, the AP further compensates for errors accumulated in the goal pose estimation and proactively mitigates future deviations. To evaluate our method, we conducted experiments across nine different scenes and three state-of-the-art approaches - ViNT, NoMaD, and NavDP - as well as four additional variants of ViNT and NoMaD. In terms of navigation performance, DevGRU significantly outperforms ViNT and NoMaD by a large margin. In addition, the proposed model has a relatively small number of trainable parameters, resulting in the fastest inference time among the baselines, particularly outperforming NavDP by 7x in model size and 17x in inference time.",
  "published": "2026-08-19",
  "updated": "2026-08-19",
  "year": "2026",
  "authors": [
   "Kyung Min Han",
   "Eunsom Kim",
   "Young J. Kim"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The proposed DevGRU navigation system employs an action predictor that generates collision-aware future trajectories, enabling effective avoidance of immediate obstacles and has a relatively small number of trainable parameters, resulting in the fastest inference time among the baselines.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Kyung Min Han",
    "id": "2337227578",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Eunsom Kim",
    "id": "2458544463",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Young J. Kim",
    "id": "2337239059",
    "h_index": 1,
    "papers": 2
   }
  ],
  "comment": "Accepted for publication in IEEE Robotics and Automation Letters (RA-L), 2026",
  "topics": [
   "navigation",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.18470v1",
  "pdf_url": "https://arxiv.org/pdf/2608.18470v1",
  "html_url": "https://arxiv.org/html/2608.18470v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.18454",
  "slug": "backward-layout-search-for-sequence-constrained-robotic-assembly",
  "title": "Backward Layout Search for Sequence-Constrained Robotic Assembly",
  "abstract": "Robotic assembly layout planning must determine the assembly site and the initial pose of each part while ensuring collision-free execution of a prescribed assembly sequence. This problem is challenging because the obstacle environment changes after each assembly step, and unassembled parts re maining in the workspace may block robot motions. We observe that the feasibility of each assembly step depends only on the initial poses of the current and later-assembled parts. Based on this dependency, we propose Backward Layout Search (BLS), which assigns initial part poses in reverse assembly order. Each expansion performs geometric, kinematic, grasp, and prescribed-motion checks, while collision masks and candidate set filtering remove infeasible initial part pose candidates. Promising partial layouts are retained through beam selection, and complete layouts are validated by full motion planning in forward assembly order. Experiments on five assembly models show that BLS produces collision-free executable layouts and reduces step evaluations and search time compared with a matched forward search.",
  "published": "2026-08-19",
  "updated": "2026-08-19",
  "year": "2026",
  "authors": [
   "Xi Zhang",
   "Jiancong Dai",
   "Hao Chen",
   "Zhengtao Hu",
   "Changcai Yang",
   "Weiwei Wan"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Backward Layout Search (BLS) is proposed, which assigns initial part poses in reverse assembly order and produces collision-free executable layouts and reduces step evaluations and search time compared with a matched forward search.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xi Zhang",
    "id": "2455804442",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Jiancong Dai",
    "id": "2458368790",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Hao Chen",
    "id": "2149050568",
    "h_index": 6,
    "papers": 20
   },
   {
    "name": "Zhengtao Hu",
    "id": "13643521",
    "h_index": 7,
    "papers": 27
   },
   {
    "name": "Changcai Yang",
    "id": "2155848341",
    "h_index": 13,
    "papers": 46
   },
   {
    "name": "Weiwei Wan",
    "id": "1717512",
    "h_index": 27,
    "papers": 297
   }
  ],
  "comment": "Submit to ROBIO2026",
  "topics": [
   "dexterous-manipulation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.18454v1",
  "pdf_url": "https://arxiv.org/pdf/2608.18454v1",
  "html_url": "https://arxiv.org/html/2608.18454v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.18446",
  "slug": "harvestpoint-act-explicit-target-selection-and-harvest-point-condition",
  "title": "HarvestPoint-ACT: Explicit Target Selection and Harvest-Point Conditioning for Robotic Fruit Harvesting under Occlusion",
  "abstract": "End-to-end imitation learning avoids hand-made robot motion for approaching and grasping, but the policy must still decide which fruit to pick and where to close the gripper. Occlusion can make the policy lose the selected fruit during harvesting, and the correct closing point is difficult to infer from pixels alone. This paper presents HarvestPoint-ACT, which makes both decisions explicit in perception and provides them to the policy. An instance segmentation front end with a keypoint branch predicts a mask and a harvest point for each visible fruit, where the harvest point specifies the location to close the gripper. A scheduler ranks detected candidates by occlusion and travel distance and selects one target. After each attempt, it redetects and reranks the candidates because the canopy may have changed. The selected fruit is encoded for an action chunking transformer as an eight-dimensional state, containing the absolute harvest point, the vector from the gripper to that point, a validity flag, and a confidence score. When the selected fruit is temporarily undetected, the system retains the last harvest point estimate in the robot base frame and marks it as stale, and aborts the attempt if the loss persists. On a canopy mock-up, HarvestPoint-ACT achieves a success rate of 88%, and of 75% under heavy occlusion.",
  "published": "2026-08-19",
  "updated": "2026-08-19",
  "year": "2026",
  "authors": [
   "Hanying Hu",
   "Weipeng Li",
   "Yikun Huang",
   "Hao Chen",
   "Zhengtao Hu",
   "Changcai Yang",
   "Weiwei Wan"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper presents HarvestPoint-ACT, which makes both decisions explicit in perception and provides them to the policy, and achieves a success rate of 88%, and of 75% under heavy occlusion.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Han Hu",
    "id": "2456540094",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Weipeng Li",
    "id": "2458546138",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Yikun Huang",
    "id": "2353665794",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Hao Chen",
    "id": "2377789016",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Zhengtao Hu",
    "id": "13643521",
    "h_index": 7,
    "papers": 27
   },
   {
    "name": "Changcai Yang",
    "id": "2155848341",
    "h_index": 13,
    "papers": 46
   },
   {
    "name": "Weiwei Wan",
    "id": "1717512",
    "h_index": 27,
    "papers": 297
   }
  ],
  "comment": "Submit to IEEE ROBIO 2026",
  "topics": [
   "dexterous-manipulation",
   "imitation-diffusion"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.18446v1",
  "pdf_url": "https://arxiv.org/pdf/2608.18446v1",
  "html_url": "https://arxiv.org/html/2608.18446v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.18433",
  "slug": "the-embodiment-gap-in-robot-foundation-models",
  "title": "The Embodiment Gap in Robot Foundation Models",
  "abstract": "Robot foundation models (RFMs), including vision-language-action (VLA) policies, are often discussed through a scaling view: more data, larger models, and broader benchmarks should improve generalization. In robotics, however, a model can generalize while work still remains before it can run on a robot with a particular body. The work required differs across methods and target robots, and those differences affect practical deployment. We call the gap between reusable models, representations, or data and their use in execution on the target robot the embodiment gap. This survey examines what can be reused across robot embodiments and what must still be implemented on a new robot. We place existing methods on a two-axis map that shows the type of shared structure and the stage at which adaptation is needed for execution on the target robot. We then examine recent work through three overlapping research directions: sharing semantics and perception, sharing robot data and interfaces, and learning correspondence across embodiments. We also propose a reporting framework for adaptation work that success rate alone does not reveal. The framework identifies the work that should be checked when comparing cross-embodiment learning and highlights work that remains on a new robot and questions for future study.",
  "published": "2026-08-19",
  "updated": "2026-08-19",
  "year": "2026",
  "authors": [
   "Yukiyasu Domae",
   "Keisuke Shirai",
   "Hanbit Oh",
   "Ryoichi Nakajo",
   "Tomohiro Motoda",
   "Koshi Makihara",
   "Masaki Murooka",
   "Takuma Yagi",
   "Yoshiaki Bando",
   "Ryo Hanai"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This survey examines what can be reused across robot embodiments and what must still be implemented on a new robot, and examines recent work through three overlapping research directions: sharing semantics and perception, sharing robot data and interfaces, and learning correspondence across embodiments.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Y. Domae",
    "id": "2512607",
    "h_index": 13,
    "papers": 128
   },
   {
    "name": "Keisuke Shirai",
    "id": "2355353380",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Hanbit Oh",
    "id": "2367643082",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Ryoichi Nakajo",
    "id": "3411577",
    "h_index": 3,
    "papers": 19
   },
   {
    "name": "Tomohiro Motoda",
    "id": "2328411976",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Koshi Makihara",
    "id": "1491238018",
    "h_index": 2,
    "papers": 22
   },
   {
    "name": "Masaki Murooka",
    "id": "2350756597",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Takuma Yagi",
    "id": "2052544968",
    "h_index": 9,
    "papers": 20
   },
   {
    "name": "Yoshiaki Bando",
    "id": "2458351749",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Ryo Hanai",
    "id": "2243398622",
    "h_index": 3,
    "papers": 17
   }
  ],
  "comment": "32 pages, 4 figures. Published in Transactions on Machine Learning Research (TMLR), August 2026",
  "topics": [
   "vla",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.18433v1",
  "pdf_url": "https://arxiv.org/pdf/2608.18433v1",
  "html_url": "https://arxiv.org/html/2608.18433v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.18401",
  "slug": "multimodal-rapport-estimation-in-real-world-hri",
  "title": "Multimodal Rapport Estimation in Real-World HRI",
  "abstract": "Evaluating interaction quality in real-world HRI is an important challenge. If interaction quality can be estimated reliably, the results can be used to improve dialogue strategies and ultimately enable robots to adapt their behavior autonomously. However, existing automatic evaluation methods have been developed primarily in controlled laboratory settings, and it remains unclear whether they can be directly applied to real-world environments, where users are free to disengage and multi-party participation may arise naturally. In this study, we investigate the automatic estimation of third-party-rated rapport scores using 62 sessions of multimodal recordings collected in a Japanese drugstore. We compare zero-shot LLMs, pretrained text, audio, and visual models, and their prediction-level fusion. The results show that, in real-world HRI, zero-shot LLMs achieve strong performance, while audio and visual models tend to provide complementary information. In particular, Gemini 2.5 Flash performs strongly as a single model, and a fusion model combining Gemini (text) with HuBERT and V-JEPA performs best overall. Further analyses showed that estimation performance varied across interaction-duration and group-size conditions. These findings suggest that rapport estimation in real-world HRI requires evaluation and model design that account for contextual variability beyond that assumed in laboratory settings.",
  "published": "2026-08-19",
  "updated": "2026-08-19",
  "year": "2026",
  "authors": [
   "Akihiro Sakuramoto",
   "Takato Hayashi",
   "Ryo Miyoshi",
   "Yuki Okafuji",
   "Shogo Okada"
  ],
  "author_count": 5,
  "categories": [
   "cs.HC",
   "cs.CL",
   "cs.RO"
  ],
  "primary_category": "cs.HC",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This study investigates the automatic estimation of third-party-rated rapport scores using 62 sessions of multimodal recordings collected in a Japanese drugstore and suggests that rapport estimation in real-world HRI requires evaluation and model design that account for contextual variability beyond that assumed in laboratory settings.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Akihiro Sakuramoto",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Takato Hayashi",
    "id": "2221854050",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Ryo Miyoshi",
    "id": "2373026175",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Yuki Okafuji",
    "id": "2266493740",
    "h_index": 4,
    "papers": 22
   },
   {
    "name": "Shogo Okada",
    "id": "48266041",
    "h_index": 10,
    "papers": 83
   }
  ],
  "comment": "9 pages, 4 figures, 3 tables. Accepted at the 28th ACM International Conference on Multimodal Interaction (ICMI 2026)",
  "topics": [
   "world-models"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.18401v1",
  "pdf_url": "https://arxiv.org/pdf/2608.18401v1",
  "html_url": "https://arxiv.org/html/2608.18401v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.21441",
  "slug": "constructing-predictive-surgical-path-for-ai-based-capsulorhexis-skill",
  "title": "Constructing Predictive Surgical Path for AI-based Capsulorhexis Skill Transfer",
  "abstract": "Automated training of surgeons is one of the most crucial factors that significantly minimize surgical training risks and expenses. With recent advances in artificial intelligence (AI) knowledge and available data from various surgeries, AI's involvement in surgical training is becoming very promising. It is recommended that at the early stages of AI development, it interferes in the surgery as a third agent alongside the trainer. As trust in AI increases, this process will lead to an AI agent acting as a trainer in the future. The first phase in which AI can intervene in the training process is to suggest an improved surgical path to the trainer. A platform must be constructed in the first step, to accomplish this task and to enhance the movement path of trainee surgeons. This paper introduces this platform along with an annotated capsulorhexis surgery dataset called the ARAS-Farabi dataset. In this research, a deep convolutional neural network is pre-trained with JIGSAWS and ARAS-Farabi surgical datasets that can extract surgical skill characteristics from surgery tool tip motion data. The proposed platform develops a reference model from the feature space of an expert surgeon's movement trajectory and proposes an improved path to enhance the skill of a novice surgeon. An optimization with two loss functions is utilized to create a path that raises the skill level of the novice surgeon's path while simultaneously predicting and preserving his/her intent. The results of this study reveal that, with the assistance of an AI agent, the trainee surgeon's movement path can be enhanced by at least 20 percent while maintaining his intentional objective. In addition to the recommended deep network, various tangible indicators have also been developed in this research to verify the level of trainee improvement.",
  "published": "2026-08-18",
  "updated": "2026-08-18",
  "year": "2026",
  "authors": [
   "Mohammad Javad Ahmadi",
   "Hamid D. Taghirad"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A deep convolutional neural network is pre-trained with JIGSAWS and ARAS-Farabi surgical datasets that can extract surgical skill characteristics from surgery tool tip motion data and creates an improved path to enhance the skill of a novice surgeon.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "M. J. Ahmadi",
    "id": "2058961780",
    "h_index": 3,
    "papers": 16
   },
   {
    "name": "H. D. Taghirad",
    "id": "2348557032",
    "h_index": 3,
    "papers": 20
   }
  ],
  "comment": "",
  "topics": [
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.21441v1",
  "pdf_url": "https://arxiv.org/pdf/2608.21441v1",
  "html_url": "https://arxiv.org/html/2608.21441v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.21440",
  "slug": "geo-vla-geometry-aware-vision-language-action-planning-via-internaliza",
  "title": "Geo-VLA: Geometry-Aware Vision-Language-Action Planning via Internalization of Map Semantics",
  "abstract": "Vision-language-action (VLA) models have advanced end-to-end autonomous driving by leveraging foundation models for semantic reasoning and long-tail generalization. However, their planning performance remains limited in complex driving environments because image-only representations inadequately capture planning-relevant road geometry and topology. In this paper, we propose Geo-VLA, a plug-and-play framework that enhances VLA models by learning geometry-aware visual representations. During training, Geo-VLA internalizes geometric map semantics to strengthen road-structure representations, while requiring no HD maps or additional lane information during inference. To support this approach, we introduce Geo-QA, a geometry-focused question-answering dataset that injects road geometry into vision-language representations through contrastive learning and instruction tuning. Experiments on NAVSIM v1 demonstrate that Geo-VLA consistently improves VLA planners with distinct action-generation architectures, achieving 92.1 PDMS and establishing a new state-of-the-art among single-camera VLA planners.",
  "published": "2026-08-18",
  "updated": "2026-08-18",
  "year": "2026",
  "authors": [
   "Ran Chen",
   "Jiaxing Ren",
   "Zhikun Zhang",
   "Yunhao Hou",
   "Junbao Zhuo",
   "Bochao Zou"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Geo-VLA is proposed, a plug-and-play framework that enhances VLA models by learning geometry-aware visual representations and introduces Geo-QA, a geometry-focused question-answering dataset that injects road geometry into vision-language representations through contrastive learning and instruction tuning.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ran Chen",
    "id": "2108368213",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Jiaxing Ren",
    "id": "2325463706",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Zhikun Zhang",
    "id": "2452138448",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yunhao Hou",
    "id": "2459159001",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Junbao Zhuo",
    "id": "26973936",
    "h_index": 12,
    "papers": 35
   },
   {
    "name": "Bochao Zou",
    "id": "1486438257",
    "h_index": 13,
    "papers": 49
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "navigation",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.21440v1",
  "pdf_url": "https://arxiv.org/pdf/2608.21440v1",
  "html_url": "https://arxiv.org/html/2608.21440v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.18364",
  "slug": "a-task-agnostic-control-strategy-for-dynamic-assistance-with-pneumatic",
  "title": "A Task-Agnostic Control Strategy for Dynamic Assistance with Pneumatically Actuated Soft Exosuits",
  "abstract": "Pneumatic artificial muscles have provided new opportunities to develop upper-extremity soft exosuits for reha- bilitation, augmentation, and assisted daily living. However, the complex dynamics and limited bandwidth of these actuators has made providing responsive assistance based on user intention a longstanding challenge. In this work, we present an inverse-plant control strategy for pneumatically actuated soft exosuits that only relies on kinematic sensing for task-agnostic and dynamic assistance during daily living. We model the human-robot system using a Hammerstein dynamic model, consisting of a Preisach hysteresis model and a linear time-invariant filter, to capture the static and dynamic behavior of the system. We personalize our model to each user using 140 s of data and approximate an inverse to integrate into our control loop. When evaluated on a test rig that emulated a soft assistive exosuit for the wrist, our controller reduced the interaction torque by up to 73% and the activation of key flexor and extensor muscles by up to 47% relative to the condition with no assistance for speeds ranging from 8\u00b0/s to 120\u00b0/s. Overall, this work presents a control strategy that can provide task-agnostic, dynamic assistance with pneumatically actuated soft exosuits without the need for physiological or force sensors to interpret user intention.",
  "published": "2026-08-18",
  "updated": "2026-08-18",
  "year": "2026",
  "authors": [
   "Anoush Sepehri",
   "Zachary Huang",
   "Raymond de Callafon",
   "Michael T. Tolley",
   "Tania K. Morimoto"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.HC"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Anoush Sepehri",
    "id": "2301244975",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Zachary Huang",
    "id": "2276576014",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Raymond de Callafon",
    "id": "2458352100",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "M. Tolley",
    "id": "48724645",
    "h_index": 41,
    "papers": 144
   },
   {
    "name": "Tania K. Morimoto",
    "id": "2300488454",
    "h_index": 6,
    "papers": 27
   }
  ],
  "comment": "",
  "topics": [
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.18364v1",
  "pdf_url": "https://arxiv.org/pdf/2608.18364v1",
  "html_url": "https://arxiv.org/html/2608.18364v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.18317",
  "slug": "reproducible-multimodal-affordance-prediction",
  "title": "Reproducible Multimodal Affordance Prediction",
  "abstract": "Affordance prediction is the identification of potential actions an agent can perform on a target object from multimodal inputs. Affordance prediction methods are difficult to evaluate and compare due to heterogeneous problem formulations, inconsistent dataset annotations, incomplete reporting of experimental protocols, and limited information about deployment conditions. These limitations challenge fair benchmarking and performance comparison. To promote transparency, we propose the Affordance Sheet, a documentation detailing task formulation with its input modalities, model architectures and training information, datasets, and experimental protocols. Affordance Sheets enable reproducible benchmarking and reliable evaluation of affordance models for real-world scenarios, including generalisation to novel conditions and human safety.",
  "published": "2026-08-18",
  "updated": "2026-08-18",
  "year": "2026",
  "authors": [
   "Tommaso Apicella",
   "Alessio Xompero",
   "Andrea Cavallaro"
  ],
  "author_count": 3,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "ECCV 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The Affordance Sheet is proposed, a documentation detailing task formulation with its input modalities, model architectures and training information, datasets, and experimental protocols, which enable reproducible benchmarking and reliable evaluation of affordance models for real-world scenarios, including generalisation to novel conditions and human safety.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "T. Apicella",
    "id": "78447413",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "A. Xompero",
    "id": "3188411",
    "h_index": 8,
    "papers": 29
   },
   {
    "name": "Andrea Cavallaro",
    "id": "2064088172",
    "h_index": 4,
    "papers": 12
   }
  ],
  "comment": "Paper accepted to Workshop on Human-Centered Multimodal Intelligence in the Wild (HCMIW) in European Conference on Computer Vision (ECCV) 2026; 18 pages, 3 figures, 7 tables. Project webpage at https://apicis.github.io/aff-sheet",
  "topics": [
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.18317v1",
  "pdf_url": "https://arxiv.org/pdf/2608.18317v1",
  "html_url": "https://arxiv.org/html/2608.18317v1",
  "code_url": "https://apicis.github.io/aff-sheet",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.5
 },
 {
  "id": "2608.18292",
  "slug": "guidefetch-a-task-coordination-framework-for-concurrent-navigation-and",
  "title": "GuideFetch: A Task Coordination Framework for Concurrent Navigation and Object Retrieval in Assistive Robot Dogs",
  "abstract": "Consider one robot guide dog escorting a blind user to a seat while a second retrieves and delivers an object. We introduce \\textsc{GuideFetch}, a framework for coordinating this concurrent guide-and-fetch mission with heterogeneous robots. A large language model (LLM) instantiates a schedule-conditioned four-action schema; deterministic normalization and validation enforce registered targets, robot capabilities, and the selected schedule, while robot and object states govern execution and completion. We record 360 simulator runs over 90 scene--seed combinations under scripted and online plan-provenance conditions. All 180 online responses validate on the first request and match their scripted references, so the plan-provenance comparison tests normalized-plan agreement rather than a distinct execution factor. A simulator-free mutation test accepts two valid controls and rejects all 32 rule-violating variants. Across 90 scene--seed cases per schedule, sequential and parallel execution achieve $72/90$ and $71/90$ operational successes. Among 56 common successes, the implemented role-reassigned parallel protocol reduces mean makespan by 41.3\\%. This system-level gain combines role assignment, action overlap, and scene geometry; state checks distinguish plan validity from verified mission completion.",
  "published": "2026-08-18",
  "updated": "2026-08-23",
  "year": "2026",
  "authors": [
   "Qian Yin",
   "Ruiping Liu",
   "Kunyu Peng",
   "Jianxiang Man",
   "Isik Baran Sandan",
   "Junwei Zheng",
   "Yufan Chen",
   "Di Wen",
   "Kailun Yang",
   "Rainer Stiefelhagen"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "ECCV 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work introduces GuideFetch, a coordination framework for concurrent navigation and object retrieval by a heterogeneous guider and fetcher team and controls role specialization and action overlap shorten completed missions, while state checks distinguish plan validity from verified mission completion.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Qiande Yin",
    "id": "2348513032",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Ruiping Liu",
    "id": "2273522374",
    "h_index": 7,
    "papers": 52
   },
   {
    "name": "Kunyu Peng",
    "id": "91549683",
    "h_index": 21,
    "papers": 108
   },
   {
    "name": "Jianxiang Man",
    "id": "2458353474",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Isik Baran Sandan",
    "id": "2365264057",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Junwei Zheng",
    "id": "2210176414",
    "h_index": 11,
    "papers": 58
   },
   {
    "name": "Yufan Chen",
    "id": "2243360598",
    "h_index": 8,
    "papers": 48
   },
   {
    "name": "Di Wen",
    "id": "2288587946",
    "h_index": 6,
    "papers": 39
   },
   {
    "name": "Kailun Yang",
    "id": "8689702",
    "h_index": 40,
    "papers": 284
   },
   {
    "name": "Rainer Stiefelhagen",
    "id": "2320597200",
    "h_index": 6,
    "papers": 42
   }
  ],
  "comment": "Accepted to the 1st Workshop on Multimodal Digital Agents (MDA) at ECCV 2026",
  "topics": [
   "sim2real",
   "navigation",
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.18292v2",
  "pdf_url": "https://arxiv.org/pdf/2608.18292v2",
  "html_url": "https://arxiv.org/html/2608.18292v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.18270",
  "slug": "transferable-tool-tissue-contact-detection-from-stereo-depth-in-robot",
  "title": "Transferable Tool-Tissue Contact Detection from Stereo Depth in Robot-Assisted Surgery",
  "abstract": "Reliable tool--tissue contact detection can support interaction-aware control and downstream force estimation in robot-assisted surgery. Most existing methods learn a contact classifier from RGB appearance, which is hard to generalize. In this work, we use the depth image generated from a stereo pair to give more information about tool--tissue contact. For each depth frame, we localize a spatially supported minimum-distance patch around the tool boundary and reduce it to a single scalar, $-\\log_{10}|d|$; this signal rises and falls in step with ground-truth contact. We formalize this observation with a fully supervised two-state hidden Markov model. We fit this model as a six-fold leave-one-session-out (LOSO) ensemble on six palpation sessions against a single silicone cup-like phantom, with the decision threshold selected from the pooled out-of-fold predictions. It is evaluated on four held-out sessions of three categories: 1. same task on same phantom; 2. same task on different phantom; 3. different task on different phantom. This model reaches held-out macro F1 $0.927$ and AUPRC $0.980$. We further compare against a reproduction of an RGB-based contact classifier from prior work. This RGB-based model achieves high performance on the first category (F1 $0.965$), but substantially lower performance on the other two, resulting in macro F1 $0.320$ across all four sessions. These results indicate that the tool--tissue distance is a strong, transferable cue for contact detection in robot-assisted surgery.",
  "published": "2026-08-18",
  "updated": "2026-08-18",
  "year": "2026",
  "authors": [
   "Mingyeung Wu",
   "Zhonghao Zhang",
   "Hao Yang",
   "Alan Kuntz",
   "Jie Ying Wu"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The results indicate that the tool--tissue distance is a strong, transferable cue for contact detection in robot-assisted surgery.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Mingyeung Wu",
    "id": "2458557491",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Zhonghao Zhang",
    "id": "2456390124",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Hao Yang",
    "id": "2330419650",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Alan Kuntz",
    "id": "2301247698",
    "h_index": 2,
    "papers": 13
   },
   {
    "name": "Jie Ying Wu",
    "id": "2323508279",
    "h_index": 4,
    "papers": 10
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.18270v1",
  "pdf_url": "https://arxiv.org/pdf/2608.18270v1",
  "html_url": "https://arxiv.org/html/2608.18270v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.18258",
  "slug": "veragmil-virtual-environment-for-scooping-granular-foods-with-imitatio",
  "title": "VERAGMIL: Virtual Environment for Scooping Granular Foods with Imitation Learning Models",
  "abstract": "Robot-Assisted Feeding (RAF) systems are essential for assisting individuals with disabilities or motor impairments in eating tasks. Manipulating granular food items, such as rice and beans, poses significant challenges due to their dynamic physical properties. Learning from human demonstrations offers a promising solution, but acquiring high-quality demonstrations is complex. To address this, we present VERAGMIL, a framework that combines a high-fidelity simulator with an intuitive Virtual Reality (VR) interface for recording demonstrations and supporting different imitation learning methods. VERAGMIL provides a realistic environment for training RAF systems to handle granular materials, including robots, sensors, and various food items with distinct physical characteristics. We evaluate VERAGMIL by training three imitation learning models, BC, BC-RNN, and BCQ, on granular scooping and transporting tasks using both VR interface and 3D space mouse demonstrations, comparing them with a human-expert baseline. The models are assessed on success rate, spillage, generalization to unseen food items, and task completion time. Results show that VR-based demonstrations significantly outperform 3D space mouse data, with BCQ achieving the best overall performance, particularly in reducing spillage and approaching human performance. These findings underscore the effectiveness of our framework for training RAF systems in granular material handling. The code for our framework is publicly available at: https://github.com/AmanuelErgogo/VERAGMIL.git.",
  "published": "2026-08-18",
  "updated": "2026-08-18",
  "year": "2026",
  "authors": [
   "Amanuel Ergogo",
   "Diego Dall'Alba",
   "Przemyslaw Korzeniowski"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS",
  "venue_source": "semantic-scholar",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "VERAGMIL provides a realistic environment for training RAF systems to handle granular materials, including robots, sensors, and various food items with distinct physical characteristics, and demonstrates the effectiveness of the framework for training RAF systems in granular material handling.",
  "doi": "10.1109/IROS60139.2025.11247362",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Amanuel Ergogo",
    "id": "2303679710",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "D. Dall'Alba",
    "id": "2277222707",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "P. Korzeniowski",
    "id": "2254320158",
    "h_index": 6,
    "papers": 23
   }
  ],
  "comment": "",
  "topics": [
   "egocentric-data",
   "sim2real",
   "imitation-diffusion"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.18258v1",
  "pdf_url": "https://arxiv.org/pdf/2608.18258v1",
  "html_url": "https://arxiv.org/html/2608.18258v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.18254",
  "slug": "gapl-grounded-action-effect-policy-learning-for-llm-based-trajectory-p",
  "title": "GAPL: Grounded Action-effect Policy Learning for LLM-Based Trajectory Planning",
  "abstract": "Trajectory planning for autonomous driving requires both high-level reasoning and precise low-level control. Large Language Models (LLMs) offer semantic-rich planning capabilities, however, their application is limited by hallucinated reasoning, poor grounding in environment dynamics, and limited numerical precision in control. We propose GAPL (Grounded Action-effect Policy Learning), a unified framework that integrates LLM-based effect estimation, simulation-based effect grounding, and policy optimization into a closed-loop system. GAPL consists of three modules: (1) an LLM-based Effect Evaluator for structured multi-dimensional action-effect estimation; (2) a Simulation-based Effect Grounder that predicts dynamics-consistent effects from simulator rollouts; and (3) an Effect-Aware Decision Maker that grounds LLM effect estimates against simulation via a distiller to guide Proximal Policy Optimization (PPO)-based policy learning. Experiments on four Highway-env scenarios demonstrate that GAPL consistently outperforms baselines, achieving average reductions of {0.76, 0.86, 2.00} in collision rate, average displacement error (ADE), and final displacement error (FDE), and an average reward gain of 1.44.",
  "published": "2026-08-18",
  "updated": "2026-08-18",
  "year": "2026",
  "authors": [
   "Zhihong Cui",
   "Hengyu Liu",
   "Zhangkai Wu",
   "Yushuai Li",
   "Tianyi Li",
   "Peiyuan Guan",
   "Amir Taherkordi",
   "Tor Skeie"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "GAPL (Grounded Action-effect Policy Learning), a unified framework that integrates LLM-based effect estimation, simulation-based effect grounding, and policy optimization into a closed-loop system, is proposed.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhihong Cui",
    "id": "2089582898",
    "h_index": 6,
    "papers": 14
   },
   {
    "name": "Hengyu Liu",
    "id": "2353364375",
    "h_index": 4,
    "papers": 18
   },
   {
    "name": "Zhangkai Wu",
    "id": "2228191197",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Yushuai Li",
    "id": "2311748417",
    "h_index": 3,
    "papers": 23
   },
   {
    "name": "Tianyi Li",
    "id": "2311781539",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Peiyuan Guan",
    "id": "2266652924",
    "h_index": 5,
    "papers": 24
   },
   {
    "name": "Amirhosein Taherkordi",
    "id": "2848485",
    "h_index": 29,
    "papers": 141
   },
   {
    "name": "Torkel Skeie",
    "id": "2341325984",
    "h_index": 1,
    "papers": 3
   }
  ],
  "comment": "11 pages, 5 figures, 6 tables",
  "topics": [
   "sim2real",
   "rl-control",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.18254v1",
  "pdf_url": "https://arxiv.org/pdf/2608.18254v1",
  "html_url": "https://arxiv.org/html/2608.18254v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.18240",
  "slug": "zero-shot-transfer-of-force-map-estimation-across-gelsight-mini-sensor",
  "title": "Zero-Shot Transfer of Force Map Estimation Across GelSight Mini Sensors",
  "abstract": "Despite the rapid industrialization of the touch sensor manufacturing process, most of these sensors are still handmade in research laboratories. This complicates standardizing their performance, requiring the repetition of data collection and training models for each unit produced. To address this problem, this paper presents a method that can generalize the estimation of 3D force maps across different GelSight Mini sensor units, regardless of the sensor version. Specifically, the method consists of two stages: a domain adaptation stage, in which the input tactile image is reconstructed as a general tactile image using a UniT-based model; and a stage for estimating 3D force maps employing a U-Net network. Our proposal achieves promising results in both steps, such as an SSIM of 0.9338 +- 0.0358 in the image reconstruction phase and an MAE_F of 1.1294 +- 1.5934(N) in the force estimation phase.",
  "published": "2026-08-18",
  "updated": "2026-08-18",
  "year": "2026",
  "authors": [
   "Julio Casta\u00f1o Amoros",
   "Pablo Gil"
  ],
  "author_count": 2,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "IEEE Sensors Letters",
  "venue_source": "semantic-scholar",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper presents a method that can generalize the estimation of 3D force maps across different GelSight Mini sensor units, regardless of the sensor version, and achieves promising results in both steps.",
  "doi": "10.1109/lsens.2026.3725966",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Julio Casta\u00f1o Amor\u00f3s",
    "id": "2166803148",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Pablo Gil",
    "id": "150259842",
    "h_index": 6,
    "papers": 17
   }
  ],
  "comment": "Accepted for publication in IEEE Sensors Letter",
  "topics": [
   "tactile",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.18240v1",
  "pdf_url": "https://arxiv.org/pdf/2608.18240v1",
  "html_url": "https://arxiv.org/html/2608.18240v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.18234",
  "slug": "gigabrain-wbc-0-5-a-behavior-world-model-for-robust-whole-body-control",
  "title": "GigaBrain-WBC-0.5: A Behavior World Model for Robust Whole-Body Control with Environment Interaction",
  "abstract": "Whole-body motion tracking policies turn a humanoid into a robust control interface: the teleoperator---or an upstream model---only supplies a coarse movement intent, while the low-level policy keeps the robot balanced and physically feasible. Existing trackers deliver this interface only on flat ground: trained in empty scenes, they never learn how contact with terrain and objects reshapes their dynamics, and they attempt to teach the policy to balance under any command by continually enlarging the reference-motion corpus, which stops working once feasible behaviors become environment-dependent. We present GigaBrain-WBC-0.5, the first Behavior World Model (BWM) for humanoid whole-body control. Rather than a purely reactive tracker, we train a causal Transformer to jointly predict its next action, next state, and the distribution over its next latent behavior command, so the network that acts also models how the environment shapes what it can do next. An automatic terrain-annotation pipeline recovers full 3D contact geometry from retargeted motion, enabling terrain annotation at the scale of existing motion datasets. The predicted distribution is reused at deployment to detect implausible commands online and retract them onto learned behaviors, so the robot attempts tasks in a \"best-effort\" manner. The result is a unified policy that takes real-time command, interacts with environment, and stays robust to implausible commands, falls, and disturbances. GigaBrain-WBC-0.5 achieves the highest success rate across all four regimes among three large-scale tracker baselines: 81.3% on terrain interaction (4.3x the strongest baseline), 83.1% under implausible commands, and 99.3% fall recovery (16.8x the strongest baseline). Hardware trials show robust interaction under missing supports and disturbances; the Unitree G1 checkpoint transfers to the Maker L01 robot with simple fine-tuning.",
  "published": "2026-08-18",
  "updated": "2026-08-23",
  "year": "2026",
  "authors": [
   "Ziyang Cheng",
   "Tianshu Tang",
   "Jinxin Lan",
   "Xinze Chen",
   "Yuhan Gong",
   "Zhichao Liu",
   "Changzhong Wu",
   "Yahao Mao",
   "Zongyan Deng",
   "Mingxuan Ma",
   "Huasen Xi",
   "Yilong Liu",
   "Yutong Wu",
   "Xiaofeng Wang",
   "Yang Wang",
   "Yun Ye",
   "Guan Huang",
   "Xiaojie Jin",
   "Zheng Zhu",
   "Jiwen Lu"
  ],
  "author_count": 20,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "GigaBrain-WBC-0.5, the first Behavior World Model for humanoid whole-body control, is presented, which trains a causal Transformer to jointly predict its next action, next state, and the distribution over its next latent behavior command, so the network that acts also models how the environment shapes what it can do next.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ziyang Cheng",
    "id": "2148838805",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Tianshu Tang",
    "id": "1455849848",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Jinxi Lan",
    "id": "2237349143",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Xinze Chen",
    "id": "1391211885",
    "h_index": 17,
    "papers": 34
   },
   {
    "name": "Yuhan Gong",
    "id": "2384076953",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Zhichao Liu",
    "id": "2387126288",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Changzhong Wu",
    "id": "2384355847",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yahao Mao",
    "id": "2458380229",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Zongyan Deng",
    "id": "2458368853",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Mingxuan Ma",
    "id": "2290129651",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Huasen Xi",
    "id": "2397759685",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yilong Liu",
    "id": "2458126279",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Yutong Wu",
    "id": "2456833714",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Xiaofeng Wang",
    "id": "2349399535",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Yang Wang",
    "id": "2373745208",
    "h_index": 6,
    "papers": 14
   },
   {
    "name": "Yun Ye",
    "id": "2384823293",
    "h_index": 6,
    "papers": 16
   },
   {
    "name": "Guan Huang",
    "id": "2256954306",
    "h_index": 16,
    "papers": 42
   },
   {
    "name": "Xiaojie Jin",
    "id": "2303433931",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Zheng Zhu",
    "id": "2109516240",
    "h_index": 17,
    "papers": 50
   },
   {
    "name": "Jiwen Lu",
    "id": "2243332262",
    "h_index": 6,
    "papers": 16
   }
  ],
  "comment": "20 pages, 8 figures, 4 tables. Technical report. Project page: https://shepherd1226.github.io/gigabrain-wbc-0.5/",
  "topics": [
   "world-models",
   "humanoids",
   "data-teleop"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2608.18234v2",
  "pdf_url": "https://arxiv.org/pdf/2608.18234v2",
  "html_url": "https://arxiv.org/html/2608.18234v2",
  "code_url": "https://shepherd1226.github.io/gigabrain-wbc-0.5/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.18227",
  "slug": "revisiting-the-push-t-robot-manipulation-task-with-agentic-robotics",
  "title": "Revisiting the \"Push-T\" Robot Manipulation Task with Agentic Robotics",
  "abstract": "Push-T is an iconic benchmark for learning manipulation policies from human demonstrations. The robot must use a single point of contact to push a T-shaped block into a target pose. In this short paper, we revisit the Push-T task in the context of emerging advances in Agentic Robotics where an LLM coding agent -- Claude Code with Fable 5 -- is prompted to create an algorithmic solution that does not require any demonstration data. We study how effective the agentic coding loop can solve the Push-T task, and compare the resulting code as policy with the visuomotor imitation learning policy. Results suggest that the agent found the 2D gym simulation online, and used sim experiments to learn push mechanics, iteratively optimizing to achieve 100% success rate using 46% fewer steps than the best diffusion policy trained with 200 human demonstrations. The coding agent also solve extensions from T to the full alphabet (Push-A to Push-Z) using a self generated curriculum and generated simulation code for the Franka and UR5 robot arms in 3D cross-embodiment simulations with visual feedback. Videos, policies and details will be posted online.",
  "published": "2026-08-18",
  "updated": "2026-08-18",
  "year": "2026",
  "authors": [
   "Shuangyu Xie",
   "Kaiyuan Chen",
   "Ken Goldberg"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This short paper revisits the Push-T task in the context of emerging advances in Agentic Robotics where an LLM coding agent -- Claude Code with Fable 5 -- is prompted to create an algorithmic solution that does not require any demonstration data.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Shuangyu Xie",
    "id": "2362315657",
    "h_index": 3,
    "papers": 14
   },
   {
    "name": "Kai-Peng Chen",
    "id": "9270658",
    "h_index": 7,
    "papers": 18
   },
   {
    "name": "Ken Goldberg",
    "id": "2321969550",
    "h_index": 5,
    "papers": 14
   }
  ],
  "comment": "",
  "topics": [
   "egocentric-data",
   "imitation-diffusion",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.18227v1",
  "pdf_url": "https://arxiv.org/pdf/2608.18227v1",
  "html_url": "https://arxiv.org/html/2608.18227v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.18178",
  "slug": "trust-as-a-field-a-macroscopic-representation-for-vehicular-networks",
  "title": "Trust as a Field: A Macroscopic Representation for Vehicular Networks",
  "abstract": "Trust assessment is a fundamental component of cooperative and connected vehicle systems. However, existing approaches operate primarily at the level of individual vehicles, making it difficult to reason about trust evolution across road segments. In this paper, we propose a spatio-temporal trust-field framework that aggregates microscopic vehicle-level trust into a continuous representation over space and time. The trust field is formally defined on road segments. We conducted simulation-based experiments using synthetic trajectories generated under controlled conditions, enabling analysis of trust-field behavior in simple road scenarios. Beyond theoretical modeling, we study an implication of the trust-field concept: reconstructing the full trust field from sparse roadside-unit (RSU) measurements. We compare (i) a coordinate-based deep learning baseline that learns a generic trust field from sparse samples and (ii) a field-informed deep learning method that treats trust as a latent quantity carried by vehicles and enforces measurement consistency through the aggregation mechanism. The field-informed approach more accurately recovers trajectory-aligned low-trust patterns and yields improved reconstruction error.",
  "published": "2026-08-18",
  "updated": "2026-08-18",
  "year": "2026",
  "authors": [
   "Md Mahmudul Islam",
   "Shaurya Agarwal"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO",
   "cs.LG",
   "math.DS"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper proposes a spatio-temporal trust-field framework that aggregates microscopic vehicle-level trust into a continuous representation over space and time, and compares a coordinate-based and field-informed deep learning method that treats trust as a latent quantity carried by vehicles and enforces measurement consistency through the aggregation mechanism.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Md Mahmudul Islam",
    "id": "2312308952",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "S. Agarwal",
    "id": "2118282975",
    "h_index": 5,
    "papers": 22
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.18178v1",
  "pdf_url": "https://arxiv.org/pdf/2608.18178v1",
  "html_url": "https://arxiv.org/html/2608.18178v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.18077",
  "slug": "hydra-0-action-flow-for-generalist-world-modeling-and-control",
  "title": "Hydra-0: Action Flow for Generalist World Modeling and Control",
  "abstract": "We introduce Hydra-0, a generalist world model conditioned on action flow, which represents robot actions as pixel motion. This shared visual interface enables generalist world modeling and control by learning action consequences across embodiments, tasks, environments, and video-generation backbones. Our best configuration achieves 90.4% lower robot-motion error and 60.2% lower object-motion error than our action-conditioned baseline, while supporting zero-shot composition and data-efficient adaptation. On the RoboLab benchmark, Hydra-0 achieves a Pearson correlation of r=0.96 between replayed and reference success rates. Finally, we uncover an emergent inverse mode of this interface: a world action model that predicts compatible robot motion from desired object flow transferred from a human demonstration. A trained action head maps the resulting latent features to executable actions without requiring task-specific expert robot demonstrations. Together, these results demonstrate the potential of action flow as a shared control interface connecting heterogeneous training data, open-loop policy evaluation, and robot control.",
  "published": "2026-08-18",
  "updated": "2026-08-18",
  "year": "2026",
  "authors": [
   "Hongyu Li",
   "Bowen Wen",
   "Xinghao Zhu",
   "Yixuan Wang",
   "Yilun Du",
   "Yunzhu Li",
   "George Konidaris",
   "Stan Birchfield",
   "Soha Pouya",
   "Chenran Li",
   "Yan Chang"
  ],
  "author_count": 11,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hongyu Li",
    "id": "2155109477",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Bowen Wen",
    "id": "2261740421",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Xinghao Zhu",
    "id": "2333440855",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Yixuan Wang",
    "id": "2108734730",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "Yilun Du",
    "id": "2258799458",
    "h_index": 12,
    "papers": 17
   },
   {
    "name": "Yunzhu Li",
    "id": "2294926592",
    "h_index": 6,
    "papers": 6
   },
   {
    "name": "G. Konidaris",
    "id": "1765407",
    "h_index": 49,
    "papers": 268
   },
   {
    "name": "Stanley T. Birchfield",
    "id": "2257232566",
    "h_index": 9,
    "papers": 24
   },
   {
    "name": "Soha Pouya",
    "id": "2653368",
    "h_index": 14,
    "papers": 38
   },
   {
    "name": "Chenran Li",
    "id": "2242186839",
    "h_index": 14,
    "papers": 69
   },
   {
    "name": "Yan Chang",
    "id": "2322368474",
    "h_index": 5,
    "papers": 9
   }
  ],
  "comment": "Project page: https://nvidia-isaac.github.io/video_to_data/hydra-0/",
  "topics": [
   "world-models",
   "egocentric-data"
  ],
  "orgs": [
   "NVIDIA"
  ],
  "abs_url": "https://arxiv.org/abs/2608.18077v1",
  "pdf_url": "https://arxiv.org/pdf/2608.18077v1",
  "html_url": "https://arxiv.org/html/2608.18077v1",
  "code_url": "https://nvidia-isaac.github.io/video_to_data/hydra-0/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.17962",
  "slug": "prism-precision-and-contact-rich-real-world-industrial-skill-dataset-w",
  "title": "PRISM: Precision and contact-rich Real-world Industrial Skill dataset with Multimodal sensing",
  "abstract": "Recent progress in robotic learning has been fueled by large-scale datasets collected in everyday environments. However, most existing datasets emphasize short-horizon, low-contact tasks such as pick-and-place, and therefore do not capture the precision control, force/torque or tactile regulation, and multimodal feedback required for industrial assembly. To address this gap, we introduce PRISM, a large-scale multimodal dataset for contact-rich industrial operations. The dataset spans more than 25 manipulation tasks (e.g., electronic components plug/unplug, conveyor-based sorting) and covers diverse mechanical constraints. PRISM includes more than 5,000 trajectories totaling 45 hours of teleoperated demonstrations, recorded using synchronized multi-view RGB-D, force/torque, tactile, and robot-state measurements. In contrast to datasets collected in household or laboratory settings, PRISM provides a realistic benchmark for multimodal perception and control under high-precision industrial constraints, and serves as a foundation for contact-rich, generalizable manipulation in real-world manufacturing environments. The dataset is open-sourced at: https://tengbo-yu.github.io/PRISM/",
  "published": "2026-08-18",
  "updated": "2026-08-18",
  "year": "2026",
  "authors": [
   "Tengbo Yu",
   "Jiahao Wu",
   "Hanning Wang",
   "Rui Chen",
   "Chuanhou Liu",
   "Chuang Sun",
   "Hangxin Liu"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "In contrast to datasets collected in household or laboratory settings, PRISM provides a realistic benchmark for multimodal perception and control under high-precision industrial constraints, and serves as a foundation for contact-rich, generalizable manipulation in real-world manufacturing environments.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Tengbo Yu",
    "id": "2334695176",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Jiahao Wu",
    "id": "2351654383",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Hanning Wang",
    "id": "2458624737",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Rui Chen",
    "id": "2332011508",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Chuanhou Liu",
    "id": "2412601571",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Chuang Sun",
    "id": "2458325027",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Hangxin Liu",
    "id": "2290007716",
    "h_index": 5,
    "papers": 22
   }
  ],
  "comment": "",
  "topics": [
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.17962v1",
  "pdf_url": "https://arxiv.org/pdf/2608.17962v1",
  "html_url": "https://arxiv.org/html/2608.17962v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.17928",
  "slug": "a-theoretical-framework-for-parallel-lifelong-mapf-using-group-decentr",
  "title": "A Theoretical Framework for Parallel Lifelong MAPF Using Group Decentralized Planning",
  "abstract": "In the Lifelong Multi-Agent Path Finding (L-MAPF) problem, agents must repeatedly move from one destination to another while avoiding obstacles and inter-agent collisions. Widely regarded as one of the highest-performing solutions to this problem is the Rolling-Horizon Collision Resolution (RHCR) framework. However, commensurate with its quality solutions, it incurs a computational cost that limits its applicability to even modest agent counts. In this paper, leveraging theoretical methods from the Locally Interdependent Multi-Agent MDP literature, we first theoretically prove the near-optimality of RHCR in a discounted MDP formulation of the L-MAPF problem. Then, we leverage these results to naturally motivate an extended framework called Group Decentralized RHCR (GD-RHCR) which incorporates a group decentralized structure that partitions agents based on a transitive communication scheme and plans for each partition of agents in parallel. We show that both RHCR and GD-RHCR achieve similar exponentially close to optimal guarantees, establishing a theoretical duality between the time based restrictions performed by vanilla RHCR and the additional space based partitioning performed by GD-RHCR. Lastly, we show that across varying maps, GD-RHCR is able to attain high throughput that scales into higher agent counts while maintaining a significantly lower per plan cost.",
  "published": "2026-08-18",
  "updated": "2026-08-18",
  "year": "2026",
  "authors": [
   "Alex DeWeese",
   "Jiaoyang Li",
   "Guannan Qu"
  ],
  "author_count": 3,
  "categories": [
   "cs.MA",
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.MA",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper theoretically prove the near-optimality of RHCR in a discounted MDP formulation of the L-MAPF problem, and naturally motivate an extended framework called Group Decentralized RHCR (GD-RHCR) which incorporates a group decentralized structure that partitions agents based on a transitive communication scheme and plans for each partition of agents in parallel.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Alex DeWeese",
    "id": "2305680446",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Jiaoyang Li",
    "id": "2294313888",
    "h_index": 7,
    "papers": 27
   },
   {
    "name": "Guannan Qu",
    "id": "2305680677",
    "h_index": 1,
    "papers": 2
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.17928v1",
  "pdf_url": "https://arxiv.org/pdf/2608.17928v1",
  "html_url": "https://arxiv.org/html/2608.17928v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.17882",
  "slug": "controlledshifts-towards-standardizing-robustness-evaluation-in-trajec",
  "title": "ControlledShifts: Towards Standardizing Robustness Evaluation in Trajectory Prediction Under Distribution Shifts",
  "abstract": "Trajectory prediction is central to safety in autonomous driving, yet learning-based predictors tend to degrade sharply when encountering scenarios poorly represented by their training data. Many methods attempt to mitigate distribution shift degradation through data-centric or test-time adaptation approaches; however, they are typically validated along fragmented axes of generalization, leaving the field without a standardized way to compare robustness across shifts a model may encounter. To address this, we introduce ControlledShifts, a framework and benchmark suite that systematically re-splits existing trajectory datasets into in-distribution (seen) and out-of-distribution (unseen) partitions, via a shared characterization-and-splitting formulation, in which a characterization function fixes the axis of variation a benchmark probes and a splitting function fixes how the tail of that axis is withheld. The suite comprises three benchmarks targeting key topological and behavioral distribution shifts. Furthermore, to aggregate multi-dimensional performance metrics across these benchmarks, we propose a unified robustness score that evaluates models along two complementary dimensions: prediction quality (relative performance gain) and prediction stability (performance preservation under shift). We showcase ControlledShifts by benchmarking prominent transformer-based architectures, exposing critical differences in how models of varying capacities handle latent relevance and environmental structure.",
  "published": "2026-08-18",
  "updated": "2026-08-21",
  "year": "2026",
  "authors": [
   "Ingrid Navarro",
   "Pablo Ortega-Kral",
   "Yutong Duan",
   "Jonathan Francis",
   "Jean Oh"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "controlledShifts, a framework and benchmark suite that systematically re-splits existing trajectory datasets into in-distribution and out-of-distribution partitions, and proposes a unified robustness score that evaluates models along two complementary dimensions: prediction quality and prediction stability.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ingrid Navarro",
    "id": "30596850",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Pablo Ortega-Kral",
    "id": "2313914951",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Yutong Duan",
    "id": "2458067924",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Jonathan Francis",
    "id": "2314826620",
    "h_index": 5,
    "papers": 15
   },
   {
    "name": "Jean Oh",
    "id": "2243322333",
    "h_index": 4,
    "papers": 10
   }
  ],
  "comment": "8 pages, 8 figures, 1 table",
  "topics": [
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.17882v2",
  "pdf_url": "https://arxiv.org/pdf/2608.17882v2",
  "html_url": "https://arxiv.org/html/2608.17882v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.17874",
  "slug": "jetson-orb-slam3-accuracy-preserving-gpu-implementation-for-edge-compu",
  "title": "Jetson-ORB-SLAM3: Accuracy-Preserving GPU Implementation for Edge Computing Devices",
  "abstract": "Visual-inertial SLAM on low-power edge platforms is constrained by the cost of dense feature extraction and loop closure. Prior GPU ports of ORB-SLAM trade accuracy for speed by approximating the ORB detector, altering the feature set and therefore the estimated trajectory. We present an accuracy-preserving GPU implementation of ORB-SLAM3 for the NVIDIA Jetson Orin Nano, whose GPU ORB front end reproduces the reference CPU detector algorithmically to 94.7% exact keypoint agreement and 99.9% descriptor bit agreement. This work also makes CNN-based loop closure edge-viable through native TensorRT. The visual front end (feature extraction) is offloaded to the GPU while the mapping and optimization back end is kept on the CPU, matching each computation to the hardware it suits. The accuracy is verified by comparing four configurations: the GPU pipeline and the unmodified CPU reference, each run on both the Jetson Orin Nano and a desktop. On EuRoC dataset, all four agree to within 0.10cm in mean absolute trajectory error (SE(3)), so neither the GPU port nor the change of hardware shifts the estimated trajectory. The GPU-versus-CPU comparison is reproducible on TUM-VI and KITTI datasets, so the acceleration is accuracy-preserving rather than approximate. The proposed implementation is competitive with published ORB-SLAM3 on EuRoC, attains sub-centimeter accuracy on five of the six TUM-VI room sequences, and reaches sub-1% relative translation error on nine of eleven KITTI sequences. For loop closure, the generic ONNX-Runtime CUDA/TensorRT execution providers are unusable with our CosPlace ResNet-50 on the embedded platform, whereas a native libnvinfer FP16 engine reduces per-query inference to 2.2ms, a 180x speedup. Learned place recognition therefore runs concurrently with tracking on a 7W device. In monocular-inertial mode the system sustains 32FPS mean over the eleven EuRoC sequences.",
  "published": "2026-08-18",
  "updated": "2026-08-18",
  "year": "2026",
  "authors": [
   "Rajat Roy",
   "Aditya Arun Kumar Yadav",
   "Hardik Jain"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "An accuracy-preserving GPU implementation of ORB-SLAM3 for the NVIDIA Jetson Orin Nano, whose GPU ORB front end reproduces the reference CPU detector algorithmically to 94.7% exact keypoint agreement and 99.9% descriptor bit agreement is presented.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Rajat Roy",
    "id": "2458159210",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "A. Yadav",
    "id": "2458167989",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Hardik Jain",
    "id": "2458159171",
    "h_index": 0,
    "papers": 0
   }
  ],
  "comment": "",
  "topics": [
   "spatial-3d"
  ],
  "orgs": [
   "NVIDIA"
  ],
  "abs_url": "https://arxiv.org/abs/2608.17874v1",
  "pdf_url": "https://arxiv.org/pdf/2608.17874v1",
  "html_url": "https://arxiv.org/html/2608.17874v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2608.17819",
  "slug": "effector-centric-nmpc-of-tiltable-multirotors-for-offset-free-omnidire",
  "title": "Effector-Centric NMPC of Tiltable-Multirotors for Offset-Free Omnidirectional Aerial Manipulation",
  "abstract": "Aerial manipulation extends robotic operations to previously inaccessible aerial environments. Unlike arm-equipped aerial systems, tiltable-multirotors can directly generate six-degree-of-freedom wrenches through their flight bases, enabling both efficient movement and omnidirectional operation by tilting the thrust direction. This work presents a design analysis and a wrench-based control framework for tiltable-multirotors in aerial manipulation. We show that a four-rotor tiltable configuration provides a balance between interference-free propeller sizing and hovering efficiency across different attitudes, and its null-space redundancy is crucial for traversing singular configurations under physical constraints. We further show that an upward end-effector placement yields a favorable trade-off between geometric clearance and available wrench. To address disturbances, we propose a dual strategy consisting of a modified integral term for model error and an acceleration-based estimator for external wrenches. Building on these insights, we develop an effector-centric nonlinear model predictive control (NMPC) framework that integrates design choices, singularity handling, and disturbance compensation into a unified formulation. The proposed framework runs fully onboard at 100 Hz on a custom-built tiltable-quadrotor. Real-world experiments, including a 90-deg step cartwheel rotation, whiteboard pushing, and continuous 360-deg valve turning, demonstrate the feasibility of wrench-based omnidirectional manipulation with singularity traversal on a one-DoF-per-arm tiltable-quadrotor.",
  "published": "2026-08-18",
  "updated": "2026-08-18",
  "year": "2026",
  "authors": [
   "Jinjie Li",
   "Yicheng Chen",
   "Johannes K\u00fcbel",
   "Haokun Liu",
   "Junichiro Sugihara",
   "Moju Zhao"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jinjie Li",
    "id": "2301504767",
    "h_index": 4,
    "papers": 25
   },
   {
    "name": "Yicheng Chen",
    "id": "2109311677",
    "h_index": 2,
    "papers": 11
   },
   {
    "name": "Johannes K\u00fcbel",
    "id": "2458161758",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Haokun Liu",
    "id": "2401077362",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "J. Sugihara",
    "id": "2211425523",
    "h_index": 4,
    "papers": 23
   },
   {
    "name": "Moju Zhao",
    "id": "3264557",
    "h_index": 18,
    "papers": 67
   }
  ],
  "comment": "22 pages, 26 figures. Accepted to IEEE Transactions on Robotics (T-RO). This arXiv version includes a two-page appendix with additional implementation details",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.17819v1",
  "pdf_url": "https://arxiv.org/pdf/2608.17819v1",
  "html_url": "https://arxiv.org/html/2608.17819v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.17799",
  "slug": "training-with-synthetic-data-for-drone-detection-in-thermal-imagery",
  "title": "Training with synthetic data for drone detection in thermal imagery",
  "abstract": "Ground-to-Air (G2A) drone detection in medium- and long-wave infrared (MWIR/LWIR) imagery is challenging due to reduced texture information, sensor noise, weak thermal contrast, and the scarcity of annotated data. This work investigates a synthetic-first training strategy that combines synthetic scene generation with fine-tuning on real data. We show that synthetic data provides an effective basis for learning initial object representations, while real in-domain thermal imagery is still essential for reliable deployment. Even small amounts of real IR data substantially reduce domain gaps. Our experiments indicate that dataset alignment has a stronger impact on performance than model scale. Finally, our analysis of the dataset suggests that semantic alignment in feature space is the strongest predictor of model performance, while radiometric properties such as entropy and dynamic range also contribute to detection robustness. This work provides a foundation for combining synthetic and real IR data for effective G2A drone detection.",
  "published": "2026-08-18",
  "updated": "2026-08-18",
  "year": "2026",
  "authors": [
   "Tanel Liiv",
   "Sander Soodla",
   "Nzamba Bignoumba",
   "Alma M. Liezenga",
   "Toomas Pruuden"
  ],
  "author_count": 5,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.ET",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work investigates a synthetic-first training strategy that combines synthetic scene generation with fine-tuning on real data and shows that synthetic data provides an effective basis for learning initial object representations, while real in-domain thermal imagery is still essential for reliable deployment.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Tanel Liiv",
    "id": "118669826",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Sander Soodla",
    "id": "2458162114",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Nzamba Bignoumba",
    "id": "2204216632",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "A. M. Liezenga",
    "id": "2303845361",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Toomas Pruuden",
    "id": "2458163333",
    "h_index": 0,
    "papers": 0
   }
  ],
  "comment": "To be presented at SPIE: Sensors + Imaging, Artificial Intelligence for Security and Defence Applications IV",
  "topics": [
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.17799v1",
  "pdf_url": "https://arxiv.org/pdf/2608.17799v1",
  "html_url": "https://arxiv.org/html/2608.17799v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.17779",
  "slug": "stability-control-for-real-world-testing-in-autonomous-racing",
  "title": "Stability Control for Real World Testing in Autonomous Racing",
  "abstract": "Controlling an autonomous vehicle at the limits of handling is a challenging task. Due to external influences, such as road conditions or weather, a vehicle can easily become unstable. Since most control algorithms assume stable vehicle behavior, they might fail in these situations. Especially when operating expensive vehicles without a safety driver on board, as in autonomous racing, this poses a significant challenge. To enable safe operation at the vehicle's dynamic limits, we present a comprehensive stability control system that safeguards motion control algorithms in autonomous driving. The proposed system consists of an electronic stability control (ESC), a slip control (SC), and a countersteer system (CS), which collectively adapt steering and brake commands from the motion controller to maintain vehicle stability. We validate our approach through both simulation and experiments on a real-world, full-scale vehicle. The results show that the stability control system maintains vehicle stability in critical situations and extends the operational feasible region. To simplify integration, we provide an open-source implementation at github.com/TUMFTM/tam-stability-control.",
  "published": "2026-08-18",
  "updated": "2026-08-18",
  "year": "2026",
  "authors": [
   "Phillip Pitschi",
   "Simon Sagmeister",
   "Frederik Werner",
   "Markus Lienkamp",
   "Boris Lohmann"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Phillip Pitschi",
    "id": "2296782965",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Simon Sagmeister",
    "id": "1395369553",
    "h_index": 4,
    "papers": 14
   },
   {
    "name": "Frederik Werner",
    "id": "2167156261",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Markus Lienkamp",
    "id": "2311545767",
    "h_index": 5,
    "papers": 24
   },
   {
    "name": "Boris Lohmann",
    "id": "2238801456",
    "h_index": 4,
    "papers": 11
   }
  ],
  "comment": "Accepted at IEEE ITSC 2026",
  "topics": [
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.17779v1",
  "pdf_url": "https://arxiv.org/pdf/2608.17779v1",
  "html_url": "https://arxiv.org/html/2608.17779v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.17744",
  "slug": "thinking-in-a-low-resource-language-what-sft-builds-what-rl-fixes-what",
  "title": "Thinking in a Low-Resource Language: What SFT Builds, What RL Fixes, What Accuracy Cannot See",
  "abstract": "Take three frontier mixture-of-experts models (Alibaba, OpenAI, NVIDIA; 3.6-4.0B active parameters each) and fine-tune them to reason in a low-resource language. On accuracy benchmarks almost nothing happens, and the benchmark itself is noise at this scale: changing only the random seed moves the score by 7.7 points, more than every data and recipe effect we measured. That null is our first result. The real changes live where accuracy cannot see. Base models never think in Greek: 0 of 1,000 reasoning traces, even when the question is Greek, so the model answers correctly while reasoning in a form its user cannot read, audit, or correct. After supervised fine-tuning (SFT), every released checkpoint reasons in the language of the question on ~98% of items, one family at 3x fewer tokens, with judged grammaticality improving on all four models and general ability within a few points of each base: nothing was forgotten, and fluency was gained. We propose six behavioural dimensions that make such changes measurable, each gated to reject any metric that correlates with output length, and we report how our own instruments lied: six failures, each caught by a control. What SFT cannot do is fix its own defects: a quarter of answers skip the requested format, answers leak into the reasoning channel, and an explicit \"think in English\" is obeyed under half the time. Reinforcement learning with verifiable rewards, pre-registered before training, fixes the first two outright (fallback 24% to 2.5%, leak 3.5% to 0.0%, both against a flat random-reward control) and moves the third (+9.1pp), while the Greek reasoning habit survives an accuracy-only gradient untouched. We release five checkpoints. The instruments, the controls and the pre-registration travel to any low-resource language; Greek is the case that let us measure them.",
  "published": "2026-08-18",
  "updated": "2026-08-18",
  "year": "2026",
  "authors": [
   "Ayoub Kirouane",
   "Christos Petrocheilos"
  ],
  "author_count": 2,
  "categories": [
   "cs.CL",
   "cs.LG",
   "cs.RO",
   "stat.ML"
  ],
  "primary_category": "cs.CL",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Take three frontier mixture-of-experts models and fine-tune them to reason in a low-resource language and propose six behavioural dimensions that make changes measurable, each gated to reject any metric that correlates with output length, and report how their own instruments lied.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ayoub Kirouane",
    "id": "2454421288",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Christos Petrocheilos",
    "id": "2454421342",
    "h_index": 0,
    "papers": 2
   }
  ],
  "comment": "",
  "topics": [
   "rl-control"
  ],
  "orgs": [
   "NVIDIA"
  ],
  "abs_url": "https://arxiv.org/abs/2608.17744v1",
  "pdf_url": "https://arxiv.org/pdf/2608.17744v1",
  "html_url": "https://arxiv.org/html/2608.17744v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2608.17717",
  "slug": "compcpz-preserving-multi-modal-intent-in-language-guided-robot-manipul",
  "title": "CompCPZ: Preserving Multi-Modal Intent in Language-Guided Robot Manipulation",
  "abstract": "A robot asked to \"place the cup near the red plate or the blue plate\" may reach the centroid between them and appear geometrically successful, while satisfying neither disjunct of the instruction. This silent semantic failure exposes a structural limitation of language-conditioned robot policies: representations that collapse a disjunctive instruction into a single connected set cannot preserve all feasible modes, and planners that commit to one action degrade under run-time mode uncertainty. We address this limitation with CompCPZ, a sound algebraic layer that language-conditioned learning systems wrap to recover multi-modal disjunctive representation, recursively composing per-primitive constrained polynomial zonotope enclosures along the language parse tree with distribution-free conformal coverage and sub-millisecond runtime. On a closed-loop ManiSkill3 tabletop-manipulation benchmark, CompCPZ outperforms convex set baselines, multi-peak decoders, and a zero-shot vision-language-action model (1,900/1,918 paired wins, p << 10^(-30)); the same compiler also transfers without retuning to planar real-robot trials on a Unitree Go2 quadruped under motion capture. These results suggest that compositional language grounding should be evaluated not only by reaching a decoded target, but by whether the represented feasibility set preserves the connected-component structure of the user's intent.",
  "published": "2026-08-18",
  "updated": "2026-08-18",
  "year": "2026",
  "authors": [
   "Zhen Zhang",
   "Ahmad Hafez",
   "Peng Xie",
   "Yanliang Huang",
   "Wenyuan Wu",
   "Amr Alanwar"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "CompCPZ, a sound algebraic layer that language-conditioned learning systems wrap to recover multi-modal disjunctive representation, recursively composing per-primitive constrained polynomial zonotope enclosures along the language parse tree with distribution-free conformal coverage and sub-millisecond runtime is presented.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhen Zhang",
    "id": "2354233844",
    "h_index": 2,
    "papers": 19
   },
   {
    "name": "Ahmad Hafez",
    "id": "2293614450",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Peng Xie",
    "id": "2354320476",
    "h_index": 3,
    "papers": 21
   },
   {
    "name": "Yanliang Huang",
    "id": "2261875071",
    "h_index": 3,
    "papers": 16
   },
   {
    "name": "Wenyu Wu",
    "id": "2354982743",
    "h_index": 1,
    "papers": 10
   },
   {
    "name": "Amr Alanwar",
    "id": "27706391",
    "h_index": 12,
    "papers": 75
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "humanoids",
   "egocentric-data"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2608.17717v1",
  "pdf_url": "https://arxiv.org/pdf/2608.17717v1",
  "html_url": "https://arxiv.org/html/2608.17717v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2608.17703",
  "slug": "dijkstra-as-an-oracle-for-online-stochastic-shortest-path-navigation-w",
  "title": "Dijkstra as an Oracle for Online Stochastic Shortest Path Navigation with Provable Guarantees",
  "abstract": "Mobile robots that operate in side by side with humans and critical facilities must reach their goals at low cost, despite often unknown true traversal costs of the map apriori and imperfect actuation. Planners that solve the underlying stochastic shortest path problem exactly, such as value iteration, require computation that grows with the diameter of the map, whereas Dijkstra's algorithm is fast but is usually considered inexact once transitions are stochastic. This study shows that Dijkstra's algorithm can remain an exact planning engine under a condition that is much weaker than the causality condition often invoked in the literature, namely nonnegativity of a reduced cost defined on the determinized map. Building on this characterization, an online learner DORA (Dijkstra Oracle Reduced-cost Algorithm) is proposed for robot navigation that calls a shortest path oracle a fixed number of times per episode, never estimates a transition kernel, and adds a logarithmic survival weight when the probability of contact with a dynamic obstacle must stay within a budget. In the numerical experiments involving three other benchmarks that cover grid world navigation, directional drilling, and drone surveillance, the learner matches optimistic value iteration that is given the true transition kernel while performing 4.5 to 19.3 times less planner work, reduces contacts during learning by a factor of seventeen relative to determinize and replan, and keeps the contact rate within budgets that span two orders of magnitude. These results indicate that shortest path search supports safe and efficient online navigation and path planning tasks.",
  "published": "2026-08-18",
  "updated": "2026-08-18",
  "year": "2026",
  "authors": [
   "Mansur M. Arief",
   "Ali Akarma",
   "Ahmad Alfan Alfian Irfan"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.AI",
   "math.OC"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This study shows that Dijkstra's algorithm can remain an exact planning engine under a condition that is much weaker than the causality condition often invoked in the literature, namely nonnegativity of a reduced cost defined on the determinized map.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "M. Arief",
    "id": "2312747501",
    "h_index": 2,
    "papers": 13
   },
   {
    "name": "Ali Akarma",
    "id": "2440827885",
    "h_index": 0,
    "papers": 5
   },
   {
    "name": "Ahmad Alfan Alfian Irfan",
    "id": "2451543950",
    "h_index": 0,
    "papers": 1
   }
  ],
  "comment": "",
  "topics": [
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.17703v1",
  "pdf_url": "https://arxiv.org/pdf/2608.17703v1",
  "html_url": "https://arxiv.org/html/2608.17703v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.17691",
  "slug": "force-based-offset-estimation-for-keyed-peg-in-hole-assembly-using-loc",
  "title": "Force-Based Offset Estimation for Keyed Peg-in-Hole Assembly Using Local Gaussian Process Regression",
  "abstract": "Key-keyway assembly tasks impose strict geometric constraints and are highly sensitive to grasp pose deviations in uncertain environments. This work presents a force-based offset estimation method for keyed peg-in-hole assembly, embedded within a perception-validation-insertion pipeline. Residual misalignment is estimated directly from wrist force/torque measurements using a local KNN-Gaussian Process hybrid regressor. The framework distinguishes between two contact regimes, hard collision and guided chamfer insertion, and routes inference to a dedicated model for each. Regime classification is achieved via a contact-window duration threshold. KNN combined with a deterministic search using the results of a post-grasp monocular visual validation contributes to an increased accuracy of the regressor model. This approach achieves accurate radial offset estimation in chamfered peg insertion, during a keypoint detection-based pick and place application. Experiments using the integrated force/torque sensor of a collaborative robot arm showed an increase in insertion success rate from 67% to 87% after the pipeline was applied.",
  "published": "2026-08-18",
  "updated": "2026-08-18",
  "year": "2026",
  "authors": [
   "Chandra Yuvesh Aubeeluck",
   "Abilash Philip Madavath",
   "Augustin Raju",
   "Nicolas Pyschny",
   "Felix Hackel\u00f6er",
   "Florian Zwanzig"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A force-based offset estimation method for keyed peg-in-hole assembly, embedded within a perception-validation-insertion pipeline, achieves accurate radial offset estimation in chamfered peg insertion, during a keypoint detection-based pick and place application.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Chandra Yuvesh Aubeeluck",
    "id": "2261550454",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Abilash Philip Madavath",
    "id": "2438945817",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Augustin Raju",
    "id": "2438945247",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "N. Pyschny",
    "id": "2465104",
    "h_index": 7,
    "papers": 22
   },
   {
    "name": "Felix Hackel\u00f6er",
    "id": "71919134",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Florian Zwanzig",
    "id": "2438945825",
    "h_index": 0,
    "papers": 1
   }
  ],
  "comment": "6 pages, 11 figures. Accepted and presented at the 2026 IEEE/ASME International Conference on Advanced Intelligent Mechatronics (AIM 2026), Genova, Italy. Awaiting publication in IEEE Xplore",
  "topics": [
   "dexterous-manipulation",
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.17691v1",
  "pdf_url": "https://arxiv.org/pdf/2608.17691v1",
  "html_url": "https://arxiv.org/html/2608.17691v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.17690",
  "slug": "collective-ranking-of-environmental-signals-through-gaussian-belief-pr",
  "title": "Collective Ranking of Environmental Signals through Gaussian Belief Propagation in a Patrolling Robot Swarm",
  "abstract": "Multi-robot patrolling requires a team to visit all areas of an environment at regular intervals, typically minimising idleness. A practical extension, motivated by security and environmental monitoring, is to additionally form a collective ranking of all patrol locations by some measured signal, a generalisation of the best-of-n problem to the many-option, continuous-valued regime. We observe that the patrol graph admits a natural dual interpretation: it is simultaneously the topology that dictates agent movement and a factor graph over which spatial beliefs can be propagated. Exploiting this equivalence, we apply Gaussian Belief Propagation (GBP), a graph-based algorithm, to collective ranking using unary measurement factors at visited nodes and pairwise smoothness factors along patrol edges. We compare GBP against simple and visit-count-weighted averaging across a range of sensor-noise conditions in simulation, and validate the approach on four Leo Rovers tracking a propagating radio signal in an office lobby. GBP outperforms both baselines on ranking accuracy, mean squared error, and time to consensus. We find that as noise increases and the task becomes harder, GBP degrades gracefully in simulation while both averaging methods degrade substantially. Hardware trials reproduce the same performance ordering on a real propagating radio signal, supporting the practical relevance of the simulated results.",
  "published": "2026-08-18",
  "updated": "2026-08-18",
  "year": "2026",
  "authors": [
   "Zachary R. Madin",
   "Connor York",
   "Jonathan Lawry",
   "Edmund R. Hunt"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Gaussian Belief Propagation (GBP), a graph-based algorithm, is applied to collective ranking using unary measurement factors at visited nodes and pairwise smoothness factors along patrol edges and it is found that as noise increases and the task becomes harder, GBP degrades gracefully in simulation while both averaging methods degrade substantially.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zachary R. Madin",
    "id": "2275352655",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Connor York",
    "id": "2289841350",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Jonathan Lawry",
    "id": "2275352165",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Edmund R. Hunt",
    "id": "2275355698",
    "h_index": 3,
    "papers": 4
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.17690v1",
  "pdf_url": "https://arxiv.org/pdf/2608.17690v1",
  "html_url": "https://arxiv.org/html/2608.17690v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.17633",
  "slug": "ovip-sg-open-vocabulary-instance-preserving-scene-graphs-for-mapping-a",
  "title": "OVIP-SG: Open-Vocabulary Instance-Preserving Scene Graphs for Mapping and Retrieval of Small, Fine-Grained Objects",
  "abstract": "Integrating open-vocabulary perception into object-level 3D scene graphs is a double-edged sword. While vision-language detectors recover long-tail categories and small, fine-grained objects overlooked by closed-set models, they also tend to fragment large surfaces and merge small objects into larger neighboring objects, compromising instance-level consistency and undermining mapping fidelity. Moreover, existing methods struggle to retrieve previously unmapped targets or determine whether a queried object is absent, hindering robust embodied open-world navigation and exploration. We present OVIP-SG, a unified framework for instance-preserving semantic mapping, functional scene partitioning, and language-guided small, fine-grained object retrieval. OVIP-SG uses a vision-language model (VLM) to enumerate scene-specific categories for robust open-world detection. Symmetric 3D Intersection over Union (IoU) association and area-weighted feature fusion preserve small independent instances, while VLM-inferred object functions partition scenes into compact functional search regions. A four-stage cascaded retrieval pipeline further incorporates voxel voting and determines target absence from exploration coverage. Under a unified evaluation protocol on Replica, OVIP-SG outperforms ConceptGraphs by 6.31 points in class-mean accuracy (mAcc) and 5.15 points in frequency-weighted mIoU (F-mIoU) while achieving a class-agnostic native-instance Panoptic Quality (PQ) of 0.398. It reduces the search area to 21.8% of the indoor floor space and reaches 0.773 balanced accuracy for object-presence classification. Real-world robotic experiments further demonstrate its practical effectiveness. Code is available at https://github.com/Agibot-Spatial-Intelligence/OVIP-SG.",
  "published": "2026-08-18",
  "updated": "2026-08-21",
  "year": "2026",
  "authors": [
   "Tianjing Hao",
   "Haiyu Lan",
   "Angsong Li",
   "Cheng Chen",
   "Enyu Li",
   "Jiarui Yang",
   "Yuning Su",
   "Peiwen Lin",
   "Wang Chuang"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "OVIP-SG is presented, a unified framework for instance-preserving semantic mapping, functional scene partitioning, and language-guided small, fine-grained object retrieval that outperforms ConceptGraphs under a unified evaluation protocol on Replica.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Tianjing Hao",
    "id": "2455323467",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Haiyu Lan",
    "id": "2455273729",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Ang Li",
    "id": "2457206786",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Cheng-Hung Chen",
    "id": "2448926318",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Enyu Li",
    "id": "2455335642",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Jiarui Yang",
    "id": "2447991900",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yuning Su",
    "id": "2455452490",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Pei Lin",
    "id": "2357017108",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Chuang Wang",
    "id": "2455563265",
    "h_index": 0,
    "papers": 1
   }
  ],
  "comment": "15 pages, 6 figures, including appendix",
  "topics": [
   "sim2real",
   "navigation",
   "video-generation"
  ],
  "orgs": [
   "AgiBot"
  ],
  "abs_url": "https://arxiv.org/abs/2608.17633v4",
  "pdf_url": "https://arxiv.org/pdf/2608.17633v4",
  "html_url": "https://arxiv.org/html/2608.17633v4",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2608.17628",
  "slug": "iterative-grasp-pose-refinement-a-deep-reinforcement-learning-approach",
  "title": "Iterative Grasp Pose Refinement: A Deep Reinforcement Learning Approach for 2D Vision",
  "abstract": "Developing robots capable of understanding and manipulating objects requires compact, interpretable, and generalizable representations. This work proposes a reinforcement learning-based framework for robotic grasp refinement, integrating keypoint-based object representations with a Deep Q-Network (DQN). Using 2D overhead images captured in a simulated environment, a geometric-based algorithm generates initial grasp candidates, which are iteratively refined by the proposed framework, transforming failed grasps into successful ones. Experiments conducted on 300 objects from the Dex-Net dataset using a UR5 manipulator demonstrate the framework's effectiveness, achieving a 100% success rate on objects previously deemed ungraspable by geometrical methods. The framework's sim-to-real transferability is further validated through physical experiments on a Delta parallel robot, where a refined grasp successfully manipulates an object that was previously ungraspable. The findings underscore the effectiveness of reinforcement learning in addressing challenges in robotic grasping, offering a scalable and adaptable solution for contact-rich manipulation tasks.",
  "published": "2026-08-18",
  "updated": "2026-08-18",
  "year": "2026",
  "authors": [
   "Amir Arsalan Nematollahi",
   "Shayan Ahmadi",
   "Mehdi Tale Masouleh",
   "Ahmad Kalhor"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG",
   "eess.IV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A reinforcement learning-based framework for robotic grasp refinement, integrating keypoint-based object representations with a Deep Q-Network (DQN), is proposed, offering a scalable and adaptable solution for contact-rich manipulation tasks.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Amir Arsalan Nematollahi",
    "id": "2441751152",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Shayan Ahmadi",
    "id": "2260998602",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "M. T. Masouleh",
    "id": "9435015",
    "h_index": 23,
    "papers": 239
   },
   {
    "name": "A. Kalhor",
    "id": "2516483",
    "h_index": 16,
    "papers": 123
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "sim2real",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.17628v1",
  "pdf_url": "https://arxiv.org/pdf/2608.17628v1",
  "html_url": "https://arxiv.org/html/2608.17628v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.17601",
  "slug": "physics-informed-sliding-window-particle-filtering-for-tactile-only-in",
  "title": "Physics-Informed Sliding-Window Particle Filtering for Tactile-Only In-Hand 6-DoF Object Pose Refinement",
  "abstract": "This paper studies tactile-only 6-DoF pose refinement and belief maintenance for grasped objects in static and short quasi-static in-hand configurations where vision is unavailable or heavily occluded. The key difficulty is tactile partial observability: whole-hand taxel contacts are sparse, intermittent, and ambiguous under limited excitation and object symmetries. We propose a physics-informed particle filter on $\\mathrm{SE}(3)$ that updates pose beliefs from dense whole-hand tactile measurements. The likelihood combines active-contact signed-distance consistency, force-normal alignment, friction-cone feasibility, zero-force negative evidence, and optional feasibility guards. A sliding-window log-likelihood fuses recent tactile frames to reduce single-frame ambiguity, while a potential-field-guided proposal steers particles away from hand--object penetration. Symmetry-aware resampling preserves multiple plausible modes. Experiments on an Allegro Hand V5 with five objects show lower normalized ADD-S than tactile-only geometric, particle-filter, and learning baselines, and ablations confirm the benefits of temporal fusion, potential guidance, and mode preservation.",
  "published": "2026-08-18",
  "updated": "2026-08-18",
  "year": "2026",
  "authors": [
   "Lingjun Shao",
   "Ying Zhang",
   "Xiangfei Li",
   "Xiangyang Li",
   "Huan Zhao",
   "Zhenyu Wang",
   "Han Ding"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper studies tactile-only 6-DoF pose refinement and belief maintenance for grasped objects in static and short quasi-static in-hand configurations where vision is unavailable or heavily occluded and proposes a physics-informed particle filter on SE that updates pose beliefs from dense whole-hand tactile measurements.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Lingjun Shao",
    "id": "2326697940",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Ying Zhang",
    "id": "2458711473",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Xiangfei Li",
    "id": "2108309721",
    "h_index": 19,
    "papers": 66
   },
   {
    "name": "Xiangyang Li",
    "id": "2458557969",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Huan Zhao",
    "id": "2323580263",
    "h_index": 5,
    "papers": 41
   },
   {
    "name": "Zhenyu Wang",
    "id": "2295957814",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Han Ding",
    "id": "2275549676",
    "h_index": 8,
    "papers": 31
   }
  ],
  "comment": "Accepted by IEEE RAL journal",
  "topics": [
   "dexterous-manipulation",
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.17601v1",
  "pdf_url": "https://arxiv.org/pdf/2608.17601v1",
  "html_url": "https://arxiv.org/html/2608.17601v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.17600",
  "slug": "libero-vifo-benchmarking-the-capability-and-safety-of-visual-cue-follo",
  "title": "LIBERO-VIFO: Benchmarking the Capability and Safety of Visual Cue Following in Vision-Language-Action Models",
  "abstract": "Visual cues are increasingly adopted to guide robot learning, but whether Vision-Language-Action (VLA) models can reliably follow authorized cues while disregarding unauthorized ones remains unclear. Existing work covers only a narrow range of cue forms and focuses on final task success, providing only a coarse assessment of cue-following capability. Treating all visual cues as authorized also leaves safety risks of unauthorized following unexplored. To address these gaps, we introduce LIBERO-VIFO, a benchmark to evaluate both the capability and safety of visual cue following in VLA models. LIBERO-VIFO defines eight visual cue families spanning diverse forms. A total of four protocols in two parts are defined: Part I tests cue understanding and authorized following, while Part II evaluates unauthorized visual cue following under language-cue conflict and empty language conditions. Evaluating seven VLA models reveals that although visual cue understanding does not reliably translate into execution, current VLAs are able to execute cue-indicated tasks without language instruction, exposing an emerging risk of unauthorized visual cue following. Extended experiments on scene-instantiated cues, safety-critical settings, and real-robot deployment corroborate these findings. LIBERO-VIFO brings both the capability and safety of visual cue following into systematic evaluation, establishing visual-centric safety as a new perspective for the VLA community.",
  "published": "2026-08-18",
  "updated": "2026-08-18",
  "year": "2026",
  "authors": [
   "Zhengyan Qian",
   "Rui Yan",
   "Alex Jinpeng Wang",
   "Jinhui Tang"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Evaluating seven VLA models reveals that although visual cue understanding does not reliably translate into execution, current VLAs are able to execute cue-indicated tasks without language instruction, exposing an emerging risk of unauthorized visual cue following.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhengyan Qian",
    "id": "2397699485",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Rui Yan",
    "id": "2289845725",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Alex Jinpeng Wang",
    "id": "2275540429",
    "h_index": 7,
    "papers": 22
   },
   {
    "name": "Jinhui Tang",
    "id": "2313357453",
    "h_index": 6,
    "papers": 17
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.17600v1",
  "pdf_url": "https://arxiv.org/pdf/2608.17600v1",
  "html_url": "https://arxiv.org/html/2608.17600v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.17596",
  "slug": "tinydsm-a-framework-for-skill-modeling-and-development-for-resource-co",
  "title": "tinyDSM: A Framework for Skill Modeling and Development for Resource-Constrained Millirobots",
  "abstract": "In this study, we investigate developmental mechanisms that enable small, resource-constrained systems such as cm-sized millirobots to autonomously explore, learn, and adapt their capabilities throughout their lifespan. Reinforcement learning algorithms guide the agent's skill acquisition and adaptation through the interplay of our proposed tinyDSM, which integrates intrinsic motivation and fitness-based assessment. We strive for minimal, hard-wired skills while encouraging the open-ended development of new skills. A key emphasis in our approach is to encode minimal a-priori general knowledge, which serves as a foundational starting point for the system as it further learns system-specific dependencies from the initial knowledge provided. Thus, by design, our approach attempts to cover very generic application domains. The methodology is based on (a) developmental mechanism with intrinsic motivation, and (b) a cognitive architecture (knowledge, reasoning, learning), while (c) utilizing minimal resources. It uses a hierarchical knowledge graph and kinematic reasoners to model and evaluate simple and advanced motion related skills. In our experiments, we use a resource-constrained millirobot with a volume of 36 cm^3 with a Raspberry Pi Pico 32-bit microcontroller (RP2040) that integrates all described features and capabilities except the camera system in 9 kB. Starting with learning the most elementary motor skills the millirobot autonomously progresses from simple linear and angular movements to complex geometric patterns within 15 minutes. To complement the physical experiments, we perform a simulation-based analysis that enables systematic comparisons across learning algorithms and intrinsic motivation parameters.",
  "published": "2026-08-18",
  "updated": "2026-08-18",
  "year": "2026",
  "authors": [
   "Markus D. Kobelrausch",
   "Michael Miedler",
   "Axel Jantsch"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This study investigates developmental mechanisms that enable small, resource-constrained systems such as cm-sized millirobots to autonomously explore, learn, and adapt their capabilities throughout their lifespan and proposes a proposed tinyDSM, which integrates intrinsic motivation and fitness-based assessment.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Markus D. Kobelrausch",
    "id": "2115012824",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Michael Miedler",
    "id": "2458132127",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "A. Jantsch",
    "id": "1878553",
    "h_index": 44,
    "papers": 498
   }
  ],
  "comment": "Manuscript submitted to IEEE Transactions on Cognitive and Developmental Systems",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.17596v1",
  "pdf_url": "https://arxiv.org/pdf/2608.17596v1",
  "html_url": "https://arxiv.org/html/2608.17596v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.17592",
  "slug": "communication-reduction-via-semantic-based-encoding-in-dmpc-using-lstm",
  "title": "Communication Reduction via Semantic-Based Encoding in DMPC Using LSTMs",
  "abstract": "The communication demands of distributed model prediction control (DMPC) can overwhelm even advanced wireless communication technologies as agents must exchange a significant amount of information at least once per time step. To semantically reduce communication demands, this work employs encoder-decoder networks built around long-short term memory (LSTM) cells in a distributed optimization algorithm. Agents publish a reduced representation of a message and receivers reconstruct the original message upon reception. In tests with reduced communication using formations of mobile robots, trained networks retain satisfactory performance and work reliably under conditions overwhelming full communication. As the results show, the usage of LSTMs either allows unprecedented reconstruction accuracy or the usage of different prediction-horizon lengths without the necessity to retrain.",
  "published": "2026-08-18",
  "updated": "2026-08-18",
  "year": "2026",
  "authors": [
   "Torben Schiz",
   "Pedro H. J. Nardelli",
   "Henrik Ebel"
  ],
  "author_count": 3,
  "categories": [
   "eess.SY",
   "cs.DC",
   "cs.LG",
   "cs.MA",
   "cs.RO"
  ],
  "primary_category": "eess.SY",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work employs encoder-decoder networks built around long-short term memory (LSTM) cells in a distributed optimization algorithm that allows unprecedented reconstruction accuracy or the usage of different prediction-horizon lengths without the necessity to retrain.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Torben Schiz",
    "id": "2354176794",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "P. H. Nardelli",
    "id": "2203277580",
    "h_index": 5,
    "papers": 19
   },
   {
    "name": "Henrik Ebel",
    "id": "31771003",
    "h_index": 14,
    "papers": 46
   }
  ],
  "comment": "13 pages, 13 figures",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.17592v1",
  "pdf_url": "https://arxiv.org/pdf/2608.17592v1",
  "html_url": "https://arxiv.org/html/2608.17592v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.17584",
  "slug": "hodagent-towards-on-demand-responsive-humanoids-for-physical-world-hum",
  "title": "HODAgent: Towards On-Demand, Responsive Humanoids for Physical World Human Interaction",
  "abstract": "We propose HODAgent, a System-2 embodied agent for humanoid robots in service settings, addressing situated intent, responsive execution, task revision, and outcome verification. Its semi-duplex architecture integrates an Env-Interactor, Planner, Executor, and hierarchical Memory to maintain coherent interaction, planning, and task state during service episodes. This allows handling new requests during motion, retaining progress, revising actions, and grounding closure in execution outcomes. A shared interface connects simulation and physical robots (Unitree G1), isolating platform-specific control. In an interactive simulation with 164 cases, HODAgent achieves 84.8% and 91.5% Joint Success under two VLM backbones, outperforming baselines by 9.8 and 18.9 points. On physical robots, pass rates are 92% (atomic), 72% (composite), and 63.3% (complete tasks). On multiple embodied benchmarks, it improves over baselines by 0.7-9.0 points. Results show a unified System-2 agent enables adaptive humanoid service across simulation and reality.",
  "published": "2026-08-18",
  "updated": "2026-08-20",
  "year": "2026",
  "authors": [
   "Wang Warren Chen",
   "Jiahao Zhang",
   "Zhenjiang Li",
   "Mingxu Wang",
   "Lei Yi",
   "Yuchen Kang",
   "Shuo Sun",
   "Ziping Chen",
   "Jie Chen"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Results show a unified System-2 agent enables adaptive humanoid service across simulation and reality.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "W. Chen",
    "id": "2457739436",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Jiahao Zhang",
    "id": "2186399698",
    "h_index": 4,
    "papers": 16
   },
   {
    "name": "Zhenjiang Li",
    "id": "2453951630",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Mingxuan Wang",
    "id": "2354288945",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Lei Yi",
    "id": "2268566173",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Yuchen Kang",
    "id": "2445709480",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Shuo Sun",
    "id": "2115305851",
    "h_index": 8,
    "papers": 20
   },
   {
    "name": "Ziping Chen",
    "id": "2458253635",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Jie Chen",
    "id": "2445573778",
    "h_index": 0,
    "papers": 4
   }
  ],
  "comment": "we have received a formal directive from our company requiring all company assets to undergo a mandatory internal review process before any public release. We are now required to immediately withdraw the paper to comply with this policy",
  "topics": [
   "humanoids",
   "safety-eval"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2608.17584v2",
  "pdf_url": "https://arxiv.org/pdf/2608.17584v2",
  "html_url": "https://arxiv.org/html/2608.17584v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2608.17553",
  "slug": "scalix-uncertainty-aware-scale-consistent-monocular-slam",
  "title": "Scalix: Uncertainty-Aware Scale-Consistent Monocular SLAM",
  "abstract": "Cameras are ubiquitous sensors in robotics due to their compact form factor and the perceptual richness captured through visual information. Monocular SLAM enables robots to understand the environment with a minimum setup, however, it inherently suffers from scale ambiguity. A common solution is to provide multi-modal sensor configurations, such as visual-inertial systems, where scale is observable unless the robot navigates under a constant-velocity motion, a common scenario in mobile robotics. With the advent of deep-learning, geometric foundation models have been used to address this problem, but the depths maps are often noisy and scale-inconsistent across frames. In this paper, we propose Scalix, a real-time monocular SLAM framework that achieves metric-scale state estimation by integrating learned depth cues into a probabilistic factor-graph formulation. By augmenting existing monocular depth models with both per-pixel depth uncertainty and per-frame scale uncertainty, Scalix treats scale predictions as independent measurements within its optimization, leading to improved scale consistency through multi-view data associations. Experiments in large-scale outdoor and indoor environments demonstrate state-of-the-art performance on both metric and up-to-scale benchmarks while maintaining real-time operation and generalization.",
  "published": "2026-08-18",
  "updated": "2026-08-18",
  "year": "2026",
  "authors": [
   "Sebastian Barbas Laina",
   "Tianyi Zhang",
   "Panagiotis Petropoulakis",
   "Simon Schaefer",
   "Simon Boche",
   "Jaehyung Jung",
   "Cedric Le Gentil",
   "Stefan Leutenegger"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Scalix is proposed, a real-time monocular SLAM framework that achieves metric-scale state estimation by integrating learned depth cues into a probabilistic factor-graph formulation, leading to improved scale consistency through multi-view data associations.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Sebasti\u00e1n Barbas Laina",
    "id": "2290010496",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Tianyi Zhang",
    "id": "2458368575",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Panagiotis Petropoulakis",
    "id": "2243336778",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Simon Schaefer",
    "id": "2237801279",
    "h_index": 7,
    "papers": 17
   },
   {
    "name": "Simon Boche",
    "id": "2179884808",
    "h_index": 6,
    "papers": 14
   },
   {
    "name": "Jaehyung Jung",
    "id": "2321686937",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "C. Gentil",
    "id": "51295768",
    "h_index": 10,
    "papers": 39
   },
   {
    "name": "Stefan Leutenegger",
    "id": "2290004858",
    "h_index": 9,
    "papers": 22
   }
  ],
  "comment": "8 pages, 5 figures and 3 tables",
  "topics": [
   "spatial-3d",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.17553v1",
  "pdf_url": "https://arxiv.org/pdf/2608.17553v1",
  "html_url": "https://arxiv.org/html/2608.17553v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.17512",
  "slug": "embodied-navigator-point-think-memorize-and-align-for-efficient-naviga",
  "title": "Embodied-Navigator: Point, Think, Memorize, and Align for Efficient Navigation",
  "abstract": "Although Large Vision-Language Models (VLMs) have significantly advanced embodied navigation, their direct deployment remains challenging, as existing methods often force VLMs into unnatural action spaces that misalign with their 2D pre-training priors, compounded by rigid reasoning schedules and inefficient memory management. To overcome these limitations, we propose TAMP-Nav, a unified framework for efficient embodied navigation. First, we introduce a Pixel-to-3D Action Formulation (Point) that reformulates navigation into 2D visual prompting. Specifically, the VLM merely selects 2D pixels, which are then projected into 3D coordinates for a low-level SLAM controller. This design naturally aligns embodied execution with the VLM's inherent 2D visual capabilities. Second, we propose an integrated Selective Reasoning and Anchor-Trajectory Memory mechanism (Think and Memorize), which dynamically triggers Chain-of-Thought and retains high-fidelity memory only at critical nodes, compressing redundant trajectories into lightweight Space-Time Indicators, thereby preserving critical historical information and enhancing spatio-temporal perception. Finally, we design an efficient Two-Level Alignment Paradigm (Align) via Group Relative Policy Optimization (GRPO). By superimposing global outcome rewards with fine-grained process rewards, this dense supervision tightly aligns the agent's cognitive planning with physical environmental feedback, endowing the model with adaptive reasoning capabilities. Experiments demonstrate that TAMP-Nav achieves state-of-the-art performance (e.g., 66.2% SR on R2R-CE) with high runtime and sample efficiency (requiring only 90k training trajectories).",
  "published": "2026-08-18",
  "updated": "2026-08-18",
  "year": "2026",
  "authors": [
   "Hongyan Feng",
   "Sunlai Chen",
   "Xuanyu Liu",
   "Miao Pan",
   "Yangfan Xie",
   "Yuxiang Cui",
   "Zhongxiang Zhou",
   "Rong Xiong",
   "Wenqi Zhang",
   "Jianwei Yin",
   "Yueting Zhuang",
   "Xuhong Zhang"
  ],
  "author_count": 12,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes TAMP-Nav, a unified framework for efficient embodied navigation that dynamically triggers Chain-of-Thought and retains high-fidelity memory only at critical nodes, compressing redundant trajectories into lightweight Space-Time Indicators, thereby preserving critical historical information and enhancing spatio-temporal perception.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hongyan Feng",
    "id": "2351408492",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Sunlai Chen",
    "id": "2458543237",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Xuanyu Liu",
    "id": "2445905950",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Miao Pan",
    "id": "2377317088",
    "h_index": 1,
    "papers": 14
   },
   {
    "name": "Yangfan Xie",
    "id": "2458342621",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Yuxiang Cui",
    "id": "2115440911",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Zhongxiang Zhou",
    "id": "2116589508",
    "h_index": 8,
    "papers": 32
   },
   {
    "name": "Rong Xiong",
    "id": "2274000143",
    "h_index": 2,
    "papers": 13
   },
   {
    "name": "Wenqi Zhang",
    "id": "2135282890",
    "h_index": 19,
    "papers": 48
   },
   {
    "name": "Jianwei Yin",
    "id": "2291142301",
    "h_index": 10,
    "papers": 33
   },
   {
    "name": "Yueting Zhuang",
    "id": "2332353437",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Xuhong Zhang",
    "id": "2389465917",
    "h_index": 1,
    "papers": 17
   }
  ],
  "comment": "",
  "topics": [
   "rl-control",
   "spatial-3d",
   "navigation",
   "foundation-pretraining",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.17512v1",
  "pdf_url": "https://arxiv.org/pdf/2608.17512v1",
  "html_url": "https://arxiv.org/html/2608.17512v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.17496",
  "slug": "calibrated-predictive-safety-for-heterogeneous-robots-an-action-condit",
  "title": "Calibrated Predictive Safety for Heterogeneous Robots: An Action-Conditioned JEPA Framework with Model-Based Safety Shields",
  "abstract": "Vision-language-action policies generalize broadly but provide no execution-time guarantees; classical model-based planners respect kinematic and geometric constraints but generalize poorly. We study whether an action-conditioned Joint-Embedding Predictive Architecture (JEPA) world model can predict, before execution, both task progress and physical risk for candidate action chunks, and whether coupling these predictions to an embodiment-specific model-based safety shield yields a deployable pipeline for heterogeneous robots. We propose a receding-horizon decision pipeline: (1) a proposer produces K candidate action chunks; (2) an action-conditioned JEPA rolls each candidate forward in a frozen-encoder latent space conditioned on an embodiment embedding; (3) calibrated risk and progress heads score each rollout and report uncertainty; (4) a deterministic per-embodiment safety shield filters inadmissible candidates; (5) a fallback ladder handles empty-admissible-set cases. The learned ranking only reorders admissible candidates; enforcement guarantees come from the deterministic shield and fallback ladder. We evaluate with a pre-registered protocol in simulation (LIBERO-Long). In 600-episode configurations the full framework improved success over a shield-only baseline and reduced collision false negatives at matched recall. Deployment-efficiency measurements on target on-robot and edge accelerators are included. Real-robot experiments and an offline reranking significance test remain future work; see the paper for disclosures.",
  "published": "2026-08-18",
  "updated": "2026-08-18",
  "year": "2026",
  "authors": [
   "Kaiming Zhong",
   "Tianhua Liu",
   "Yue Wang"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work studies whether an action-conditioned Joint-Embedding Predictive Architecture world model can predict, before execution, both task progress and physical risk for candidate action chunks, and whether coupling these predictions to an embodiment-specific model-based safety shield yields a deployable pipeline for heterogeneous robots.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Kaiming Zhong",
    "id": "2458133561",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Tianhua Liu",
    "id": "2458332806",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Yue Wang",
    "id": "2458706961",
    "h_index": 0,
    "papers": 0
   }
  ],
  "comment": "17 pages, 9 figures. Simulation-only empirical results on LIBERO-Long (no real-robot experiments). Source, figure-generation scripts and reproducibility checklist included. Level-3 offline reranking significance test not executed; see Sec. 7 (Scope and honesty statement) for detailed disclosure",
  "topics": [
   "world-models",
   "vla",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.17496v1",
  "pdf_url": "https://arxiv.org/pdf/2608.17496v1",
  "html_url": "https://arxiv.org/html/2608.17496v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.17484",
  "slug": "reuse-before-you-retrieve-diagnosing-headroom-and-complementarity-for",
  "title": "Reuse Before You Retrieve: Diagnosing Headroom and Complementarity for Test-Time Augmentation of Embodied Multimodal Policies",
  "abstract": "Frozen vision-language-action (VLA) policies are increasingly improved at test time by sampling additional policy behaviors or introducing external demonstrations. Yet there is little guidance for deciding which intervention a deployed policy actually needs. Additional sampling is useful only when better behavior already exists within the policy's stochastic rollouts and can be identified, whereas retrieval is most useful when the relevant action prior is not reliably represented by the policy. We study this decision through two measurable factors, recoverable headroom and retrieval complementarity, which characterize how much useful behavior is already available to recover and whether an external action prior fills a measurable gap. We evaluate an episode-level retry selector under retryable or parallel execution, together with retrieval across multiple frozen VLA policies and environments. The selector consistently recovers substantial latent capability across all tested VLA backbones on LIBERO, with gains of up to 21.0 success-rate points that closely track recoverable headroom. It also transfers to a different robot and simulator and remains effective under degraded observations, while experiments with autoregressive OpenVLA illustrate the distinction between available headroom and the ability to rank candidate rollouts. Retrieval behaves differently, improving the policy with the largest measured action-prior gap and providing further gains when combined with selection. Together, these results provide an empirical basis for characterizing test-time augmentation opportunities by separating capability that can be recovered from the frozen policy from behavioral priors that may need to be introduced externally.",
  "published": "2026-08-18",
  "updated": "2026-08-18",
  "year": "2026",
  "authors": [
   "Yuhwan Jeong",
   "Kuk-Jin Yoon"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "ECCV 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The results provide an empirical basis for characterizing test-time augmentation opportunities by separating capability that can be recovered from the frozen policy from behavioral priors that may need to be introduced externally.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yuhwan Jeong",
    "id": "2279930538",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Kuk-Jin Yoon",
    "id": "2279721731",
    "h_index": 12,
    "papers": 31
   }
  ],
  "comment": "Accepted to ECCV 2026 workshop",
  "topics": [
   "vla",
   "sim2real"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.17484v1",
  "pdf_url": "https://arxiv.org/pdf/2608.17484v1",
  "html_url": "https://arxiv.org/html/2608.17484v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.17455",
  "slug": "optimal-control-of-a-swimming-robot-based-on-purcell-s-microswimmer-mo",
  "title": "Optimal control of a swimming robot based on Purcell's microswimmer model",
  "abstract": "Purcell's swimmer is a well-known planar model of a swimming microorganism, governed by low Reynolds number hydrodynamics, which is comprised of three rigid links connected by actuated rotary joints. This model has been analyzed as a robotic locomotion system governed by first-order nonlinear dynamics with a periodic input (gait) of the two joint angles. In this work, we present a robotic macro-scale realization of this three-link swimmer moving in a highly viscous fluid. We propose a simple variant of Purcell's theoretical model with non-slender links and a central rigid sphere which represents the added drag of the robot's central flotation block, and calibrate the model's parameters to fit experimental measurements. Next, we apply optimal control formulation based on Pontryagin's Maximum Principle (PMP) in order to find optimal gaits that maximize the displacement per cycle under bounds on the joint angles. Employing a differential geometric method that transforms the problem to area integral enclosed by the gait trajectory in the plane of joint angles, enables visual interpretation which explains topological changes in displacement-optimal gaits upon varying the bound on the joint angles. We then apply PMP formulation to the problem of maximizing Lighthill's energy efficiency in order to obtain a boundary value problem (BVP) whose solution gives efficiency-optimal gaits for Purcell's swimmer model, as well as its variant with a central sphere. Finally, we utilize numerical methods such as parameterizing the input gait as a truncated Fourier series, as well as GPOPS-II solver, to produce sufficient initial guess values for solving the BVPs and obtaining efficiency-optimal gaits.",
  "published": "2026-08-18",
  "updated": "2026-08-18",
  "year": "2026",
  "authors": [
   "Noam Berkovich Lahav",
   "Oren Wiezel",
   "Yizhar Or"
  ],
  "author_count": 3,
  "categories": [
   "physics.flu-dyn",
   "cs.RO",
   "math.OC"
  ],
  "primary_category": "physics.flu-dyn",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Noam Berkovich Lahav",
    "id": "2458131649",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "O. Wiezel",
    "id": "7361812",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Y. Or",
    "id": "1781217",
    "h_index": 21,
    "papers": 97
   }
  ],
  "comment": "",
  "topics": [
   "humanoids"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.17455v1",
  "pdf_url": "https://arxiv.org/pdf/2608.17455v1",
  "html_url": "https://arxiv.org/html/2608.17455v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.17453",
  "slug": "eatr-stereo-embodiment-aware-token-routing-of-paired-stereo-evidence-f",
  "title": "EATR-Stereo: Embodiment-Aware Token Routing of Paired Stereo Evidence for Humanoid Vision-Language-Action Control",
  "abstract": "Long-horizon humanoid vision--language--action (VLA) control with head-mounted stereo cameras requires visual interfaces that can exploit complementary views while maintaining compatibility with pretrained representations. Existing interfaces often discard complementary stereo evidence or fuse additional observations without preserving the native primary-view pathway and adapting auxiliary information to robot embodiment. We present EATR-Stereo, an embodiment-aware token-routing framework that retains primary-view tokens and constructs primary-aligned Cross-View Auxiliary Tokens (CVATs) by querying the synchronized auxiliary-view token sequence. A body-segmented proprioceptive encoder further conditions token-wise auxiliary usage on robot configuration history, enabling selective incorporation of stereo evidence during action generation. The routed auxiliary stream augments the language and primary-visual context of a pretrained VLA while keeping its vision--language model frozen. On a 33-DoF physical humanoid with a 37-D proprioceptive state, we evaluate nine configurations in over-100-s search--approach--grasp--place--return tasks. EATR-Stereo achieves 60.0% full-task success, 100.0% grasp success, and 80.0% stage success. Under severe asymmetric occlusion, it improves recovery to 80% compared with 30% for CVAT alone. Ablation studies further show the importance of preserving primary tokens and combining cross-view auxiliary features with structured proprioceptive routing. These results demonstrate that selectively routed paired stereo evidence improves spatial grounding for reliable long-horizon humanoid VLA control.",
  "published": "2026-08-18",
  "updated": "2026-08-20",
  "year": "2026",
  "authors": [
   "Songwei Wu",
   "Rui Zhao",
   "Fan Yang",
   "Zhongqiang Nie",
   "Zhiduo Jiang",
   "Wandong Sun",
   "Yuwei Li",
   "Jian Hu",
   "Yang Liu",
   "Hong Liu"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "EATR-Stereo is presented, an embodiment-aware token-routing framework that retains primary-view tokens and constructs primary-aligned Cross-View Auxiliary Tokens (CVATs) by querying the synchronized auxiliary-view token sequence.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Songwei Wu",
    "id": "48915105",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Rui Zhao",
    "id": "2153291683",
    "h_index": 12,
    "papers": 20
   },
   {
    "name": "Fan Yang",
    "id": "2307191608",
    "h_index": 4,
    "papers": 15
   },
   {
    "name": "Zhongqiang Nie",
    "id": "2458132953",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Zhiduo Jiang",
    "id": "2363618131",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Wandong Sun",
    "id": "2276039680",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Yuwei Li",
    "id": "2434454841",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Jianda Hu",
    "id": "2441656809",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yang Liu",
    "id": "2292155018",
    "h_index": 4,
    "papers": 23
   },
   {
    "name": "Hong Liu",
    "id": "2311336998",
    "h_index": 3,
    "papers": 11
   }
  ],
  "comment": "8 pages, 5 figures",
  "topics": [
   "vla",
   "dexterous-manipulation",
   "humanoids"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.17453v3",
  "pdf_url": "https://arxiv.org/pdf/2608.17453v3",
  "html_url": "https://arxiv.org/html/2608.17453v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.17432",
  "slug": "unireflex-plug-and-play-force-control-for-pretrained-generative-polici",
  "title": "UniReflex: Plug-and-Play Force Control for Pretrained Generative Policies via Fast-Slow Reflex",
  "abstract": "Generative imitation learning policies excel at trajectory planning but lack closed-loop force regulation, while directly incorporating force modalities often requires redesigning or retraining the network. We present UniReflex, a universal plug-and-play framework that equips frozen generative policies with variable impedance control (VIC) for contact regulation, guided by force-direction intent collected during demonstration, without further slow-backbone fine-tuning. By non-invasively intercepting deep latent representations from the action head, UniReflex drives a fast reflex network that decouples active force exertion from external interaction response. This scheme predicts normalized anisotropic stiffness directions for directional compliance allocation. Furthermore, UniReflex integrates an adaptive gating mechanism that enables seamless transitions between position-dominant planning and force-dominant execution. Real-world bimanual experiments demonstrate that UniReflex significantly improves contact stability and success rates while preserving original position accuracy. Our approach achieves 25-66x lower per-step backward latency relative to joint training strategies on the evaluated backbones.",
  "published": "2026-08-18",
  "updated": "2026-08-18",
  "year": "2026",
  "authors": [
   "Yan Huang",
   "Shoujie Li",
   "Ziwu Song",
   "Wenbo Ding"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "UniReflex is a universal plug-and-play framework that equips frozen generative policies with variable impedance control (VIC) for contact regulation, guided by force-direction intent collected during demonstration, without further slow-backbone fine-tuning.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yan Huang",
    "id": "2283305599",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Shoujie Li",
    "id": "2155443623",
    "h_index": 11,
    "papers": 49
   },
   {
    "name": "Ziwu Song",
    "id": "2148838682",
    "h_index": 9,
    "papers": 18
   },
   {
    "name": "Wenbo Ding",
    "id": "2265840777",
    "h_index": 3,
    "papers": 6
   }
  ],
  "comment": "",
  "topics": [
   "imitation-diffusion"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.17432v1",
  "pdf_url": "https://arxiv.org/pdf/2608.17432v1",
  "html_url": "https://arxiv.org/html/2608.17432v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.17423",
  "slug": "prism-grpo-faster-vla-policy-optimization-via-splitting-same-outcome-g",
  "title": "Prism-GRPO: Faster VLA Policy Optimization via Splitting Same-outcome Groups",
  "abstract": "GRPO is increasingly used for reinforcement learning of vision-language-action (VLA) policies because, unlike PPO, it does not require training a critic. This simplification comes with a sampling cost: group-relative advantages require multiple rollouts from each scene. Under binary success rewards, groups whose rollouts all succeed or all fail have zero advantage and are discarded by dynamic sampling. These groups are especially common early in training, when most rollouts fail, wasting much of the expensive robotic rollout budget. We introduce Prism-GRPO, which augments binary outcome reward with a weighted trajectory-level execution-quality score. By splitting same-outcome groups into a quality spectrum, Prism-GRPO recovers training signal while ensuring that every success still outranks every failure. Quality scores can be derived from simulator contacts, executed actions, or visual observations, avoiding task-specific progress rewards. We prove that Prism-GRPO never increases the probability that a sampled group is discarded for having zero advantages, and derive a gradient-alignment condition under which its combined update remains a local ascent direction for task success. Across four RoboTwin tasks spanning different horizons and coordination patterns, Prism-GRPO improves success and quality at matched rollout budgets and reaches target success rates with up to 56% fewer rollouts. It also suppresses a reward-hacking shortcut, with the cleaner behavior transferring under direct deployment to a real robot. Through ablations, we show consistent gains across contact-, smoothness-, and VLM-derived quality signals.",
  "published": "2026-08-18",
  "updated": "2026-08-18",
  "year": "2026",
  "authors": [
   "Zeyun Deng",
   "Yuzhe Lu",
   "Yawei Wang",
   "Linbo Liu",
   "Qing Ping",
   "Han Ding",
   "Guande Wu",
   "Panpan Xu",
   "Jun Huan"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Across four RoboTwin tasks spanning different horizons and coordination patterns, Prism-GRPO improves success and quality at matched rollout budgets and reaches target success rates with up to 56% fewer rollouts.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zeyun Deng",
    "id": "2342368680",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Yuzhe Lu",
    "id": "2269475690",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Yawei Wang",
    "id": "2385732217",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Linbo Liu",
    "id": "2243894780",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Qing Ping",
    "id": "2280574284",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Han Ding",
    "id": "2267679737",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Guande Wu",
    "id": "2153300135",
    "h_index": 9,
    "papers": 19
   },
   {
    "name": "Panpan Xu",
    "id": "2248954229",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Jun Huan",
    "id": "2316996088",
    "h_index": 3,
    "papers": 8
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "sim2real",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.17423v1",
  "pdf_url": "https://arxiv.org/pdf/2608.17423v1",
  "html_url": "https://arxiv.org/html/2608.17423v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.17416",
  "slug": "bi-layer-ant-colony-optimization-for-multi-robot-task-allocation-and-r",
  "title": "Bi-Layer Ant Colony Optimization for Multi-Robot Task Allocation and Routing in Delivery Applications",
  "abstract": "This paper addresses the multi-robot task allocation (MRTA) problem, which is essential for delivery and logistics applications. Our approach first defines a new cost function that transforms the MRTA into a unified optimization problem capturing both task assignment and routing. A bi-layer ant colony optimization (ACO) algorithm is then introduced, integrating two interdependent decision layers within a single colony process to solve the problem. This hierarchical framework enables simultaneous optimization of task allocation and route planning across multiple robots. Comparative experiments with mixed-integer linear programming (MILP) and particle swarm optimization (PSO) demonstrate that the proposed bi-layer ACO achieves the shortest total travel distance and fastest completion time across all task sizes. Specifically, it reduces total travel distance by up to 17.7% and completion time by nearly 20% compared with baseline methods. These results confirm the efficiency, scalability, and reliability of the proposed bi-layer ACO for multi-robot delivery tasks.",
  "published": "2026-08-18",
  "updated": "2026-08-18",
  "year": "2026",
  "authors": [
   "Le Na Nguyen",
   "Thanh Long Nguyen",
   "Thanh Thao Ton Nu",
   "Quan Le",
   "Manh Duong Phung"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Comparative experiments with mixed-integer linear programming (MILP) and particle swarm optimization (PSO) demonstrate that the proposed bi-layer ACO achieves the shortest total travel distance and fastest completion time across all task sizes.",
  "doi": "10.1145/3805862.3805901",
  "oa_pdf": "https://doi.org/10.1145/3805862.3805901",
  "s2_authors": [
   {
    "name": "Le-Minh-Kha Nguyen",
    "id": "2445088790",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Thanh Long Nguyen",
    "id": "2326019201",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Thanh Thao Ton Nu",
    "id": "2458056771",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Q. Le",
    "id": "1631718115",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "M. D. Phung",
    "id": "3102919",
    "h_index": 17,
    "papers": 70
   }
  ],
  "comment": "6 pages. Accepted at 2026 11th International Conference on Intelligent Information Technology (ICIIT 2026)",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.17416v1",
  "pdf_url": "https://arxiv.org/pdf/2608.17416v1",
  "html_url": "https://arxiv.org/html/2608.17416v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.17386",
  "slug": "maniguard-a-benchmark-and-data-suite-for-specification-grounded-safety",
  "title": "MANIGUARD: A Benchmark and Data Suite for Specification-Grounded Safety Evaluation and Improvement of Robotic Manipulation",
  "abstract": "Foundation-model policies for robotic manipulation are advancing rapidly on task success, but rigorous evaluation of whether they succeed safely is still lacking. We introduce ManiGuard, a specification-grounded framework for evaluating and improving the safety of foundation-model manipulation, comprising the ManiGuard-Bench task suite and a paired safety-annotated trajectory-generation pipeline. ManiGuard-Bench organizes six contact-rich household task families into 200 locked base tasks along a skill $\\times$ constraint taxonomy, with safety specified independently of task success. Each task is evaluated under one in-distribution and four single-axis out-of-distribution perturbations that hold the safety specification fixed, giving 1,000 locked scenarios. Every rollout is runtime-checked by LTL$_f$-grounded automaton monitors over physics-grounded predicates rather than learned classifiers or LLM judges, in simulation and on a physical Franka platform. The pipeline pairs an automated motion-planning generator with human teleoperation, annotated by the same per-step monitor, and directly supports safety-aware fine-tuning; we release 8,000 safety-annotated demonstrations, 40 per base task. Benchmarking zero-shot and fine-tuned VLAs across more than 23,000 rollouts, we find: (i) safety must be evaluated independently of task success, as 6-21% of successful rollouts violate the specification; (ii) fine-tuning on our suite raises safe task completion from near zero to 7.5-29.8% and engaged-and-safe behavior from 16-40% to 51-72%; but (iii) a gap remains that scaling demonstrations does not close, with 21-42% of engaged rollouts still violating, two of six families below 2% safe success for every policy, and these failures persisting under distribution shift and on hardware.",
  "published": "2026-08-18",
  "updated": "2026-08-18",
  "year": "2026",
  "authors": [
   "Yiyan Peng",
   "Philip Wang",
   "Simon Sinong Zhan",
   "Yiqi Lyu",
   "Zhenyang Ni",
   "Jixin Yan",
   "Fiorelli Wong",
   "Ruochen Jiao",
   "Hang Yin",
   "Xinyu Cao",
   "Huajie Shao",
   "Manling Li",
   "Ruohan Zhang",
   "Qi Zhu"
  ],
  "author_count": 14,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yi Peng",
    "id": "2457280273",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Philip Wang",
    "id": "2324800564",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "S. Zhan",
    "id": "2186740041",
    "h_index": 5,
    "papers": 15
   },
   {
    "name": "Yiqi Lyu",
    "id": "2328006223",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Zhenyang Ni",
    "id": "2162837304",
    "h_index": 7,
    "papers": 12
   },
   {
    "name": "Jixin Yan",
    "id": "2458315725",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Fiorelli Wong",
    "id": "2458133703",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Ruochen Jiao",
    "id": "2371991057",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Hang Yin",
    "id": "2292126834",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Xinyu Cao",
    "id": "2386391634",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Huajie Shao",
    "id": "2282958256",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Manling Li",
    "id": "2386131426",
    "h_index": 3,
    "papers": 18
   },
   {
    "name": "Ruohan Zhang",
    "id": "2344793490",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Qi Zhu",
    "id": "2385785912",
    "h_index": 2,
    "papers": 3
   }
  ],
  "comment": "",
  "topics": [
   "tactile",
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.17386v1",
  "pdf_url": "https://arxiv.org/pdf/2608.17386v1",
  "html_url": "https://arxiv.org/html/2608.17386v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.17347",
  "slug": "repetition-as-reinforcement-enhancing-sample-efficiency-via-instant-ep",
  "title": "Repetition as Reinforcement: Enhancing Sample Efficiency via Instant Episode Repetition in Reinforcement Learning",
  "abstract": "Repetition is a fundamental mechanism in human learning, where revisiting successful experiences strengthens memory, consolidates skills, and improves future performance. Motivated by this biological principle, we introduce Instant Episode Repetition (IER), a simple and novel mechanism that improves sample efficiency by immediately repeating action sequences from successful episodes during environment interaction. Unlike conventional approaches such as Experience Replay and Self-Imitation Learning (SIL), which passively reuse past experience during training updates, IER directly influences the data collection process. Upon identifying a high-reward episode, the agent repeats its action sequence for a fixed number of subsequent episodes, reinforcing valuable behaviors through renewed interaction with the environment. We integrate IER into state-of-the-art SAC and TD3 algorithms and evaluate its effectiveness on continuous-control benchmarks, including MuJoCo, the DeepMind Control Suite, and a real-world dynamic object translation task with a robotic manipulator. Experimental results demonstrate that this simple mechanism improves learning performance over standard and self-imitation-based baselines.",
  "published": "2026-08-18",
  "updated": "2026-08-18",
  "year": "2026",
  "authors": [
   "Hoda Yamani",
   "Yuning Xing",
   "Koen van Rijnsoever",
   "Bruce A. MacDonald",
   "Henry Williams"
  ],
  "author_count": 5,
  "categories": [
   "cs.LG",
   "cs.RO"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Instant Episode Repetition (IER) is introduced, a simple and novel mechanism that improves sample efficiency by immediately repeating action sequences from successful episodes during environment interaction by directly influences the data collection process.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hoda Yamani",
    "id": "2342961154",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yuning Xing",
    "id": "2314930035",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Koen van Rijnsoever",
    "id": "2317012169",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Bruce A. MacDonald",
    "id": "2248194039",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Henry Williams",
    "id": "2300088987",
    "h_index": 3,
    "papers": 5
   }
  ],
  "comment": "23 pages, 12 figures. Accepted at RLC 2026; to appear in Reinforcement Learning Journal (RLJ) 2026. Code: https://github.com/UoA-CARES/instant-episode-repetition",
  "topics": [
   "imitation-diffusion",
   "rl-control",
   "data-teleop"
  ],
  "orgs": [
   "Google DeepMind"
  ],
  "abs_url": "https://arxiv.org/abs/2608.17347v1",
  "pdf_url": "https://arxiv.org/pdf/2608.17347v1",
  "html_url": "https://arxiv.org/html/2608.17347v1",
  "code_url": "https://github.com/UoA-CARES/instant-episode-repetition",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.17324",
  "slug": "reconfiguration-complete-motion-primitives-with-constructive-planning",
  "title": "Reconfiguration-Complete Motion Primitives with Constructive Planning for Deformable Planar Modular Robots",
  "abstract": "The continuously deformable geometry of modular robots makes it difficult to define a fixed representation for reconfiguration planning and analysis. This letter introduces a square-cell abstraction that maps deformable rhombus modules to fixed-size grid cells while retaining physically interpretable local motions through two primitives, pivoting and shearing. Under this abstraction, we prove that every non-straight edge-connected configuration with $N \\geq 7$ can be transformed to a fixed canonical staircase using only admissible primitive motions. Since these motions are reversible, any two configurations in this class are mutually reconfigurable. The proof is constructive and directly yields a staircase-canonicalization planner that transports removable boundary modules while preserving connectivity. As a practical enhancement, we further introduce a boundary-to-delivery lookahead selector that ranks admissible high level choices without affecting the completeness guarantee. Experiments demonstrate the constructive reconfiguration process and show that the selector substantially reduces planning time, while reference comparisons indicate lower planning times than the prior framework over the shared module counts.",
  "published": "2026-08-18",
  "updated": "2026-08-18",
  "year": "2026",
  "authors": [
   "Jie Gu",
   "Tingting Wang",
   "Hongrun Gao",
   "Yirun Sun",
   "Zhihao Xia",
   "Chunxu Tian",
   "Dan Zhang"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This letter introduces a square-cell abstraction that maps deformable rhombus modules to fixed-size grid cells while retaining physically interpretable local motions through two primitives, pivoting and shearing, and proves that every non-straight edge-connected configuration can be transformed to a fixed canonical staircase.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jie Gu",
    "id": "2347843325",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Tingting Wang",
    "id": "2454983140",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Hong Gao",
    "id": "143666865",
    "h_index": 4,
    "papers": 22
   },
   {
    "name": "Yirun Sun",
    "id": "2407127073",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Zhihao Xia",
    "id": "2072881082",
    "h_index": 7,
    "papers": 19
   },
   {
    "name": "Chunxu Tian",
    "id": "101603779",
    "h_index": 14,
    "papers": 47
   },
   {
    "name": "Dan Zhang",
    "id": "2408224872",
    "h_index": 2,
    "papers": 4
   }
  ],
  "comment": "Jie Gu and Tingting Wang contributed equally to this work",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.17324v1",
  "pdf_url": "https://arxiv.org/pdf/2608.17324v1",
  "html_url": "https://arxiv.org/html/2608.17324v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.17323",
  "slug": "orpa-online-residual-policy-adaptation-for-robot-manipulation-control",
  "title": "ORPA: Online Residual Policy Adaptation for Robot Manipulation Control with Human Feedback",
  "abstract": "Robotic manipulation policies trained via imitation learning, such as Action Chunking with Transformers (ACT), can achieve strong performance under ideal conditions but often remain sensitive to small execution errors and distribution shifts. Correcting these failures typically requires dataset aggregation and full-policy retraining, which is computationally expensive and unsuitable for real-time deployment. In this work, we propose Online Residual Policy Adaptation (ORPA), a framework that enables immediate, feedback-driven correction of robot actions without modifying the underlying policy parameters. ORPA augments a pretrained control policy with a lightweight, feedback-conditioned module that predicts residual adjustments directly in joint space, allowing the system to adapt its behavior at runtime. We evaluate ORPA on a set of precision-sensitive manipulation tasks using the ALOHA platform, demonstrating improvements in success rate and recovery from small perturbations compared to baseline control policies and rule-based inverse kinematics corrections.",
  "published": "2026-08-18",
  "updated": "2026-08-18",
  "year": "2026",
  "authors": [
   "Muhammad A. Muttaqien",
   "Tomohiro Motoda",
   "Ryo Hanai",
   "Yukiyasu Domae"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes Online Residual Policy Adaptation (ORPA), a framework that enables immediate, feedback-driven correction of robot actions without modifying the underlying policy parameters.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Muhammad A. Muttaqien",
    "id": "2345017657",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Tomohiro Motoda",
    "id": "2328411976",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Ryo Hanai",
    "id": "2243398622",
    "h_index": 3,
    "papers": 17
   },
   {
    "name": "Y. Domae",
    "id": "2512607",
    "h_index": 13,
    "papers": 128
   }
  ],
  "comment": "",
  "topics": [
   "imitation-diffusion"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.17323v1",
  "pdf_url": "https://arxiv.org/pdf/2608.17323v1",
  "html_url": "https://arxiv.org/html/2608.17323v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.17320",
  "slug": "robust-brachiation-on-a-life-sized-dual-arm-robot-using-waypoint-guide",
  "title": "Robust Brachiation on a Life-Sized Dual-Arm Robot Using Waypoint-Guided Reinforcement Learning",
  "abstract": "Brachiation is a form of locomotion in which primates move primarily using their arms, enabling traversal in environments without footholds. However, this motion requires highly coordinated whole-body movement and precise timing control for bar grasping and release. As a result, achieving robust behavior on life-sized robotic platforms remains challenging. In this study, we present a reinforcement learning-based method to realize brachiation on a life-sized dual-arm robot. The core of the proposed approach is Waypoint-Guided Reinforcement Learning (WGRL), a learning framework for inducing non-linear and complex motions. For high-difficulty tasks where imitation learning data are unavailable, WGRL guides behavior acquisition by sparsely specifying waypoints for the end-effector trajectory, while whole-body motion is generated through reinforcement learning. In addition, by integrating the waypoint-following guidance with rewards based on task success and mechanical energy, and training in an environment designed for Sim-to-Real transfer, the proposed method achieves both forward progression and motion stability. The acquired behavior is evaluated through Sim-to-Sim experiments under monkey-bar environments with geometric variations and hardware experiments, confirming robust brachiation including failure recovery behavior. This study provides effective learning design guidelines for realizing arm-based locomotion on life-sized robotic hardware and expanding the traversable workspace of robots.",
  "published": "2026-08-18",
  "updated": "2026-08-18",
  "year": "2026",
  "authors": [
   "Ayumu Iwata",
   "Kento Kawaharazuka",
   "Keita Yoneda",
   "Takahiro Hattori",
   "Kei Okada"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Effective learning design guidelines for realizing arm-based locomotion on life-sized robotic hardware and expanding the traversable workspace of robots are provided.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ayumu Iwata",
    "id": "2376541386",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Kento Kawaharazuka",
    "id": "8308607",
    "h_index": 17,
    "papers": 220
   },
   {
    "name": "Keita Yoneda",
    "id": "2365839163",
    "h_index": 1,
    "papers": 15
   },
   {
    "name": "Takahiro Hattori",
    "id": "2365802659",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Kei Okada",
    "id": "2248244895",
    "h_index": 5,
    "papers": 68
   }
  ],
  "comment": "Accepted to 2026 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS)",
  "topics": [
   "dexterous-manipulation",
   "humanoids",
   "sim2real",
   "imitation-diffusion",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.17320v1",
  "pdf_url": "https://arxiv.org/pdf/2608.17320v1",
  "html_url": "https://arxiv.org/html/2608.17320v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.17318",
  "slug": "if-then-otherwise-diagnosing-conditional-branching-in-vision-language",
  "title": "If, Then, Otherwise: Diagnosing Conditional Branching in Vision-Language Navigation",
  "abstract": "Vision-language navigation agents are often evaluated on their ability to follow route-like instructions toward a fixed goal. Yet, real navigation instructions often depend on observed states of the environment: if a condition holds, then follow one path, otherwise take another. Such instructions require an agent to evaluate scene evidence, select the correct logical branch, and execute the corresponding navigation behavior. Existing evaluations provide limited control over conditional branch execution, making it difficult to determine whether agents fail because of perception, grounding, navigation, or logical decision-making. We introduce CondVLN, a scene-graph-grounded benchmark for diagnosing conditional branching in vision-language navigation. CondVLN programmatically generates instructions whose branch conditions are grounded in verifiable 3D scene-graph predicates, with controlled variation in branch depth, dependency chain length, spatial composition, evidence observability, and instruction horizon. CondVLN contains over 11,500 generated conditional instructions across AI2-THOR, Matterport3D, Gibson, and ReplicaCAD, and evaluates agents using standard VLN metrics and branch-specific diagnostics: Branch Selection Accuracy and Conditional Success Rate. Evaluating four state-of-the-art VLN agents (VLN-Zero, NaVid, NaVILA, and Open-Nav) shows that conditional branching exposes failures that are not captured by standard success rate or path length alone: agents can navigate plausibly while committing to a branch inconsistent with the observed scene condition. We also present a lightweight neurosymbolic branch-selection model that separates condition grounding from navigation execution, improving performance by 2x. CondVLN provides a reusable testbed for measuring whether embodied agents can not only follow instructions, but follow the right instruction under the right condition.",
  "published": "2026-08-18",
  "updated": "2026-08-18",
  "year": "2026",
  "authors": [
   "Seoyoung Lee",
   "Neel P. Bhatt",
   "Pranay Samineni",
   "Cong Liu",
   "S P Sharan",
   "Timothy Barclay",
   "Gregory M. Wagner",
   "Daniel Milan",
   "Sandeep Chinchali",
   "Ufuk Topcu",
   "Atlas Wang"
  ],
  "author_count": 11,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Evaluating four state-of-the-art VLN agents shows that conditional branching exposes failures that are not captured by standard success rate or path length alone: agents can navigate plausibly while committing to a branch inconsistent with the observed scene condition.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Seoyoung Lee",
    "id": "2310053219",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "N. Bhatt",
    "id": "2042522543",
    "h_index": 11,
    "papers": 37
   },
   {
    "name": "Pranay Samineni",
    "id": "2381731710",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Cong Liu",
    "id": "2455459818",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "S. Sharan",
    "id": "2421233871",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Timothy Barclay",
    "id": "2405811202",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Gregory M. Wagner",
    "id": "2458133552",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "D. Milan",
    "id": "2329098703",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Sandeep P. Chinchali",
    "id": "2277751990",
    "h_index": 14,
    "papers": 111
   },
   {
    "name": "U. Topcu",
    "id": "3199888",
    "h_index": 54,
    "papers": 610
   },
   {
    "name": "Atlas Wang",
    "id": "2375863912",
    "h_index": 3,
    "papers": 4
   }
  ],
  "comment": "11 pages, 1 figure, 3 tables. Project page: https://condvln.github.io/",
  "topics": [
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.17318v1",
  "pdf_url": "https://arxiv.org/pdf/2608.17318v1",
  "html_url": "https://arxiv.org/html/2608.17318v1",
  "code_url": "https://condvln.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2608.17209",
  "slug": "teach-and-grow-an-agent-centered-architecture-for-general-robot-learni",
  "title": "Teach and Grow: An Agent-Centered Architecture for General Robot Learning",
  "abstract": "End-to-end vision-language-action (VLA) and world-action models offer an elegant route to general-purpose robotics, but their reliability is bounded by validated physical coverage. When an unfamiliar object, sensor, embodiment, or contact falls outside that coverage and no validated fallback exists, correcting the failure requires new robot data, a policy update, and regression testing. This recurring burden is the retraining tax. Unlike text, embodied data must often be created by operating machines. We present Teach-and-Grow Learning (TGL), an agent-centered architecture for general robot learning. In its general form, a multimodal agent turns a few successful demonstrations into reusable Skill Blocks: closed-loop behaviors for meaningful subgoals. In a new scene, the agent grounds and composes these blocks, selects learned or geometric tools, observes the physical outcome, and revises the route when execution departs from intent. A Skill Library stores executable behavior, while structured Experience Memory carries forward success, failure, and repair. New tasks are acquired without task-specific policy retraining. Our LIBERO evaluation attains state-of-the-art performance; controlled studies expose skill induction, persistent reuse, and agent-directed adaptation. Finally, we propose the Teach-and-Grow scaling-law hypothesis: if X denotes effective reusable experience, future-task error and teaching demand should approach irreducible floors as power laws in X. The architecture therefore treats deployment as a period of continued learning, in which one task can make the next easier.",
  "published": "2026-08-17",
  "updated": "2026-08-17",
  "year": "2026",
  "authors": [
   "Chang Nie",
   "Zhe Liu",
   "Hesheng Wang"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work presents Teach-and-Grow Learning (TGL), an agent-centered architecture for general robot learning, and proposes the Teach-and-Grow scaling-law hypothesis: if X denotes effective reusable experience, future-task error and teaching demand should approach irreducible floors as power laws in X.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Chang Nie",
    "id": "2354334874",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Zhe Liu",
    "id": "2274060921",
    "h_index": 8,
    "papers": 40
   },
   {
    "name": "Hesheng Wang",
    "id": "2316674851",
    "h_index": 7,
    "papers": 28
   }
  ],
  "comment": "",
  "topics": [
   "vla"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.17209v1",
  "pdf_url": "https://arxiv.org/pdf/2608.17209v1",
  "html_url": "https://arxiv.org/html/2608.17209v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.17146",
  "slug": "pddl-art-autonomous-symbolic-abstraction-from-demonstration-for-long-h",
  "title": "PDDL-ART: Autonomous Symbolic Abstraction From Demonstration For Long-Horizon Robotic Manipulation Using Vision-Language Models",
  "abstract": "Symbolic planning with PDDL offers a principled framework for long-horizon robot manipulation, but constructing accurate PDDL domain and problem descriptions remains a significant bottleneck, typically requiring substantial domain expertise. We present a Vision-Language Model (VLM)-based approach called PDDL-ART, a framework that autonomously generates task-specific PDDL domain and problem descriptions from a single expert demonstration, a natural language task description, and a library of available high-level action names. PDDL-ART does not require any domain templates, action signatures, or fine-tuning. To ensure the generated descriptions are not only syntactically valid but semantically aligned with the demonstrated task, PDDL-ART introduces a multi-stage correction pipeline operating at syntactic, semantic, and execution levels. A key component of execution-guided correction is symbolic predicate grounding. Instead of relying solely on visual observations, PDDL-ART leverages the tool-use capabilities of modern VLMs to incorporate geometric and temporal reasoning for evaluating relational predicates that are not directly discernible from images alone. Critically, the model autonomously determines when to invoke these tools and how to interpret their outputs. We evaluate PDDL-ART on challenging manipulation tasks in engine maintenance and household domains, including tasks that require memory, abstract predicate inference, and goal states that are visually indistinguishable from the initial state. PDDL-ART achieves an average success rate of 93.3%, compared to 78.3% for a baseline VLM-based planner.",
  "published": "2026-08-17",
  "updated": "2026-08-17",
  "year": "2026",
  "authors": [
   "Disha Kamale",
   "Dmitry Berenson"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "PDDL-ART is a framework that autonomously generates task-specific PDDL domain and problem descriptions from a single expert demonstration, a natural language task description, and a library of available high-level action names, which leverages the tool-use capabilities of modern VLMs to incorporate geometric and temporal reasoning for evaluating relational predicates that are not directly discernible from images alone.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Disha Kamale",
    "id": "1994434963",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "D. Berenson",
    "id": "1747706",
    "h_index": 40,
    "papers": 131
   }
  ],
  "comment": "",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.17146v1",
  "pdf_url": "https://arxiv.org/pdf/2608.17146v1",
  "html_url": "https://arxiv.org/html/2608.17146v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.17129",
  "slug": "probe-manipulation-grounded-visual-question-answering-with-vlm-agents",
  "title": "PROBE: Manipulation-Grounded Visual Question Answering with VLM Agents",
  "abstract": "Vision-language Models (VLMs) excel at 2D grounding, spatial reasoning and agentic tool-based planning in static scenes. However, consider asking a home robot \"Is my medication still in the cabinet?\" The answer may be physically hidden behind a row of containers that must first be moved aside. Answering such questions in real-world cluttered environments requires reasoning in dynamic scenes: distractors must be manipulated to reveal occluded objects, and each action changes the scene the model must reason over. We formalize this setting as Manipulation-Grounded Visual Question Answering (MG-VQA) and introduce PROBE, a framework for benchmarking and finetuning VLM agents on such tasks. We first develop PROBE-Sim, a high-fidelity tabletop simulator with everyday objects and a robot manipulator equipped with grasping and pushing tools. PROBE-Sim is used to create PROBE-Bench: an evaluation suite of 150 tasks across 6 question types on cluttered tabletop scenes, where a VLM perceives, picks up or pushes objects before answering. We observe consistent trend across all frontier VLMs: agentic tool-based methods outperform their perception-only baselines (8.0% on average) across all task types. We further design PROBE-Agent, a finetuning recipe to distill successful trajectories from a powerful teacher foundation model to a smaller open-weight model using a mixed data recipe that encourages manipulation-efficient question answering. PROBE Agent finetuned models outperform their off-the-shelf agent baseline (11.5% on average) and demonstrate positive transfer to unseen objects and a held-out task. We validate sim-to-real transfer by deploying PROBE-Agent finetuned policies in real-world tabletop environments.",
  "published": "2026-08-17",
  "updated": "2026-08-17",
  "year": "2026",
  "authors": [
   "Vineet Bhat",
   "Siyi Chen",
   "Alex Zook",
   "Xuning Yang",
   "Stan Birchfield",
   "Valts Blukis",
   "Jonathan Tremblay"
  ],
  "author_count": 7,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work develops PROBE, a framework for benchmarking and finetuning VLM agents on cluttered tabletop scenes, and designs PROBE-Agent, a finetuning recipe to distill successful trajectories from a powerful teacher foundation model to a smaller open-weight model using a mixed data recipe that encourages manipulation-efficient question answering.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "V. Bhat",
    "id": "2289874504",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Siyi Chen",
    "id": "2319784944",
    "h_index": 7,
    "papers": 19
   },
   {
    "name": "Alex Zook",
    "id": "2322986327",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Xuning Yang",
    "id": "9575840",
    "h_index": 9,
    "papers": 22
   },
   {
    "name": "Stanley T. Birchfield",
    "id": "2257232566",
    "h_index": 9,
    "papers": 24
   },
   {
    "name": "Valts Blukis",
    "id": "2235737945",
    "h_index": 12,
    "papers": 22
   },
   {
    "name": "Jonathan Tremblay",
    "id": "2294175839",
    "h_index": 6,
    "papers": 21
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "sim2real",
   "spatial-3d",
   "foundation-pretraining",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.17129v1",
  "pdf_url": "https://arxiv.org/pdf/2608.17129v1",
  "html_url": "https://arxiv.org/html/2608.17129v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.17038",
  "slug": "terrain-aware-local-path-planning-with-global-dem-data-integration-for",
  "title": "Terrain-Aware Local Path Planning with Global DEM Data Integration for Autonomous UGV Navigation",
  "abstract": "Autonomous navigation in complex outdoor terrains presents critical challenges for unmanned ground vehicles (UGVs) due to the inherent disconnect between global mapping and real-time sensor feedback. This work proposes a hybrid framework that integrates low-resolution Digital Elevation Model (DEM) data with real-time LiDAR-based obstacle detection and terrain analysis for efficient path planning. A global path is initially computed using a preprocessed DEM-based A* algorithm. Subsequently, local sensor data drives adaptive path correction, enabling the UGV to negotiate sudden environmental changes while maintaining safety and efficiency. Simulation results in Gazebo demonstrate significant improvements over a baseline approach, achieving a 95\\% obstacle avoidance rate and reducing the average encountered slope from $8^\\circ$ to $2.7^\\circ$ in custom terrain. This integration enhances path efficiency and terrain traversability and supports robust real-time adaptation, paving the way for more reliable autonomous navigation in dynamic outdoor environments.",
  "published": "2026-08-17",
  "updated": "2026-08-17",
  "year": "2026",
  "authors": [
   "Devender Singh",
   "Issah Nazif Suleiman",
   "Paul Mitten",
   "Glenn Cutler",
   "Vinicius Prado da Fonseca",
   "Matthew Hamilton"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes a hybrid framework that integrates low-resolution Digital Elevation Model (DEM) data with real-time LiDAR-based obstacle detection and terrain analysis for efficient path planning and supports robust real-time adaptation, paving the way for more reliable autonomous navigation in dynamic outdoor environments.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Devender Singh",
    "id": "2458330471",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Issah N. Suleiman",
    "id": "103678771",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Paul Mitten",
    "id": "2458133197",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Glenn Cutler",
    "id": "2458133458",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Vinicius Prado da Fonseca",
    "id": "11245008",
    "h_index": 9,
    "papers": 28
   },
   {
    "name": "M. Hamilton",
    "id": "2376340650",
    "h_index": 2,
    "papers": 10
   }
  ],
  "comment": "",
  "topics": [
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.17038v1",
  "pdf_url": "https://arxiv.org/pdf/2608.17038v1",
  "html_url": "https://arxiv.org/html/2608.17038v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.17030",
  "slug": "lambda-hold-control-human-like-movement-emerges-from-a-minimal-task-re",
  "title": "Lambda-Hold Control: Human-Like Movement Emerges from a Minimal Task Reward in Predictive Musculoskeletal Simulation",
  "abstract": "The massive overactuation in the human musculoskeletal system makes it challenging to train musculoskeletal models to generate human-like motion via reinforcement learning, primarily because exploration in the resulting high-dimensional and redundant action space is extremely inefficient. To address this problem, we propose the $\u03bb$-hold controller, inspired by the equilibrium-point (EP) hypothesis, which has been widely supported by extensive evidence from human motor control studies. The policy's control variable is the per-muscle EP threshold length $\u03bb$, from which a stretch-reflex recruitment law computes the muscle excitations automatically. Holding each $\u03bb$ over an interval of the gait phase also sharply reduces the frequency at which the policy must be queried. Consequently, the controller, to our knowledge for the first time, enables a muscle-actuated skeletal model to learn human-like sprinting using only a minimal reward within an hour of training. The efficient exploration through the proposed $\u03bb$-hold controller is not merely an engineering trick but an approach grounded in physiology, bringing together the EP hypothesis, intermittent control, and optimal feedback control. Beyond encapsulating human-like behavior in predictive simulation, this achievement contributes to developing a learnable model of the human motor controller.",
  "published": "2026-08-17",
  "updated": "2026-08-17",
  "year": "2026",
  "authors": [
   "Jun Hyuk Lee",
   "Chihyeong Lee",
   "Jooeun Ahn"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.GR",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jun Hyuk Lee",
    "id": "2232272612",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Chihyeong Lee",
    "id": "2249533637",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Jooeun Ahn",
    "id": "2239194904",
    "h_index": 5,
    "papers": 20
   }
  ],
  "comment": "19 pages, 8 figures, 1 table. Project page and video demos: https://lee-jun-hyuk-37.github.io/projects/lambda-hold/",
  "topics": [
   "rl-control",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.17030v1",
  "pdf_url": "https://arxiv.org/pdf/2608.17030v1",
  "html_url": "https://arxiv.org/html/2608.17030v1",
  "code_url": "https://lee-jun-hyuk-37.github.io/projects/lambda-hold/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2608.17027",
  "slug": "fetchman-learning-visual-humanoid-loco-manipulation-policies-from-simu",
  "title": "FetchMan: Learning Visual Humanoid Loco-Manipulation Policies from Simulated Experiences",
  "abstract": "Visual loco-manipulation policies that can generalize to novel scenes and objects have long been a goal of robotics research. However, today's data-hungry algorithms make collecting sufficient demonstrations a struggle for tabletop manipulation, and even more so for humanoids that must also walk and balance. Learning from simulated data and transferring that behavior to the real world, as is commonly done in locomotion, sidesteps this struggle, so we replicate that recipe for loco-manipulation. In doing so, we find that cloning synthetic demonstrations results in a low performance ceiling no matter the amount of training data. Reinforcement learning breaks through it, and refining the cloned policy with Flow-GRPO on a single sparse reward yields performance that synthetic behavior cloning cannot match. Together, these stages form our end-to-end sim-to-real pipeline spanning more than 150,000 scenes, which we use to train FetchMan. We evaluate it on FetchMan-Bench, a simulation benchmark we release, and deploy it zero-shot on a real Unitree G1, where our single-object reach-and-pick policy walks to and grasps a target across unseen scenes at 73.3% success. Finally, we extend this recipe to multi-object training, a first step toward loco-manipulation generalist policies at this data scale.",
  "published": "2026-08-17",
  "updated": "2026-08-17",
  "year": "2026",
  "authors": [
   "Omar Rayyan",
   "Zhi Li",
   "Max Argus",
   "Yuxin Jiang",
   "Chang Yu",
   "Chenfanfu Jiang",
   "Yuchen Cui"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This recipe for loco-manipulation generalist policies is replicated and extended to multi-object training, a first step toward loco-manipulation generalist policies at this data scale.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Omar Rayyan",
    "id": "2409827952",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Zhi Li",
    "id": "2458549386",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Max Argus",
    "id": "49965376",
    "h_index": 16,
    "papers": 34
   },
   {
    "name": "Yuxin Jiang",
    "id": "2361654632",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Chang Yu",
    "id": "2281793578",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Chenfanfu Jiang",
    "id": "2267866705",
    "h_index": 16,
    "papers": 77
   },
   {
    "name": "Yuchen Cui",
    "id": "2238151901",
    "h_index": 9,
    "papers": 23
   }
  ],
  "comment": "Project website: https://orayyan.com/fetchman",
  "topics": [
   "dexterous-manipulation",
   "humanoids",
   "sim2real",
   "imitation-diffusion",
   "rl-control"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2608.17027v1",
  "pdf_url": "https://arxiv.org/pdf/2608.17027v1",
  "html_url": "https://arxiv.org/html/2608.17027v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2608.16978",
  "slug": "vlcp-vision-language-control-policy-closed-loop-code-replanning-for-ro",
  "title": "VLCP: Vision Language Control Policy Closed-Loop Code Replanning for Robot Manipulation",
  "abstract": "Turning a frontier vision-language model into a robot policy usually means fine-tuning it to emit an action representation it never saw in pretraining, which throws away much of the reasoning that made the model worth reaching for. We go the other way and keep the VLM frozen. It writes the policy as a short Python control function, with no demonstrations and no fine-tuning. Writing that code once is open-loop, though. Existing closed-loop methods react at the wrong level: they retry a fixed policy or pick a different subtask, but never rewrite the code that failed. VLCP closes the loop where the failure actually lives, on the control code, within a single episode. Every $K$ steps the VLM re-observes the scene from multi-view RGB, proprioceptive state, and a state delta, then rewrites the control function from what it just saw, so a failure is caught before it compounds. We evaluate on a 57-task MuJoCo/RoboVerse sweep. This training-free policy reaches $35.1\\%$ pooled success, against $3.5\\%$ for the identical system queried once per episode. That tenfold gap holds with non-overlapping confidence intervals in every scene family. The gain traces to a $27.3\\%$ within-episode recovery rate on failed grasps: a miss an open-loop controller would carry to the end of the episode gets re-observed and fixed at the next replan. And the loop stays cheap. A median $84\\%$ of input tokens hit cache, an episode needs only about $10$ compact queries, and control blocks written during any replan persist to a cross-episode skill library reused in later prompts.",
  "published": "2026-08-17",
  "updated": "2026-08-17",
  "year": "2026",
  "authors": [
   "Dhia Naouali",
   "Minghan Wu",
   "Claudia Wong",
   "Abhinav Puthran",
   "Omar G. Younis"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "VLCP closes the loop where the failure actually lives, on the control code, within a single episode, and keeps the VLM frozen, which is a training-free policy with a tenfold gap between pooled success and confidence intervals in every scene family.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Dhia Naouali",
    "id": "2458131444",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Ming Wu",
    "id": "2449458192",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Claudia Wong",
    "id": "2458167640",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Abhinav Puthran",
    "id": "2458131272",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Omar G. Younis",
    "id": "2376192401",
    "h_index": 2,
    "papers": 5
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "foundation-pretraining",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.16978v1",
  "pdf_url": "https://arxiv.org/pdf/2608.16978v1",
  "html_url": "https://arxiv.org/html/2608.16978v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.16966",
  "slug": "multi-observer-vehicle-localization-case-study-with-roadside-radar-and",
  "title": "Multi-Observer Vehicle Localization Case Study with Roadside Radar and Connected Vehicle Sensing",
  "abstract": "In modern intelligent transportation systems, it is essential to accurately estimate vehicle positions, especially in mixed traffic conditions where both connected and conventional vehicles coexist. Roadside infrastructure and connected vehicles can provide complementary observations of the same traffic scene, but real-world evidence on decision-level fusion between these sources remains limited. This paper proposes a multi-observer vehicle localization framework that fuses compact object-level detections from a static roadside radar and a dynamic LiDAR-equipped connected vehicle. We evaluate the framework with real-world data collected at an urban intersection in Helsinki, Finland, with a separately instrumented target vehicle used as the reference trajectory. Two extended Kalman filter based strategies for the localization task were benchmarked. The performance of the radar and LiDAR sensors were evaluated separately, and the two fusion strategies were explored under nominal sensing conditions, reduced LiDAR update rates, simulated LiDAR occlusions, and different target-vehicle motion states. The results show that, under full LiDAR availability, fusion performance is dominated by the LiDAR observations, while the less accurate and less consistent radar observations provide only limited additional improvement. Nevertheless, AEKF achieves small gains over the LiDAR-only baseline, and object-level connected vehicle observations remain useful when shared at reduced update rates. These findings indicate that decision-level fusion provides scenario-dependent benefits rather than automatic improvement over a strong single-sensor baseline. We release the dataset and implementation on Github to support further research: https://github.com/AppuriAalto/multi-observer-vehicle-tracking",
  "published": "2026-08-17",
  "updated": "2026-08-17",
  "year": "2026",
  "authors": [
   "Aleksi Pippuri",
   "Nilusha Jayawickrama",
   "Risto Ojala"
  ],
  "author_count": 3,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Findings indicate that decision-level fusion provides scenario-dependent benefits rather than automatic improvement over a strong single-sensor baseline, and AEKF achieves small gains over the LiDAR-only baseline, and object-level connected vehicle observations remain useful when shared at reduced update rates.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Aleksi Pippuri",
    "id": "2249560823",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "N. Jayawickrama",
    "id": "40951209",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Risto Ojala",
    "id": "1389964698",
    "h_index": 7,
    "papers": 28
   }
  ],
  "comment": "12 pages, 7 figures and 8 tables",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.16966v1",
  "pdf_url": "https://arxiv.org/pdf/2608.16966v1",
  "html_url": "https://arxiv.org/html/2608.16966v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.16889",
  "slug": "don-t-drop-the-baton-long-horizon-robot-manipulation-via-agentic-subta",
  "title": "Don't Drop the BATON: Long-Horizon Robot Manipulation via Agentic Subtask Exploration and Transition-aware Memory",
  "abstract": "Long-horizon robot manipulation chains many contact-rich skills into one multi-stage task. Vision-language-action (VLA) models increasingly master the individual skills, yet the chain still fails: errors compound beyond the policy's ability to correct, and one subtask silently constrains the next. A promising recipe freezes the VLA and puts an LLM agent in charge: it plans in language, moves in free space with analytic primitives, invokes the VLA only for contact-rich segments, and writes adaptation into language memory. Applied to long horizons, it breaks twice. (1) Competence comes from whole-task exploration at test time, whose cost is multiplicative in stages: if one stage needs T episodes, a K-stage task needs about T^K, and a failure does not reveal which stage caused it. (2) It has no representation of transitions: the VLA primitive carries an exit but no entry condition, so a subtask can succeed in a form its successor cannot use. We present BATON. Against (1), BATON makes the subtask the unit of exploration: each is explored in the cheap short-horizon regime and its solution stored in memory; a long-horizon trajectory is then composed from these solutions rather than discovered whole. Cost becomes additive (T*K) and every failure is attributed to a single stage. Against (2), BATON equips exploration with a transition-aware memory. Within a subtask, a verifier agent governs the invocation transition: the VLA is called only after the wrist view confirms the scene is ready. Across subtasks, a handoff transition restores an entry state disturbed by the predecessor's residue, and a lookahead transition selects the strategy whose outcome the successor can inherit. No parameters are updated. On the long-horizon benchmark RoboMemArena, BATON improves task success by 11.6% and cumulative success by 14.9% over the SoTA.",
  "published": "2026-08-17",
  "updated": "2026-08-17",
  "year": "2026",
  "authors": [
   "Bingxin Xu",
   "Yuzhang Shang",
   "Emilio Ferrara"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "BATON improves task success by 11.6% and cumulative success by 14.9% over the SoTA and improves task success by 11.6% over the RoboMemArena on the long-horizon benchmark RoboMemArena.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Bingxin Xu",
    "id": "2293281293",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Yuzhang Shang",
    "id": "2380471827",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Emilio Ferrara",
    "id": "2384134381",
    "h_index": 3,
    "papers": 12
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "tactile",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.16889v1",
  "pdf_url": "https://arxiv.org/pdf/2608.16889v1",
  "html_url": "https://arxiv.org/html/2608.16889v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.16885",
  "slug": "0-vla-a-hierarchical-robot-foundation-model-with-world-model-guided-te",
  "title": "$\u03c4_0$-VLA: a Hierarchical Robot Foundation Model with World-Model-Guided Test-Time Computation",
  "abstract": "Long-horizon robot manipulation requires a robot to both execute individual skills reliably and sequence them coherently over extended tasks. Most hierarchical vision-language-action (VLA) models make each such decision with a single forward pass, leaving no mechanism to allocate additional computation to difficult or consequential choices. We introduce $\u03c4_0$-VLA, a hierarchical robot foundation model that formulates high-level subtask generation as a compute-scalable inference problem through world-model-guided test-time computation. At each inference step, the high-level policy uses execution memory to generate a subtask and, when needed, searches over alternatives before committing to its output. A low-level policy then executes the generated subtask across multiple robot embodiments. The policy is trained on 40,115 hours of heterogeneous real-world data with multimodal co-training. Across in-domain and distribution-shifted settings, allocating additional test-time computation substantially improves next-subtask prediction accuracy, and these gains translate into higher closed-loop success on long-horizon robot manipulation tasks.",
  "published": "2026-08-17",
  "updated": "2026-08-17",
  "year": "2026",
  "authors": [
   "Xiaowei Cai",
   "Yunuo Cai",
   "Bingao Chen",
   "Jingxiao Chen",
   "Zhi Chen",
   "Siyuan Feng",
   "Tengyu Hou",
   "Jingshun Huang",
   "Han Jiang",
   "Runkun Ju",
   "Dong Li",
   "Mingxiang Li",
   "Shaowei Li",
   "Xinchen Li",
   "Yifan Li",
   "Yi Liu",
   "Zhongyuan Liu",
   "Jianlan Luo",
   "Junwen Miao",
   "Ruiqi Ni",
   "Buqing Nie",
   "Mingjie Pan",
   "Xinlin Ren",
   "Jianheng Song",
   "Jiaxu Wang",
   "Peiqi Wang",
   "Sen Wang",
   "Xiaoyan Wang",
   "Dafeng Wei",
   "Dongming Wu",
   "Pengwei Xie",
   "Pu Yang",
   "Hangjian Ye",
   "Xiangyu Yue",
   "Jinyu Zhang",
   "Qinglin Zhang",
   "Xueyong Zhao",
   "Pengfei Zhou",
   "Yue Zhou"
  ],
  "author_count": 39,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Across in-domain and distribution-shifted settings, allocating additional test-time computation substantially improves next-subtask prediction accuracy, and these gains translate into higher closed-loop success on long-horizon robot manipulation tasks.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xiaowei Cai",
    "id": "2239093156",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Yunuo Cai",
    "id": "2217984829",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Bin Chen",
    "id": "2456358171",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Jingxiao Chen",
    "id": "2257310624",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Zhi Chen",
    "id": "2379888799",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Siyuan Feng",
    "id": "2350330740",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Teng Hou",
    "id": "2266913219",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Jingshun Huang",
    "id": "2356004147",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Han Jiang",
    "id": "2387249927",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Runkun Ju",
    "id": "2393924169",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Dong Li",
    "id": "2320496083",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Mingxi Li",
    "id": "2451672957",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Shaowei Li",
    "id": "2448942345",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Xinchen Li",
    "id": "2387115706",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Yifan Li",
    "id": "2281904596",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Yi Liu",
    "id": "2402998392",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Zhongyuan Liu",
    "id": "2455661740",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Jianlan Luo",
    "id": "2349439364",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Junwen Miao",
    "id": "2398811201",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Ruiqi Ni",
    "id": "51152285",
    "h_index": 8,
    "papers": 18
   },
   {
    "name": "Buqing Nie",
    "id": "2086828401",
    "h_index": 4,
    "papers": 16
   },
   {
    "name": "Mingjie Pan",
    "id": "2338889337",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Xinlin Ren",
    "id": "2163652354",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Jia-Yi Song",
    "id": "2306357090",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Jiaxu Wang",
    "id": "2242768739",
    "h_index": 10,
    "papers": 38
   },
   {
    "name": "Peiqi Wang",
    "id": "2390224074",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Sen Wang",
    "id": "2448250408",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Xiaoyan Wang",
    "id": "2454484924",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Dafeng Wei",
    "id": "2349556277",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Dongming Wu",
    "id": "2269424517",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Pengwei Xie",
    "id": "2380031706",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Pu Yang",
    "id": "2295534837",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Hang Ye",
    "id": "2453276750",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Xiangyu Yue",
    "id": "2364030943",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Jinyu Zhang",
    "id": "2315062043",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Qingli Zhang",
    "id": "2258535380",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Xue Zhao",
    "id": "2453395647",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Pengfei Zhou",
    "id": "2338818174",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Yue Zhou",
    "id": "2306083274",
    "h_index": 5,
    "papers": 11
   }
  ],
  "comment": "18 pages, 5 figures. Project page: https://tau0-vla.github.io/",
  "topics": [
   "world-models",
   "vla",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.16885v1",
  "pdf_url": "https://arxiv.org/pdf/2608.16885v1",
  "html_url": "https://arxiv.org/html/2608.16885v1",
  "code_url": "https://tau0-vla.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2608.16853",
  "slug": "flexworm-primitive-augmented-hybrid-contact-motion-planning-for-suctio",
  "title": "FlexWorm: Primitive-augmented Hybrid Contact-motion Planning for Suction-based Multi-segment Deformable Robots",
  "abstract": "Multi-segment suction-based soft robots are promising for inspection and maintenance in confined or fragile environments, but existing approaches still depend heavily on manually designed gaits and environment-specific motion scripts. This work presents a planning framework for serial multi-segment soft robots with deformable body segments and boundary suction pads. The formulation targets full 3D navigation on complex surfaces and explicitly handles discrete adhesion switching and continuous body deformation under geometric, collision, and quasi-static feasibility constraints, while remaining agnostic to the specific actuation realization used to produce segment deformation. Its core, block-wise IK hybrid search (IKHS), performs best-first search over feasible adhesion transitions while solving inverse kinematics only on induced free blocks. On top of IKHS, primitive-augmented hybrid search (PaHS) uses a learned observation--primitive embedding to retrieve short validated motion segments for fast local proposal, with fallback to standard IKHS branching when retrieval fails. In simulation, the framework consistently outperforms controlled baselines in planning success, transition quality, and efficiency across diverse terrains. PaHS matches IKHS in success rate while substantially reducing planning time. Repeated hardware experiments on a pneumatic multi-segment soft robot further demonstrate executability and online recovery under actuation and adhesion uncertainty.",
  "published": "2026-08-17",
  "updated": "2026-08-20",
  "year": "2026",
  "authors": [
   "Zili Tang",
   "Tiecheng Guo",
   "Qinyue Zhang",
   "Meng Guo"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work presents a planning framework for serial multi-segment soft robots with deformable body segments and boundary suction pads that consistently outperforms controlled baselines in planning success, transition quality, and efficiency across diverse terrains.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zili Tang",
    "id": "2301560169",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Tiecheng Guo",
    "id": "2374201745",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Qinyu Zhang",
    "id": "2452935631",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Meng Guo",
    "id": "2301175313",
    "h_index": 3,
    "papers": 4
   }
  ],
  "comment": "13 pages, 18 figures, accepted for publication in IEEE Robotics and Automation Letters (RA-L), 2026. Supplementary video: https://youtu.be/OQR5Sx5Bwnc",
  "topics": [
   "navigation",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.16853v2",
  "pdf_url": "https://arxiv.org/pdf/2608.16853v2",
  "html_url": "https://arxiv.org/html/2608.16853v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.16843",
  "slug": "security-of-foundation-model-powered-embodied-agents-attack-surfaces-a",
  "title": "Security of Foundation-Model-Powered Embodied Agents: Attack Surfaces, Attacks, Defenses, and Evaluation",
  "abstract": "Foundation models are increasingly used for perception, reasoning, planning, and action generation in embodied agents, creating security risks that can propagate from digital inputs to physical behavior. Existing surveys often organize threats by mechanisms such as jailbreaks, prompt injection, backdoors, poisoning, or adversarial examples, but these categories do not consistently identify where an adversary first enters the embodied control loop. We present a trust-boundary-centric survey of foundation-model-powered embodied-agent security. Using a first-compromised-trust-boundary principle, we separate attack surface from attack mechanism and organize the system into five layers and twelve attack surfaces spanning the model supply chain, user instructions, context and memory, physical semantic environments, multimodal perception, world state, internal reasoning, task planning, action interfaces, middleware, multi-agent communication, and execution control. Based on 58 attack records and 61 defense records collected through August 15, 2026, we analyze representative attacks, cross-layer propagation, defense placement, and evaluation practices. Our quantitative analysis shows that attack research is concentrated on multimodal perception and action interfaces, while defenses are especially concentrated on action-level and runtime protection. Context and long-term memory, middleware and networking, world-state integrity, and multi-agent trust remain comparatively underexplored. We conclude with open challenges in state provenance, compositional defenses, long-horizon attack propagation, physical realizability, Byzantine multi-robot behavior, and unified closed-loop evaluation.",
  "published": "2026-08-17",
  "updated": "2026-08-17",
  "year": "2026",
  "authors": [
   "Jiawei Liu",
   "Jiacheng Guo",
   "Tian Zhang",
   "Yiwei Xu",
   "Juan Wang",
   "Jinlin Fan",
   "Bowen Xiao"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work presents a trust-boundary-centric survey of foundation-model-powered embodied-agent security, and shows that attack research is concentrated on multimodal perception and action interfaces, while defenses are especially concentrated on action-level and runtime protection.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jiawei Liu",
    "id": "2324835288",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Jiacheng Guo",
    "id": "2319146416",
    "h_index": 9,
    "papers": 36
   },
   {
    "name": "Tianwei Zhang",
    "id": "2311284370",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Yiwei Xu",
    "id": "153018841",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Juan Wang",
    "id": "2288040612",
    "h_index": 6,
    "papers": 18
   },
   {
    "name": "Jinlin Fan",
    "id": "2297057761",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Bowen Xiao",
    "id": "2378713940",
    "h_index": 2,
    "papers": 6
   }
  ],
  "comment": "",
  "topics": [
   "foundation-pretraining",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.16843v1",
  "pdf_url": "https://arxiv.org/pdf/2608.16843v1",
  "html_url": "https://arxiv.org/html/2608.16843v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.16837",
  "slug": "haf-adapting-generalist-vlas-to-humanoid-whole-body-loco-manipulation",
  "title": "HAF: Adapting Generalist VLAs to Humanoid Whole-Body Loco-manipulation via Hierarchical Action Flow and Spectral Latent RL",
  "abstract": "Humanoid robots hold great promise as general-purpose agents in human-centered environments, yet generalist vision-language-action (VLA) foundation models are not readily applicable to humanoid whole-body loco-manipulation. The high dimensionality and interdependence of humanoid motions make it challenging for conventional single-stage VLA architectures to coordinate locomotion, waist posture, and dual-arm manipulation effectively. Moreover, policies trained through offline behavior cloning can remain suboptimal during real-world deployment. Although online reinforcement learning can refine policies through real-world interaction, directly tuning large VLA backbones demands excessive computation and may introduce safety risks during real-robot exploration. To address these bottlenecks, we introduce HAF (Humanoid Adaptation Framework), a two-part framework consisting of HAF-VLA and HAF-Steer that transfers off-the-shelf generalist VLA foundation models to humanoid whole-body loco-manipulation. HAF-VLA is a hierarchical action-flow generator built on a pretrained flow-matching VLA. It splits full-body action denoising into three sequential stages with stage embeddings and cross-stage KV caches that retain kinematic dependencies, avoiding incoherent whole-body actions from one-shot generation. On top of the frozen HAF-VLA, HAF-Steer is a latent offline-to-online RL pipeline that leverages flow-matching invertibility and DCT-based dimensionality reduction to restrict RL optimization to a compact noise subspace and train a regularized SAC policy. This avoids updating the large VLA backbone and enables efficient real-world policy refinement. Evaluated on seven real-world humanoid loco-manipulation tasks, HAF surpasses vanilla single-stage VLA baselines and improves whole-body coordination and task performance. Project website: https://grange007.github.io/HAF .",
  "published": "2026-08-17",
  "updated": "2026-08-17",
  "year": "2026",
  "authors": [
   "Langzhe Gu",
   "Chengkai Hou",
   "Meng Li",
   "Xinhua Wang",
   "Jiaming Liu",
   "Xinyuan Lv",
   "Bowei Zhang",
   "Shuanghao Bai",
   "Guangrun Li",
   "Jingyang He",
   "Gaole Dai",
   "Ziluo Ding",
   "Zhiyuan Xu",
   "Kuan Cheng",
   "Jian Tang",
   "Zhengping Che",
   "Shanghang Zhang"
  ],
  "author_count": 17,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "HAF (Humanoid Adaptation Framework), a two-part framework consisting of HAF-VLA and HAF-Steer that transfers off-the-shelf generalist VLA foundation models to humanoid whole-body loco-manipulation, surpasses vanilla single-stage VLA baselines and improves whole-body coordination and task performance.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Langzhe Gu",
    "id": "2378192104",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Chengkai Hou",
    "id": "2335857967",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Meng Li",
    "id": "2336577709",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Xinhua Wang",
    "id": "2308136082",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Jiaming Liu",
    "id": "2401539139",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Xinyuan Lv",
    "id": "2239061274",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Bowei Zhang",
    "id": "2257373058",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Shuanghao Bai",
    "id": "2274938328",
    "h_index": 10,
    "papers": 23
   },
   {
    "name": "Guangrun Li",
    "id": "2363550799",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Jingyang He",
    "id": "2307452468",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Gaole Dai",
    "id": "2218469002",
    "h_index": 8,
    "papers": 22
   },
   {
    "name": "Ziluo Ding",
    "id": "1742314398",
    "h_index": 13,
    "papers": 30
   },
   {
    "name": "Zhiyuan Xu",
    "id": "48559420",
    "h_index": 27,
    "papers": 68
   },
   {
    "name": "Kuan Cheng",
    "id": "2376044478",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Jian Tang",
    "id": "2367267184",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Zhengping Che",
    "id": "1939695",
    "h_index": 26,
    "papers": 88
   },
   {
    "name": "Shan-Shan Zhang",
    "id": "2257020189",
    "h_index": 5,
    "papers": 6
   }
  ],
  "comment": "Project page: https://grange007.github.io/HAF",
  "topics": [
   "vla",
   "humanoids",
   "imitation-diffusion",
   "rl-control",
   "navigation",
   "foundation-pretraining",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.16837v1",
  "pdf_url": "https://arxiv.org/pdf/2608.16837v1",
  "html_url": "https://arxiv.org/html/2608.16837v1",
  "code_url": "https://grange007.github.io/HAF",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2608.16822",
  "slug": "adaptive-repulsive-pheromone-clustering-for-foraging-robot-swarms",
  "title": "Adaptive Repulsive Pheromone Clustering for Foraging Robot Swarms",
  "abstract": "The Central Place Foraging Algorithm (CPFA) combines site fidelity, pheromone-guided navigation, and uninformed random search to enable decentralized resource collection in robot swarms. However, CPFA often revisits previously explored regions while leaving other areas insufficiently searched, reducing efficiency as resources become scarce. In this paper, we propose Adaptive Repulsive Pheromone Clustering (ARPC), a bio-inspired method in which robots deposit repulsive pheromone waypoints to mark previously explored locations. These waypoints are clustered around the nest to estimate low-value search regions, allowing robots to be redirected toward likely unvisited areas. By integrating the exploitation of known resources with systematic avoidance of redundant exploration, ARPC improves search diversity and resource discovery efficiency. Extensive simulations in ARGoS across varying arena sizes, resource densities, and clustered, random, and power-law spatial distributions demonstrate that ARPC consistently outperforms CPFA and the Grid-Based CPFA (GPFA). In particular, ARPC yields significant gains during both early discovery (10\\%) and late-stage (up to 60\\%) collection, where conventional methods typically degrade. These results indicate that ARPC provides a scalable and robust strategy for large-scale heterogeneous swarm foraging environments.",
  "published": "2026-08-17",
  "updated": "2026-08-17",
  "year": "2026",
  "authors": [
   "Carlos Pena-Caballero",
   "Constantine Tarawneh",
   "Qi Lu"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Adaptive Repulsive Pheromone Clustering (ARPC), a bio-inspired method in which robots deposit repulsive pheromone waypoints to mark previously explored locations, provides a scalable and robust strategy for large-scale heterogeneous swarm foraging environments.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Carlos Pena-Caballero",
    "id": "2458106607",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Constantine Tarawneh",
    "id": "2340290586",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Qi Lu",
    "id": "2377948847",
    "h_index": 1,
    "papers": 12
   }
  ],
  "comment": "14 pages, 5 figures, The 18th International Symposium on Distributed Autonomous Robotic Systems",
  "topics": [
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.16822v1",
  "pdf_url": "https://arxiv.org/pdf/2608.16822v1",
  "html_url": "https://arxiv.org/html/2608.16822v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.16806",
  "slug": "breaking-planner-integrity-boundary-enviroment-state-text-injection-at",
  "title": "Breaking Planner Integrity Boundary: Enviroment State-Text Injection Attack on LLM-Driven Embodied Agents",
  "abstract": "Large language model (LLM)-driven embodied agents rely on environment states to interpret scenes, generate high-level plans, and drive physical execution, making planner-visible state representations a critical security boundary. Existing attacks primarily manipulate user instructions, prompt contexts, model behavior, or perceptual inputs, while paying limited attention to whether environment-state text itself can serve as deceptive task evidence and propagate beyond planning to affect execution outcomes. Because embodied tasks are constrained by entity grounding, action preconditions, spatial relations, and environmental constraints, planning deviation alone does not guarantee adversarial execution. To address this gap, we investigate environment-state text as an independent attack surface and present the first closed-loop Environment State-Text Injection (ESTI) attack for LLM-driven embodied agents. Without modifying the original user instruction, model parameters, or executor, ESTI reformulates an adversarial objective as false state evidence compatible with the current environment and influences planning and execution through object properties, spatial relations, affordances, task-stage rules, and execution feedback. We further develop ESTI-Bench to evaluate attack propagation across the planning-to-execution closed loop and compare ESTI with Vanilla IPI, EIRAD, and BADROBOT across ProgPrompt/VirtualHome, VoxPoser/RLBench, and AI2-THOR/iTHOR. ESTI consistently outperforms existing baselines, improving planning-level and execution-level attack success rates by up to 89.32\\% and 43.69\\%, respectively. Further analysis shows that grounding, consistency, and executability jointly determine whether manipulated state evidence can propagate through the embodied closed loop and produce verifiable environmental changes.",
  "published": "2026-08-17",
  "updated": "2026-08-18",
  "year": "2026",
  "authors": [
   "Jiawei Liu",
   "Jiacheng Guo",
   "Tian Zhang",
   "Yiwei Xu",
   "Juan Wang",
   "Jinlin Fan",
   "Bowen Xiao",
   "Chi Guo",
   "Keyan Guo",
   "Hongxin Hu"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Investigation of environment-state text as an independent attack surface and the first closed-loop Environment State-Text Injection (ESTI) attack for LLM-driven embodied agents show that grounding, consistency, and executability jointly determine whether manipulated state evidence can propagate through the embodied closed loop and produce verifiable environmental changes.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jiawei Liu",
    "id": "2324835288",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Jiacheng Guo",
    "id": "2319146416",
    "h_index": 9,
    "papers": 36
   },
   {
    "name": "Tianke Zhang",
    "id": "2312667943",
    "h_index": 8,
    "papers": 24
   },
   {
    "name": "Yiwei Xu",
    "id": "2362512026",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Juan Wang",
    "id": "2288040612",
    "h_index": 6,
    "papers": 18
   },
   {
    "name": "Jinlin Fan",
    "id": "2297057761",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Bowen Xiao",
    "id": "2378713940",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Chi Guo",
    "id": "2110228497",
    "h_index": 11,
    "papers": 26
   },
   {
    "name": "Keyan Guo",
    "id": "2165583728",
    "h_index": 7,
    "papers": 22
   },
   {
    "name": "Hongxin Hu",
    "id": "2288043385",
    "h_index": 6,
    "papers": 13
   }
  ],
  "comment": "submitted to USENIX Security 2027",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.16806v2",
  "pdf_url": "https://arxiv.org/pdf/2608.16806v2",
  "html_url": "https://arxiv.org/html/2608.16806v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.16794",
  "slug": "neurosymbolic-embodied-agents",
  "title": "Neurosymbolic Embodied Agents",
  "abstract": "Language and vision-language models generate plausible embodied plans but do not guarantee executability, as their outputs can violate environment dynamics or act on incorrectly grounded entities. We present a neurosymbolic agent that factors long-horizon household tasks into task-directed visual exploration and constrained symbolic planning. In the first phase, a vision-language model and exploration harness acquire goal-relevant predicates and instance bindings from egocentric observations and grounded interactions, producing a symbolic initial state. In the second, a PDDL transition model restricts decoding to tokens that extend applicable actions. Monte Carlo tree search then evaluates executable continuations using a domain-independent planning heuristic. The resulting plans are executable by construction under the transition model, with transfer to the environment conditioned on correct visual grounding. On VirtualHome and ALFWorld, open 4B-27B models exceed 90% success in both environments, and our smallest agent substantially outperforms a 27B direct visual policy in each. Constraints and search prove complementary rather than interchangeable: in ALFWorld either alone solves under a third of tasks, whereas their combination solves over 95%. The method also uses several times fewer generated tokens than extended thinking and far fewer model-visible images than direct interaction, and residual failures localize to state acquisition rather than plan generation without any specialized training.",
  "published": "2026-08-17",
  "updated": "2026-08-18",
  "year": "2026",
  "authors": [
   "Mohammad Albinhassan",
   "Yuming Feng",
   "Alessandra Russo",
   "Pranava Madhyastha"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CL"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A neurosymbolic agent that factors long-horizon household tasks into task-directed visual exploration and constrained symbolic planning and evaluates executable continuations using a domain-independent planning heuristic is presented.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Mohammad Albinhassan",
    "id": "2321534472",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Yuming Feng",
    "id": "2453025513",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Alessandra Russo",
    "id": "2321580533",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "P. Madhyastha",
    "id": "3238408",
    "h_index": 15,
    "papers": 81
   }
  ],
  "comment": "",
  "topics": [
   "egocentric-data",
   "navigation",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.16794v2",
  "pdf_url": "https://arxiv.org/pdf/2608.16794v2",
  "html_url": "https://arxiv.org/html/2608.16794v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.16741",
  "slug": "semantic-and-density-aware-planning-for-accessibility-preserving-multi",
  "title": "Semantic- and Density-Aware Planning for Accessibility-Preserving Multi-Object Placement",
  "abstract": "Long-term manipulation planning requires robots to reason not only about immediate task success but also about how current decisions affect future interactions with the environment. In this context, household service robots may need to organize groceries in partially occupied shelves while using limited storage space efficiently and preserving access for subsequent placements. In this paper, we consider an online multi-object shelf-placement setting in which future objects arrivals are unknown. Existing approaches do not jointly address semantic organization, dense space utilization, and manipulator accessibility during sequential shelf filling. To address this gap, we propose Semantic-Dense Placement Planning (SDPP), an accessibility-preserving approach that ranks candidate poses using a semantic-density score combining inter-object semantic similarity with spatial proximity. An Accessibility Map (AM) further filters candidates unlikely to be reachable before motion planning and penalizes placements that reduce the remaining accessible workspace. Simulation experiments show that SDPP significantly improves semantic placement quality over state-of-the-art baselines and achieves the highest average shelf density, while the AM substantially reduces the time required to identify feasible placement poses. A qualitative real-world experiment demonstrates the applicability of our pipeline in a domestic shelf-storage scenario.",
  "published": "2026-08-17",
  "updated": "2026-08-17",
  "year": "2026",
  "authors": [
   "Benno Wingender",
   "Nils Dengler",
   "Nicolas Busch",
   "Sicong Pan",
   "Maren Bennewitz"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Semantic-Dense Placement Planning (SDPP) is proposed, an accessibility-preserving approach that ranks candidate poses using a semantic-density score combining inter-object semantic similarity with spatial proximity and significantly improves semantic placement quality over state-of-the-art baselines.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Benno Wingender",
    "id": "2386019053",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Nils Dengler",
    "id": "1381675037",
    "h_index": 9,
    "papers": 31
   },
   {
    "name": "Nicolas Busch",
    "id": "2458105338",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Sicong Pan",
    "id": "10456004",
    "h_index": 9,
    "papers": 26
   },
   {
    "name": "Maren Bennewitz",
    "id": "2249760470",
    "h_index": 7,
    "papers": 44
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.16741v1",
  "pdf_url": "https://arxiv.org/pdf/2608.16741v1",
  "html_url": "https://arxiv.org/html/2608.16741v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.16728",
  "slug": "design-optimization-for-large-high-force-soft-robot-manipulators-under",
  "title": "Design Optimization for Large High-Force Soft Robot Manipulators Under Gravitational Loads",
  "abstract": "Designing large soft robots capable of generating high forces for physical human-robot interaction remains a significant challenge in soft robotics. Prior work in large soft robots has focused on proof-of-concept prototypes, and no systematic framework exists for determining the suitability of a design paradigm for a desired task. This manuscript introduces a method for optimizing the geometry of a soft robot limb, maximizing its blocking force subject to an anti-bucking constraint under its own gravitational loading. We demonstrate that an explicit solution exists to the proposed optimization problem under certain assumptions. Experiments with three geometries of a large, soft, pneumatically-actuated manipulator demonstrate that the method correctly predicts which designs meet constraints and which produces the largest end-effector forces. This method, with its closed-form solution, can allow designers to determine a-priori if an intended class of soft manipulators is an appropriate choice for physical interaction at large size scales.",
  "published": "2026-08-17",
  "updated": "2026-08-17",
  "year": "2026",
  "authors": [
   "Isara Cholaseuk",
   "Penelope Llibre",
   "Alexa Kyriacou",
   "Audrey Wang",
   "Akua K. Dickson",
   "Ran Jing",
   "Juan C. Pacheco Garcia",
   "Andrew P. Sabelhaus"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Isara Cholaseuk",
    "id": "2458107370",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Penelope Llibre",
    "id": "2458107308",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Alexa Kyriacou",
    "id": "2458108832",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Audrey X. Wang",
    "id": "2406938890",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Akua K. Dickson",
    "id": "2277741833",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Ran Jing",
    "id": "2268529888",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Juan C. Pacheco Garcia",
    "id": "2283820361",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Andrew P. Sabelhaus",
    "id": "1817742",
    "h_index": 16,
    "papers": 49
   }
  ],
  "comment": "8 pages, 8 figures",
  "topics": [
   "hardware-codesign",
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.16728v1",
  "pdf_url": "https://arxiv.org/pdf/2608.16728v1",
  "html_url": "https://arxiv.org/html/2608.16728v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.16715",
  "slug": "matchingpolicy-correspondence-aware-policy-enables-cross-object-in-con",
  "title": "MatchingPolicy: Correspondence-Aware Policy Enables Cross-Object In-Context Learning",
  "abstract": "In-context imitation learning enables few-shot policy generalization but struggles to maintain performance on unseen objects and novel scenarios. To address this, we introduce MatchingPolicy, a correspondence-driven framework that explicitly decouples demonstration-to-scene matching from policy learning. Central to our method is a correspondence-aware diffusion policy that conditions robotic actions directly on dense semantic correspondences. This architectural separation resolves the inherent conflict between correspondence identification and action adaptation, enabling robust out-of-distribution transfer. Our framework integrates vision foundation models with a novel two-stage matching algorithm to dynamically establish reliable correspondences. Extensive evaluations on RLBench and real-world manipulation tasks confirm that MatchingPolicy achieves superior few-shot performance, generalizing reliably across unseen object instances and semantic categories.",
  "published": "2026-08-17",
  "updated": "2026-08-17",
  "year": "2026",
  "authors": [
   "Qijin She",
   "Hanyang Yu",
   "Zeming Li",
   "Ping Tan"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work introduces MatchingPolicy, a correspondence-driven framework that explicitly decouples demonstration-to-scene matching from policy learning, and central to this method is a correspondence-aware diffusion policy that conditions robotic actions directly on dense semantic correspondences.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Qijin She",
    "id": "1768846972",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Hanyang Yu",
    "id": "2324225754",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Zeming Li",
    "id": "2364935004",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Ping Tan",
    "id": "2334748460",
    "h_index": 1,
    "papers": 5
   }
  ],
  "comment": "",
  "topics": [
   "imitation-diffusion",
   "foundation-pretraining",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.16715v1",
  "pdf_url": "https://arxiv.org/pdf/2608.16715v1",
  "html_url": "https://arxiv.org/html/2608.16715v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.16712",
  "slug": "h-pac-hand-control-oriented-modeling-and-tendon-elasticity-compensatio",
  "title": "H-PAC Hand: Control-Oriented Modeling and Tendon-Elasticity Compensation for an Underactuated Robotic Hand",
  "abstract": "Underactuated tendon-driven hands offer compact actuation and passive compliance, but tendon elongation under restoring-spring loading introduces configuration-dependent joint deviations. This paper presents H-PAC, a modular 6-actuator, 15-DoF robotic hand with a control-oriented modeling and implementation framework. A sparse analytical actuator-joint model is derived from the tendon-routing geometry, and a mechanics-based compensation model is developed to account for tendon-elasticity-induced joint errors. The proposed method is implemented in a hierarchical architecture: a host computer performs workspace-constrained posture mapping and compensation, while an ESP32 generates synchronized commands for six position-controlled servos. The same control parameters and execution strategy are used across all tasks without task-specific retuning. Monotonic servo-sweep experiments show that the compensation substantially improves joint-angle prediction. The MAE of the index DIP joint decreases from 1.15 degrees to 0.18 degrees, and all nine evaluated joints achieve an MAE below 0.23 degrees. Representative postures and grasping configurations are further executed using the same control pipeline without external joint or force sensing in the control loop. The results demonstrate a practical approach to improving posture reproducibility in compact underactuated robotic end-effectors.",
  "published": "2026-08-17",
  "updated": "2026-08-17",
  "year": "2026",
  "authors": [
   "Teng Yan",
   "Jiongxu Chen",
   "Teng Wang",
   "Yue Yu",
   "Qixiang Hua",
   "Zihang Wang",
   "Yongru Chen",
   "Bingzhuo Zhong"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Tengfei Yan",
    "id": "2368482044",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Jiongxu Chen",
    "id": "2381780873",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Teng Wang",
    "id": "2448898258",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yue Yu",
    "id": "2362568120",
    "h_index": 0,
    "papers": 5
   },
   {
    "name": "Qixiang Hua",
    "id": "2425459195",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Zihang Wang",
    "id": "2425463299",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yongru Chen",
    "id": "2458117599",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "B. Zhong",
    "id": "2425457433",
    "h_index": 0,
    "papers": 2
   }
  ],
  "comment": "7 pages, 6 figures. Extended preprint",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.16712v1",
  "pdf_url": "https://arxiv.org/pdf/2608.16712v1",
  "html_url": "https://arxiv.org/html/2608.16712v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.16686",
  "slug": "closing-the-affective-loop-multimodal-speaker-listener-emotion-dynamic",
  "title": "Closing the Affective Loop: Multimodal Speaker-Listener Emotion-Dynamics-Aware Empathetic Social Robots",
  "abstract": "Empathetic social robots should respond not only to what users say, but also to how their emotions dynamically evolve during interaction. However, existing empathetic dialogue systems are often text-centered and primarily model empathy as a one-way mapping from the user's emotion to the system response, limiting their ability to capture embodied speaker--listener affective exchange. We present AffectLoop, a multimodal speaker-listener emotion-dynamics-aware spoken dialogue system implemented on the Misty II robot. The system tracks the speaker's verbal and facial affective dynamics, estimates the robot listener's own verbal and behavioral affective state, and conditions LLM-based response generation on both affective streams. The robot then generates a short spoken empathetic response together with emotionally congruent embodied behavior, forming a closed speaker--listener affective loop. We evaluate the system in a pilot within-subject study with five participants, comparing it with an otherwise identical utterance-conditioned baseline that omits the speaker- and listener-affective-state inputs. The proposed system received higher overall impression ratings, especially for empathetic response and user satisfaction. Post-hoc log analysis further showed higher speaker-listener affective alignment and stronger valence-based distress recovery. These preliminary results suggest that explicitly modeling both speaker emotional dynamics and listener affective state can improve embodied empathetic interaction.",
  "published": "2026-08-17",
  "updated": "2026-08-17",
  "year": "2026",
  "authors": [
   "Zi Haur Pang",
   "Casey Kennington",
   "Tatsuya Kawahara"
  ],
  "author_count": 3,
  "categories": [
   "cs.HC",
   "cs.CL",
   "cs.RO"
  ],
  "primary_category": "cs.HC",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Preliminary results suggest that explicitly modeling both speaker emotional dynamics and listener affective state can improve embodied empathetic interaction.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Z. Pang",
    "id": "2284760440",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "C. Kennington",
    "id": "2274960812",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Tatsuya Kawahara",
    "id": "2254443223",
    "h_index": 6,
    "papers": 40
   }
  ],
  "comment": "This paper has been accepted for presentation at APSIPA ASC 2026",
  "topics": [
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.16686v1",
  "pdf_url": "https://arxiv.org/pdf/2608.16686v1",
  "html_url": "https://arxiv.org/html/2608.16686v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.16658",
  "slug": "x-2-localizer-cross-grained-alignment-for-progressive-cross-view-video",
  "title": "X$^2$Localizer: Cross-grained Alignment for Progressive Cross-view Video Geo-localization",
  "abstract": "Cross-view Video Geo-localization (CVG) aims to localize ground-view videos by retrieving their corresponding geo-tagged aerial images. However, CVG approaches rely on fixed-length inputs and post-hoc refinement, hindering online-oriented localization under partial or dynamic observations. In this work, we formulate Progressive Cross-view Video Geo-localization (PCVG) as a deployment-oriented extension and evaluation protocol of CVG, enabling localization under varying temporal budgets, prefix-based inference, random-start evaluation, and long-range localization with interruptions. To explore PCVG, we introduce X$^2$Localizer, a cross-grained alignment framework that jointly supervises global prefix-to-aerial retrieval and token-aggregated frame--aerial-tile matching with a budget-dependent asymmetric objective. Furthermore, we introduce a Sliding-Window Re-Localization (SWRL) strategy that dynamically refreshes candidate regions for failure recovery and long-range deployment without full-sequence reprocessing. Extensive experiments show that X$^2$Localizer preserves conventional full-video performance, with marginal gains of +0.1 Recall@1 and +0.3 Recall@10, while substantially improving early localization. In the challenging single-frame setting, X$^2$Localizer improves coarse retrieval by +4.7 Recall@1 and +11.5 Recall@10 over the previous state-of-the-art method. With SWRL, our approach further enables robust progressive localization under random-start and long-distance scenarios, narrowing the gap between benchmark evaluation and real-world deployment.",
  "published": "2026-08-17",
  "updated": "2026-08-17",
  "year": "2026",
  "authors": [
   "Zichao Zeng",
   "Weijia Fan",
   "Yufan Chen",
   "June Moh Goo",
   "Junwei Zheng",
   "Ruiping Liu",
   "Kunyu Peng",
   "Jiaming Zhang",
   "Rainer Stiefelhagen",
   "Jan Boehm"
  ],
  "author_count": 10,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work forms Progressive Cross-view Video Geo-localization (PCVG) as a deployment-oriented extension and evaluation protocol of CVG, enabling localization under varying temporal budgets, prefix-based inference, random-start evaluation, and long-range localization with interruptions.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zichao Zeng",
    "id": "2238401424",
    "h_index": 5,
    "papers": 25
   },
   {
    "name": "Weijia Fan",
    "id": "2302559038",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Yufan Chen",
    "id": "2243360598",
    "h_index": 8,
    "papers": 48
   },
   {
    "name": "June Moh Goo",
    "id": "2274931483",
    "h_index": 3,
    "papers": 19
   },
   {
    "name": "Junwei Zheng",
    "id": "2210176414",
    "h_index": 11,
    "papers": 58
   },
   {
    "name": "Ruiping Liu",
    "id": "2273522374",
    "h_index": 7,
    "papers": 52
   },
   {
    "name": "Kunyu Peng",
    "id": "91549683",
    "h_index": 21,
    "papers": 108
   },
   {
    "name": "Jiaming Zhang",
    "id": "2313700347",
    "h_index": 5,
    "papers": 28
   },
   {
    "name": "Rainer Stiefelhagen",
    "id": "2320597200",
    "h_index": 6,
    "papers": 42
   },
   {
    "name": "Jan Boehm",
    "id": "2238221583",
    "h_index": 5,
    "papers": 16
   }
  ],
  "comment": "Accepted to The 37th British Machine Vision Conference (BMVC 2026)",
  "topics": [
   "sim2real"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.16658v1",
  "pdf_url": "https://arxiv.org/pdf/2608.16658v1",
  "html_url": "https://arxiv.org/html/2608.16658v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.16651",
  "slug": "orbit-planner-towards-latent-world-models-for-on-orbit-obstacle-avoida",
  "title": "Orbit-Planner: Towards Latent World Models for On-Orbit Obstacle Avoidance of Satellite Agents",
  "abstract": "Satellite agents for on-orbit navigation tasks need to predict collision risks using limited onboard observations. However, conventional planners often rely on predefined maps and fixed environmental assumptions, limiting their adaptability in dynamic on-orbit scenarios. In this paper, we propose Orbit-Planner, a two-stage latent world model for on-orbit obstacle avoidance. Orbit-Planner learns action-conditioned spacecraft dynamics to perform future-state rollouts in latent space, and introduces a Physics Probe to decode physical state changes from imagined latent trajectories. Experiments demonstrate that Orbit-Planner can perform long-horizon latent rollouts and recover physical states from imagined trajectories. In closed-loop obstacle-avoidance navigation in Isaac Sim, it attains a success rate of 91.7%. Code is available at https://github.com/ZhijianLi2003/Orbit_Planner.",
  "published": "2026-08-17",
  "updated": "2026-08-17",
  "year": "2026",
  "authors": [
   "Zhijian Li",
   "Chao Ren",
   "Peijin Wang",
   "Xian Sun"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Experiments demonstrate that Orbit-Planner can perform long-horizon latent rollouts and recover physical states from imagined trajectories and in closed-loop obstacle-avoidance navigation in Isaac Sim, it attains a success rate of 91.7%.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhijia Li",
    "id": "2455437380",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Chao Ren",
    "id": "2455254245",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Peijin Wang",
    "id": "152702629",
    "h_index": 19,
    "papers": 48
   },
   {
    "name": "Xiang Sun",
    "id": "2455652770",
    "h_index": 0,
    "papers": 1
   }
  ],
  "comment": "4 pages, 6 figures. Accepted to AP-GARSS 2026. Project page: https://zhijianli2003.github.io/Orbit_Planner/",
  "topics": [
   "world-models",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.16651v1",
  "pdf_url": "https://arxiv.org/pdf/2608.16651v1",
  "html_url": "https://arxiv.org/html/2608.16651v1",
  "code_url": "https://zhijianli2003.github.io/Orbit_Planner/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2608.16642",
  "slug": "throwing-a-tight-spiral-american-football-by-a-humanoid-robot",
  "title": "Throwing a Tight Spiral American Football by a Humanoid Robot",
  "abstract": "Accurate throwing of the American football requires precise regulation of release conditions, where coupled linear and angular momentum determine flight stability and targeting accuracy. While prior work on robotic object throwing has largely focused on generating dynamically feasible release velocities using open-gripper paradigms, explicit control of spin injection at detachment remains underexplored, particularly for aerodynamically anisotropic objects like the American football. In this paper, we present the spin-stabilized controlled tight spiral throw of an American football by a humanoid robot. Achieving this requires (i) accurately reaching the desired coupled momentum, which often involves high degrees-of-freedom (DoF) movements completed within approximately half a second, and (ii) managing the complex transient contact dynamics that arise during the sub-100-millisecond release phase, when the football is effectively underactuated as it moves partially across the fingers. To this end, we develop a coupled whole-body control strategy where the lower body is performing informed stabilization while the upper body is further divided into two phases with (i) a throw phase accelerating the football to a target state through trajectory optimization and tracking, and (ii) a follow-through phase utilizing model predictive control to actively control the wrist and remaining in-contact fingers. The proposed framework is empirically validated on a 29-DoF Unitree G1 humanoid equipped with a 7-DoF Dex3-1 three-fingered gripper. The thrown American football reaches up to 93.6% spin efficiency and a 0.286 radians linear-velocity-to-nose-alignment (nose-angle) error (where an ``ideal'' tight spiral corresponds to 100 % spin efficiency and 0 radians nose-angle error) at up to a 5.35 m/s linear velocity and an angular velocity of 14.5 rad/s.",
  "published": "2026-08-17",
  "updated": "2026-08-17",
  "year": "2026",
  "authors": [
   "Zaid Mahboob",
   "Bowen Weng"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zaid Mahboob",
    "id": "2268080671",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Bowen Weng",
    "id": "51497959",
    "h_index": 11,
    "papers": 39
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "rl-control"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2608.16642v1",
  "pdf_url": "https://arxiv.org/pdf/2608.16642v1",
  "html_url": "https://arxiv.org/html/2608.16642v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2608.16640",
  "slug": "dpnet-efficient-dead-end-prediction-and-avoidance-for-vision-based-uav",
  "title": "DPNet: Efficient Dead-End Prediction and Avoidance for Vision-Based UAV Navigation",
  "abstract": "Vision-based Unmanned Aerial Vehicles (UAVs) often suffer from navigation failures in dead ends due to limited sensing accuracy and range. To address this challenge, this paper proposes a systematic solution for efficient dead-end prediction and avoidance. The proposed method introduces a lightweight neural network to predict the relative distance and bearing of potential dead ends within the current field of view using RGB-D inputs. These predictions prune a predefined, compact trajectory library, enabling the planner to proactively avoid dead ends while maintaining navigational smoothness. Notably, our approach transfers across real-world scenarios without manual annotation or fine-tuning on real-world data. The system achieves high-frequency replanning at 50 Hz onboard. Extensive simulation benchmarks demonstrate superior performance in success rate, flight time, and trajectory length, and real-world experiments further validate its effectiveness in complex scenarios.",
  "published": "2026-08-17",
  "updated": "2026-08-17",
  "year": "2026",
  "authors": [
   "Ruibin Zhang",
   "Lun Pan",
   "Zelong Xia",
   "Jialiang Hou",
   "Fei Gao"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The proposed method introduces a lightweight neural network to predict the relative distance and bearing of potential dead ends within the current field of view using RGB-D inputs, enabling the planner to proactively avoid dead ends while maintaining navigational smoothness.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ruibin Zhang",
    "id": "2126417701",
    "h_index": 8,
    "papers": 14
   },
   {
    "name": "Lun Pan",
    "id": "2458081473",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Zelong Xia",
    "id": "2458067171",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Jialiang Hou",
    "id": "2157186184",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Fei Gao",
    "id": "2379731930",
    "h_index": 1,
    "papers": 3
   }
  ],
  "comment": "",
  "topics": [
   "sim2real",
   "navigation",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.16640v1",
  "pdf_url": "https://arxiv.org/pdf/2608.16640v1",
  "html_url": "https://arxiv.org/html/2608.16640v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.16590",
  "slug": "zetta-an-efficient-closed-loop-embodied-harness-for-self-evolving-phys",
  "title": "Zetta $\u03b6$: An Efficient Closed-Loop Embodied Harness for Self-Evolving Physical Intelligence",
  "abstract": "Embodied agents are increasingly used to close the gap left by end-to-end policy models. Yet the agentic path has not realized closed-loop learning in physical execution: existing harnesses remain largely open-loop, following fixed skills during rollout and reflecting only after an episode completes. Such post-hoc reflection cannot govern execution as it unfolds, because physical interaction requires decisions to track rapidly changing robot-environment states at a frequency beyond today's large agentic models. We present Zetta, a closed-loop embodied harness that evolves code-based runtime critics and recovery skills online while keeping the base policy frozen. Through three timescale-separated loops, Zetta provides action-frequency governance, rollout-level critic-recovery proposal, and validation-gated skill updates. Together with Z-Infra, a rollout infrastructure decoupling agent logic from heterogeneous execution resources, Zetta achieves state-of-the-art success on LIBERO-Pro and RoboCasa under our current rollout budget, reaching 90.8% and 93.6%, with an 11.1x inference speedup; success continues to scale with self-exploration experience; learned skills transfer zero-shot, and clear robotic \"Aha Moments\" emerge. These results show that closed-loop harness self-evolution opens a scaling path for reliable physical intelligence.",
  "published": "2026-08-17",
  "updated": "2026-08-17",
  "year": "2026",
  "authors": [
   "Xin Ding",
   "Liang Mi",
   "Mingzhe Huang",
   "Zixuan Wang",
   "Chao Zhang",
   "Zixu Hao",
   "Fu Chen",
   "Xiangyu Li",
   "Yikai Zheng",
   "Yaoyu Guo",
   "Weijun Wang",
   "Kun Li",
   "Hao Wu",
   "Yunxin Liu",
   "Ting Cao"
  ],
  "author_count": 15,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Zetta is presented, a closed-loop embodied harness that evolves code-based runtime critics and recovery skills online while keeping the base policy frozen, and shows that closed-loop harness self-evolution opens a scaling path for reliable physical intelligence.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xin Ding",
    "id": "2350151012",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Liang Mi",
    "id": "2276423894",
    "h_index": 4,
    "papers": 15
   },
   {
    "name": "Mingzhe Huang",
    "id": "2449156186",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Zixuan Wang",
    "id": "2457763428",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Chao Zhang",
    "id": "2457967765",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Zixu Hao",
    "id": "2328256168",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Fu Chen",
    "id": "2349549141",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Xiangyu Li",
    "id": "2303907684",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Yikai Zheng",
    "id": "2394177297",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Yao Guo",
    "id": "2455802456",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Weijun Wang",
    "id": "2346275858",
    "h_index": 2,
    "papers": 12
   },
   {
    "name": "Kun Li",
    "id": "2327910782",
    "h_index": 4,
    "papers": 17
   },
   {
    "name": "Hao Wu",
    "id": "2349380257",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Yunxin Liu",
    "id": "2440971123",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Ting Cao",
    "id": "2326975040",
    "h_index": 4,
    "papers": 13
   }
  ],
  "comment": "",
  "topics": [
   "navigation"
  ],
  "orgs": [
   "Physical Intelligence"
  ],
  "abs_url": "https://arxiv.org/abs/2608.16590v1",
  "pdf_url": "https://arxiv.org/pdf/2608.16590v1",
  "html_url": "https://arxiv.org/html/2608.16590v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2608.16572",
  "slug": "vihateleop-a-low-cost-lightweight-visual-haptic-teleoperation-system-f",
  "title": "ViHaTeleop: A Low-Cost, Lightweight Visual-Haptic Teleoperation System for Dexterous Manipulation Learning",
  "abstract": "Learning from demonstration is a promising approach for dexterous manipulation, but collecting high-quality contact-critical demonstrations remains difficult with low-cost teleoperation hardware. We present ViHaTeleop, a lightweight (0.7 kg), low-cost (\\$550) visual-haptic teleoperation system with SLAM-based wrist tracking, camera-based hand tracking, and finger-wise vibrotactile feedback through Linear Resonant Actuators (LRA). The system includes several design choices (LED illumination, fisheye hand camera, and tactile-aware retargeting constraints) and is deployed on Franka + LEAP Hand + 9DTact in both real and simulated environments. Under matched with/without-haptic conditions with nine participants across six contact-critical tasks, haptics improved success rates across all tasks (+2.2 to +15.6 percentage points), while completion-time effects were task-dependent. Subjective ratings showed significant gains in contact clarity and grasp confidence in both simulation and real-world settings (Wilcoxon signed-rank, $p<0.05$). We also integrate a lightweight depth-camera-based tactile proxy in Isaac Sim, enabling a full pipeline from multi-modal demonstration collection to visual-tactile policy training. Preliminary downstream validation by training visual-tactile policies from collected demonstrations shows tactile cues benefit contact-critical subtasks (peg-in-hole: +17 percentage points over vision-only).",
  "published": "2026-08-17",
  "updated": "2026-08-17",
  "year": "2026",
  "authors": [
   "Fucai Zhu",
   "Yanhou Lai",
   "Paul Maestre",
   "Koichi Hashimoto"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Preliminary downstream validation by training visual-tactile policies from collected demonstrations shows tactile cues benefit contact-critical subtasks (peg-in-hole: +17 percentage points over vision-only) and preliminary downstream validation by training visual-tactile policy training shows tactile cues benefit contact-critical subtasks.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Fucai Zhu",
    "id": "30772053",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Yanhou Lai",
    "id": "2458076887",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Paul Maestre",
    "id": "2458107109",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "K. Hashimoto",
    "id": "1696089",
    "h_index": 30,
    "papers": 286
   }
  ],
  "comment": "Accepted to the 2026 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS). 8 pages",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "spatial-3d",
   "data-teleop",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.16572v1",
  "pdf_url": "https://arxiv.org/pdf/2608.16572v1",
  "html_url": "https://arxiv.org/html/2608.16572v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.16555",
  "slug": "co-design-of-neural-and-muscle-network-based-on-embodied-perceptron-re",
  "title": "Co-design of Neural and Muscle Network based on Embodied Perceptron Representation",
  "abstract": "Recent advances in AI technologies have enabled the advanced design of complex control policies. In contrast, focusing on the body, many robots still employ simple bodies that can limit adaptability to environments. Studies in embodied robotics have shown that well-designed bodies can partially replace the role of control and computation with physical body-environment interactions, yet such designs still depend heavily on expert intuition. There is a need for a systematic theoretical framework for body design, as well as a method for joint optimization of the body and controller. To address this, we introduce the Embodied Perceptron, a theoretical framework that unifies neural networks and physical body systems. In this view, the body itself acts as a perceptron: mechanical parameters correspond to weights, and physical nonlinearities play the role of activation functions. By representing physical constraints as weights and nonlinear properties as activation functions, a physical body can be modeled in neural-network form. The system representation enables us to explicitly and theoretically explain that the body can substitute for part of the neural control. As an application, we co-optimize control policy and muscle configuration in a musculoskeletal robot and show that the resulting embodied intelligence can provide inherent stability, improve learning efficiency, and drastically reduce model size-even with a single-neuron controller. The results bridge the informational and physical worlds and provide a pathway toward understanding and systematic design of embodied AI systems.",
  "published": "2026-08-17",
  "updated": "2026-08-17",
  "year": "2026",
  "authors": [
   "Siyuan Tao",
   "Yoichi Masuda",
   "Hiroyuki Nabae",
   "Masato Ishikawa"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work co-optimize control policy and muscle configuration in a musculoskeletal robot and shows that the resulting embodied intelligence can provide inherent stability, improve learning efficiency, and drastically reduce model size\u2014even with a single-neuron controller.",
  "doi": "10.1109/SII64115.2026.11404727",
  "oa_pdf": "https://doi.org/10.48550/arxiv.2608.16555",
  "s2_authors": [
   {
    "name": "Siyuan Tao",
    "id": "2370358723",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Yoichi Masuda",
    "id": "2352515142",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Hiroyuki Nabae",
    "id": "2355813699",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Masato Ishikawa",
    "id": "2297558705",
    "h_index": 1,
    "papers": 9
   }
  ],
  "comment": "10 pages, 7 figures, 2026 IEEE/SICE International Symposium on System Integration (SII)",
  "topics": [
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.16555v1",
  "pdf_url": "https://arxiv.org/pdf/2608.16555v1",
  "html_url": "https://arxiv.org/html/2608.16555v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.16503",
  "slug": "nebulavla-a-dual-frequency-vision-language-action-model-with-guide-act",
  "title": "NebulaVLA: A Dual-Frequency Vision-Language-Action Model With Guide Action for Robotic Manipulation",
  "abstract": "Real-world deployment of Vision-Language-Action (VLA) models is often bottlenecked by efficiency-performance trade-offs, cross-embodiment generalization, and execution smoothness. We present NebulaVLA, an asynchronous dual-frequency architecture that decouples high-level semantic reasoning from low-level action control, optimizing computational resources and modularity. To bridge semantic gaps across heterogeneous robots, we introduce GESTURE-7, a unified language-grounded action representation. Furthermore, our Guide Action algorithm enforces kinematic continuity via mask-based smoothness constraints. Comprehensive evaluations demonstrate that NebulaVLA significantly outperforms synchronous baselines, achieving an 85.5\\% average success rate on LIBERO-Plus and accelerating action generation by \\textasciitilde 2.7$\\times$. This asynchronous design enables highly efficient and responsive control for practical robotics.",
  "published": "2026-08-17",
  "updated": "2026-08-17",
  "year": "2026",
  "authors": [
   "Cong Zhao",
   "Shuai Tian",
   "Xu Zhang",
   "Baocheng Ni",
   "Xinguo Song",
   "Xueying Sun",
   "Shu Jiang",
   "Shouchang Yang",
   "Bo Tang",
   "Jin Deng",
   "Ge Zhu",
   "YongCheng Wang",
   "Jin Xu",
   "Ri Yang"
  ],
  "author_count": 14,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "NebulaVLA is presented, an asynchronous dual-frequency architecture that decouples high-level semantic reasoning from low-level action control, optimizing computational resources and modularity and introduces GESTURE-7, a unified language-grounded action representation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Congyu Zhao",
    "id": "2454991312",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Shuai Tian",
    "id": "2345921369",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Xu Zhang",
    "id": "2384869575",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Baocheng Ni",
    "id": "2458099108",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Xinguo Song",
    "id": "2458115867",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Xueying Sun",
    "id": "2109229020",
    "h_index": 6,
    "papers": 18
   },
   {
    "name": "Shu Jiang",
    "id": "2456833010",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Shouchang Yang",
    "id": "2458317465",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Bosch Tang",
    "id": "2360166253",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Jin Deng",
    "id": "2450169509",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Ge Zhu",
    "id": "2458109398",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Yongchen Wang",
    "id": "2456391728",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Jin Xu",
    "id": "2385500422",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Richa Yang",
    "id": "2452461788",
    "h_index": 0,
    "papers": 1
   }
  ],
  "comment": "14 pages, 5 figures",
  "topics": [
   "vla",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.16503v1",
  "pdf_url": "https://arxiv.org/pdf/2608.16503v1",
  "html_url": "https://arxiv.org/html/2608.16503v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.16499",
  "slug": "occamview-object-conditioned-view-selection-for-frame-budgeted-active",
  "title": "OccamView: Object-Conditioned View Selection for Frame-Budgeted Active 3D Gaussian Reconstruction",
  "abstract": "Active 3D Gaussian reconstruction fundamentally relies on selecting informative next-best views under limited sensing budgets. Existing active 3DGS methods primarily plan viewpoints according to geometric information gain, treating object-induced hidden regions in the same manner as general unexplored space. Under tight frame budgets, such geometry-driven strategies may prioritize global scene coverage while leaving partially observed objects incompletely reconstructed. To address this limitation, we propose OccamView, an object-conditioned view-selection framework for frame-budgeted active 3D Gaussian reconstruction. Rather than predicting unseen object geometry or performing shape completion, OccamView maintains an online object memory from open-vocabulary detections grounded in measured RGB-D observations and represents unresolved local occupancy around detected objects as conservative hidden-region proxies. Candidate viewpoints are then evaluated using an occlusion-aware proxy-coverage score. Furthermore, we introduce a Geo-Floor mechanism that restricts object-conditioned re-ranking to geometrically competitive candidates, allowing object-conditioned cues to guide complementary observations while preserving the geometry-driven exploration behavior of the underlying planner. Experiments on Replica and Matterport3D under a unified frame-budgeted protocol show that OccamView consistently reduces Completion and improves Completion Ratio across five frame budgets, with particularly pronounced gains under limited frame budgets. These results demonstrate that lightweight object-conditioned cues effectively complement geometry-driven active view planning.",
  "published": "2026-08-17",
  "updated": "2026-08-17",
  "year": "2026",
  "authors": [
   "Hongbo Gao",
   "Wei Zhang",
   "Zeyu Ni",
   "Dihao Zhu",
   "Ruifeng Li",
   "Yunke Wang",
   "Chang Xu"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Experiments show that OccamView consistently reduces Completion and improves Completion Ratio across five frame budgets, with particularly pronounced gains under limited frame budgets, and demonstrate that lightweight object-conditioned cues effectively complement geometry-driven active view planning.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hongbo Gao",
    "id": "2293140848",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Wei Zhang",
    "id": "2456387104",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Zeyu Ni",
    "id": "2458081663",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Di Zhu",
    "id": "2375301481",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Ruifeng Li",
    "id": "2458437042",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Yunke Wang",
    "id": "2119215768",
    "h_index": 10,
    "papers": 43
   },
   {
    "name": "Chang Xu",
    "id": "2292018438",
    "h_index": 6,
    "papers": 20
   }
  ],
  "comment": "7 pages, 5 figures. Preprint",
  "topics": [
   "spatial-3d",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.16499v1",
  "pdf_url": "https://arxiv.org/pdf/2608.16499v1",
  "html_url": "https://arxiv.org/html/2608.16499v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.16476",
  "slug": "exposing-the-long-tail-in-embodied-urban-navigation-via-scalable-learn",
  "title": "Exposing the Long-tail in Embodied Urban Navigation via Scalable Learning from In-the-Wild Videos",
  "abstract": "Learning embodied urban navigation policies from real-world data is constrained by the cost of task-specific data collection and the limited coverage of rare yet safety-critical scenarios. To address these challenges, we present a scalable framework for learning point-goal urban navigation from web-scale in-the-wild egocentric videos while systematically exposing its long tail. The framework automatically annotates uncurated web videos with metric trajectories and structured navigation semantics, which are then used to train a vision-language-action policy for interpretable navigation planning. We characterize the long tail based on model performance and the distribution of perception-motion patterns, and employ reflection-based analysis to diagnose recurring failure modes. Experiments on web-video data and real-world urban navigation tasks demonstrate effective knowledge transfer from unconstrained videos and reveal coherent long-tail structures beyond aggregate navigation performance.",
  "published": "2026-08-17",
  "updated": "2026-08-17",
  "year": "2026",
  "authors": [
   "Bingyi Xia",
   "Han Bao",
   "Zhewei Chen",
   "Hanjing Ye",
   "Jingwen Yu",
   "Yuhan Pang",
   "Wenjun Xu",
   "Jiankun Wang"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work presents a scalable framework for learning point-goal urban navigation from web-scale in-the-wild egocentric videos while systematically exposing its long tail, and characterize the long tail based on model performance and the distribution of perception-motion patterns.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Bingyi Xia",
    "id": "2348400978",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Han Bao",
    "id": "47469612",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Zhewei Chen",
    "id": "2458250530",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Hanjing Ye",
    "id": "2160747744",
    "h_index": 7,
    "papers": 27
   },
   {
    "name": "Jingwen Yu",
    "id": "2180798953",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Yuhan Pang",
    "id": "2198793157",
    "h_index": 2,
    "papers": 13
   },
   {
    "name": "Wenjun Xu",
    "id": "2379807474",
    "h_index": 1,
    "papers": 11
   },
   {
    "name": "Jiankun Wang",
    "id": "51068901",
    "h_index": 25,
    "papers": 126
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "egocentric-data",
   "navigation",
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.16476v1",
  "pdf_url": "https://arxiv.org/pdf/2608.16476v1",
  "html_url": "https://arxiv.org/html/2608.16476v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.16447",
  "slug": "harecap-habitual-action-grounding-for-recursive-large-language-model-a",
  "title": "HaReCAP: Habitual-action Grounding for Recursive Large Language Model Agents",
  "abstract": "Long-horizon embodied tasks require LLM agents to iteratively decompose high-level goals, revise plans in response to environmental feedback, and ground leaf-level subgoals into valid executable actions. Recursive context-management methods such as ReCAP improve planning stability through multi-level task decomposition and parent-node refinement, but still repeatedly invoke the LLM at leaf nodes to ground atomic subtasks into exact valid actions. We refer to this final grounding step as last-mile grounding redundancy, which accumulates into substantial LLM-call and token overhead during long-horizon execution. To mitigate this issue, we propose HaReCAP (Habitual-action Grounded ReCAP), a low-intrusion leaf grounding extension for ReCAP. HaReCAP extracts frequent leaf decisions from successful trajectories and compiles them offline into auditable and abstainable one-step leaf-reflex rules. At runtime, it skips the leaf LLM call only when a rule can uniquely determine a legal action in the current valid-action set; otherwise, it falls back to the original ReCAP. This design avoids repeatedly carrying the full recursive context into the LLM for routine leaf action grounding, while preserving the original recursive control flow. We evaluate HaReCAP on Robotouille and ALFWorld with Qwen3.5-27B as the main model. On tasks solved by both ReCAP and HaReCAP, HaReCAP reduces token consumption by 14.67%, 17.93%, and 20.08% on Robotouille synchronous, Robotouille asynchronous, and ALFWorld, respectively. The results show that HaReCAP can serve as a low-intrusion extension to ReCAP-style recursive context-management frameworks, reducing last-mile grounding redundancy across environments and models on commonly successful trajectories.",
  "published": "2026-08-17",
  "updated": "2026-08-17",
  "year": "2026",
  "authors": [
   "Shen Liu",
   "Zhenguo Xu",
   "Shaopu Wang",
   "Yike Gao",
   "Chunlei Wang"
  ],
  "author_count": 5,
  "categories": [
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.AI",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The results show that HaReCAP can serve as a low-intrusion extension to ReCAP-style recursive context-management frameworks, reducing last-mile grounding redundancy across environments and models on commonly successful trajectories.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Shengchang Liu",
    "id": "2450734415",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Zhenguo Xu",
    "id": "2458128233",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Shaopu Wang",
    "id": "2347552445",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Yike Gao",
    "id": "2388581436",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Chunlei Wang",
    "id": "2345879840",
    "h_index": 1,
    "papers": 9
   }
  ],
  "comment": "15 pages, 3 figures",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.16447v1",
  "pdf_url": "https://arxiv.org/pdf/2608.16447v1",
  "html_url": "https://arxiv.org/html/2608.16447v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.16442",
  "slug": "observation-constrained-joint-space-viewpoint-optimization-for-robotic",
  "title": "Observation-Constrained Joint-Space Viewpoint Optimization for Robotic Inspection of Cylindrical Cavities",
  "abstract": "Inspection is a core capability in many mobile robotics applications, including industrial facility monitoring, infrastructure maintenance, agriculture, and search and rescue. Observing the bottom of a cylindrical cavity, as required by ASTM search-task benchmarks for response robots, presents a representative challenge: the robot must position its camera precisely while satisfying visibility, kinematic, and collision constraints. This paper presents a fully autonomous method for observation-constrained inspection of cylindrical cavities in robot joint space. Rather than prescribing a single Cartesian camera pose, the method represents the inspection objective as a set of valid viewing geometries, thereby avoiding the rejection of reachable viewpoints and configurations with poor joint-limit margins. An RGB perception front end estimates the opening center and directed cavity axis from semantic masks using arc-supported ellipse fitting together with body and side-generator cues. These estimates parameterize constraints on camera-axis alignment, lateral offset, and axial standoff. A multistart derivative-free search then optimizes robot joint configurations with lexicographic priority given to constraint satisfaction; feasible configurations are ranked according to motion economy, joint-limit margin, and view quality. The resulting candidates are evaluated by a collision-aware motion planner, and the executed camera pose is verified geometrically and using a ray-based estimate of bottom visibility. In Isaac Sim, the proposed method successfully completes 92 of 100 target configurations and attains 91.65% mean bottom visibility among executed trials, compared with 76 of 100 and 84.3% for a multistart coordinate-search baseline. Tabletop and Unitree A2-mounted experiments demonstrate the complete perception-planning-execution pipeline.",
  "published": "2026-08-17",
  "updated": "2026-08-17",
  "year": "2026",
  "authors": [
   "Yuezhong Wang",
   "Rongshen Yin",
   "Bichi Zhang",
   "S\u00f6ren Schwertfeger"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yuezhong Wang",
    "id": "2458699815",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Rongshen Yin",
    "id": "2458092804",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Bichi Zhang",
    "id": "2213712686",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "S\u00f6ren Schwertfeger",
    "id": "2267572355",
    "h_index": 4,
    "papers": 12
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2608.16442v1",
  "pdf_url": "https://arxiv.org/pdf/2608.16442v1",
  "html_url": "https://arxiv.org/html/2608.16442v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2608.16433",
  "slug": "robot-body-aware-traversal-risk-graph-planning-for-wheeled-legged-robo",
  "title": "Robot-Body-Aware Traversal Risk Graph Planning for Wheeled-Legged Robots in Complex Terrain",
  "abstract": "Traversal Risk Graphs (TRGs) provide a compact, terrain-aware representation for global navigation, but native TRG costs are computed over circular node neighborhoods and edge-aligned terrain regions rather than the robot's oriented body footprint. For wheeled-legged robots, this abstraction can miss partial support loss and body-terrain interference, especially during turns. We present Robot-Body-Aware TRG planning (RB-TRG), which builds on the sparse TRG representation and lifts edge-wise terrain-risk search to heading- and turn-aware body-risk transitions. An oriented rectangular footprint is sampled along graph edges and yaw sweeps to measure longitudinal support variation, lateral inclination, terrain interference, and exposure to untrusted map regions. Mean-and-upper-tail features are incorporated into transition costs, whose accumulated value is minimized by A* over ordered node-pair states, preserving TRG construction and its planning interface. We evaluate RB-TRG in a same-graph study on four scanned terrain environments and in paired closed-loop MuJoCo trials. RB-TRG reduces the three core geometric body-placement metrics and increases end-to-end success from 51.5% to 68.5%, while increasing mean path length by 2.3%. A Go2-W deployment further demonstrates RB-TRG with a full LiDAR navigation stack, which received the Best Autonomy and Best Mobility awards at the IEEE ICRA 2026 Legged Robot Challenges. The code for RB-TRG is released at https://github.com/ZhiqiaoGuo/RB-TRG.",
  "published": "2026-08-17",
  "updated": "2026-08-17",
  "year": "2026",
  "authors": [
   "Zhiqiao Guo",
   "Bichi Zhang",
   "S\u00f6ren Schwertfeger"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work presents Robot-Body-Aware TRG planning (RB-TRG), which builds on the sparse TRG representation and lifts edge-wise terrain-risk search to heading- and turn-aware body-risk transitions and reduces the three core geometric body-placement metrics and increases end-to-end success.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhiqiao Guo",
    "id": "2458112533",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Bichi Zhang",
    "id": "2213712686",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "S\u00f6ren Schwertfeger",
    "id": "2267572355",
    "h_index": 4,
    "papers": 12
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.16433v1",
  "pdf_url": "https://arxiv.org/pdf/2608.16433v1",
  "html_url": "https://arxiv.org/html/2608.16433v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.16351",
  "slug": "arm-aware-guided-dexterous-grasp-generation-with-arm-agnostic-grasp-mo",
  "title": "Arm-Aware Guided Dexterous Grasp Generation with Arm-Agnostic Grasp Models",
  "abstract": "Dexterous grasp generation that considers arm-related constraints is crucial in real-world scenarios involving arm environment collision avoidance, workspace boundary grasps, and consecutive grasping. Existing hand-centric grasp models, which primarily focus on the floating hand's pose, are insufficient for such cases. Conventional arm-aware methods either rely on rejection sampling to discard infeasible samples or require retraining on arm-specific data, leading to low sample efficiency under adverse conditions or limited generalization across different robots and environments. To overcome these limitations, this letter presents an arm-aware dexterous grasp generation framework that leverages pretrained arm-agnostic grasp models while integrating arm and environmental information only at inference time. Specifically, we formulate arm-aware constrained grasp generation as a joint optimization of hand pose and arm configuration, and derive closed-form gradients for arm-related constraints. Assuming the hand pose distribution is represented by a diffusion model, we prove that gradient-based optimization is equivalent to guided diffusion sampling, steering near-feasible samples toward the feasible region. Through comprehensive evaluation involving 10k objects across 6 scenarios, we demonstrate that the proposed framework generates feasible grasps in highly constrained settings with significantly higher probability, highlighting its advantages in real-world applications. Supplementary materials and appendix are available at https://arm-aware-dexgrasp.github.io/.",
  "published": "2026-08-17",
  "updated": "2026-08-17",
  "year": "2026",
  "authors": [
   "Yongyi Jia",
   "Yongpeng Jiang",
   "Kangchen Lv",
   "Yi Ren",
   "Mingrui Yu",
   "Xiang Li"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "RA-L",
  "venue_source": "semantic-scholar",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "This letter forms arm-aware constrained grasp generation as a joint optimization of hand pose and arm configuration, and derive closed-form gradients for arm-related constraints, and proves that gradient-based optimization is equivalent to guided diffusion sampling, steering near-feasible samples toward the feasible region.",
  "doi": "10.1109/LRA.2026.3674025",
  "oa_pdf": "https://doi.org/10.48550/arxiv.2608.16351",
  "s2_authors": [
   {
    "name": "Yongyi Jia",
    "id": "2198480005",
    "h_index": 5,
    "papers": 17
   },
   {
    "name": "Yongpeng Jiang",
    "id": "2258802294",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Kangchen Lv",
    "id": "1648745587",
    "h_index": 8,
    "papers": 15
   },
   {
    "name": "Yi Ren",
    "id": "2276490366",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Xiang Li",
    "id": "2328522222",
    "h_index": 3,
    "papers": 9
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "data-teleop",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.16351v1",
  "pdf_url": "https://arxiv.org/pdf/2608.16351v1",
  "html_url": "https://arxiv.org/html/2608.16351v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.8
 },
 {
  "id": "2608.16335",
  "slug": "readiness-barrier-functions-forward-invariant-control-authority-for-ov",
  "title": "Readiness Barrier Functions: Forward-Invariant Control Authority for Overactuated Multirotor Allocation",
  "abstract": "Allocation schemes that greedily maximize a readiness metric over the actuator fiber bundle of an overactuated multirotor produce commands that jump between disconnected optimal strata, demanding actuator rates no motor can deliver; effort-minimizing schemes are continuous but cannot guarantee that wrench-rate authority stays above any certified level. We reconcile the two by treating authority as a forward-invariant quantity: a control barrier function on the log-determinant of the drag-aware actuator-authority co-metric, enforced at torque level by a quadratic program in the allocation null space. A single design inequality renders the certified set compact and strictly interior to the actuator box, with the readiness cost of any rotor deactivation given in closed form as $\\ln(n/(n{-}m))$ for symmetric designs. Tracking is sacrificed only through an explicit alignment ratio, with wrench error bounded by $\\mathcal{O}(\u03c1^{-1/2})$ and a robust variant handles motor-parameter uncertainty with a closed-form floor shift independent of the airframe matrix. On a hexarotor and a fully-actuated octorotor the closed-form gap matches simulation to machine precision; in the authority-scarce regime greedy maximization violates the certified floor and commits wrench errors up to eighty times larger than the proposed filter, which holds invariance of the certified set at negligible tracking cost.",
  "published": "2026-08-17",
  "updated": "2026-08-17",
  "year": "2026",
  "authors": [
   "Giuseppe Silano"
  ],
  "author_count": 1,
  "categories": [
   "cs.RO",
   "eess.SY",
   "math.OC"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Giuseppe Silano",
    "id": "49552909",
    "h_index": 11,
    "papers": 32
   }
  ],
  "comment": "This work has been submitted to the IEEE for possible publication",
  "topics": [
   "rl-control",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.16335v1",
  "pdf_url": "https://arxiv.org/pdf/2608.16335v1",
  "html_url": "https://arxiv.org/html/2608.16335v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.16281",
  "slug": "marker-constrained-pose-graph-correction-for-cross-platform-georeferen",
  "title": "Marker-Constrained Pose-Graph Correction for Cross-Platform Georeferencing in GNSS-Denied Environments",
  "abstract": "Autonomous operation in GNSS-denied environments requires heterogeneous mapping pipelines to maintain a consistent spatial reference. This paper presents a framework using camouflage-matched fiducial markers fabricated from Cholesteric Spherical Reflectors (CSRs) as pre-surveyed visual anchors. The anchors georeference both a lightweight LiDAR-odometry trajectory and a dense RTAB-Map reconstruction, allowing their outputs to be expressed in a common LUREF frame (geodetic coordinate reference system used in Luxembourg) without requiring GNSS measurements during operation. The method combines coarse similarity alignment with marker-constrained pose-graph optimization. We evaluate it using two handheld acquisition sessions with ground-level and elevated motion profiles emulating UGV and UAV operation. A single iMarker was relocated among six surveyed positions, with the first position revisited to quantify drift correction. Marker-anchor correction reduced revisit inconsistency by 97.9% and 99.1% for the UAV- and UGV-emulating sessions, respectively, and improved held-out anchor prediction compared with one-time alignment. Separately georeferenced dense reconstructions achieved a median cross-session nearest-neighbour distance of 58 cm without explicit cross-session registration. Marker processing operated in real time, while trajectory correction required less than 0.25 s per session. These results demonstrate a proof of concept for georeferencing lightweight odometry and dense reconstructions using visually unobtrusive, pre-surveyed anchors during GNSS-denied operation.",
  "published": "2026-08-17",
  "updated": "2026-08-17",
  "year": "2026",
  "authors": [
   "Marco Giberna",
   "Jose Luis Sanchez Lopez",
   "Holger Voos"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.MA"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Marco Giberna",
    "id": "2314828583",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Jos\u00e9 Luis S\u00e1nchez L\u00f3pez",
    "id": "145089408",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Holger Voos",
    "id": "2060086740",
    "h_index": 12,
    "papers": 66
   }
  ],
  "comment": "14 pages, 5 figures, 5 tables, submitted to SPIE Security + Defence conference",
  "topics": [
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.16281v1",
  "pdf_url": "https://arxiv.org/pdf/2608.16281v1",
  "html_url": "https://arxiv.org/html/2608.16281v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.16264",
  "slug": "cyclops-lidar-as-a-camera-that-dreams-in-color",
  "title": "Cyclops: LiDAR as a Camera That Dreams in Color",
  "abstract": "Conventionally, robotic perception relies heavily on cameras due to the rich semantic texture they provide. However, their performance degrades significantly in low-light or high-dynamic-range environments. Conversely, while Light Detection and Ranging (LiDAR) captures illumination-invariant geometric and intensity properties, the resulting data are typically single-channel and sparse, creating a significant modality gap when applying vision models pre-trained on RGB datasets. In this paper, we propose Cyclops, a framework that translates sparse Non-Repetitive Scanning LiDAR (NRS-LiDAR) intensity into RGB video, enabling camera-free inference for all-day perception tasks. Our approach first converts sparse LiDAR intensity projections into dense representations via a frozen pre-trained densification module, serving as a geometrically rich source condition. The dense intensity latent is then transported toward the target RGB distribution through Latent Bridge Matching (LBM) with a learned velocity field in a few ODE integration steps. To mitigate inter-frame flickering, we inject prior-frame context via temporal attention layers and further formulate the velocity field as a policy optimized by a differentiable terminal reward that encourages terminal fidelity through backpropagation along the ODE trajectory. Extensive experiments demonstrate that the synthesized RGB, including those generated under near-dark conditions, enable standard RGB-based perception models to substantially outperform both LiDAR baselines and conventional cameras on semantic segmentation, lane detection, and point cloud colorization across diverse lighting conditions.",
  "published": "2026-08-17",
  "updated": "2026-08-17",
  "year": "2026",
  "authors": [
   "Wei Gao",
   "Jian Shu",
   "Mingle Zhao",
   "Maani Ghaffari",
   "David Kong",
   "Chengzhong Xu",
   "Hui Kong"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Cyclops is proposed, a framework that translates sparse Non-Repetitive Scanning LiDAR intensity into RGB video, enabling camera-free inference for all-day perception tasks and mitigating inter-frame flickering.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Wei Gao",
    "id": "2365438547",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Jian Shu",
    "id": "2333593897",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Mingle Zhao",
    "id": "2282159819",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Maani Ghaffari",
    "id": "1389560593",
    "h_index": 17,
    "papers": 108
   },
   {
    "name": "David Kong",
    "id": "2458079860",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Chengzhong Xu",
    "id": "2153079122",
    "h_index": 10,
    "papers": 34
   },
   {
    "name": "Hui Kong",
    "id": "2312754624",
    "h_index": 3,
    "papers": 8
   }
  ],
  "comment": "",
  "topics": [
   "spatial-3d",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.16264v1",
  "pdf_url": "https://arxiv.org/pdf/2608.16264v1",
  "html_url": "https://arxiv.org/html/2608.16264v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.16229",
  "slug": "planner-conditioned-diffusion-for-coordinated-multi-agent-exploration",
  "title": "Planner-Conditioned Diffusion for Coordinated Multi-Agent Exploration",
  "abstract": "Coordinated multi-agent exploration requires not only efficient individual coverage but also non-redundant coverage across agents over extended planning horizons. Conventional approaches rely on hand-crafted coordination rules, while end-to-end multi-agent learning methods are difficult to scale and train. Diffusion-based planners such as DARE offer a promising alternative by generating long-horizon trajectories instead of single-step actions, but existing methods are trained on a narrow planner distribution, limiting behavioral diversity and inference-time controllability. We propose a Planner-Conditioned Diffusion Policy (PCDP) for graph-based multi-agent exploration. PCDP is trained on demonstrations from multiple planner styles with planner identity as an explicit conditioning input, enabling a single shared model to learn a multimodal trajectory distribution and generate diverse, controllable trajectory candidates from the same observation. Rather than learning coordination end-to-end, we reuse this multimodal single-agent policy across all agents and introduce coordination through local reranking, in which nearby agents jointly select the trajectory combination with minimal predicted overlap. We evaluate PCDP against classical and diffusion-based baselines on 100 held-out maps in a four-agent simulation setting. PCDP matches the perfect success rate of the diffusion-based baselines while improving mean max-agent travel, total team travel, and agent imbalance. Crucially, reranking alone over a single-planner baseline yields only marginal gains, indicating that planner-conditioned multimodality is the main contributor to improved coordination. Qualitative simulation results and real-robot experiments with two agents further validate that diverse long-horizon trajectory generation produces emergent spatial separation between agents without any explicit repulsion mechanism.",
  "published": "2026-08-17",
  "updated": "2026-08-17",
  "year": "2026",
  "authors": [
   "Marcus Yu Siong Teo",
   "Jeric Lew",
   "Tanishq Duhan",
   "Guillaume Sartoretti"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A Planner-Conditioned Diffusion Policy (PCDP) is proposed, trained on demonstrations from multiple planner styles with planner identity as an explicit conditioning input, enabling a single shared model to learn a multimodal trajectory distribution and generate diverse, controllable trajectory candidates from the same observation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Marcus Yu Siong Teo",
    "id": "2458109609",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Jeric Lew",
    "id": "2327050458",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "T. Duhan",
    "id": "2257346752",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "G. Sartoretti",
    "id": "2292917033",
    "h_index": 11,
    "papers": 67
   }
  ],
  "comment": "Code and models are available at https://github.com/marmotlab/PCDP",
  "topics": [
   "imitation-diffusion",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.16229v1",
  "pdf_url": "https://arxiv.org/pdf/2608.16229v1",
  "html_url": "https://arxiv.org/html/2608.16229v1",
  "code_url": "https://github.com/marmotlab/PCDP",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2608.16222",
  "slug": "hiphi-a-large-scale-benchmark-for-high-precision-human-motion-and-obje",
  "title": "HiPHI: A Large-Scale Benchmark for High-Precision Human Motion and Object-Interaction",
  "abstract": "Humanoid intelligence requires learning over an extremely diverse space of whole-body motions and physically grounded interactions. However, existing embodied datasets remain fundamentally limited: internet-scale video data lack precise physical states and interaction grounding, while laboratory motion datasets provide high fidelity but only narrow behavioral coverage. This mismatch creates a critical bottleneck for scalable humanoid policy learning. We present HiPHI, a 600+ hour scale high-fidelity whole-body human motion dataset designed to systematically maximize coverage of the human motion and interaction manifold. HiPHI is theoretically guided by FrameNet, a linguistic framework organizing human primitives. Created using an optical motion capture pipeline, HiPHI provides sub-millimeter spatial marker tracking accuracy for full-body human motion and mesh-level object trajectories. We further introduce a benchmark suite evaluating motion-space diversity, interaction grounding, object consistency, and physical AI applications. Our analyses demonstrate that HiPHI significantly expands motion coverage compared to existing motion datasets while maintaining high-fidelity interaction quality, and establishes a scalable data foundation for training, evaluating, and generalizing humanoid policies in real-world embodied tasks, where similar extensions are also applicable to motion prior models in computer graphics.",
  "published": "2026-08-17",
  "updated": "2026-08-17",
  "year": "2026",
  "authors": [
   "Jiahao Ji",
   "Ji Ma",
   "Runhan Zhang",
   "Runyi Yu",
   "Wenjia Wang",
   "Weiheng Chi",
   "Qianqian Peng",
   "Weichao Yan",
   "Yongfei Gu",
   "Ye Tian",
   "Ting Wu",
   "Longwei Li",
   "Chun Yuan",
   "Ruoli Dai",
   "Lei Han"
  ],
  "author_count": 15,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work presents HiPHI, a 600+ hour scale high-fidelity whole-body human motion dataset designed to systematically maximize coverage of the human motion and interaction manifold, and introduces a benchmark suite evaluating motion-space diversity, interaction grounding, object consistency, and physical AI applications.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jiahao Ji",
    "id": "2444911497",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Ji Ma",
    "id": "2336920494",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "Runhan Zhang",
    "id": "2445708812",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Runyi Yu",
    "id": "2352918099",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Wenjia Wang",
    "id": "2269835875",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Weiheng Chi",
    "id": "2290768210",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Qianqian Peng",
    "id": "2458102477",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Weichao Yan",
    "id": "1387715363",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Yongfei Gu",
    "id": "2458117157",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Ye Tian",
    "id": "2346053075",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Tingfan Wu",
    "id": "2254158966",
    "h_index": 12,
    "papers": 20
   },
   {
    "name": "Longwei Li",
    "id": "2447886895",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Chun Yuan",
    "id": "2455954838",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Ruoli Dai",
    "id": "2347118701",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Lei Han",
    "id": "2362318590",
    "h_index": 2,
    "papers": 6
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "egocentric-data"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.16222v1",
  "pdf_url": "https://arxiv.org/pdf/2608.16222v1",
  "html_url": "https://arxiv.org/html/2608.16222v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.16221",
  "slug": "deep-probabilistic-indoor-gas-source-localization-via-physical-depende",
  "title": "Deep Probabilistic Indoor Gas Source Localization via Physical Dependency-Guided Sequential Inference",
  "abstract": "Reliable gas source localization (GSL) is critical to safety in industrial and urban environments, yet remains challenging indoors because walls and obstacles interact with airflow to create complex gas dispersion. High-fidelity models such as computational fluid dynamics and filament models can capture these effects, but their computational cost limits online use. We propose a deep probabilistic framework that infers the source posterior from sparse and noisy measurements collected by a mobile robot. Unlike end-to-end models that directly infer source estimates from measurements, the proposed method incorporates physical dependencies of indoor gas transport, where wind and source location govern the concentration field. These dependencies are embedded through sequential conditional inference, in which inferred wind and concentration fields guide source posterior estimation. This structure improves localization under sparse and noisy observations. Evaluations show that the proposed method outperforms representative GSL baselines and enables accurate and efficient active GSL in simulations. Real-robot experiments demonstrate the feasibility of online operation on an embedded GPU.",
  "published": "2026-08-17",
  "updated": "2026-08-17",
  "year": "2026",
  "authors": [
   "Seunghwan Kim",
   "Hyungjin Kim",
   "Junhee Lee",
   "Hyondong Oh"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A deep probabilistic framework that infers the source posterior from sparse and noisy measurements collected by a mobile robot, which enables accurate and efficient active GSL in simulations and demonstrates the feasibility of online operation on an embedded GPU.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Seunghwan Kim",
    "id": "2323766679",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Hyungjin Kim",
    "id": "2457956274",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Junhee Lee",
    "id": "2361553434",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Hyondong Oh",
    "id": "2290862867",
    "h_index": 5,
    "papers": 30
   }
  ],
  "comment": "18 pages, 22 figures, 5 tables. Submitted to IEEE Transactions on Robotics",
  "topics": [
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.16221v1",
  "pdf_url": "https://arxiv.org/pdf/2608.16221v1",
  "html_url": "https://arxiv.org/html/2608.16221v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.16195",
  "slug": "robostriker-latent-space-strategic-games-for-autonomous-humanoid-boxin",
  "title": "RoboStriker: Latent-Space Strategic Games for Autonomous Humanoid Boxing",
  "abstract": "Achieving human-level competitive intelligence and physical agility in humanoid robots remains a profound challenge, particularly in contact-rich and highly dynamic tasks such as boxing. While Multi-Agent Reinforcement Learning offers a principled framework for strategic interaction, its direct application to unstructured raw motor spaces inevitably leads to joint-level physical collapse, preventing the emergence of any viable combat tactics. To resolve this fundamental conflict between strategic exploration and physical feasibility, we formulate the humanoid combat task as a novel two-player latent-space zero-sum Markov game. Under standard regularity and approximate best-response assumptions, we show that the latent formulation induces an equivalent game over the decoder-reachable action manifold, providing an approximate-Nash interpretation of the resulting self-play dynamics. To instantiate this theoretical formulation, we propose RoboStriker, a hierarchical framework that decouples high-level reasoning from low-level execution. It first distills the tracking expertise of predefined boxing motions into a topologically bounded latent manifold. This structured latent foundation subsequently drives multi-agent co-evolution via Latent-Space Neural Fictitious Self-Play. Extensive experimental results demonstrate that gaming within this structured latent space substantially outperforms direct exploration. By constraining strategic exploration through a pretrained motion decoder, RoboStriker substantially reduces the catastrophic balance failures observed in raw action-space methods and achieves superior tactical performance in both competitive win rates and striking efficiency. Finally, we successfully deploy and validate our learned combat policies on real-world humanoid robots. Our code and video and supplementary materials are available at RoboStriker.",
  "published": "2026-08-17",
  "updated": "2026-08-17",
  "year": "2026",
  "authors": [
   "Kangning Yin",
   "Kaige Liu",
   "Zhe Cao",
   "Wentao Dong",
   "Weishuai Zeng",
   "Tianyi Zhang",
   "Qiang Zhang",
   "Jingbo Wang",
   "Jiangmiao Pang",
   "Yang Li",
   "Ming Zhou",
   "Weinan Zhang"
  ],
  "author_count": 12,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "By constraining strategic exploration through a pretrained motion decoder, RoboStriker substantially reduces the catastrophic balance failures observed in raw action-space methods and achieves superior tactical performance in both competitive win rates and striking efficiency.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Kangning Yin",
    "id": "2289841621",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Kaige Liu",
    "id": "2146387179",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Zhe Cao",
    "id": "2198283035",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Wentao Dong",
    "id": "2256991109",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Weishuai Zeng",
    "id": "2315923491",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Tianyi Zhang",
    "id": "2146333670",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Qiang Zhang",
    "id": "2374148137",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Jingbo Wang",
    "id": "2363513787",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "Jiangmiao Pang",
    "id": "2405891293",
    "h_index": 2,
    "papers": 17
   },
   {
    "name": "Yang Li",
    "id": "2321328048",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Ming Zhou",
    "id": "2152174952",
    "h_index": 9,
    "papers": 19
   },
   {
    "name": "Weinan Zhang",
    "id": "2344034124",
    "h_index": 5,
    "papers": 16
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "tactile",
   "rl-control",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.16195v1",
  "pdf_url": "https://arxiv.org/pdf/2608.16195v1",
  "html_url": "https://arxiv.org/html/2608.16195v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.16172",
  "slug": "sparkvla-stop-aware-hierarchical-vla-with-adaptive-action-chunking-for",
  "title": "SparkVLA: Stop-Aware Hierarchical VLA with Adaptive Action Chunking for Long-Horizon Manipulation",
  "abstract": "At every re-observation point in a hierarchical Vision-Language-Action (VLA) system, two interface decisions must be made: when to terminate the current subtask and how far to execute the proposed action chunk. These decisions are mutually dependent---the optimal stopping point depends on what the executor plans to do, while the optimal execution length depends on where the subtask boundary lies---yet existing architectures evaluate them in isolation, an asymmetry neither module can overcome alone. We present SparkVLA, a stop-aware hierarchical VLA that resolves this mutual dependency by formulating both decisions as a single ranking: Stop competes against every action-prefix length in a unified candidate set, and the system selects the highest-scoring option, eliminating threshold tuning and requiring only offline ordinal preferences. An Anchor-Conditioned Context Encoding module caches a history-aware subtask anchor encoding onset-state memory and goal semantics, guiding visual-token pruning toward task-relevant regions; a Stop-Aware Action-Prefix Selection head scores all candidates via full self bnattention at chunk boundaries for efficiency. On RoboCerebra, SparkVLA achieves 47.12% success rate, surpassing the official hierarchical baseline by 30.57% and the strongest reproducible method by 26.83% Real-robot experiments on multi-step tasks further validate these gains on physical hardware.",
  "published": "2026-08-17",
  "updated": "2026-08-17",
  "year": "2026",
  "authors": [
   "Xunyao Lei",
   "Renjun Wu",
   "Tianlin Huo",
   "Xuesong Li"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "SparkVLA is presented, a stop-aware hierarchical VLA that resolves this mutual dependency by formulating both decisions as a single ranking: Stop competes against every action-prefix length in a unified candidate set, and the system selects the highest-scoring option, eliminating threshold tuning and requiring only offline ordinal preferences.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xunyao Lei",
    "id": "2458075587",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Renjun Wu",
    "id": "2410853913",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Tianlin Huo",
    "id": "2455190689",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Xuesong Li",
    "id": "2346896847",
    "h_index": 1,
    "papers": 10
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "imitation-diffusion"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.16172v1",
  "pdf_url": "https://arxiv.org/pdf/2608.16172v1",
  "html_url": "https://arxiv.org/html/2608.16172v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.16164",
  "slug": "trajectory-level-automatic-curriculum-learning-for-legged-locomotion-o",
  "title": "Trajectory-Level Automatic Curriculum Learning for Legged Locomotion on Unstructured Terrain",
  "abstract": "Training locomotion policies for complex unstructured terrain requires a curriculum to avoid early exploration failures. However, since unstructured terrain lacks explicit difficulty ordering for curriculum design, existing methods resort to heuristic curricula over parameterized terrains. This abstraction limits generalization, as policies can overadapt to near-fixed perceptual patterns. To address this, we propose \\textbf{\\ourname{}}, an \\textbf{T}rajectory-level \\textbf{A}utomatic \\textbf{C}urriculum \\textbf{L}earning framework that generates training tasks directly from unstructured terrain maps. At each curriculum update, the evaluator learns a difficulty function for the current policy that maps a given trajectory task to a difficulty score. The sampler then proposes new trajectories guided by the learned evaluator as the curriculum for the next policy update. This forms a closed loop in which the curriculum is iteratively matched to the evolving policy. Quantitative and qualitative experiments show that \\ourname{} continuously provides effective curricula on unstructured terrain, improving trajectory success rate by \\(56.3\\%\\) over direct training without curriculum. Compared with handcrafted curriculum learning, our method improves success rate by \\(18.5\\%\\) on the hardest terrain tasks and by up to \\(39.74\\%\\) when evaluating traversal from diverse approach directions on the same obstacle type.",
  "published": "2026-08-17",
  "updated": "2026-08-17",
  "year": "2026",
  "authors": [
   "Rocky Liu",
   "Tengyu Liu",
   "Baoxiong Jia",
   "Fangwei Zhong",
   "Xinyi Tong",
   "Hongzhao Xie",
   "Siyuan Huang"
  ],
  "author_count": 7,
  "categories": [
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.AI",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Quantitative and qualitative experiments show that this proposed curriculum framework continuously provides effective curricula on unstructured terrain, improving trajectory success rate by \\(56.3\\%\\) over direct training without curriculum.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Rocky Liu",
    "id": "2458124082",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Tengyu Liu",
    "id": "2110032600",
    "h_index": 21,
    "papers": 31
   },
   {
    "name": "Baoxiong Jia",
    "id": "26663607",
    "h_index": 27,
    "papers": 61
   },
   {
    "name": "Fangwei Zhong",
    "id": "27093563",
    "h_index": 19,
    "papers": 40
   },
   {
    "name": "Xinyi Tong",
    "id": "2330237213",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Hongzhao Xie",
    "id": "2330382531",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Siyuan Huang",
    "id": "2264375840",
    "h_index": 13,
    "papers": 26
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.16164v1",
  "pdf_url": "https://arxiv.org/pdf/2608.16164v1",
  "html_url": "https://arxiv.org/html/2608.16164v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.16153",
  "slug": "unified-condition-action-modeling-for-accurate-one-step-action-generat",
  "title": "Unified Condition-Action Modeling for Accurate One-Step Action Generation",
  "abstract": "Robot manipulation requires policies that are both accurate and efficient, as robot control must respond to changing observations under tight latency constraints. Recent diffusion and flow policies are promising, but they often treat conditions as auxiliary signals rather than jointly evolving them with action trajectories. We find that this limitation can be effectively mitigated by a \\textbf{simple yet effective unified condition-action modeling design} that represents conditions and actions in a shared token space, allowing a compact model to achieve high performance while improving both inference speed and accuracy. Therefore, we propose UCA-Flow, a unified condition-action modeling framework for accurate one-step action generation. Our method unifies observation conditions, timestep conditions, interval conditions, and action tokens into a single sequence, and processes them with a Unified Condition-Action Transformer for joint condition-action representation learning. As a result, condition representations are dynamically reconstructed according to the current generation stage, highlighting information most relevant for action refinement. Furthermore, we introduce an improved dual-pass supervision scheme over $u$ and $v$ for stronger optimization of unified condition-action modeling. UCA-Flow improves the average success rate by 9.3 percentage points over the strongest baseline, while achieving $45.6\\times$ and $33.4\\times$ speedups over DP3 and Simple DP3, and remaining $4.3\\times$ and $2.3\\times$ faster than one-step FlowPolicy and MP1, respectively.",
  "published": "2026-08-17",
  "updated": "2026-08-19",
  "year": "2026",
  "authors": [
   "Xinyu Zhou",
   "Zikun Cai",
   "Kuangji Zuo",
   "Gen Li",
   "Boyu Ma",
   "Yanshuo Lu",
   "Yutong Song",
   "Mingqi Yuan",
   "Jiayu Chen",
   "Jianfei Yang"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes UCA-Flow, a unified condition-action modeling framework for accurate one-step action generation that unifies observation conditions, timestep conditions, interval conditions, and action tokens into a single sequence, and processes them with a Unified Condition-Action Transformer for joint condition-action representation learning.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xinyu Zhou",
    "id": "2265357596",
    "h_index": 2,
    "papers": 12
   },
   {
    "name": "Zikun Cai",
    "id": "2457708591",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Kuangji Zuo",
    "id": "98207822",
    "h_index": 1,
    "papers": 12
   },
   {
    "name": "Gen Li",
    "id": "2306977914",
    "h_index": 2,
    "papers": 17
   },
   {
    "name": "Boyu Ma",
    "id": "2275779097",
    "h_index": 4,
    "papers": 29
   },
   {
    "name": "Yanshuo Lu",
    "id": "2441685214",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Yutong Song",
    "id": "2454309568",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Mingqi Yuan",
    "id": "2304366633",
    "h_index": 5,
    "papers": 21
   },
   {
    "name": "Jiayu Chen",
    "id": "2371088879",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Jianfei Yang",
    "id": "2404007795",
    "h_index": 2,
    "papers": 26
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.16153v2",
  "pdf_url": "https://arxiv.org/pdf/2608.16153v2",
  "html_url": "https://arxiv.org/html/2608.16153v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.16074",
  "slug": "us-vla-an-ultrasound-vision-language-action-model-for-embodied-abdomin",
  "title": "US-VLA: An Ultrasound Vision-Language-Action Model for Embodied Abdomina",
  "abstract": "Artificial intelligence-assisted ultrasound scanning enhances diagnostic reliability and efficiency by providing real-time guidance for standardized image acquisition and reducing operator dependence. However, existing reinforcement learning and learning-assisted ultrasound scanning methods typically rely on carefully designed reward functions or extensive interaction data, which limits their generalization ability and stability across different devices, patient populations, and complex clinical scenarios. To address these challenges, we propose an ultrasound vision-language-action model (US-VLA) for automated ultrasound scanning that explicitly encodes clinical semantic goals and generates sequential probe manipulation actions under real-time ultrasound feedback. In particular, we first design an ultrasound-aware expert fusion module to jointly integrate ultrasound observations with auxiliary contextual information, enabling semantic ultrasound feedback to effectively guide the scanning process. Then, we construct US-VLA-Data, a real-world dataset covering liver and kidney examinations, which includes five clinically defined standard planes and comprises 320 expert scanning trajectories with approximately 80,000 synchronized timesteps. Extensive experiments demonstrate that US-VLA achieves competitive performance in ultrasound probe manipulation tasks, indicating its effectiveness and promising generalization within the evaluated abdominal ultrasound setting. The source code is available at https://github.com/VMVLab/US-VLA.",
  "published": "2026-08-17",
  "updated": "2026-08-17",
  "year": "2026",
  "authors": [
   "Cheng Zhang",
   "Xingzheng Wu",
   "Guihao Yan",
   "Xifeng Hu",
   "Zhi Liu",
   "Mei Wu",
   "Qing Cai"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "An ultrasound vision-language-action model (US-VLA) for automated ultrasound scanning that explicitly encodes clinical semantic goals and generates sequential probe manipulation actions under real-time ultrasound feedback is proposed.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Cheng Zhang",
    "id": "2404674401",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Xingzhen Wu",
    "id": "2351727363",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Guihao Yan",
    "id": "2395714066",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Xifeng Hu",
    "id": "2333156418",
    "h_index": 2,
    "papers": 11
   },
   {
    "name": "Zhi Liu",
    "id": "2333520867",
    "h_index": 3,
    "papers": 14
   },
   {
    "name": "Mei Wu",
    "id": "2349339706",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Qing Cai",
    "id": "2339477781",
    "h_index": 1,
    "papers": 4
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.16074v1",
  "pdf_url": "https://arxiv.org/pdf/2608.16074v1",
  "html_url": "https://arxiv.org/html/2608.16074v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.16058",
  "slug": "surgvil-scaling-surgical-robot-imitation-learning-with-open-source-sur",
  "title": "SurgVIL: Scaling Surgical Robot Imitation Learning with Open-source Surgical Videos",
  "abstract": "Learning-based surgical robot autonomy requires large-scale demonstrations with synchronized videos and robot actions, but such data are exceedingly rare in clinical or realistic tissue settings because robot kinematics are typically inaccessible outside controlled research systems. In contrast, phantom data collected on research platforms provide accurate action labels but lack the visual diversity of real tissue. We propose SurgVIL, a framework for scaling surgical robot imitation learning using open-source surgical videos. SurgVIL combines kinematically labeled phantom robot demonstrations with surgical videos from open-source datasets and online sources for policy learning. Since these videos lack robot motion labels, we estimate approximate kinematics as weak supervision. We evaluate SurgVIL on two da Vinci robot tasks: needle pick-up and cholecystectomy cutting. Across ACT, $\u03c0_0$, and GR00T-H backbones, adding surgical videos substantially improves generalization to real-tissue and out-of-distribution settings, suggesting a scalable path from phantom training toward generalizable surgical robot policies.",
  "published": "2026-08-17",
  "updated": "2026-08-17",
  "year": "2026",
  "authors": [
   "Xinhao Chen",
   "JuoTung Chen",
   "Nigel Nelson",
   "Antony Goldenberg",
   "Jesse Haworth",
   "Sean D. Huver",
   "Axel Krieger"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "SurgVIL, a framework for scaling surgical robot imitation learning using open-source surgical videos that combines kinematically labeled phantom robot demonstrations with surgical videos from open-source datasets and online sources for policy learning, is proposed.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xinhao Chen",
    "id": "2307218135",
    "h_index": 3,
    "papers": 13
   },
   {
    "name": "Juo-Tung Chen",
    "id": "2361665894",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Nigel Nelson",
    "id": "2356549198",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Antony Goldenberg",
    "id": "2361499954",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Jesse Haworth",
    "id": "2221117772",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Sean Huver",
    "id": "10419594",
    "h_index": 8,
    "papers": 29
   },
   {
    "name": "Axel Krieger",
    "id": "2256991202",
    "h_index": 8,
    "papers": 19
   }
  ],
  "comment": "",
  "topics": [
   "imitation-diffusion",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.16058v1",
  "pdf_url": "https://arxiv.org/pdf/2608.16058v1",
  "html_url": "https://arxiv.org/html/2608.16058v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.16049",
  "slug": "pluralistic-human-robot-interaction-designing-for-robot-interaction-wi",
  "title": "Pluralistic Human-Robot Interaction: Designing for Robot Interaction with Diverse Communities",
  "abstract": "Social robots are being developed for homes, schools, and other environments where they will interact with diverse users. While Human-Robot Interaction (HRI) research often emphasizes natural communication, engagement, personalization, and task success, these goals do not fully address the social complexity of real-world deployment. This paper proposes \\emph{Pluralistic HRI}, a framework for designing social robots that treat human diversity as a foundational design concern. The framework brings together pluralism, civic dialogue, perspective-taking, empathy, intercultural competence, cultural humility, and moral imagination to guide inclusive, adaptive, and ethically grounded interaction. We outline how pluralistic HRI can inform design, evaluation, and deployment in diverse human communities.",
  "published": "2026-08-17",
  "updated": "2026-08-17",
  "year": "2026",
  "authors": [
   "Raj Korpan"
  ],
  "author_count": 1,
  "categories": [
   "cs.HC",
   "cs.CY",
   "cs.RO"
  ],
  "primary_category": "cs.HC",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper proposes pluralistic HRI, a framework for designing social robots that treat human diversity as a foundational design concern, and outlines how pluralistic HRI can inform design, evaluation, and deployment in diverse human communities.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Raj Korpan",
    "id": "26336279",
    "h_index": 6,
    "papers": 29
   }
  ],
  "comment": "Accepted to the Broadening the Users - A Cross-Disciplinary Roadmap for Social Humanoid Interaction (BU-SHI) Workshop at IEEE RO-MAN 2026",
  "topics": [
   "humanoids",
   "data-teleop",
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.16049v1",
  "pdf_url": "https://arxiv.org/pdf/2608.16049v1",
  "html_url": "https://arxiv.org/html/2608.16049v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.16041",
  "slug": "scenariocharacterization-a-modular-toolkit-for-characterizing-safety-a",
  "title": "ScenarioCharacterization: A Modular Toolkit for Characterizing Safety across Trajectory Datasets",
  "abstract": "We introduce ScenarioCharacterization, an open-source framework for automated, dataset-agnostic profiling of driving scenarios in trajectory datasets. Our framework is packaged as a modular, configuration-driven pipeline of three layers: a dataset adapter that maps custom datasets onto an open Scenario representation, a characterizer that performs feature extraction, behavior probing, and criticality scoring at scenario and agent levels, and an analysis layer for scenario visualization and feature, score, and probe analyses. Because the layers communicate only through Pydantic-validated schemas composed via configurations, a new dataset can easily plug in without rewriting the characterization and analysis stack. This technical report describes the design and APIs, shows example outputs on Waymo Open Motion, Argoverse2, and nuPlan, and discusses downstream uses of the approach. The framework is available at https://github.com/navarrs/ScenarioCharacterization.",
  "published": "2026-08-17",
  "updated": "2026-08-17",
  "year": "2026",
  "authors": [
   "Ingrid Navarro",
   "Yutong Duan",
   "Jonathan Francis",
   "Jean Oh"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "This technical report describes the design and APIs of ScenarioCharacterization, shows example outputs on Waymo Open Motion, Argoverse2, and nuPlan, and discusses downstream uses of the approach.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ingrid Navarro",
    "id": "30596850",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Yutong Duan",
    "id": "2458067924",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Jonathan Francis",
    "id": "2314826620",
    "h_index": 5,
    "papers": 15
   },
   {
    "name": "Jean Oh",
    "id": "2243322333",
    "h_index": 4,
    "papers": 10
   }
  ],
  "comment": "9 pages, 7 figures, 3 tables",
  "topics": [
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.16041v1",
  "pdf_url": "https://arxiv.org/pdf/2608.16041v1",
  "html_url": "https://arxiv.org/html/2608.16041v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2608.16030",
  "slug": "benchmarking-identity-sensitive-llm-outputs-for-surveillance-and-secur",
  "title": "Benchmarking Identity-Sensitive LLM Outputs for Surveillance and Security Robots",
  "abstract": "Large language models (LLMs) are increasingly used to generate textual robot design specifications, interaction policies, and risk assessments during early-stage robot development. Such outputs may influence how surveillance and security robots are conceptualized, documented, and ultimately implemented. This paper evaluates whether identity-conditioned prompts produce systematic differences in LLM-generated surveillance and security robot design descriptions. Using 236 demographic identity labels across single-label and model-augmented prompt conditions, we analyze readability as an initial benchmark for evaluating accessibility and identity-conditioned variation in generated robot design descriptions. The results show significant differences in readability across prompt conditions, design dimensions, and demographic identities. Although readability cannot determine whether an output is fair or socially appropriate, it provides an interpretable baseline within a broader benchmarking framework that also includes lexical, semantic, sentiment, syntactic, and fairness-focused analyses.",
  "published": "2026-08-17",
  "updated": "2026-08-17",
  "year": "2026",
  "authors": [
   "Nneka Hyman",
   "Jasmine Khan",
   "Raj Korpan"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.CY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper evaluates whether identity-conditioned prompts produce systematic differences in LLM-generated surveillance and security robot design descriptions and analyzes readability as an initial benchmark for evaluating accessibility and identity-conditioned variation in generated robot design descriptions.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Nneka Hyman",
    "id": "2458106902",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Jasmine Khan",
    "id": "2458071606",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Raj Korpan",
    "id": "26336279",
    "h_index": 6,
    "papers": 29
   }
  ],
  "comment": "Accepted to the Foundation Models in the Ro-Man Age (FoRMA) Workshop at IEEE RO-MAN 2026",
  "topics": [
   "foundation-pretraining",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.16030v1",
  "pdf_url": "https://arxiv.org/pdf/2608.16030v1",
  "html_url": "https://arxiv.org/html/2608.16030v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.15995",
  "slug": "learning-varying-physical-therapist-patient-interactions-for-robot-med",
  "title": "Learning Varying Physical Therapist-Patient Interactions for Robot-mediated Upper Limb Task-Specific Training",
  "abstract": "Upper extremity motor function recovery is positively linked to Task-Specific Training (TST) and sufficient therapy dosage. Rehabilitation robots can increase TST dosage via controlled, repetitive treatment and free therapists to simultaneously manage other patients, but it has yet to demonstrate significant benefits over conventional treatment. This is potentially linked to inaccurate robotic representation of personalised physical therapist-patient interaction and lack of practice variability during TST. Hence, we advocate for robotic interventions that preserve the personalised physical therapist-patient interactions when delivering TST for patients across varying practise conditions. We propose a Learning-from-Demonstration framework using Task-Parameterised Gaussian Mixture Models (TPGMM) to learn personalised physical therapist-patient interaction in Task-Specific exercises, mapping patient joint kinematics to therapist-applied torques using few demonstrations. The model is generalised to reconstruct therapist torques in new task variations. The framework was evaluated on physical interactions from 14 mock \"therapist-patient\" pairs over three tasks of increasing complexity, each with six variations. A benchmark comparison against a Look-Up Table was conducted. The results show both methods reproducing interactions in unseen task variations that deviate slightly from the actual interaction, with TPGMM slightly outperforming LUT. Both methods reproduced interactions that gets increasingly closer to the actual interaction as task complexity increases.",
  "published": "2026-08-17",
  "updated": "2026-08-17",
  "year": "2026",
  "authors": [
   "Jia Quan Loh",
   "Vincent Crocher",
   "Marlena Klaic",
   "Denny Oetomo",
   "Ying Tan"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes a Learning-from-Demonstration framework using Task-Parameterised Gaussian Mixture Models (TPGMM) to learn personalised physical therapist-patient interaction in Task-Specific exercises, mapping patient joint kinematics to therapist-applied torques using few demonstrations.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jia Quan Loh",
    "id": "2370942433",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Vincent Crocher",
    "id": "50417061",
    "h_index": 16,
    "papers": 46
   },
   {
    "name": "M. Klaic",
    "id": "3409927",
    "h_index": 12,
    "papers": 59
   },
   {
    "name": "D. Oetomo",
    "id": "2794163",
    "h_index": 29,
    "papers": 283
   },
   {
    "name": "Ying Tan Human Robotics Laboratory",
    "id": "2458104952",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Department of Mechanical Engineering",
    "id": "2458107350",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "The University of Melbourne Melbourne School of Health Sciences",
    "id": "2458106886",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "The University of Melbourne",
    "id": "103157296",
    "h_index": 8,
    "papers": 16
   }
  ],
  "comment": "12 pages, 4 figures, 3 tables Submitted to:IEEE Transactions on Neural Systems and Rehabilitation Engineering",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.15995v1",
  "pdf_url": "https://arxiv.org/pdf/2608.15995v1",
  "html_url": "https://arxiv.org/html/2608.15995v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.15968",
  "slug": "tabletop-pen-manipulation-with-a-vision-guided-4-dof-arm",
  "title": "Tabletop Pen Manipulation With a Vision-Guided 4-DoF Arm",
  "abstract": "Low-cost four-degree-of-freedom (DoF) arms are among the most accessible robotic platforms. But they are, in theory, underactuated for picking up in situations where objects are at arbitrary orientations, a task that appears to require five degrees of freedom: the planar position (x and y), the height (z), a wrist rotation to align the gripper with the object, and gripper actuation, of which a four-DoF arm lacks the wrist rotation. This work shows that perception and motion planning can enable such an arm, a roughly $200 Waveshare RoArm-M2-S, under a fixed overhead camera to detect and color-sort writing utensils without that joint. A YOLO11n-OBB (You Only Look Once, oriented bounding box) detector locates each writing utensil; camera intrinsics and an ArUco reference pose convert its pixel coordinates to robot coordinates; and a color classifier labels it. The detected orientation angle determines the motion strategy: utensils close to the arm's fixed approach direction are picked up directly, and those at steeper angles are reoriented via corrective sweeps until they are graspable, after which they are picked up and sorted into the assigned color bin. Across 326 logged motions on seven writing utensils, the arm made 196 direct grasps and 130 corrective sweep passes, correcting misalignments up to 90 degrees, suggesting that clever task-informed engineering can compensate for a missing degree of freedom on tasks like this one.",
  "published": "2026-08-16",
  "updated": "2026-08-16",
  "year": "2026",
  "authors": [
   "Anirudh Rangarajan",
   "Bibit Bianchini"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work shows that perception and motion planning can enable a four-degree-of-freedom arm, a roughly $200 Waveshare RoArm-M2-S, under a fixed overhead camera to detect and color-sort writing utensils without that joint.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Anirudh Rangarajan",
    "id": "2458112095",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Bibit Bianchini",
    "id": "2122787416",
    "h_index": 4,
    "papers": 9
   }
  ],
  "comment": "19 pages, 8 figures",
  "topics": [
   "dexterous-manipulation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.15968v1",
  "pdf_url": "https://arxiv.org/pdf/2608.15968v1",
  "html_url": "https://arxiv.org/html/2608.15968v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.15946",
  "slug": "rotate-disks-to-reach-farther-design-and-modeling-of-a-novel-reconfigu",
  "title": "Rotate Disks to Reach Farther: Design and Modeling of a Novel Reconfigurable Tendon Driven Manipulator",
  "abstract": "Rerouting the tendon path in tendon driven continuum manipulators (TDCMs) enables a broad range of deformation modes. This work presents a Reconfigurable TDCM design which allows independent rotation of intermediate spacer disks, thereby locally rerouting the tendon and achieving non-trivial backbone spatial deformations. Two such designs, (a) Manual Disk Locked (MDL) and (b) Continuous Disk Rotor (CDR) manipulators are presented to achieve disk rotations before and during operation, respectively. A predictive static model based on the piecewise constant strain (PCS) assumption is developed within a potential energy minimization framework, incorporating (a) disk rotations, (b) discrete tendon paths between disk segments, (c) rigid thickness of spacer disks, and (d) elasticity of the tendons. The model is validated against experimental results, demonstrating an average tip error of $1.2\\%$ of the manipulator's total length for parallel tendon routing and around $3\\%$ for the case when multiple disks are rotated. The computation time is an order of magnitude lower than the state of the art Cosserat rod solver.",
  "published": "2026-08-16",
  "updated": "2026-08-18",
  "year": "2026",
  "authors": [
   "Sabyasachi Dash",
   "Yangkun Liu",
   "Will Hunter",
   "John Golden",
   "Girish Krishnan"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Sabyasachi Dash",
    "id": "2316063430",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Yangkun Liu",
    "id": "2458569876",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Will Hunter",
    "id": "2458108056",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "John Golden",
    "id": "2268166347",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Girish Krishnan",
    "id": "2273992041",
    "h_index": 3,
    "papers": 17
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.15946v2",
  "pdf_url": "https://arxiv.org/pdf/2608.15946v2",
  "html_url": "https://arxiv.org/html/2608.15946v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.15938",
  "slug": "revisiting-open-loop-execution-in-robotics-toward-reactive-higher-perf",
  "title": "Revisiting Open-Loop Execution in Robotics: Toward Reactive, Higher-Performing Policies",
  "abstract": "Action chunking --- the practice of predicting a sequence of actions and executing a prefix open-loop --- has emerged as a key enabler of recent progress in imitation learning for robotic manipulation. However, executing long open-loop prefixes reduces reactivity, limiting policies' ability to correct for errors. Further, the mechanisms underlying these performance benefits remain poorly understood: prior works cite mitigating compounding errors, absorbing inference latency, or smoothing motions, but provide limited controlled evidence or guidance for preserving reactivity. In this work, we argue that long open-loop execution primarily helps short-context policies imitate \"non-Markovian demonstrations\". Across four simulation and two real-world tasks, we show that expert non-Markovianity strongly shapes the relationship between task success and open-loop execution horizon. Further, we investigate the impact of compounding errors --- the prevailing explanation for long open-loop execution in prior work --- and find that while they matter, expert non-Markovianity has a much stronger impact in our experimental setting. Finally, we show that when policies are provided with a sufficiently long context, open-loop execution is no longer beneficial and the most reactive, closed-loop policies perform best. While imitation learning has seen great success using long open-loop execution, our findings motivate long-context, reactive policies as a more principled and performant paradigm.",
  "published": "2026-08-16",
  "updated": "2026-08-19",
  "year": "2026",
  "authors": [
   "Michael Zeng",
   "Abhinav Agarwal",
   "Ajay Bati",
   "Brian Lee",
   "Siddharth Ancha",
   "Russ Tedrake"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "It is argued that long open-loop execution primarily helps short-context policies imitate\"non-Markovian demonstrations\" and that when policies are provided with a sufficiently long context, open-loop execution is no longer beneficial and the most reactive, closed-loop policies perform best.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Michael Zeng",
    "id": "2283846854",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Abhinav Agarwal",
    "id": "2387770128",
    "h_index": 2,
    "papers": 13
   },
   {
    "name": "Ajay Bati",
    "id": "2334867788",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Brian Lee",
    "id": "2458117295",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Siddharth Ancha",
    "id": "3422311",
    "h_index": 8,
    "papers": 15
   },
   {
    "name": "Russ Tedrake",
    "id": "2263905014",
    "h_index": 14,
    "papers": 35
   }
  ],
  "comment": "",
  "topics": [
   "imitation-diffusion"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.15938v2",
  "pdf_url": "https://arxiv.org/pdf/2608.15938v2",
  "html_url": "https://arxiv.org/html/2608.15938v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.15924",
  "slug": "rapac-dp-response-aligned-pending-action-compensation-for-diffusion-po",
  "title": "RAPAC-DP: Response-Aligned Pending-Action Compensation for Diffusion Policies under Delayed Execution",
  "abstract": "Cloud-side inference gives imitation-learning policies access to greater computational resources, but communication and computation delays can degrade control performance. To compensate for these delays, we propose RAPAC-DP, a response-aligned pending-action compensation framework designed for both diffusion- and flow-based action generators. RAPAC-DP encodes the actions already scheduled for execution before the cloud response arrives into a pending-action sequence that serves as the conditioning input to a parameter-efficient compensation pathway. When delay effects are negligible, bypassing this pathway exactly recovers the frozen base policy. For training, RAPAC-DP constructs delay-conditioned samples from delay-free demonstrations, requiring neither explicit system dynamics nor additional delayed demonstrations. At the largest fixed delay tested on Kinetix, RAPAC-DP retained 81.4% of its overall delay-free performance. At the largest fixed delay tested on each RoboMimic task, it achieved a mean success rate of 0.633 across the three tasks. These results demonstrate the effectiveness of pending-action compensation for cloud-deployed imitation-learning policies.",
  "published": "2026-08-16",
  "updated": "2026-08-16",
  "year": "2026",
  "authors": [
   "Tao Wang",
   "Wei Wang",
   "Jianhui Wang",
   "Qi Wang",
   "Weidi Huang",
   "Bing Xu"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "RAPAC-DP encodes the actions already scheduled for execution before the cloud response arrives into a pending-action sequence that serves as the conditioning input to a parameter-efficient compensation pathway, demonstrating the effectiveness of pending-action compensation for cloud-deployed imitation-learning policies.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Taozhi Wang",
    "id": "2457752149",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Wei Wang",
    "id": "2457759967",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Jianhui Wang",
    "id": "2458596292",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Qi Wang",
    "id": "2358236936",
    "h_index": 4,
    "papers": 15
   },
   {
    "name": "Weidi Huang",
    "id": "2276311472",
    "h_index": 7,
    "papers": 33
   },
   {
    "name": "Bing Xu",
    "id": "2267711829",
    "h_index": 8,
    "papers": 36
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.15924v1",
  "pdf_url": "https://arxiv.org/pdf/2608.15924v1",
  "html_url": "https://arxiv.org/html/2608.15924v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.15917",
  "slug": "pre-training-visual-dexterity-in-simulation",
  "title": "Pre-training Visual Dexterity in Simulation",
  "abstract": "Large-scale pre-training has made robot policy fine-tuning increasingly data-efficient, but this progress has largely been driven by datasets and embodiments built around simple parallel-jaw grippers. Dexterous, multi-fingered hands remain comparatively data-starved because real teleoperation is costly to scale, while human hand video is off-embodiment and requires lossy pose estimation and retargeting. We introduce Simulation Pre-training for Dexterity (SPD), a pre-training framework for dexterous manipulation that uses data entirely collected in simulation. In SPD, humans manipulate virtual objects inside a VR headset, enabling on-embodiment trajectories and robot-free collection. With the help of five operators, we collect 75 hours of multi-task dexterous manipulation over one week, and use it to pre-train a causal transformer on a sequence modeling objective. We study the benefits of simulation pre-training on real-world tasks by fine-tuning on 1-2 hours of physical demonstrations on a 56-DoF bimanual dexterous setup. We find that our approach outperforms training behavior cloning policies from scratch, showing that simulation teleoperation is a viable pre-training source for real-world dexterous manipulation. We perform ablation studies, measuring the benefits of history conditioning and short action chunks for reactive control.",
  "published": "2026-08-16",
  "updated": "2026-08-16",
  "year": "2026",
  "authors": [
   "Sarthak Kamat",
   "Adam Rashid",
   "Satvik Sharma",
   "Aseem Doriwala",
   "Chelsea Finn",
   "Phillip Isola",
   "C. Karen Liu"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Simulation Pre-training for Dexterity (SPD) is introduced, a pre-training framework for dexterous manipulation that uses data entirely collected in simulation and outperforms training behavior cloning policies from scratch, showing that simulation teleoperation is a viable pre-training source for real-world dexterous manipulation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Sarthak Kamat",
    "id": "2069938522",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Adam Rashid",
    "id": "2241320067",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Satvik Sharma",
    "id": "2116532955",
    "h_index": 9,
    "papers": 18
   },
   {
    "name": "Aseem Doriwala",
    "id": "2428626130",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Chelsea Finn",
    "id": "2387218305",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Phillip Isola",
    "id": "2348263167",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "C. K. Liu",
    "id": "2325109818",
    "h_index": 3,
    "papers": 8
   }
  ],
  "comment": "Project page: https://spd.bot",
  "topics": [
   "dexterous-manipulation",
   "imitation-diffusion",
   "foundation-pretraining",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.15917v1",
  "pdf_url": "https://arxiv.org/pdf/2608.15917v1",
  "html_url": "https://arxiv.org/html/2608.15917v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.15897",
  "slug": "tactile-sim2real-without-tactile-simulation-via-bottlenecked-latent-re",
  "title": "Tactile Sim2Real without Tactile Simulation via Bottlenecked Latent Reconstruction",
  "abstract": "Robot sensor designs, particularly tactile sensors, are highly diverse and evolve rapidly. Modeling each sensor in simulation demands substantial domain expertise and computational approximations can degrade the fidelity of the simulated signals. We propose Sim2Real via Bottlenecked Latent Reconstruction (SBLR), a framework that avoids sensor-specific simulation entirely by (1) training policies on a simulator-native oracle sensor that is easy to construct without modeling any particular sensor (e.g. we use a point-cloud and finger-tip forces as a tactile oracle), and (2) aligning real sensor latent embeddings to those of the oracle sensor at inference time. Policy training proceeds in two-stage: the policy first learns from the oracle sensor latents, then a bottlenecked latent reconstruction adapts it to the information loss expected when using the real sensor instead of the oracle. The alignment between oracle and real sensor is learned from unpaired random-play data collected in both simulation and the real world, using rectified-flow-based transformation networks trained on nearest-neighbor pseudo-pairs. Simulation experiments on three contact-rich tasks show that SBLR matches or approaches the performance of an oracle with direct access to tactile simulation. Hardware experiments on Peg Insertion and Gear Meshing with GelSight Mini and DIGIT sensors demonstrate 85-97.5% zero-shot success without requiring any sensor-specific modeling or calibration, outperforming a physics-based tactile simulation baseline by 7.5-15%.",
  "published": "2026-08-16",
  "updated": "2026-08-19",
  "year": "2026",
  "authors": [
   "Fan Yang",
   "Youngsun Wi",
   "Jinhao Yu",
   "Nima Fazeli",
   "Dmitry Berenson"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Sim2Real via Bottlenecked Latent Reconstruction (SBLR), a framework that avoids sensor-specific simulation entirely by training policies on a simulator-native oracle sensor that is easy to construct without modeling any particular sensor, is proposed.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Fan Yang",
    "id": "2292401187",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Youngsun Wi",
    "id": "2152020759",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Jinhao Yu",
    "id": "2458127883",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Nima Fazeli",
    "id": "2718552",
    "h_index": 19,
    "papers": 83
   },
   {
    "name": "Dmitry Berenson",
    "id": "2284773296",
    "h_index": 2,
    "papers": 6
   }
  ],
  "comment": "",
  "topics": [
   "tactile",
   "sim2real"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.15897v2",
  "pdf_url": "https://arxiv.org/pdf/2608.15897v2",
  "html_url": "https://arxiv.org/html/2608.15897v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.15884",
  "slug": "grouping-auction-consensus-algorithm-for-decentralized-task-allocation",
  "title": "Grouping Auction-Consensus Algorithm for Decentralized Task Allocation in Multi-Robot Systems",
  "abstract": "Decentralized multi-robot task allocation (MRTA) is essential for scalable and resilient autonomous systems. The Consensus-Based Bundle Algorithm (CBBA) is a widely adopted decentralized baseline. However, its individual task-level bidding is poorly aligned with the min-sum objective of minimizing total team travel distance, leading to suboptimal allocations in spatially distributed environments. This paper introduces the Grouping Auction-Consensus Algorithm (GACA). This decentralized MRTA framework adopts the two-phase auction-consensus architecture of CBBA while fundamentally redesigning its bidding mechanism to reason over groups of spatially proximate tasks. A nearest-neighbor preprocessing step partitions tasks into spatially coherent groups before allocation. Agents then iteratively propose structured group-level actions: claiming unassigned groups, acquiring partial groups, or contesting groups held by other agents. Competing actions are resolved through a consensus phase. Operating in the MT-SR-IA problem class, GACA is evaluated against CBBA using a Mixed-Integer Linear Program as the ground-truth optimality reference. Across four swarm sizes and 4,000 test worlds, GACA achieves a median percent optimality of approximately 97% compared to 81--84% for CBBA, while converging in equal or fewer iterations. A scalability evaluation over 3,280 additional problem instances spanning swarm sizes of 5 to 20 agents and task counts of 10 to 50 confirms that these gains generalize robustly across a wide range of problem configurations.",
  "published": "2026-08-16",
  "updated": "2026-08-16",
  "year": "2026",
  "authors": [
   "Jose Rodriguez",
   "Sven Koenig",
   "Wenjie Dong",
   "Qi Lu"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper introduces the Grouping Auction-Consensus Algorithm (GACA), a decentralized MRTA framework that adopts the two-phase auction-consensus architecture of CBBA while fundamentally redesigning its bidding mechanism to reason over groups of spatially proximate tasks.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "J. Rodriguez",
    "id": "2407644944",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Sven Koenig",
    "id": "2336955684",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Wenjie Dong",
    "id": "2281617568",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Qi Lu",
    "id": "2377948847",
    "h_index": 1,
    "papers": 12
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.15884v1",
  "pdf_url": "https://arxiv.org/pdf/2608.15884v1",
  "html_url": "https://arxiv.org/html/2608.15884v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.15875",
  "slug": "gigabrain-0-7-scaling-embodied-foundation-models-to-emergent-capabilit",
  "title": "GigaBrain-0.7: Scaling Embodied Foundation Models to Emergent Capabilities with a Three-System Architecture",
  "abstract": "Vision-language-action (VLA) models have become a dominant paradigm for generalist embodied agents, demonstrating strong complex and long-horizon task completion in structured settings. Yet it remains an open question whether current VLA systems can benefit from more effective architectural design, scale to substantially larger and more heterogeneous data regimes, and achieve broader generalization across tasks and embodiments. To this end, we present GigaBrain-0.7, an embodied foundation model with substantially improved generalization across diverse robot embodiments. Specifically, GigaBrain-0.7 unifies understanding, prediction, and action through a three-system architecture, scales pretraining to over 37,000 hours of heterogeneous embodied data, and introduces one-stage alignment training that jointly optimizes vision-language understanding and multi-embodiment action generation. Compared with the preceding GigaBrain-0 series and prior state-of-the-art models including $\u03c0_{0.5}$, GigaBrain-0.7 achieves substantial improvements in foundation zero-shot capabilities, language-conditioned instruction following, and post-training task success rates. In particular, on our in-house Maker H01 platform and mainstream robot embodiments, GigaBrain-0.7 demonstrates strong task adaptability and completion ability across both home and industrial scenarios. All training code and pretrained model weights will be released.",
  "published": "2026-08-16",
  "updated": "2026-08-16",
  "year": "2026",
  "authors": [
   " GigaBrain Team",
   "Angen Ye",
   "Axiang Sun",
   "Can Jin",
   "Chenxi Cheng",
   "Chong Shi",
   "Dengke Shang",
   "Dingqian Zhang",
   "Guan Huang",
   "Guangqiang Wang",
   "Guangqing Ding",
   "Guo Li",
   "Hangcong Li",
   "Hengyu Zhong",
   "Hongtao Lu",
   "Jianbo Qin",
   "Jiming Mao",
   "Jing Zhu",
   "Jindi Lv",
   "Jingzhi Cui",
   "Junjie Xie",
   "Junyi Bao",
   "Kai Liu",
   "Lei Yuan",
   "Limin Long",
   "Lv Feng",
   "Mingming Yu",
   "Peng Li",
   "Pengfei Yi",
   "Qi Li",
   "Qianli Zhang",
   "Qingfang Li",
   "Qitang Hu",
   "Rui Zhang",
   "Shaoyan Sun",
   "Shibo Sun",
   "Shiying Duan",
   "Tenghui Chen",
   "Tianze Liu",
   "Weijie Ke",
   "Wenyao Xue",
   "Xiaofeng Wang",
   "Xiaoyu Tian",
   "Xinyu Liu",
   "Xinze Chen",
   "Yang Wang",
   "Yankai Wang",
   "Yejun Zeng",
   "Yifan Li",
   "Yifei Nie",
   "Yilong Li",
   "Yilong Liu",
   "Yongchao Feng",
   "Yumeng Wang",
   "Yun Ye",
   "Zhichao Liu",
   "Ziheng He",
   "Zonghai Yang",
   "Zheng Zhu"
  ],
  "author_count": 59,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work presents GigaBrain-0.7, an embodied foundation model with substantially improved generalization across diverse robot embodiments, and introduces one-stage alignment training that jointly optimizes vision-language understanding and multi-embodiment action generation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "GigaBrain Team",
    "id": "2458083664",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Angen Ye",
    "id": "2373031539",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Axiang Sun",
    "id": "2458083293",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Can Jin",
    "id": "2396680673",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Chen Cheng",
    "id": "2453950841",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Chong Shi",
    "id": "2280911924",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Dengke Shang",
    "id": "2265752712",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Di Zhang",
    "id": "2381837030",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Guan Huang",
    "id": "2256954306",
    "h_index": 16,
    "papers": 42
   },
   {
    "name": "Guangqiang Wang",
    "id": "2458128288",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Guangqing Ding",
    "id": "2458083880",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Guo Li",
    "id": "2184570319",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Han Li",
    "id": "2443351326",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Hengyu Zhong",
    "id": "2336752741",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Hongtao Lu",
    "id": "2115863606",
    "h_index": 10,
    "papers": 53
   },
   {
    "name": "Ji-Chen Qin",
    "id": "2456178201",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Jiming Mao",
    "id": "2458075364",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Jingru Zhu",
    "id": "2448363288",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Jindi Lv",
    "id": "2087059132",
    "h_index": 5,
    "papers": 15
   },
   {
    "name": "Jingzhi Cui",
    "id": "2399926529",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Junjie Xie",
    "id": "2445238076",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Jun Bao",
    "id": "2457858840",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Kai Liu",
    "id": "2377763386",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Lei Yuan",
    "id": "49785134",
    "h_index": 16,
    "papers": 69
   },
   {
    "name": "L. Long",
    "id": "2334947795",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Lv Feng",
    "id": "2387111501",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Mingming Yu",
    "id": "2387333550",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Peng Li",
    "id": "2421793186",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Pengfei Yi",
    "id": "2323509253",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Qi Li",
    "id": "2391049894",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "QianLi Zhang",
    "id": "2456521060",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Qingfang Li",
    "id": "2458435100",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Qitang Hu",
    "id": "2450003998",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Rui Zhang",
    "id": "2449181652",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Shaoyan Sun",
    "id": "3141359",
    "h_index": 10,
    "papers": 19
   },
   {
    "name": "Shibo Sun",
    "id": "2338700617",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Shiying Duan",
    "id": "2098945415",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Te Chen",
    "id": "2454161250",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Tianze Liu",
    "id": "2276646806",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Weijie Ke",
    "id": "2448002938",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Wenyao Xue",
    "id": "2328335515",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Xiaofeng Wang",
    "id": "2349399535",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Xiaoyu Tian",
    "id": "2149326880",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Xinyu Liu",
    "id": "2445899668",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Xinze Chen",
    "id": "1391211885",
    "h_index": 17,
    "papers": 34
   },
   {
    "name": "Yang Wang",
    "id": "2373745208",
    "h_index": 6,
    "papers": 14
   },
   {
    "name": "Yankai Wang",
    "id": "2108737729",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Yejun Zeng",
    "id": "2388179036",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Yifan Li",
    "id": "2336158403",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Yifei Nie",
    "id": "2410365217",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Yilong Li",
    "id": "2322363269",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Yilong Liu",
    "id": "2458126279",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Yongchao Feng",
    "id": "2267307854",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "Yumeng Wang",
    "id": "2457311505",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yun Ye",
    "id": "2384823293",
    "h_index": 6,
    "papers": 16
   },
   {
    "name": "Zhichao Liu",
    "id": "2387126288",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Ziheng He",
    "id": "2338038073",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Zonghai Yang",
    "id": "2458248989",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Zheng Zhu",
    "id": "2109516240",
    "h_index": 17,
    "papers": 50
   }
  ],
  "comment": "https://gigaai.cc/blog/gigabrain07",
  "topics": [
   "vla",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.15875v1",
  "pdf_url": "https://arxiv.org/pdf/2608.15875v1",
  "html_url": "https://arxiv.org/html/2608.15875v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.15863",
  "slug": "scaling-manual-grounded-appliance-manipulation-with-data-synthesis-and",
  "title": "Scaling Manual-Grounded Appliance Manipulation with Data Synthesis and Unified Planning",
  "abstract": "Operating household appliances requires long-horizon planning that is state-dependent and robust to disturbances, yet existing large models fall short, as no sufficiently diverse, task-oriented dataset exists to support such planning. To bridge this gap, we propose MAGE, a scalable data synthesis pipeline that introduces a novel Hierarchical Appliance Graph (HAG) to automatically generate part grounding, long-horizon planning, and closed-loop recovery data from appliance manuals. With MAGE, we build UseAppliance, the first large-scale dataset for manual-grounded appliance manipulation planning, spanning 22 appliance categories with 89K+ part annotations, 53K+ manipulation tasks, and 33K+ closed-loop adjustment steps. Built on UseAppliance, we develop AppliancePlan, an end-to-end model for manual-grounded appliance manipulation planning. On RealAppliance-Bench, AppliancePlan with only 7B parameters achieves over 10x the best baseline on open-loop planning and consistently outperforms state-of-the-art models across all tasks. Real-robot experiments on six household appliances further confirm effective sim-to-real transfer, marking an important step toward general-purpose household robotics.",
  "published": "2026-08-16",
  "updated": "2026-08-16",
  "year": "2026",
  "authors": [
   "Yuxing Long",
   "Lei Kang",
   "Ziyan Yu",
   "Yuzheng Gao",
   "Bin Cheng",
   "Jiyao Zhang",
   "Xiaoqi Li",
   "Haolin Yang",
   "Dongjiang Li",
   "Hui Shen",
   "Hao Dong"
  ],
  "author_count": 11,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CL",
   "cs.CV",
   "cs.MM"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "MAGE is proposed, a scalable data synthesis pipeline that introduces a novel Hierarchical Appliance Graph (HAG) to automatically generate part grounding, long-horizon planning, and closed-loop recovery data from appliance manuals for manual-grounded appliance manipulation planning.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yuxing Long",
    "id": "2242982685",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Lei Kang",
    "id": "2387922919",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Ziyan Yu",
    "id": "30679544",
    "h_index": 10,
    "papers": 29
   },
   {
    "name": "Yuzheng Gao",
    "id": "2314745618",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Bin Cheng",
    "id": "2458104842",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Jiyao Zhang",
    "id": "2239160954",
    "h_index": 10,
    "papers": 23
   },
   {
    "name": "Xiaoqi Li",
    "id": "2243138342",
    "h_index": 8,
    "papers": 21
   },
   {
    "name": "Haolin Yang",
    "id": "2450232787",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Dongjiang Li",
    "id": "2108809564",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Hui Shen",
    "id": "2342472659",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Hao Dong",
    "id": "2243529948",
    "h_index": 4,
    "papers": 8
   }
  ],
  "comment": "Accepted by ACM MM 26",
  "topics": [
   "sim2real",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.15863v1",
  "pdf_url": "https://arxiv.org/pdf/2608.15863v1",
  "html_url": "https://arxiv.org/html/2608.15863v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.15816",
  "slug": "vitar-visuo-tactile-residual-adaptation-for-foundation-vla-manipulatio",
  "title": "ViTaR: Visuo-Tactile Residual Adaptation for Foundation VLA Manipulation",
  "abstract": "As Vision-Language-Action (VLA) models scale toward real-world deployment, contact-rich manipulation exposes a critical blind spot: these policies encode broad visual-semantic priors yet remain unaware of local contact events, producing identical actions whether contact is established, lost, or destabilized. Existing remedies either modify VLA internals, risking catastrophic forgetting, or demand online reinforcement under near-failure contact conditions. Both grant tactile unbounded influence over action generation, conflicting with the priors that make VLAs generalizable. We introduce ViTaR, which reframes tactile feedback from an action-generating perceptual input to an execution modulator that selects and scales bounded residual corrections atop a frozen VLA, preserving pretrained capabilities by construction. ViTaR decomposes adaptation into two stages: Effect-Guided Modeling determines whether and which correction is locally justified via outcome-grounded preference evidence, and Residual Action Modulation converts this evidence into a residual choice with continuously scaled gain from real-time visuotactile observations. On the UniVTAC benchmark spanning seven contact-rich tasks, ViTaR achieves 61.3% average success, a 30.6 percentage-point improvement over its frozen VLA base that also surpasses purpose-built tactile baselines. Physical-robot experiments confirm that bounded tactile modulation transfers to real sensor noise and dynamics.",
  "published": "2026-08-16",
  "updated": "2026-08-16",
  "year": "2026",
  "authors": [
   "Yi Wang",
   "Renjun Wu",
   "Jinyan Liu",
   "Xuesong Li"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "ViTaR is introduced, which reframes tactile feedback from an action-generating perceptual input to an execution modulator that selects and scales bounded residual corrections atop a frozen VLA, preserving pretrained capabilities by construction.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yi Wang",
    "id": "2363937133",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Renjun Wu",
    "id": "2410853913",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Jinyan Liu",
    "id": "2352006318",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Xuesong Li",
    "id": "2346896847",
    "h_index": 1,
    "papers": 10
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "tactile",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.15816v1",
  "pdf_url": "https://arxiv.org/pdf/2608.15816v1",
  "html_url": "https://arxiv.org/html/2608.15816v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.15784",
  "slug": "reliable-piezoresistive-strain-sensing-through-physical-limits-and-unc",
  "title": "Reliable Piezoresistive Strain Sensing Through Physical Limits and Uncertainty Monitoring",
  "abstract": "Soft piezoresistive strain sensors are one of the most common sensing solutions for wearable and soft robotic applications due to their flexibility and compliance. However, their resistance response is nonlinear and hysteretic, and a sensor can be pushed past its calibrated workspace or misbehave inside it, carrying that error into a decision or control loop. Probabilistic regressors track confidence but ignore those limits. A predictive mean can look unremarkable even when the reading comes from a sensor outside its admissible range or already failing internally, so a confident-looking estimate is not the same as a trustworthy one. This paper proposes a reliability framework pairing a physics-informed probabilistic inverse model, built on physics-guided input features, with a risk factor fusing uncertainty with strain and strain-rate limits into a three-state monitor. Tests on a Nitinol wire and a silver-coated polyamide thread with a Gaussian Process raised fit scores to 0.90-0.95 (RMSE 0.26%-0.15%) and a 96% empirical coverage against the 95% target. The monitor caught 95% of out-of-range and 100% of abnormal conditions while staying reliable under nominal operation. A sensor that reports confidence alongside its estimate lets a system withhold action instead, since it needs no labeled failure examples, which are hard to collect for soft materials.",
  "published": "2026-08-16",
  "updated": "2026-08-16",
  "year": "2026",
  "authors": [
   "Carmen Ballester",
   "V\u00edctor Mu\u00f1oz",
   "Dorin Copaci",
   "Dolores Blanco"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Carmen Ballester",
    "id": "2213628361",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "V\u00edctor Mu\u00f1oz",
    "id": "2280837198",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "D. Copaci",
    "id": "2675254",
    "h_index": 13,
    "papers": 38
   },
   {
    "name": "D. Blanco",
    "id": "145293807",
    "h_index": 21,
    "papers": 72
   }
  ],
  "comment": "",
  "topics": [
   "hardware-codesign",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.15784v1",
  "pdf_url": "https://arxiv.org/pdf/2608.15784v1",
  "html_url": "https://arxiv.org/html/2608.15784v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.15766",
  "slug": "tac4loco-learning-spatiotemporal-plantar-pressure-representations-for",
  "title": "Tac4Loco: Learning Spatiotemporal Plantar Pressure Representations for Humanoid Locomotion",
  "abstract": "Humanoid robots are expected to traverse complex terrains, where the plantar support may vary dramatically due to foot placement errors, ground properties, and transient dynamics. To achieve robust locomotion, the robots are required to adapt to uneven terrain and uncertain foot--ground interactions. Existing locomotion policies rely primarily on proprioception or exteroceptive terrain perception, where the former provides only indirect evidence of plantar support, while the latter predicts contact conditions before touchdown but cannot observe the actual support in real-time. Although some studies incorporate plantar contacts as an auxiliary perception, they rely mainly on summary statistics, overlooking the spatial topology of plantar pressure, which provides a more direct characterization of the realized contact state. To bridge this gap, we present Tac4Loco, a tactile-perceptive framework that incorporates multi-array plantar pressure as direct feedback for humanoid locomotion. We formulate a topology-preserving ordinal representation to map simulated and physical sensor signals into a shared observation space, with a dual-branch encoder for extracting their spatial and temporal representations. Subsequently, the learned spatiotemporal features are integrated with augmented proprioception including terrain estimation cues, and provided to an asymmetric actor-critic architecture for policy learning. Extensive simulation and real-world experiments demonstrate improved tracking performance and support adaptation on terrains with inclined, partial, asymmetric, and changing support. We further demonstrate its zero-shot deployment on unseen compliant and unstructured terrains, including a foam platform and a gravel road. All code and experimental configurations will be released as open-source to facilitate reproducibility.",
  "published": "2026-08-16",
  "updated": "2026-08-16",
  "year": "2026",
  "authors": [
   "Ziyun Liu",
   "Sikai Guo",
   "Zheng Li",
   "Jiahang Cao",
   "Haichao Liu",
   "Pei Qu",
   "Yinghong Zhang",
   "Jinni Zhou",
   "Jun Ma"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Tac4Loco is presented, a tactile-perceptive framework that incorporates multi-array plantar pressure as direct feedback for humanoid locomotion that is integrated with augmented proprioception including terrain estimation cues, and provided to an asymmetric actor-critic architecture for policy learning.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ziyun Liu",
    "id": "2335629640",
    "h_index": 1,
    "papers": 9
   },
   {
    "name": "Sikai Guo",
    "id": "2348494205",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Zheng Li",
    "id": "2146248434",
    "h_index": 10,
    "papers": 18
   },
   {
    "name": "Jiahang Cao",
    "id": "2348487240",
    "h_index": 3,
    "papers": 14
   },
   {
    "name": "Haichao Liu",
    "id": "2239158847",
    "h_index": 7,
    "papers": 31
   },
   {
    "name": "Pei Qu",
    "id": "2380527655",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Yinghong Zhang",
    "id": "2306055255",
    "h_index": 5,
    "papers": 24
   },
   {
    "name": "Jinni Zhou",
    "id": "2381351586",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Jun Ma",
    "id": "2380748551",
    "h_index": 1,
    "papers": 4
   }
  ],
  "comment": "9 pages,6 figures",
  "topics": [
   "humanoids",
   "tactile",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.15766v1",
  "pdf_url": "https://arxiv.org/pdf/2608.15766v1",
  "html_url": "https://arxiv.org/html/2608.15766v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.15748",
  "slug": "making-two-action-heads-agree-coordination-mechanisms-and-a-runtime-co",
  "title": "Making two action heads agree: coordination mechanisms and a runtime collapse certificate for flow-matching policies",
  "abstract": "A dual-representation flow-matching policy decodes each predicted motion into joint and end-effector spaces, and the residual between the two kinematically equivalent decodings provides a physically interpretable runtime signal. On multimodal tasks, however, independently sampled branches may choose different valid modes, causing false alarms. We study how to coordinate the two branches and at what cost. Across two robot environments and a non-robotic testbed, the tested mechanisms fall into four classes. An auxiliary latent shared by both branches but absent from the flow-matching construction is erased at the population optimum, a provable dead end confirmed within a prespecified 2% equivalence band. Sharing source noise can coordinate or anti-coordinate: its effect changes sign with the representation map and tracks the alignment of decoder mode basins. Consistency regularization gives intermediate coordination but reduces the valid-pair rate, while training-supported discrete partitions achieve near-ceiling coordination robustly. We further derive a chance-corrected coordination bound based only on each branch's Gini-Simpson diversity, yielding an attainable region and a label-free certificate that separates coordination from collapse when zero mismatch is ambiguous. On LIBERO-Plus, benign multimodality adds 1.57 percentage points of false alarms to the residual, which remains the strongest evaluated failure signal; the preregistered token intervention does not meet its false-alarm criterion or produce a seed-robust detection change. Code, models, and per-run configurations are available at https://github.com/kimo423/dual-head-coordination.",
  "published": "2026-08-16",
  "updated": "2026-08-16",
  "year": "2026",
  "authors": [
   "Jinhui Sun",
   "Wei Zhou",
   "Bowen Yang",
   "Xinliang Xiao",
   "Li Yang"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A chance-corrected coordination bound is derived based only on each branch's Gini-Simpson diversity, yielding an attainable region and a label-free certificate that separates coordination from collapse when zero mismatch is ambiguous.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jinhui Sun",
    "id": "2458130259",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Wei Zhou",
    "id": "2446890176",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Bowen Yang",
    "id": "2370441971",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Xinliang Xiao",
    "id": "2458151223",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Li Yang",
    "id": "2458344875",
    "h_index": 0,
    "papers": 0
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.15748v1",
  "pdf_url": "https://arxiv.org/pdf/2608.15748v1",
  "html_url": "https://arxiv.org/html/2608.15748v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.15741",
  "slug": "some-modifications-to-our-end-to-end-uav-planner",
  "title": "Some Modifications to Our End-to-End UAV Planner",
  "abstract": "The one-stage planner YOPO maps a single depth image and the robot state directly to a set of candidate trajectories, trained by backpropagating through differentiable trajectory costs. This yields dense, geometrically informative supervision, but inherits the pathologies of soft-constrained optimization: the safety cost competes with the smoothness and goal-reaching terms, is non-convex across homotopy classes, and the single-piece polynomial is limited in expressiveness. In this report, we summarize several effective modifications. We adopt a two-piece MINCO parameterization, trading time for smoothness without altering the trajectory's spatial profile. We further lift YOPO's multi-modal prediction to span distinct homotopy classes, treating each motion primitive as a homotopy anchor that confines the trajectory to a feasible basin - without explicit safe-flight-corridor construction or front-end search. For dynamic feasibility, we impose barrier penalties on velocity and acceleration together with a curvature-dependent speed limit whose gradient acts only on the velocity, producing an adaptive-speed behavior that decelerates in cluttered regions or sharp turns. We replace score regression with a ranking loss, preventing small score errors from reordering the candidate set. These yield richer trajectory representations, safer obstacle avoidance, and more direct flight paths.",
  "published": "2026-08-16",
  "updated": "2026-08-16",
  "year": "2026",
  "authors": [
   "Junjie Lu",
   "Bailing Tian"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This report adopts a two-piece MINCO parameterization, trading time for smoothness without altering the trajectory's spatial profile, and replaces score regression with a ranking loss, preventing small score errors from reordering the candidate set.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Junjie Lu",
    "id": "2151166328",
    "h_index": 8,
    "papers": 22
   },
   {
    "name": "Bailing Tian",
    "id": "2242943785",
    "h_index": 4,
    "papers": 10
   }
  ],
  "comment": "",
  "topics": [
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.15741v1",
  "pdf_url": "https://arxiv.org/pdf/2608.15741v1",
  "html_url": "https://arxiv.org/html/2608.15741v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.15707",
  "slug": "gains-leveraging-inconsistent-human-intervention-signals-in-reinforcem",
  "title": "GAINS: Leveraging Inconsistent Human Intervention Signals in Reinforcement Learning",
  "abstract": "Correcting robot manipulation policies through human intervention holds great promise for real-world deployment, yet human operators are inherently imperfect in both the actions they provide and the timing of their intervention signals. While the former has been extensively discussed in reinforcement learning (RL), the latter remains underexplored. At high control frequencies, human intervention signals are often delayed and inconsistent across time and state space. In this work, we present GAINS, a framework for leveraging inconsistent human intervention signals in RL. At the core of GAINS, we employ distributional RL with quantile Q-networks to model the return variability induced by sparse task rewards and inconsistent human interventions. Building on this distributional representation, we introduce a pessimistic exploration strategy that promotes safe and sample-efficient learning under human corrections. We evaluate GAINS on four diverse simulated manipulation tasks and two challenging real-world scenarios against state-of-the-art intervention-based methods. GAINS achieves a 22% higher task success rate than RLIF and improves recovery success by up to 43% in failure scenarios. These results highlight the importance of modeling return variability induced by human imperfection for real-world deployment of intervention-based learning.",
  "published": "2026-08-16",
  "updated": "2026-08-16",
  "year": "2026",
  "authors": [
   "Xinyi Zhang",
   "Yinuo Zhao",
   "Pei Ren",
   "Lechun Jiang",
   "Huiqian Jin",
   "Lei Sun",
   "Dapeng Wu",
   "Zhengping Che",
   "Chi Harold Liu",
   "Jian Tang"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "GAINS is presented, a framework for leveraging inconsistent human intervention signals in RL with quantile Q-networks to model the return variability induced by sparse task rewards and inconsistent human interventions and introduces a pessimistic exploration strategy that promotes safe and sample-efficient learning under human corrections.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xinyi Zhang",
    "id": "2374349762",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Yinuo Zhao",
    "id": "1720832487",
    "h_index": 10,
    "papers": 18
   },
   {
    "name": "Pei Ren",
    "id": "2365229417",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Lechun Jiang",
    "id": "2401896176",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Hui Jin",
    "id": "2109929634",
    "h_index": 9,
    "papers": 19
   },
   {
    "name": "Lei Sun",
    "id": "2349398734",
    "h_index": 10,
    "papers": 33
   },
   {
    "name": "Dapeng Wu",
    "id": "2316004043",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Zhengping Che",
    "id": "1939695",
    "h_index": 26,
    "papers": 88
   },
   {
    "name": "Chi Harold Liu",
    "id": "2353311969",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Jian Tang",
    "id": "2152779004",
    "h_index": 12,
    "papers": 33
   }
  ],
  "comment": "",
  "topics": [
   "rl-control",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.15707v1",
  "pdf_url": "https://arxiv.org/pdf/2608.15707v1",
  "html_url": "https://arxiv.org/html/2608.15707v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.15680",
  "slug": "robo-dopamine-2-0-history-conditioned-and-ood-aware-process-reward-mod",
  "title": "Robo-Dopamine 2.0: History-Conditioned and OOD-Aware Process Reward Modeling for Robotic Manipulation",
  "abstract": "Vision-language-action (VLA) models improve robotic manipulation but remain vulnerable to compounding errors, scene changes, and off-trajectory states. Reinforcement learning can refine pretrained VLA policies, yet sparse success signals hinder exploration, while engineered dense rewards are costly and task-specific. Existing learned visual reward models often rely on static before-after observations, causing temporal ambiguity and weak discrimination between robustness-preserving variations and task-invalid failures under out-of-distribution (OOD) execution. We introduce Robo-Dopamine 2.0, a history- and OOD-aware process reward model with a pairwise prediction interface. It combines (1) history-conditioned pairwise rewards that use source-aligned reference panels for synthetic OOD queries and observed rollout history for online queries, while preserving the queried endpoints, and (2) an OOD-aware signed progress space that represents valid progress, robustness, failure, and recovery. A Signed-Hop Curriculum with transition-aware replay learns coarse execution ordering before fine-grained progress calibration. We also construct an OOD trajectory dataset and a five-family benchmark. Reference panels improve mean visual order consistency (VOC) from 0.967 to 0.986 and OOD-robust VOC from 0.906 to 0.958. With the same 400K pairwise-reward budget, Signed-Hop training with 25% replay reaches 0.9872 mean VOC, compared with 0.9858 for a matched-pool shuffled control. In downstream reinforcement learning, the full model achieves 86.8% mean RoboTwin success and 71/80 successful real-world insertions.",
  "published": "2026-08-16",
  "updated": "2026-08-16",
  "year": "2026",
  "authors": [
   "Yijie Xu",
   "Haopeng Jin",
   "Run Zhou",
   "Shengbang Liu",
   "Sixiang Chen",
   "Hongyang Cheng",
   "Sicheng Hu",
   "Peterson Co",
   "Jinwen Luo",
   "Huajie Tan",
   "Shanghang Zhang"
  ],
  "author_count": 11,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work introduces Robo-Dopamine 2.0, a history- and OOD-aware process reward model with a pairwise prediction interface that combines history-conditioned pairwise rewards that use source-aligned reference panels for synthetic OOD queries and observed rollout history for online queries, while preserving the queried endpoints.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yijie Xu",
    "id": "2456838237",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Haopeng Jin",
    "id": "2408472091",
    "h_index": 1,
    "papers": 10
   },
   {
    "name": "Run Zhou",
    "id": "2458075504",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Shengbang Liu",
    "id": "2397743399",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Sixiang Chen",
    "id": "2303462098",
    "h_index": 9,
    "papers": 12
   },
   {
    "name": "Hongyang Cheng",
    "id": "2372562573",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Sichen Hu",
    "id": "2403530766",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "P. Co",
    "id": "2183781780",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Jinwen Luo",
    "id": "2373683804",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Huajie Tan",
    "id": "2348888831",
    "h_index": 8,
    "papers": 15
   },
   {
    "name": "Shanghang Zhang",
    "id": "2346116279",
    "h_index": 16,
    "papers": 51
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "rl-control",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.15680v1",
  "pdf_url": "https://arxiv.org/pdf/2608.15680v1",
  "html_url": "https://arxiv.org/html/2608.15680v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.15636",
  "slug": "algorithm-architecture-co-design-for-efficient-vla-inference-via-specu",
  "title": "Algorithm-Architecture Co-Design for Efficient VLA Inference via Speculative Inference and Verification",
  "abstract": "Vision-Language-Action (VLA) models have demonstrated remarkable capabilities in the field of embodied AI, but their high computational cost and limited predicted action length hinder real-time deployment. Although Dadu-Corki, a dedicated accelerator for efficient embodied AI, has been introduced, it does not exploit the inherent interaction patterns between the robot and its environment, which results in a relatively short predicted action length. We observe that robotic environments naturally alternate between active states-where precise actions are crucial-and inactive states-where actions have limited impact on task success. This insight enables a new scheduling opportunity: long-action-length speculative prediction in inactive states, paired with selective verification in active states. We propose SpecVLA, an algorithm-system co-design framework that adaptively balances action length, inference latency, and task reliability. On the algorithm side, SpecVLA introduces a state-aware VLA inference execution paradigm and a hardware-friendly construction of a smaller verification model (sVLA) using differential residuals and block-wise mixed-precision quantization. On the system side, we develop a heterogeneous architecture consisting of a GPU and a robotic-specific hardware module, along with a speculative dataflow that decouples VLA and sVLA through parallel execution. Comprehensive evaluations on OpenVLA and RDT across LIBERO and ManiSkill benchmarks show that SpecVLA reduces end-to-end latency significantly while preserving task success rate. By enabling long-action-length speculative prediction with timely verification, SpecVLA achieves real-time robotic manipulation with both high efficiency and reliability.",
  "published": "2026-08-16",
  "updated": "2026-08-16",
  "year": "2026",
  "authors": [
   "Chunyu Qi",
   "Zhuoran Song",
   "Jian Weng",
   "Haozhe Jiang",
   "Xueyuan Liu",
   "Naifeng Jing",
   "Guanghui He",
   "Xiaoyao Liang",
   "Haibing Guan"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Comprehensive evaluations on OpenVLA and RDT across LIBERO and ManiSkill benchmarks show that SpecVLA reduces end-to-end latency significantly while preserving task success rate, and achieves real-time robotic manipulation with both high efficiency and reliability.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Chunyu Qi",
    "id": "2275908561",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Zhuoran Song",
    "id": "123930868",
    "h_index": 12,
    "papers": 68
   },
   {
    "name": "Jian Weng",
    "id": "1836379365",
    "h_index": 10,
    "papers": 27
   },
   {
    "name": "Haozhe Jiang",
    "id": "2332739107",
    "h_index": 2,
    "papers": 13
   },
   {
    "name": "Xueyuan Liu",
    "id": "2278389429",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Naifeng Jing",
    "id": "2314376586",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Guanghui He",
    "id": "2368698006",
    "h_index": 1,
    "papers": 9
   },
   {
    "name": "Xiaoyao Liang",
    "id": "2153398018",
    "h_index": 11,
    "papers": 68
   },
   {
    "name": "Haibing Guan",
    "id": "2351064704",
    "h_index": 4,
    "papers": 24
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "hardware-codesign",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.15636v1",
  "pdf_url": "https://arxiv.org/pdf/2608.15636v1",
  "html_url": "https://arxiv.org/html/2608.15636v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.15573",
  "slug": "not-all-history-helps-velocity-aware-selective-memory-for-long-horizon",
  "title": "Not All History Helps: Velocity-Aware Selective Memory for Long-Horizon End-to-End Autonomous Driving",
  "abstract": "Reliable long-horizon planning remains a key challenge in end-to-end autonomous driving. By accounting for future motion evolution and potential consequences, it provides forward-looking guidance for safe and consistent driving in evolving traffic environments. Existing methods use historical planning states as temporal context. Self-generated history may become stale or conflict with the current motion stage, introducing unreliable priors. We propose StableDrive to address cross-cycle historical reliability and within-horizon motion-stage evolution. Selective Momentum Memory (SMM), implemented with a Mamba selective state-space operator, controls the influence of the preceding self-predicted planning state on the current cycle. Motion-Stage Training Scaffold (MSTS) uses motion-stage, long-horizon trajectory, and longitudinal-motion supervision to guide stage-aware future motion learning and is removed before inference. A fixed parameter midpoint between two architecture-aligned endpoints yields a single deployable SMM planner without model ensembling or extra inference-time computation. On nuScenes under the MomAD evaluation protocol, StableDrive achieves SOTA performance across all reported planning metrics from 1 to 6 s, reducing average collision rate by 23.3%, TPC by 30.9%, and L2 by 11.8% over the best previously reported value for each metric. On the curated Longitudinal-Transition nuScenes (LT-nuScenes), StableDrive reduces 6-s collision rate by 23.81%, TPC by 10.90%, and L2 by 6.37%. On NAVSIM v1 and v2, StableDrive achieves the highest PDMS/EPDMS in all three reported settings, including a 5.7-point EPDMS gain on v2 navhard over the previous best.",
  "published": "2026-08-16",
  "updated": "2026-08-16",
  "year": "2026",
  "authors": [
   "Yuchen Liu",
   "Ziying Song",
   "Shengkai Zhang",
   "Jiannan Chen",
   "Peiliang Wu",
   "Lei Yang",
   "Bin Sun",
   "Yan Gong",
   "Li Wang"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "StableDrive is proposed to address cross-cycle historical reliability and within-horizon motion-stage evolution, and achieves the highest PDMS/EPDMS in all three reported settings, including a 5.7-point EPDMS gain on v2 navhard over the previous best.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yuchen Liu",
    "id": "2326072548",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Ziying Song",
    "id": "2367119325",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Shengkai Zhang",
    "id": "2315625136",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Jiannan Chen",
    "id": "2458320022",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Peiliang Wu",
    "id": "2391119240",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Lei Yang",
    "id": "2257380929",
    "h_index": 11,
    "papers": 16
   },
   {
    "name": "Bin Sun",
    "id": "2284302666",
    "h_index": 4,
    "papers": 17
   },
   {
    "name": "Yan Gong",
    "id": "2357734834",
    "h_index": 2,
    "papers": 12
   },
   {
    "name": "Li Wang",
    "id": "2278451900",
    "h_index": 6,
    "papers": 15
   }
  ],
  "comment": "14 pages, 7 figures",
  "topics": [
   "sim2real",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.15573v1",
  "pdf_url": "https://arxiv.org/pdf/2608.15573v1",
  "html_url": "https://arxiv.org/html/2608.15573v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.15560",
  "slug": "reforce-learning-force-aware-retargeting-for-dexterous-manipulation",
  "title": "ReForce: Learning Force-aware Retargeting for Dexterous Manipulation",
  "abstract": "Human demonstrations offer a scalable data source for dexterous manipulation, but transferring them to robot actions remains challenging due to the embodiment gap. Today's retargeting is mostly kinematic, yet manipulation is decided by force, which governs how the hand interacts with the object and how the object moves. In this paper, we present ReForce, a Force-aware Retargeting method that turns human motion and forces into robot actions that reproduce the intended contact. ReForce predicts a residual on the kinematically retargeted action to reach the desired force, using a general force tracker trained on large-scale simulation interactions. It supports both online force-aware teleoperation and offline data translation. In simulation and on real hardware, ReForce achieves lower force-tracking error and stronger multi-finger contact engagement on contact-rich tasks such as paper-cup grasping and tongs manipulation.",
  "published": "2026-08-16",
  "updated": "2026-08-16",
  "year": "2026",
  "authors": [
   "Yuhang Wu",
   "Lingqi Zeng",
   "Changwei Jing",
   "Jianglong Ye",
   "Xiaolong Wang"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "ReForce is presented, a Force-aware Retargeting method that turns human motion and forces into robot actions that reproduce the intended contact and supports both online force-aware teleoperation and offline data translation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yuhang Wu",
    "id": "2335578066",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Lingqi Zeng",
    "id": "2269761659",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Changwei Jing",
    "id": "2392866450",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Jianglong Ye",
    "id": "2153258399",
    "h_index": 9,
    "papers": 16
   },
   {
    "name": "Xiaolong Wang",
    "id": null,
    "h_index": 0,
    "papers": 0
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "egocentric-data",
   "tactile",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.15560v1",
  "pdf_url": "https://arxiv.org/pdf/2608.15560v1",
  "html_url": "https://arxiv.org/html/2608.15560v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.15549",
  "slug": "mistypilot-enabling-social-robot-control-through-multi-agent-llm-skill",
  "title": "MistyPilot: Enabling Social-Robot Control through Multi-Agent LLM Skill Orchestration",
  "abstract": "Programming small social robots from natural-language instructions requires more than invoking isolated APIs. Interactive tasks combine reactive physical behaviors with stateful social behaviors, while existing interfaces often require developers to manually compose APIs into skills, configure their parameters, bind sensor events to skills, and manage task states at runtime. We present MistyPilot, a multi-agent LLM framework that interprets high-level natural-language instructions and orchestrates the corresponding skills on the Misty social robot. A Task Router dispatches each instruction to one of two specialized agents: a Physically Interactive Agent for sensor-triggered robot control and direct skill invocation, and a Social Interaction Agent for dialogue-oriented task-state management and context-dependent multimodal response generation. To improve efficiency, the Social Interaction Agent reuses previously generated results when applicable and invokes full generation otherwise. We evaluate MistyPilot on five component-level suites, with sensor bindings and skill invocations executed on the physical Misty robot, and a preliminary user study with 12 participants. MistyPilot attains high accuracy on routing, sensor-skill binding, task-state parsing, result reuse, and skill extension up to 100 skills, and lower variance than an otherwise identical single-agent baseline, while participants report positive perceptions of usability and interaction quality. The code will be made publicly available via the project page.",
  "published": "2026-08-16",
  "updated": "2026-08-16",
  "year": "2026",
  "authors": [
   "Xiao Wang",
   "Lu Dong",
   "Ifeoma Nwogu",
   "Srirangaraj Setlur",
   "Venu Govindaraju"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "ECCV 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "MistyPilot is presented, a multi-agent LLM framework that interprets high-level natural-language instructions and orchestrates the corresponding skills on the Misty social robot and attains high accuracy on routing, sensor-skill binding, task-state parsing, result reuse, and skill extension up to 100 skills, and lower variance than an otherwise identical single-agent baseline.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xiao Wang",
    "id": "2345236606",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Lu Dong",
    "id": "2304144489",
    "h_index": 9,
    "papers": 19
   },
   {
    "name": "Ifeoma Nwogu",
    "id": "1841118",
    "h_index": 16,
    "papers": 92
   },
   {
    "name": "Srirangaraj Setlur",
    "id": "1800513",
    "h_index": 24,
    "papers": 139
   },
   {
    "name": "Venugopal Govindaraju",
    "id": "2193297520",
    "h_index": 10,
    "papers": 40
   }
  ],
  "comment": "Accepted at the ECCV 2026 ACVR Workshop",
  "topics": [
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.15549v1",
  "pdf_url": "https://arxiv.org/pdf/2608.15549v1",
  "html_url": "https://arxiv.org/html/2608.15549v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.15541",
  "slug": "contact-modes-are-strata-what-geometric-structure-buys-in-discrete-con",
  "title": "Contact Modes Are Strata: What Geometric Structure Buys in Discrete-Continuous Planning",
  "abstract": "Contact-rich manipulation poses a discrete question and a continuous one at once, namely which contacts are active and how to move while they hold. The two are coupled by a change of dimension, since each contact that a robot maintains confines its motion to a lower-dimensional manifold. We make that coupling the explicit object of planning by observing that a contact mode is not merely analogous to a stratum of the configuration space; it is one. A plan is then a walk over strata whose within-stratum segments are geodesics. On two contact-rich manipulation tasks in simulation, pushing a T-shaped block around obstacles and reorienting a cube in a dexterous hand, our planner returns solutions within seconds with no mode, contact sequence, or stratum given in advance.",
  "published": "2026-08-16",
  "updated": "2026-08-16",
  "year": "2026",
  "authors": [
   "Phone Thiha Kyaw",
   "Jonathan Kelly"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "On two contact-rich manipulation tasks in simulation, pushing a T-shaped block around obstacles and reorienting a cube in a dexterous hand, the planner returns solutions within seconds with no mode, contact sequence, or stratum given in advance.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Phone Thiha Kyaw",
    "id": "2008300601",
    "h_index": 9,
    "papers": 16
   },
   {
    "name": "Jonathan Kelly",
    "id": "2279755152",
    "h_index": 2,
    "papers": 4
   }
  ],
  "comment": "Submitted to IROS 2026 Workshop on Geometric Representations in Robotics",
  "topics": [
   "dexterous-manipulation",
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.15541v1",
  "pdf_url": "https://arxiv.org/pdf/2608.15541v1",
  "html_url": "https://arxiv.org/html/2608.15541v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.15532",
  "slug": "degenerate-in-whose-frame-an-equivariance-condition-for-degeneracy-det",
  "title": "Degenerate in Whose Frame? An Equivariance Condition for Degeneracy Detection in LiDAR Registration",
  "abstract": "Degeneracy detectors for LiDAR registration commonly return six per-axis binary labels. We ask whether these labels are properties of the scene. Under a body-frame change, the point-to-plane information matrix transforms by congruence, H' = Ad(T)^T H Ad(T), not similarity. Congruence preserves nullity and, through the adjoint reparameterization, identifies the same physical twist subspace; the per-axis footprint and a thresholded spectrum need not be invariant. In a noise-free circular tunnel, shifting the origin by one metre changes which degrees of freedom are flagged. A generalized criterion Hv = lambda Mv is universally frame-independent over positive-semidefinite information forms if and only if its metric rule is equivariant. No fixed metric qualifies, while a rig-adapted one exists only at zero screw pitch, met in one of nineteen surveyed calibrations. The equivariant point-displacement metric M = sum_i J_i^T J_i yields dimensionless, scene-scale-invariant generalized eigenvalues. They are invariant to body frame, consistent changes of length unit and scene scales; the threshold also transfers empirically across sequences. Across 365 frame pairs from four public sequences, labels rarely change at practical extrinsic magnitudes, yet a remapping estimator's correction differs between body-frame choices on 44.5-69.5% of pairs, with a median of 0.7-4.0 mm and a maximum of 0.87 m. The per-axis footprint changes even under the equivariant metric, placing the fundamental issue in the reported quantity.",
  "published": "2026-08-16",
  "updated": "2026-08-16",
  "year": "2026",
  "authors": [
   "Yujie Zhang",
   "Chunlei Zhao",
   "Yuzong Lin",
   "Yuxuan Guo",
   "Xiaohui Jia",
   "Jinyue Liu"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yujie Zhang",
    "id": "2380803030",
    "h_index": 0,
    "papers": 6
   },
   {
    "name": "Chunlei Zhao",
    "id": "2456539395",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yuzong Lin",
    "id": "2456536840",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yuxuan Guo",
    "id": "2449364409",
    "h_index": 0,
    "papers": 5
   },
   {
    "name": "Xiaohui Jia",
    "id": "2242175580",
    "h_index": 7,
    "papers": 38
   },
   {
    "name": "Jinyue Liu",
    "id": "2155695702",
    "h_index": 8,
    "papers": 54
   }
  ],
  "comment": "8 pages, 4 figures, 5 tables. Submitted to IEEE Robotics and Automation Letters (RA-L)",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.15532v1",
  "pdf_url": "https://arxiv.org/pdf/2608.15532v1",
  "html_url": "https://arxiv.org/html/2608.15532v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.15509",
  "slug": "temporal-logic-guided-universal-task-representations-for-reinforcement",
  "title": "Temporal Logic Guided Universal Task Representations for Reinforcement Learning",
  "abstract": "Task guided agents demonstrate strong performance in a wide range of complex tasks. However, most existing task representation algorithms are tailored to specific contexts and struggle to generalize across diverse scenarios. Moreover, they typically depend on gradient signals from reinforcement learning controllers to update their weights, which can degrade both representation quality and learning efficiency. To overcome these limitations, we propose LOTUS, a temporal logic inspired universal task representation framework that can be seamlessly integrated into any RL algorithm to enhance agent performance across diverse task settings. Specifically, we design a novel task representation architecture capable of modeling relationships and extracting task semantics from LTL formulas. We further introduce a more effective update mechanism that treats the LTL encoder as a policy, thereby improving representation capacity. To enhance stability and robustness, LOTUS leverages the bisimulation metric, which provides theoretical guarantees for LTL representation, including behavioral equivalence, optimality fidelity, and trajectory robustness. Experimental results show that LOTUS outperforms most existing methods in learning efficiency, generalization capability, and representation quality. Specifically, LOTUS accelerates convergence over 20% in single-task scenarios, achieves a 15%-45% higher success rate in unseen manipulation tasks, and improves generalization performance over 25% in complex multi-task environments with increased sub-goal depth or conjunctions. The corresponding code, videos, and appendix are available at: https://lotus-website.github.io/.",
  "published": "2026-08-16",
  "updated": "2026-08-16",
  "year": "2026",
  "authors": [
   "Hao Zhang",
   "Zhangli Zhou",
   "Zhen Kan"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.FL",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work designs a novel task representation architecture capable of modeling relationships and extracting task semantics from linear temporal logic (LTL) formulas and introduces a more effective update mechanism that treats the LTL encoder as a policy, thereby improving representation capacity.",
  "doi": "10.1109/TNNLS.2026.3698967",
  "oa_pdf": "https://doi.org/10.48550/arxiv.2608.15509",
  "s2_authors": [
   {
    "name": "Hao Zhang",
    "id": "2292211939",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Zhangli Zhou",
    "id": "1500385565",
    "h_index": 8,
    "papers": 29
   },
   {
    "name": "Zhen Kan",
    "id": "2293404312",
    "h_index": 3,
    "papers": 14
   }
  ],
  "comment": "Accepted by IEEE Transactions on Neural Networks and Learning Systems (Early Access). Project page: https://lotus-website.github.io/",
  "topics": [
   "rl-control",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.15509v1",
  "pdf_url": "https://arxiv.org/pdf/2608.15509v1",
  "html_url": "https://arxiv.org/html/2608.15509v1",
  "code_url": "https://lotus-website.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2608.15502",
  "slug": "ecovla-energy-efficient-device-edge-co-inference-for-vision-language-a",
  "title": "EcoVLA: Energy-Efficient Device-Edge Co-Inference for Vision-Language-Action Models under Real-Time Constraints",
  "abstract": "Vision-Language-Action (VLA) models have emerged as a promising foundation for Embodied AI, but their high inference cost poses significant challenges for deployment in robotic systems. In practice, on-device inference is constrained by limited compute capacity and energy budgets, struggling to simultaneously satisfy real-time control and energy efficiency requirements. Alternatively, offloading the inference workload to an edge server is susceptible to fluctuations in system conditions, introducing unpredictable latency risks. Device-edge co-inference offers a promising solution, but systematic research tailored to VLA models remains scarce, particularly a unified co-inference framework that jointly addresses real-time constraints and system-level energy efficiency. Thus, we propose EcoVLA, an adaptive device-edge co-inference framework for VLA models that maximizes system energy efficiency under real-time constraints. EcoVLA first introduces a unified stage-level abstraction over different VLA paradigms, establishing an architecture-agnostic co-inference design space. It then formulates a joint device-edge-network latency and energy prediction model to enable rapid runtime evaluation of candidate co-inference schemes. Building on this, EcoVLA continuously selects the energy-optimal scheme satisfying real-time constraints with millisecond-level overhead, adapting to runtime variations in network and system states. Furthermore, EcoVLA incorporates a lightweight transmission mechanism for inter-stage intermediate tensors to reduce the communication overhead incurred by cross-device collaboration. Experimental results across VLA models show that EcoVLA improves system energy efficiency by up to 236% over existing co-inference approaches under a 20 Hz action output frequency constraint, while consistently maintaining SLO satisfaction under dynamic network and edge workload conditions.",
  "published": "2026-08-16",
  "updated": "2026-08-16",
  "year": "2026",
  "authors": [
   "Ao Zhou",
   "Bo Dai",
   "Le Yu",
   "Xingyu Liu",
   "Zeyu Hao",
   "Lingkun Long",
   "Chunming Hu",
   "Jianlei Yang"
  ],
  "author_count": 8,
  "categories": [
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.AI",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "EcoVLA is proposed, an adaptive device-edge co-inference framework for VLA models that maximizes system energy efficiency under real-time constraints and incorporates a lightweight transmission mechanism for inter-stage intermediate tensors to reduce the communication overhead incurred by cross-device collaboration.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ao Zhou",
    "id": "2064473548",
    "h_index": 6,
    "papers": 14
   },
   {
    "name": "Bo Dai",
    "id": "2372440169",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Le Yu",
    "id": "2458116881",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Xingyu Liu",
    "id": "2454981102",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Zeyu Hao",
    "id": "2453638463",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Lingkun Long",
    "id": "2265384543",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Chunming Hu",
    "id": "2261276468",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Jianlei Yang",
    "id": "2109723583",
    "h_index": 11,
    "papers": 46
   }
  ],
  "comment": "Accepted by APPT 2026",
  "topics": [
   "vla",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.15502v1",
  "pdf_url": "https://arxiv.org/pdf/2608.15502v1",
  "html_url": "https://arxiv.org/html/2608.15502v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.15490",
  "slug": "vision-based-tactile-intelligence-for-robotics-sensing-learning-and-em",
  "title": "Vision-Based Tactile Intelligence for Robotics: Sensing, Learning, and Embodied Manipulation",
  "abstract": "Tactile sensing is essential for robots in contact-rich tasks, yet many tactile sensors still provide sparse, low-dimensional signals that do not capture sufficient information for complex robotic perception and interaction. Vision-based tactile sensors (VBTSs) offer a powerful alternative by con-verting contact-induced deformation of a soft interface into im-ages. The image-based formulation gives VBTSs high-resolution, information-rich tactile observations that enable complex robotic tasks. This review surveys the full VBTS pipeline and treats sensing hardware, learning methods, simulation, and datasets as an integrated sensing-and-learning system. We 1) organize representative VBTSs into a hardware taxonomy structured by deformable elastomer design, sensor size and shape, and optical system design to guide future sensor development; 2) present a hierarchical view of learning-based tactile intelligence from low-level signal understanding to task-level policies and foundation models; and 3) examine simulation platforms and tactile datasets as a scaling layer, together with sim-to-real transfer and cross-sensor adaptation for training, benchmarking, and deployment. Finally, we identify open challenges and future directions for VBTSs in robotics. By providing a holistic view of how hardware, AI architectures, simulation, and datasets interact, this review aims to advance tactile intelligence for contact-rich robotic tasks.",
  "published": "2026-08-16",
  "updated": "2026-08-16",
  "year": "2026",
  "authors": [
   "Peng Zhou",
   "Jun Hu",
   "Sihan Chen",
   "Zeqing Zhang",
   "Haofei Ma",
   "Zhenyu Lu",
   "Sichao Liu",
   "Xueqian Wang",
   "Pai Zheng",
   "Xiang Li",
   "Shan Luo",
   "Jia Pan",
   "David Navarro-Alarcon",
   "Chenguang Yang",
   "Michael Yu Wang"
  ],
  "author_count": 15,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This review surveys the full VBTS pipeline and treats sensing hardware, learning methods, simulation, and datasets as an integrated sensing-and-learning system to advance tactile intelligence for contact-rich robotic tasks.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Peng Zhou",
    "id": "2276487843",
    "h_index": 10,
    "papers": 29
   },
   {
    "name": "Jun Hu",
    "id": "2404341658",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Sihan Chen",
    "id": "2335781176",
    "h_index": 3,
    "papers": 13
   },
   {
    "name": "Zeqing Zhang",
    "id": "2374290695",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Haofei Ma",
    "id": "2238913295",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Zhenyu Lu",
    "id": "2338469366",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Sichao Liu",
    "id": "2023843509",
    "h_index": 23,
    "papers": 47
   },
   {
    "name": "Xueqian Wang",
    "id": "2458703129",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Pai Zheng",
    "id": "2280139310",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Xiang Li",
    "id": "2377564463",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Shan Luo",
    "id": "2307566527",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Jia Pan",
    "id": "2258308591",
    "h_index": 8,
    "papers": 20
   },
   {
    "name": "D. Navarro-Alarc\u00f3n",
    "id": "1401318419",
    "h_index": 23,
    "papers": 168
   },
   {
    "name": "Chenguang Yang",
    "id": "2260627497",
    "h_index": 15,
    "papers": 127
   },
   {
    "name": "Michael Yu Wang",
    "id": "2248999920",
    "h_index": 2,
    "papers": 6
   }
  ],
  "comment": "",
  "topics": [
   "tactile",
   "sim2real",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.15490v1",
  "pdf_url": "https://arxiv.org/pdf/2608.15490v1",
  "html_url": "https://arxiv.org/html/2608.15490v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.15461",
  "slug": "detachable-wire-drive-reconfigurable-robot-architecture-with-shared-ac",
  "title": "Detachable Wire Drive : Reconfigurable Robot Architecture with Shared Actuators",
  "abstract": "Reconfigurable robots offer significant potential for adapting to diverse tasks; however, conventional centralized architectures often require dedicated actuators for each module, leading to substantial increases in overall system weight, volume, and cost. To address these challenges, this paper presents the \"Detachable Wire Drive,\" a reconfigurable robotic system that enables the sharing of heavy and expensive actuators across various morphologies. The core of this system is the \"Wire Detach Unit,\" a mechanism designed to physically split and reconnect wire drive paths, allowing motors to be consolidated into a common base unit. We demonstrate the versatility of this approach by developing a 2-DOF rigid arm, a continuum arm, and two distinct grippers, all of which are interchangeably attached to, and driven by, a single shared actuator set. Experimental results validate the mechanical reliability of the detachment process and the control framework's ability to seamlessly manage transitions between configurations, highlighting a path toward more efficient and multi-functional robotic systems.",
  "published": "2026-08-16",
  "updated": "2026-08-16",
  "year": "2026",
  "authors": [
   "Takahiro Hattori",
   "Kento Kawaharazuka",
   "Kei Okada"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Takahiro Hattori",
    "id": "2365802659",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Kento Kawaharazuka",
    "id": "8308607",
    "h_index": 17,
    "papers": 220
   },
   {
    "name": "Kei Okada",
    "id": "2248244895",
    "h_index": 5,
    "papers": 68
   }
  ],
  "comment": "Accepted at IROS2026, website - https://hatofly.github.io/detachable-wire-drive/",
  "topics": [
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.15461v1",
  "pdf_url": "https://arxiv.org/pdf/2608.15461v1",
  "html_url": "https://arxiv.org/html/2608.15461v1",
  "code_url": "https://hatofly.github.io/detachable-wire-drive/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2608.15446",
  "slug": "guider-evaluating-goal-free-human-intent-inference-for-teleoperated-ma",
  "title": "GUIDER: Evaluating Goal-Free Human Intent Inference for Teleoperated Manipulation on Real-Robot Data",
  "abstract": "This paper presents an evaluation of a goal-free probabilistic framework for human intent inference during robotic manipulation. We deploy the Global User Intent Dual-phase Estimation for Robots (GUIDER) on data collected from a robotic arm to test the manipulation phase across various assistance scenarios, including making tea and fetching medicine. To support operation, we add online probability updates, workspace limits, support-plane filtering, and a grasping mode that prioritizes feasible grasp regions, all of which are tested on the recorded data while preserving its original temporal conditions. Across 20 manipulation steps in three scenarios, GUIDER estimated human intent within the correct grasp-candidate set in all cases and achieved a time to confident prediction of 3.7 s, a remaining time before first grasp of 49.6 s, a prediction stability of 96.4%, and a runtime of 4.857/4.474 s (mean/median) per perceptual phase of intent.",
  "published": "2026-08-15",
  "updated": "2026-08-15",
  "year": "2026",
  "authors": [
   "Nicholas Kenny",
   "Cesar Alan Contreras",
   "Basile Ouedraogo",
   "Rustam Stolkin",
   "Manolis Chiou",
   "Maria Kyrarini"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.HC"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The Global User Intent Dual-phase Estimation for Robots (GUIDER) is deployed on data collected from a robotic arm to test the manipulation phase across various assistance scenarios, including making tea and fetching medicine.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Nicholas Kenny",
    "id": "2458109382",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Cesar Alan Contreras",
    "id": "2279910470",
    "h_index": 6,
    "papers": 14
   },
   {
    "name": "Basile Ouedraogo",
    "id": "2458108011",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Rustam Stolkin",
    "id": "1490453006",
    "h_index": 12,
    "papers": 74
   },
   {
    "name": "Manolis Chiou",
    "id": "3134567",
    "h_index": 10,
    "papers": 34
   },
   {
    "name": "M. Kyrarini",
    "id": "2287968661",
    "h_index": 4,
    "papers": 24
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.15446v1",
  "pdf_url": "https://arxiv.org/pdf/2608.15446v1",
  "html_url": "https://arxiv.org/html/2608.15446v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.15440",
  "slug": "accelerating-mixed-discrete-continuous-motion-planning-via-neural-grap",
  "title": "Accelerating Mixed Discrete-Continuous Motion Planning via Neural Graphs of Convex Sets",
  "abstract": "Motion planning problems such as collision-free navigation and contact-rich manipulation can be naturally formulated as optimization problems that couple discrete decisions with continuous trajectories. The Graphs of Convex Sets (GCS) framework offers a practical solution to these problems. It represents discrete decisions as nodes of a graph and encodes continuous trajectories in the edges connecting them. However, the resulting optimization subproblems can become computationally prohibitive for online replanning. In this work, we propose a learning-based strategy to mitigate this limitation. Specifically, we replace the costly convex relaxation step required by nominal GCS with a single forward pass through a Graph Attention Network that predicts a set of highly probable candidate paths through the graph. A lightweight ranking network then orders these candidates by their estimated trajectory cost. Evaluating them in this order, we terminate our search early while still recovering a near-optimal motion plan. We validate the resulting pipeline across diverse robotic tasks, including collision-free motion planning for a 3D quadrotor and a 7-DoF manipulator, and planning through contact for planar pushing. Across both convex and non-convex cost and constraint settings, our approach yields up to two orders of magnitude speedup over nominal GCS while maintaining a 100% success rate, at the cost of some suboptimality in the recovered solutions. Code implementations and video demonstrations can be found at https://neural-gcs.github.io/.",
  "published": "2026-08-15",
  "updated": "2026-08-15",
  "year": "2026",
  "authors": [
   "Ananya Trivedi",
   "Sarvesh Prajapati",
   "Mohamed Khalid M Jaffar",
   "Zhexin Xu",
   "David Rosen",
   "Taskin Padir"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work replaces the costly convex relaxation step required by nominal GCS with a single forward pass through a Graph Attention Network that predicts a set of highly probable candidate paths through the graph, and generates a lightweight ranking network that orders these candidates by their estimated trajectory cost.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ananya Trivedi",
    "id": "2228171663",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Sarvesh Prajapati",
    "id": "2287922847",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "M. K. M. Jaffar",
    "id": "122699322",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Zhexi Xu",
    "id": "2410374718",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "David M. Rosen",
    "id": "2242000854",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "T. Pad\u0131r",
    "id": "1709510",
    "h_index": 22,
    "papers": 211
   }
  ],
  "comment": "",
  "topics": [
   "tactile",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.15440v1",
  "pdf_url": "https://arxiv.org/pdf/2608.15440v1",
  "html_url": "https://arxiv.org/html/2608.15440v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.15437",
  "slug": "mm-bev-enhancing-timeliness-by-computing-where-and-when-it-matters",
  "title": "MM-BEV: Enhancing Timeliness by Computing Where and When it Matters",
  "abstract": "Multimodal bird's-eye-view (BEV) perception combines LiDAR depth accuracy with dense camera semantics, but its high computational cost and imperfect sensing conditions make real-time deployment challenging. Existing methods largely compress individual detectors and overlook three opportunities: structured sparsity within camera and LiDAR inputs, timing misalignment between modalities, and the fact that many detected objects do not affect the planner's immediate action. We present MM-BEV, a real-time multimodal BEV system guided by a simple principle: compute where and when it matters. MM-BEV divides perception into mandatory work for safety-critical objects within braking distance of the ego vehicle and with short time-to-collision (TTC), and optional work for less urgent regions. It prioritizes mandatory work and reduces or sheds optional work under tight compute budgets. MM-BEV integrates four mechanisms: (1) a criticality-ranked temporal ROI selector based on motion-extrapolated detections from prior frames; (2) sparse, ROI-aware feature extraction using shared-shape camera crops at context-adaptive resolution and ROI-aware LiDAR voxelization; (3) a latency-aware coordinator that adapts LiDAR sweeps, image resolution, and keyframes according to scene dynamics and TTC; and (4) an asynchronous scheduler that decouples sensing from inference and skips stale frames. On nuScenes, MM-BEV reduces inference latency by 1.96x and end-to-end latency by 2.93x, with no loss in geometry-critical recall and only a 0.2 percentage-point drop in safety-critical recall. On a Clearpath Husky A300 equipped with an Ouster-128 LiDAR, BEV cameras, and a Jetson AGX Orin, MM-BEV further reduces mean latency by 2.11x, demonstrating its potential for real-world autonomous systems.",
  "published": "2026-08-15",
  "updated": "2026-08-15",
  "year": "2026",
  "authors": [
   "Liangkai Liu",
   "Kang G. Shin"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO",
   "cs.CV",
   "cs.DC",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "MM-BEV divides perception into mandatory work for safety-critical objects within braking distance of the ego vehicle and with short time-to-collision (TTC), and optional work for less urgent regions, which prioritizes mandatory work and reduces or sheds optional work under tight compute budgets.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Liangkai Liu",
    "id": "2326067289",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Kang G. Shin",
    "id": "2333453881",
    "h_index": 1,
    "papers": 6
   }
  ],
  "comment": "12 pages, 20 figures",
  "topics": [
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.15437v1",
  "pdf_url": "https://arxiv.org/pdf/2608.15437v1",
  "html_url": "https://arxiv.org/html/2608.15437v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.15410",
  "slug": "floodreasonbench-benchmarking-vlm-reasoning-segmentation-for-embodied",
  "title": "FloodReasonBench: Benchmarking VLM Reasoning Segmentation for Embodied Flood Response at the Edge",
  "abstract": "Reasoning segmentation enables vision-language models (VLMs) to translate mission-relevant language requests into pixel-level visual grounding, offering a natural perception interface for embodied agents. However, existing benchmarks largely focus on generic visual scenes and overlook the domain and resource constraints encountered in flood-response platforms. We present FloodReasonBench, a benchmark for VLM reasoning segmentation for embodied flood response at the edge. At its core, FloodReasonBench introduces FloodResponseSeg, a flood-specific reasoning-segmentation dataset constructed from real-world scenes and response-relevant targets. Beyond task accuracy, the benchmark characterizes reasoning-segmentation pipelines under lightweight visual encoding, hierarchical split inference, and compressed intermediate representations. We observe strong partition-dependent accuracy variation in the generic pre-adaptation setting, while the flood-adapted target-workload design space exhibits a substantially more compact accuracy range across partitions. Evaluation on an NVIDIA Jetson AGX Xavier further exposes the tradeoffs among reasoning-segmentation accuracy, edge-side latency, energy, and communication footprint, enabling quality-constrained selection of edge operating points. Together, these results provide a task- and system-level characterization of reasoning segmentation for resource-constrained embodied flood response at the edge.",
  "published": "2026-08-15",
  "updated": "2026-08-15",
  "year": "2026",
  "authors": [
   "Rajat Bhattacharjya",
   "Yoomee Jung",
   "Minwoo Kim",
   "Sing-Yao Wu",
   "Eli Bozorgzadeh",
   "Nalini Venkatasubramanian",
   "Nikil Dutt"
  ],
  "author_count": 7,
  "categories": [
   "cs.DC",
   "cs.AI",
   "cs.CV",
   "cs.RO",
   "eess.SY"
  ],
  "primary_category": "cs.DC",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work presents FloodReasonBench, a benchmark for VLM reasoning segmentation for embodied flood response at the edge, and introduces FloodResponseSeg, a flood-specific reasoning-segmentation dataset constructed from real-world scenes and response-relevant targets.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Rajat Bhattacharjya",
    "id": "1919420336",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Yoo-Min Jung",
    "id": "2316957294",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Minwoo Kim",
    "id": "2458115896",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Sing-Yao Wu",
    "id": "2107775466",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "E. Bozorgzadeh",
    "id": "81166981",
    "h_index": 5,
    "papers": 15
   },
   {
    "name": "N. Venkatasubramanian",
    "id": "1732742",
    "h_index": 40,
    "papers": 435
   },
   {
    "name": "Nikil D. Dutt",
    "id": "2286954342",
    "h_index": 2,
    "papers": 9
   }
  ],
  "comment": "Paper is currently under review. The code and dataset will be made public upon acceptance",
  "topics": [
   "video-generation"
  ],
  "orgs": [
   "NVIDIA"
  ],
  "abs_url": "https://arxiv.org/abs/2608.15410v1",
  "pdf_url": "https://arxiv.org/pdf/2608.15410v1",
  "html_url": "https://arxiv.org/html/2608.15410v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2608.15380",
  "slug": "adaptive-bridge-a-proxy-based-decoupling-layer-for-mitigating-dds-back",
  "title": "Adaptive Bridge: A Proxy-Based Decoupling Layer for Mitigating DDS Backpressure in ROS 2",
  "abstract": "In ROS 2 systems using DDS, a single slow subscriber on a RELIABLE topic can cause backpressure that degrades throughput and latency for all subscribers sharing the same publisher, including safety-critical local nodes. We present Adaptive Bridge, a proxy-based decoupling layer that isolates critical subscribers from noncritical ones through topic splitting and adaptive rate control. The proxy subscribes to the original topic and republishes onto two independent DDS writers, one RELIABLE for critical consumers and one BEST_EFFORT for noncritical consumers, breaking the causal chain of backpressure propagation. A probe-based classifier monitors subscriber health with hysteresis and adjusts noncritical rate limits in real time. We evaluate the system under Gilbert-Elliot bursty wireless loss using a reproducible Docker-based harness. Results show the bridge reduces critical subscriber tail latency from up to 15 seconds to under 2 milliseconds at p95 and preserves publisher throughput at 30 Hz regardless of impairment severity.",
  "published": "2026-08-15",
  "updated": "2026-08-15",
  "year": "2026",
  "authors": [
   "Kaushalraj Puwar"
  ],
  "author_count": 1,
  "categories": [
   "cs.NI",
   "cs.RO"
  ],
  "primary_category": "cs.NI",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Adaptive Bridge is presented, a proxy-based decoupling layer that isolates critical subscribers from noncritical ones through topic splitting and adaptive rate control, and reduces critical subscriber tail latency from up to 15 seconds to under 2 milliseconds at p95 and preserves publisher throughput at 30 Hz regardless of impairment severity.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Kaushalraj Puwar",
    "id": "2458107252",
    "h_index": 0,
    "papers": 0
   }
  ],
  "comment": "6 pages, 5 figures",
  "topics": [
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.15380v1",
  "pdf_url": "https://arxiv.org/pdf/2608.15380v1",
  "html_url": "https://arxiv.org/html/2608.15380v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.15375",
  "slug": "admissibility-preserving-control-for-strict-feedback-nonlinear-systems",
  "title": "Admissibility-Preserving Control for Strict-Feedback Nonlinear Systems with Asymmetric Actuator Constraints",
  "abstract": "This paper develops Admissibility-Preserving Control (APC), a realization-centered safety-critical control framework for strict-feedback systems subject to asymmetric actuator limits, time-varying output constraints, and actuator-rate limitations. APC denotes the overall control architecture, whereas an Admissibility-Preserving Input Realization (APIR) denotes its constraint-realization module. Therein, the APIR dynamically generates the physical plant input while rendering its prescribed asymmetric actuator set forward invariant. In contrast to algebraic clipping and post-design saturation compensation, the actuator limits are embedded directly in a continuously differentiable dynamic realization with user-selectable regularity and interpretable tuning parameters. The APIR is integrated with recursive backstepping by treating the realized plant input as an additional state. The resulting design does not require an input-to-state stability assumption on the uncontrolled plant. Instead, the nonlinear drift terms are compensated recursively, subject to an explicit compatibility condition between the desired motion, the available control authority, and the APIR interior gain. The framework is further extended to time-varying output-safe tracking through a smooth asymmetric logarithmic barrier coordinate and its associated Lyapunov function and to simultaneous actuator-magnitude and rate constraints through a cascaded APIR. Rigorous Lyapunov and invariance analyses establish regional asymptotic tracking, forward invariance of the compatible admissible sets, and boundedness of all closed-loop signals. Numerical studies illustrate asymmetric actuator utilization, output-safety preservation, and magnitude-rate constraint enforcement.",
  "published": "2026-08-15",
  "updated": "2026-08-15",
  "year": "2026",
  "authors": [
   "Saurabh Kumar",
   "Shashi Ranjan Kumar",
   "Abhinav Sinha"
  ],
  "author_count": 3,
  "categories": [
   "eess.SY",
   "cs.RO",
   "math.DS"
  ],
  "primary_category": "eess.SY",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Saurabh Kumar",
    "id": "2144041592",
    "h_index": 5,
    "papers": 29
   },
   {
    "name": "S. R. Kumar",
    "id": "2109680801",
    "h_index": 24,
    "papers": 158
   },
   {
    "name": "Abhinav Sinha",
    "id": "143668057",
    "h_index": 19,
    "papers": 133
   }
  ],
  "comment": "",
  "topics": [
   "hardware-codesign",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.15375v1",
  "pdf_url": "https://arxiv.org/pdf/2608.15375v1",
  "html_url": "https://arxiv.org/html/2608.15375v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.15289",
  "slug": "score-shape-conforming-regions-for-flight-in-enclosed-degraded-environ",
  "title": "SCORE: Shape-Conforming Regions for Flight in Enclosed, Degraded Environments",
  "abstract": "Autonomous UAVs enter enclosed environments such as caves and collapsed structures that confine the vehicle and degrade perception. Conformal prediction provides a distribution-free guarantee by calibrating how far an obstacle keep-out must expand to absorb perception error at a target coverage level. However, existing keep-out regions use convex primitives whose bulges consume narrow passages and grow as perception degrades. Our main contribution defines the nonconformity score on a signed distance field (SDF). This produces a non-convex keep-out that tightly follows obstacle geometry and avoids the unnecessary bulging of equal-margin convex regions. Two supporting components keep this geometry usable as perception degrades. First, a voxelwise union of complementary sensor observations certifies voxels that any single sensor misses. Second, the margin around the obstacle adapts to measured visibility without weather labels or the online ground-truth feedback that single-pass flight cannot provide. Results on real subterranean data show that the resulting distribution-free, shape-conforming keep-out retains more usable free space than convex baselines at the same certified coverage, and produces safer closed-loop flight.",
  "published": "2026-08-15",
  "updated": "2026-08-15",
  "year": "2026",
  "authors": [
   "Eric Minwoo Kim",
   "Jong-Kook Kim"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work defines the nonconformity score on a signed distance field (SDF) and produces a non-convex keep-out that tightly follows obstacle geometry and avoids the unnecessary bulging of equal-margin convex regions.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Eric Minwoo Kim",
    "id": "2458115806",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Jong-Kook Kim",
    "id": "2458124819",
    "h_index": 0,
    "papers": 0
   }
  ],
  "comment": "8 pages, 4 figures. Technical appendix available on request",
  "topics": [
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.15289v1",
  "pdf_url": "https://arxiv.org/pdf/2608.15289v1",
  "html_url": "https://arxiv.org/html/2608.15289v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.15285",
  "slug": "phaselora-control-regime-conditioned-low-rank-adaptation-for-continuou",
  "title": "PhaseLoRA: Control-Regime-Conditioned Low-Rank Adaptation for Continuous-Action Vision-Language-Action Policies",
  "abstract": "Parameter-efficient fine-tuning (PEFT) is a natural way to adapt pretrained vision-language-action (VLA) policies, but most adapter designs apply temporally static updates throughout a control rollout, overlooking the phase-dependent nature of continuous-action manipulation. Such policies traverse distinct regimes, including approach, contact transition, grasping, transport, and placement, each requiring different adaptation behaviors. We propose \\textbf{PhaseLoRA}, a lightweight LoRA parameterization that conditions adaptation at each action-chunk prediction step using two weakly supervised descriptors: fine-control tendency and event/boundary intensity. PhaseLoRA modulates the LoRA left factor in the action expert, allowing the effective low-rank update direction to vary over time while keeping the backbone largely frozen. On LIBERO, PhaseLoRA improves average success rate by 12.2 points over a matched-parameter high-rank LoRA baseline and outperforms stronger LoRA variants. Ablations show that random temporal modulation and scalar gating do not reproduce the performance of the full model, while update-direction analyses reveal structured temporal variation associated with the predicted control descriptors. These results establish within-trajectory conditioning as an effective lightweight PEFT axis for continuous-action VLA policies.",
  "published": "2026-08-15",
  "updated": "2026-08-15",
  "year": "2026",
  "authors": [
   "Yufei Guo",
   "Yinan Wu",
   "Haoran Duan",
   "Guiguang Ding",
   "Jungong Han"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A lightweight LoRA parameterization that conditions adaptation at each action-chunk prediction step using two weakly supervised descriptors: fine-control tendency and event/boundary intensity is proposed, establishing within-trajectory conditioning as an effective lightweight PEFT axis for continuous-action VLA policies.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yufei Guo",
    "id": "2276101257",
    "h_index": 6,
    "papers": 22
   },
   {
    "name": "Yinan Wu",
    "id": "48607579",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Haoran Duan",
    "id": "2283933352",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Guiguang Ding",
    "id": "38329336",
    "h_index": 59,
    "papers": 205
   },
   {
    "name": "Jungong Han",
    "id": "2350491677",
    "h_index": 3,
    "papers": 9
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "dexterous-manipulation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.15285v1",
  "pdf_url": "https://arxiv.org/pdf/2608.15285v1",
  "html_url": "https://arxiv.org/html/2608.15285v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.15284",
  "slug": "vtinstructor-visual-trajectory-prompting-for-navigation-instruction-ge",
  "title": "VTInstructor: Visual Trajectory Prompting for Navigation Instruction Generation in Continuous Environments",
  "abstract": "Navigation instruction generation from ego-centric RGB video in continuous environments is an important yet challenging task for human-robot interaction and scalable dataset construction. Prior instruction generators assume discrete viewpoint graphs with panoramic observations, where trajectory structure is explicit; in continuous environments, however, the agent receives only a dense RGB stream, making trajectory cues difficult to recover. We propose VTInstructor, the first VLN instruction generation framework for continuous environments. Our key idea is to convert implicit trajectory geometry into explicit visual trajectory prompts: EDTC condenses long RGB trajectories into navigation-critical keyframes, VTP overlays path, turn, and goal cues onto these anchors, VTMod injects the resulting trajectory signals into the visual encoder, and VT-GRPO further calibrates this spatial injection during training, all without requiring a navigation graph, pre-built map, or scene reconstruction. On the challenging R2R-CE and RxR-CE Val Unseen benchmarks, VTInstructor sets a new state of the art across all standard NLG metrics, surpassing the strongest baseline by +0.357 CIDEr and +0.109 CIDEr, respectively. Beyond automatic metrics, VTInstructor-generated instructions raise a frozen follower's success rate to 63.3%, a +14.7 percentage-point gain over the best competing instruction source, and provide consistent data augmentation gains of +3 SR points on downstream navigation tasks.",
  "published": "2026-08-15",
  "updated": "2026-08-15",
  "year": "2026",
  "authors": [
   "Haolin Yang",
   "Yuxing Long",
   "Zihan Yang",
   "Hao Dong"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CL",
   "cs.CV",
   "cs.MM"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "VTInstructor is proposed, the first VLN instruction generation framework for continuous environments that converts implicit trajectory geometry into explicit visual trajectory prompts, and sets a new state of the art across all standard NLG metrics.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Haolin Yang",
    "id": "2363728163",
    "h_index": 3,
    "papers": 15
   },
   {
    "name": "Yuxing Long",
    "id": "2242982685",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Zihan Yang",
    "id": "2376121157",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Hao Dong",
    "id": "2243529948",
    "h_index": 4,
    "papers": 8
   }
  ],
  "comment": "accepted by ACM MM 2026",
  "topics": [
   "navigation",
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.15284v1",
  "pdf_url": "https://arxiv.org/pdf/2608.15284v1",
  "html_url": "https://arxiv.org/html/2608.15284v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.15269",
  "slug": "remember-smarter-visual-history-compressor-and-hyperbolic-experience-s",
  "title": "Remember Smarter: Visual History Compressor and Hyperbolic Experience Space for Robotic Memory",
  "abstract": "Long-horizon robot policies require compact access to recent observations and reusable experience without expanding the vision-language-action (VLA) context. We introduce Remember Smarter (RS), a plug-and-play module with complementary visual-history and hyperbolic experience-memory branches. Its visual branch compresses multi-view patch histories using bidirectional spatial Mamba and causal temporal Mamba, then exposes the resulting memory to action-facing hidden states through residual cross-attention while leaving the VLM visual-token stream unchanged. Its experience branch stores successful final-layer VLM states in a Poincare VAE space, organizes them hierarchically, and asynchronously converts retrieved experience into geodesic prompt tokens without blocking action inference. When adapted to pi0, RS increases total success on LIBERO-Plus from 53.6% to 70.6% and achieves substantial performance gains in real-robot experiments designed to evaluate memory retention and experience utilization.",
  "published": "2026-08-15",
  "updated": "2026-08-15",
  "year": "2026",
  "authors": [
   "Dai Zhou",
   "Jiexi Yan",
   "Tong Li",
   "Yuxuan Wang",
   "Cheng Deng"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Remember Smarter (RS), a plug-and-play module with complementary visual-history and hyperbolic experience-memory branches that achieves substantial performance gains in real-robot experiments designed to evaluate memory retention and experience utilization.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Dai Zhou",
    "id": "2458113048",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Jiexi Yan",
    "id": "2112299425",
    "h_index": 9,
    "papers": 36
   },
   {
    "name": "Tong Li",
    "id": "2458113876",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Yuxuan Wang",
    "id": "2311827747",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Cheng Deng",
    "id": "2293444950",
    "h_index": 4,
    "papers": 20
   }
  ],
  "comment": "19 pages, 7 pages",
  "topics": [
   "vla"
  ],
  "orgs": [
   "Physical Intelligence"
  ],
  "abs_url": "https://arxiv.org/abs/2608.15269v1",
  "pdf_url": "https://arxiv.org/pdf/2608.15269v1",
  "html_url": "https://arxiv.org/html/2608.15269v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2608.15259",
  "slug": "uav-video-deblurring-via-motion-aware-diffusion-a-path-to-robust-targe",
  "title": "UAV Video Deblurring via Motion-Aware Diffusion: A Path to Robust Target Detection",
  "abstract": "Unmanned Aerial Vehicles (UAVs) play a crucial role in various scenarios ranging from disaster response to traffic surveillance. However, aerial video footage often suffers from severe motion blur due to rapid flight maneuvers, vibrations, and camera panning, which can significantly degrade downstream tasks such as target detection. Our goal is to explore a computationally-efficient and effective video deblurring approach to enhance UAV target detection performance. To reduce computational cost, we first propose an Adaptive Latent Scale Selector that dynamically adjusts the latent space resolution according to the intensity of UAV motion, thus balancing detail preservation with inference efficiency. To ensure temporal consistency, we introduce a Multi-Frame Alignment and Learnable Gating module to warp and gate the preceding frames, allowing the model to fuse only relevant temporal information and suppress misaligned or uninformative features. Our method can effectively recover sharp details from the UAV video stream. Extensive experiments on real UAV benchmarks demonstrate that our method not only yields superior deblurring performance but also significantly boosts target detection accuracy, making it highly applicable to robust aerial vision tasks.",
  "published": "2026-08-15",
  "updated": "2026-08-15",
  "year": "2026",
  "authors": [
   "Zhiqiang Hu",
   "Shouren Huang",
   "Masatoshi Ishikawa"
  ],
  "author_count": 3,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "IROS",
  "venue_source": "semantic-scholar",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes an Adaptive Latent Scale Selector that dynamically adjusts the latent space resolution according to the intensity of UAV motion, thus balancing detail preservation with inference efficiency and introducing a Multi-Frame Alignment and Learnable Gating module to ensure temporal consistency.",
  "doi": "10.1109/IROS60139.2025.11246531",
  "oa_pdf": "https://doi.org/10.48550/arxiv.2608.15259",
  "s2_authors": [
   {
    "name": "Zhiqiang Hu",
    "id": "2239440449",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Shouren Huang",
    "id": "2205923626",
    "h_index": 10,
    "papers": 55
   },
   {
    "name": "Masatoshi Ishikawa",
    "id": "2280688574",
    "h_index": 4,
    "papers": 20
   }
  ],
  "comment": "8 pages, 8 figures. Published in the 2025 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS 2025)",
  "topics": [
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.15259v1",
  "pdf_url": "https://arxiv.org/pdf/2608.15259v1",
  "html_url": "https://arxiv.org/html/2608.15259v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.15175",
  "slug": "lapf-llm-agent-based-path-finder-using-the-uavscenes-dataset",
  "title": "LAPF: LLM-Agent-Based Path Finder Using the UAVScenes Dataset",
  "abstract": "Uncrewed aerial vehicles (UAVs) are increasingly deployed for autonomous navigation in complex outdoor environments, where dynamic conditions and mission requirements require intelligent adaptive decision-making. Existing optimization-based, Machine Learning (ML), and Reinforcement Learning (RL) approaches often rely on predefined models or task-specific training, limiting their generalization and adaptability in uncertain scenarios. Recent Large Language Model (LLM)-assisted approaches offer promising reasoning capabilities but remain constrained by limited agentic functionality, including insufficient memory, planning, and tool interaction mechanisms.This paper proposes an LLM-Agent-Based Path Finder (LAPF) framework for autonomous UAV navigation in town-scale outdoor environments. LAPF extends LLM-assisted navigation by integrating perception, memory, planning, and action modules into a closed-loop cognitive architecture. The proposed agent leverages prior navigation experiences, performs Chain-of-Thought (CoT) reasoning, couples each detected hazard to a bounded corrective action, and dynamically refines waypoint decisions based on environmental feedback.The three independent trials per method demonstrate that LAPF achieves mean path lengths of 512.83 m and 506.37 m, compared to the straight-line optimum of 497.33 m, corresponding to path length reductions of 17.2% and 15.6% relative to CoT prompting and absolute path efficiencies of 97.1% and 98.1% in open-field and obstacle-injected scenarios, respectively. Furthermore, LAPF is the only evaluated approach that couples every detected hazard to a bounded, metric-neutral corrective action while maintaining near-goal stability, with zero clamp events in both scenarios, whereas CoT prompting increases from 9.7 to 14.0 events.",
  "published": "2026-08-15",
  "updated": "2026-08-15",
  "year": "2026",
  "authors": [
   "Yousef Emami",
   "Mohammadhossein Homaei",
   "Hao Zhou",
   "Miguel Guti\u00e9rrez Gait\u00e1n",
   "Atefeh Hajijamali Arani",
   "Rui Zhang"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "LAPF is the only evaluated approach that couples every detected hazard to a bounded, metric-neutral corrective action while maintaining near-goal stability, with zero clamp events in both scenarios, whereas CoT prompting increases from 9.7 to 14.0 events.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yousef Emami",
    "id": "75031832",
    "h_index": 9,
    "papers": 27
   },
   {
    "name": "MohammadHossein Homaei",
    "id": "2264994466",
    "h_index": 7,
    "papers": 37
   },
   {
    "name": "Hao Zhou",
    "id": "2340385252",
    "h_index": 5,
    "papers": 16
   },
   {
    "name": "M. Gait\u00e1n",
    "id": "89377424",
    "h_index": 5,
    "papers": 27
   },
   {
    "name": "Atefeh Hajijamali Arani",
    "id": "3365415",
    "h_index": 9,
    "papers": 27
   },
   {
    "name": "Rui Zhang",
    "id": "2455642336",
    "h_index": 0,
    "papers": 1
   }
  ],
  "comment": "15 pages",
  "topics": [
   "rl-control",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.15175v1",
  "pdf_url": "https://arxiv.org/pdf/2608.15175v1",
  "html_url": "https://arxiv.org/html/2608.15175v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.15156",
  "slug": "low-rank-dynamics-effective-latent-carriers-for-counterfactual-rollout",
  "title": "Low-Rank Dynamics-Effective Latent Carriers for Counterfactual Rollout in Learned World Models",
  "abstract": "World models may predict the future without making clear which parts of their hidden state actually drive those predictions. We ask whether a small, directly addressable hidden-state change can place a learned world model on the intended counterfactual trajectory and then let the model continue that future on its own. We study a recurrent world model with a 192-dimensional hidden state in a controlled two-object, two-dimensional collision environment. For a bounded family of local velocity edits, we first verify that the model can natively represent and roll out the edited future. We then construct candidate low-rank carriers from training-only factual-to-counterfactual hidden differences and learn a map from the factual state and requested edit to carrier coefficients. On the registered rank grid, rank 4 is the smallest tested rank that satisfies the full development-panel criteria. A single rank-4 patch at the anchor is sufficient to redirect a 12-step autonomous rollout, with no future observations, teacher forcing, or repeated correction. The frozen procedure satisfies the preregistered replication rule across independently trained checkpoints and remains usable across nearby intervention times. Random equal-norm, wrong-object, and wrong-time controls do not explain the effect. A position-edit stress test provides a negative contrast: the intended position patch can pass the raw rollout criteria, but no-patch and random controls can pass the same criteria, and wrong-object specificity is not established. Thus, successful editing alone is not enough. We use dynamics-effective to describe an intervention that changes the model's future computation in a sustained and target-specific way under autonomous rollout. The rank-4 result identifies a compact intervention interface for the tested velocity-edit family, not a closed four-dimensional state or an intrinsic state dimension.",
  "published": "2026-08-15",
  "updated": "2026-08-19",
  "year": "2026",
  "authors": [
   "Yang Liu",
   "Yuming Chen"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Dynamics-effective is used to describe an intervention that changes the model's future computation in a sustained and target-specific way under autonomous rollout under autonomous rollout.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yang Liu",
    "id": "2445687641",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Yuming Chen",
    "id": "2109307440",
    "h_index": 5,
    "papers": 20
   }
  ],
  "comment": "Revised version: removed an inconclusive development-only event-relative phase analysis; the main rank-4 carrier, fresh-checkpoint replication, B1/B2 temporal reuse, position-edit, and joint-edit conclusions are unchanged",
  "topics": [
   "world-models"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.15156v2",
  "pdf_url": "https://arxiv.org/pdf/2608.15156v2",
  "html_url": "https://arxiv.org/html/2608.15156v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.15139",
  "slug": "structrl-structured-action-space-exploration-for-flow-based-vlas",
  "title": "StructRL: Structured Action-Space Exploration for Flow-Based VLAs",
  "abstract": "Flow-based Vision-Language-Action (VLA) models are now widely used for continuous robotic manipulation, and online reinforcement learning (RL) is emerging as a key technique for adapting them to new tasks. Existing RL methods typically inject stochasticity inside the denoising chain, often through isotropic or temporally independent noise. However, effective robot exploration calls for structured noise: temporally smooth and scaled differently across action groups. We show that simply switching the in-chain noise to a structured form does not suffice: noise added at an intermediate flow time can be weakened by the remaining denoising steps before execution, a phenomenon we call \\emph{Structured Noise Dilution}. We propose \\textbf{StructRL}, which avoids dilution by relocating policy stochasticity to the action space via three coupled choices: (i) a deterministic ODE decoder, (ii) structured noise injected directly in the action space, and (iii) last-step replay, where policy-gradient updates avoid assigning likelihoods to intermediate denoising states. This keeps structured exploration tied to the executed action while providing a tractable training signal for the flow decoder. Across three flow-based VLA models on multiple simulated manipulation benchmarks and two real-world tasks, StructRL improves exploration efficiency and OOD performance over prior in-chain baselines, demonstrating the effectiveness of structured action-space exploration for adapting flow-based VLA with RL. \\textbf{Project page:} https://flyfaerss.github.io/structrl/",
  "published": "2026-08-15",
  "updated": "2026-08-15",
  "year": "2026",
  "authors": [
   "Jiarui Yang",
   "Bin Zhu",
   "Jingjing Chen",
   "Na Zou",
   "Yanwei Fu",
   "Jianggang Zhu",
   "Yu-Gang Jiang"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Across three flow-based VLA models on multiple simulated manipulation benchmarks and two real-world tasks, StructRL improves exploration efficiency and OOD performance over prior in-chain baselines, demonstrating the effectiveness of structured action-space exploration for adapting flow-based VLA with RL.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jiarui Yang",
    "id": "2327698705",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Bin Zhu",
    "id": "2289341222",
    "h_index": 6,
    "papers": 23
   },
   {
    "name": "Jingjing Chen",
    "id": "2108536365",
    "h_index": 37,
    "papers": 168
   },
   {
    "name": "Nanpeng Zou",
    "id": "2440000249",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yanwei Fu",
    "id": "2405950194",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Jianggang Zhu",
    "id": "2173635030",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Yu-Gang Jiang",
    "id": "2345304992",
    "h_index": 27,
    "papers": 76
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "rl-control",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.15139v1",
  "pdf_url": "https://arxiv.org/pdf/2608.15139v1",
  "html_url": "https://arxiv.org/html/2608.15139v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.15088",
  "slug": "max-q-selective-imitation-for-human-in-the-loop-online-robot-learning",
  "title": "Max-Q Selective Imitation for Human-in-the-Loop Online Robot Learning",
  "abstract": "Human-in-the-loop (HIL) online reinforcement learning for real robots must absorb human interventions quickly while continuing to improve beyond the human prior. We present a training method for this setting based on two components. First, an \\emph{MC Q-chunk} critic regresses chunk-level action values onto Monte Carlo returns from the replay buffer, performing sample-average (behavior) policy evaluation so that intervention trajectories are credited directly rather than diluted by current-policy TD backups. Second, \\emph{max-Q selective imitation} updates the actor by imitating, at each state, the higher-$Q$ action between the current policy action and a buffer sample under a hard winner-take-all rule. This rule automatically switches between learning from interventions and on-policy self-improvement: when the autonomous policy is stronger, targets align with the policy distribution, reducing the policy--target-sample gap that otherwise induces execution-time distribution shift. In practice we score candidates with a standard critic ensemble mean to reduce comparison noise, without softening targets or introducing score-gap thresholds. On a real USB pick-and-insertion task with 20 demonstrations, ACT QChunk-MCBC attains 99\\% success within 30 minutes of HIL training, whereas HIL-SERL requires about 5 hours to converge. In simulation on Peg Insertion and Square, ACT/Flow Q-chunk variants similarly reach $\\ge$96\\% success within roughly half an hour of effective training, outperforming HIL-SERL, EXPO, and E2HiL on the success--time frontier.",
  "published": "2026-08-15",
  "updated": "2026-08-15",
  "year": "2026",
  "authors": [
   "Zihang Wang",
   "Yishan Wang"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A training method for HIL online reinforcement learning for real robots that automatically switches between learning from interventions and on-policy self-improvement, reducing the policy--target-sample gap that otherwise induces execution-time distribution shift.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zihang Wang",
    "id": "2334215224",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Yishan Wang",
    "id": "2458610350",
    "h_index": 0,
    "papers": 0
   }
  ],
  "comment": "",
  "topics": [
   "rl-control",
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.15088v1",
  "pdf_url": "https://arxiv.org/pdf/2608.15088v1",
  "html_url": "https://arxiv.org/html/2608.15088v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.15060",
  "slug": "egotac-in-the-wild-tactile-prediction-from-egocentric-vision",
  "title": "EgoTac: In-the-wild Tactile Prediction from Egocentric Vision",
  "abstract": "Touch is fundamental to dexterous manipulation, yet most egocentric human data increasingly used for robot learning lacks tactile information. Directly collecting large-scale tactile data is challenging due to sensor limitations, while human video data is abundant, contact-rich, and easily scalable. This motivates a natural question: can tactile signals be inferred purely from vision? To address this, we introduce EgoTac, a generalizable model that predicts rich tactile information directly from egocentric human videos. EgoTac is trained on a unified corpus of over 5.7M image-tactile pairs, covering both continuous force measurements and binary contacts. By learning from this diverse dataset, EgoTac captures nuanced touch dynamics across varied interactions. Experiments demonstrate strong performance: in-domain prediction achieves an average force error below 0.06N. On out-of-domain contact prediction benchmarks, EgoTac consistently outperforms the state-of-the-art contact estimator. It also captures the rise and fall patterns of real tactile data and enables zero-shot predictions on unconstrained real-world videos. Scaling analyses further reveal that both data diversity and volume improve performance steadily. Overall, EgoTac provides a scalable pathway to extract tactile priors from egocentric human videos, enabling broadly applicable tactile-aware robot learning.",
  "published": "2026-08-15",
  "updated": "2026-08-15",
  "year": "2026",
  "authors": [
   "Wenkang Zhang",
   "Chengbo Yuan",
   "Zicheng Zhang",
   "Zhengxue Cheng",
   "Yang Gao"
  ],
  "author_count": 5,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Overall, EgoTac provides a scalable pathway to extract tactile priors from egocentric human videos, enabling broadly applicable tactile-aware robot learning.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Wenkang Zhang",
    "id": "2359302514",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Chengbo Yuan",
    "id": "2280187985",
    "h_index": 7,
    "papers": 12
   },
   {
    "name": "Zicheng Zhang",
    "id": "2344793839",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Zhengxue Cheng",
    "id": "2310296090",
    "h_index": 9,
    "papers": 63
   },
   {
    "name": "Yang Gao",
    "id": "2330756947",
    "h_index": 4,
    "papers": 7
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "egocentric-data",
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.15060v1",
  "pdf_url": "https://arxiv.org/pdf/2608.15060v1",
  "html_url": "https://arxiv.org/html/2608.15060v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.15026",
  "slug": "pace-phase-progress-aware-credit-for-long-horizon-embodied-manipulatio",
  "title": "PACE: Phase-Progress-Aware Credit for Long-Horizon Embodied Manipulation",
  "abstract": "Post-training of vision-language-action (VLA) models typically relies on expert demonstrations and policy interaction trajectories. However, in long-horizon manipulation, a single episode often spans hundreds of control steps and multiple phases, while success or failure is only revealed at episode termination. Policy improvement therefore requires step-level credit signals to distinguish behaviors that advance the task from those that stall or regress. We present PACE, a credit-assignment framework for post-training on long-horizon manipulation, centered on a phase-progress-aware critic. PACE consists of two key modules: (1) the Global-Local Cooperative Value-Correction Critic (GLC-Critic) aggregates visual and motion-difference features within local temporal windows to infer the phase and intra-phase progress of each step, and applies residual correction to a discretized remaining-cost distribution accordingly, enabling step-level credit assignment; (2) Progressive Policy Distillation (PPD) converts credit into positive and negative conditions via task-wise thresholds and trains a credit-conditioned action generation policy: it first protects the pretrained policy with high-credit positive samples, then incorporates all positive and negative credits to learn the quality boundary, and at inference amplifies high-credit behaviors through the difference between conditional outputs. Extensive simulation experiments and diverse real-world robotic-arm experiments demonstrate that PACE consistently achieves significant improvements over the strongest baseline.",
  "published": "2026-08-15",
  "updated": "2026-08-15",
  "year": "2026",
  "authors": [
   "Chengye Song",
   "Jiawei Zhang",
   "Rui Song",
   "Shengqi Wang",
   "Xiangrong Zhang",
   "Ziyi Wang",
   "Huanbin Zhou",
   "Hongzhou Wang"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "PACE, a credit-assignment framework for post-training on long-horizon manipulation, centered on a phase-progress-aware critic, is presented, demonstrating that PACE consistently achieves significant improvements over the strongest baseline.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Chengye Song",
    "id": "2361657367",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Jiawei Zhang",
    "id": "2457343235",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Ruike Song",
    "id": "2456859750",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Shengqi Wang",
    "id": "2458128379",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Xiangrong Zhang",
    "id": "2458629016",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Ziyi Wang",
    "id": "2446079759",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Huanbin Zhou",
    "id": "2458124444",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Hongzhou Wang",
    "id": "2458701157",
    "h_index": 0,
    "papers": 0
   }
  ],
  "comment": "9 pages, 6 figures",
  "topics": [
   "vla"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.15026v1",
  "pdf_url": "https://arxiv.org/pdf/2608.15026v1",
  "html_url": "https://arxiv.org/html/2608.15026v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.15024",
  "slug": "motiongs-slam-event-modulated-gaussian-splatting-for-motion-blur-robus",
  "title": "MotionGS-SLAM: Event-Modulated Gaussian Splatting for Motion-Blur Robust SLAM",
  "abstract": "Current Vision-based SLAM systems fail catastrophically when motion blur corrupts the visual input, as they attempt the ill-posed inverse problem of recovering sharp content from degraded observations. We present MotionGS-SLAM, which fundamentally reimagines motion blur handling through a paradigm shift: rather than removing blur artifacts, we reformulate the challenge as a well-constrained forward problem that generatively models blur formation within the rendering pipeline. By leveraging event cameras' microsecond temporal resolution and immunity to motion blur, we introduce a novel event-modulated Gaussian kernel that dynamically adapts each Gaussian's rasterization based on precise motion cues. Our dual-modulation mechanism transforms 2D Gaussian projections from isotropic dots into anisotropic, motion-aligned elliptical brush strokes (spatial modulation) while adaptively varying exposure integral sampling density based on local velocity (temporal modulation). This physics-based approach enables joint optimization of intra-exposure camera trajectories and 3D scene geometry through blur-aware photometric and event-based constraints. Extensive experiments demonstrate significant improvements over state-of-the-art methods in trajectory accuracy and map quality under severe high-motion conditions.",
  "published": "2026-08-15",
  "updated": "2026-08-15",
  "year": "2026",
  "authors": [
   "Zhiqiang Hu",
   "Shouren Huang",
   "Masatoshi Ishikawa"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work fundamentally reimagines motion blur handling through a paradigm shift: rather than removing blur artifacts, the challenge is reformulate the challenge as a well-constrained forward problem that generatively models blur formation within the rendering pipeline.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhiqiang Hu",
    "id": "2239440449",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Shouren Huang",
    "id": "2205923626",
    "h_index": 10,
    "papers": 55
   },
   {
    "name": "Masatoshi Ishikawa",
    "id": "2280688574",
    "h_index": 4,
    "papers": 20
   }
  ],
  "comment": "8 pages, 5 figures. Published in the 2026 IEEE International Conference on Robotics and Automation",
  "topics": [
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.15024v1",
  "pdf_url": "https://arxiv.org/pdf/2608.15024v1",
  "html_url": "https://arxiv.org/html/2608.15024v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.15009",
  "slug": "forceu-vla-a-force-aware-vision-language-action-model-for-embodied-ult",
  "title": "ForceU-VLA: A Force-Aware Vision-Language-Action Model for Embodied Ultrasound Scanning",
  "abstract": "Embodied intelligent ultrasound scanning enables the automation and standardization of the ultrasound examination process by integrating perception, decision-making, and execution capabilities. However, existing methods suffer from loosely coupled modeling between force and ultrasound modalities and lack awareness of scanning stages, which limits their ability to capture dynamic probe-tissue interactions. To address these issues, we propose ForceU-VLA, a force-aware Vision-Language-Action model for autonomous embodied ultrasound scanning, which leverages force signals and ultrasound image feedback throughout the scanning process to enable accurate and high-quality ultrasound acquisition. Firstly, we propose a Force-Ultrasound Synergistic Fusion Module (FUSFM) that synergistically fuses ultrasound visual and force-feedback information to provide stable, reliable guidance for probe motion. Secondly, a Stage-Adaptive Modulation Mechanism (SAMM) is proposed to accommodate the task requirements across different scanning stages by adaptively modulating multimodal features to enhance their representation quality. Additionally, we introduce ForceU-VLA-Data, a real-world, force-aware embodied ultrasound dataset that integrates visual, force, and action signals, including data from two organs across five representative clinical scanning views, and comprising 450 expert-collected trajectories with approximately 100,000 synchronized multimodal frames. Extensive experimental results demonstrate that ForceU-VLA significantly improves contact stability and probe pressure regulation in embodied ultrasound scanning, thereby effectively enhancing task execution quality and overall system reliability. The source code is available at https://github.com/VMVLab/ForceU-VLA.",
  "published": "2026-08-15",
  "updated": "2026-08-15",
  "year": "2026",
  "authors": [
   "Xingzheng Wu",
   "Cheng Zhang",
   "Guihao Yan",
   "Xifeng Hu",
   "Zhi Liu",
   "Qing Cai"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "ForceU-VLA is proposed, a force-aware Vision-Language-Action model for autonomous embodied ultrasound scanning, which leverages force signals and ultrasound image feedback throughout the scanning process to enable accurate and high-quality ultrasound acquisition.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xingzhen Wu",
    "id": "2351727363",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Cheng Zhang",
    "id": "2404674401",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Guihao Yan",
    "id": "2395714066",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Xifeng Hu",
    "id": "2333156418",
    "h_index": 2,
    "papers": 11
   },
   {
    "name": "Zhi Liu",
    "id": "2333520867",
    "h_index": 3,
    "papers": 14
   },
   {
    "name": "Qing Cai",
    "id": "2339477781",
    "h_index": 1,
    "papers": 4
   }
  ],
  "comment": "",
  "topics": [
   "vla"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.15009v1",
  "pdf_url": "https://arxiv.org/pdf/2608.15009v1",
  "html_url": "https://arxiv.org/html/2608.15009v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.15002",
  "slug": "npu-offloading-of-a-frozen-visual-encoder-for-robot-policy-training",
  "title": "NPU Offloading of a Frozen Visual Encoder for Robot Policy Training",
  "abstract": "When a robot policy is trained for a new task or dataset, its visual encoder can be frozen and only its action generation module trained, reducing training cost. Freezing removes the encoder's backward pass, but its forward pass must still run at every training step because the input images change, so it keeps consuming GPU compute. We therefore ask whether moving this computation to a low power AI accelerator such as an NPU can reduce total energy despite the added data transfer and longer training time, and how it affects policy performance. We built an asynchronous training pipeline that uses both a GPU and an NPU for the AR-Actor specialist. The frozen visual encoder runs in A8W8 INT8 on a Mobilint Aries2 NPU, while the FP32 action expert is trained on an NVIDIA GeForce RTX 5060 Ti GPU. We compared a GPU-only baseline with four conditions, L1 to L4, which gradually extend NPU offloading from one to four Transformer encoder layers. Each condition was trained for 30,000 steps with three random seeds. We measured GPU board power for the GPU-only condition and combined GPU and NPU board power for the NPU conditions. Energy per sample decreased by 17.1% in L1, which offloaded ResNet18 and the first encoder layer, and by 27.9% in L4, which offloaded ResNet18 and all four encoder layers. In contrast, training time per sample increased by 15.2% in L1 and 37.7% in L4, and peak allocated GPU memory decreased by 19.8 to 20.7%. The 15 resulting policies were each evaluated with the same 300 environment seeds, for a total of 4,500 simulator rollouts. The combined success rate was 93.33% for GPU-only and 91.44 to 92.89% for the NPU conditions. These results show that NPU offloading of a frozen visual encoder can reduce training energy, but it increases training time and lowers policy success rate by 0.44 to 1.89 percentage points compared with GPU-only training.",
  "published": "2026-08-15",
  "updated": "2026-08-15",
  "year": "2026",
  "authors": [
   "Hyojun Yun",
   "Seungjae Won",
   "Hyungpil Moon"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.AR",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "NPU offloading of a frozen visual encoder can reduce training energy, but it increases training time and lowers policy success rate by 0.44 to 1.89 percentage points compared with GPU-only training.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hyojun Yun",
    "id": "2401897988",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Seungjae Won",
    "id": "2334641178",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Hyungpil Moon",
    "id": "2279601496",
    "h_index": 6,
    "papers": 28
   }
  ],
  "comment": "6 pages, 4 figures, 4 tables",
  "topics": [
   "sim2real",
   "data-teleop"
  ],
  "orgs": [
   "NVIDIA"
  ],
  "abs_url": "https://arxiv.org/abs/2608.15002v1",
  "pdf_url": "https://arxiv.org/pdf/2608.15002v1",
  "html_url": "https://arxiv.org/html/2608.15002v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2608.14996",
  "slug": "hp2-slam-adaptive-hybrid-icp-for-robust-and-efficient-lidar-slam",
  "title": "HP2-SLAM: Adaptive Hybrid ICP for Robust and Efficient LiDAR SLAM",
  "abstract": "Achieving robustness, accuracy, and efficiency simultaneously remains a central challenge in light detection and ranging (LiDAR) simultaneous localization and mapping (SLAM). While learning-based approaches deliver strong benchmark performance, they often require extensive training, substantial computational resources, and struggle to generalize to unseen or degenerate environments. Geometry-based methods are efficient and interpretable, yet their performance degrades in planar or repetitive scenes due to limitations of standard iterative closest point (ICP) formulations. We present HP2-SLAM, a minimalist yet robust LiDAR SLAM framework built around a neighborhood-size adaptive hybrid ICP. Our key insight is a planarity-aware adaptive threshold that dynamically classifies correspondences based on local geometric structure and density, thereby enabling a principled balance between point-to-plane and point-to-point residuals. This formulation stabilizes alignment in both structured and degenerate environments without feature engineering, learning modules, or dataset-specific tuning. Integrated into a complete SLAM pipeline with submap management, loop closure detection, and pose graph optimization, HP2-SLAM consistently outperforms strong geometry-based baselines across publicly available datasets while maintaining real-time performance on commodity hardware. Our results demonstrate that carefully designed geometric adaptation can achieve strong generalization and robustness without sacrificing simplicity or efficiency.",
  "published": "2026-08-15",
  "updated": "2026-08-15",
  "year": "2026",
  "authors": [
   "Nam Tran",
   "Thu Tran",
   "Hieu Phan",
   "Thai Luu",
   "Toan Nguyen",
   "William J. Beksi",
   "Tuan Dang"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "HBP2-SLAM is presented, a minimalist yet robust LiDAR SLAM framework built around a neighborhood-size adaptive hybrid ICP that dynamically classifies correspondences based on local geometric structure and density, thereby enabling a principled balance between point-to-plane and point-to-point residuals.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "N. Tr\u1ea7n",
    "id": "2367911156",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "T. Tran",
    "id": "2454695701",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Hieu Phan",
    "id": "2233584764",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Thai Luu",
    "id": "2458034239",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Toan Nguyen",
    "id": "2458118641",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "William J. Beksi",
    "id": "2552647",
    "h_index": 12,
    "papers": 37
   },
   {
    "name": "T. Dang",
    "id": "2284680630",
    "h_index": 3,
    "papers": 12
   }
  ],
  "comment": "",
  "topics": [
   "spatial-3d",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.14996v1",
  "pdf_url": "https://arxiv.org/pdf/2608.14996v1",
  "html_url": "https://arxiv.org/html/2608.14996v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.14986",
  "slug": "gaussmemory-task-driven-3d-gaussian-scene-memory-for-long-horizon-robo",
  "title": "GaussMemory: Task-Driven 3D Gaussian Scene Memory for Long-Horizon Robotic Manipulation",
  "abstract": "Long-horizon robotic manipulation fundamentally relies on persistent spatial memory. However, existing 3D memory systems function merely as passive recorders: they store observations using fixed, hand-crafted rules, treating every scene element--whether a critical grasp target or an irrelevant background wall--with equal importance. In this paper, we propose a paradigm shift from passive storage to active, task-driven spatial memory. We argue that a robot's memory should not simply record what it sees, but actively learn how to remember--discovering which objects to track precisely, how aggressively to update them, and what to discard, all learned end-to-end without hand-designed rules. Crucially, this active paradigm is realized by unifying memory update and readout as two sides of the same cognitive process, enabling bidirectional flow where task needs shape update strategies and vice versa. To instantiate this vision, we introduce GaussMemory, which leverages 3D Gaussian Splatting as a persistent geometric substrate. On LIBERO, GaussMemory outperforms MemoryVLA on Goal and Long-10; on VLABench, it surpasses $\u03c0_0$-FAST by +5.2% (Track 1) and +6.0% (Track 6).",
  "published": "2026-08-15",
  "updated": "2026-08-15",
  "year": "2026",
  "authors": [
   "Zhiqiang Hu",
   "Shouren Huang",
   "Masatoshi Ishikawa"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "It is argued that a robot's memory should not simply record what it sees, but actively learn how to remember--discovering which objects to track precisely, how aggressively to update them, and what to discard, all learned end-to-end without hand-designed rules.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhiqiang Hu",
    "id": "2239440449",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Shouren Huang",
    "id": "2205923626",
    "h_index": 10,
    "papers": 55
   },
   {
    "name": "Masatoshi Ishikawa",
    "id": "2280688574",
    "h_index": 4,
    "papers": 20
   }
  ],
  "comment": "8 pages, 10 figures. Accepted to the 2026 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS 2026)",
  "topics": [
   "vla",
   "dexterous-manipulation",
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.14986v1",
  "pdf_url": "https://arxiv.org/pdf/2608.14986v1",
  "html_url": "https://arxiv.org/html/2608.14986v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.14952",
  "slug": "evidence-of-absence-cross-modal-abductive-risk-perception-to-sustain-w",
  "title": "Evidence of Absence: Cross-Modal Abductive Risk Perception to Sustain World Models When Vision Fails",
  "abstract": "A structured world-state (entities, relations, context, and predictive cues) is designed to preserve prediction-critical content when perception degrades, but it presumes observations to populate it; when the primary visual modality is occluded or degraded, those observations may be missing. We address how to sustain the world model from a complementary modality by treating the absence of expected co-evidence as evidence of a hidden cause. The abductive framework is modality-agnostic; this article instantiates it acoustically. A microphone-array front-end estimates the bearing of engine and tire sources and extracts approach-rate evidence (Doppler when a stable tone exists, a broadband looming readout otherwise); the event \"signature present, visual co-evidence absent\" then triggers abductive inference of a hidden road user, emitting a calibrated risk advisory rather than a control command. Recoverability of the hidden state is analyzed as an identifiability question separating shared from modality-unique information, and cueing is cast as Neyman-Pearson detection under an explicit false-alarm budget. On real occluded-approach recordings at blind junctions, the method warns a mean 1.7 seconds before line-of-sight entry, matches the sustained-window variant of the published acoustic baseline's detection rate with 42% fewer false alarms, localizes to 3.4 degrees median once in view, is well calibrated (expected calibration error 0.034), and keeps hazard awareness above 0.87 under staged vision degradation that collapses a vision-only channel to 0.03. We also measure the method's limits: calibration transfers to an unseen junction almost losslessly, the signature classifier does not, and moving-ego noise is the binding deployment constraint.",
  "published": "2026-08-15",
  "updated": "2026-08-15",
  "year": "2026",
  "authors": [
   "Cong Xu",
   "Ravi Sankar"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO",
   "cs.CV",
   "eess.SP"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This article addresses how to sustain the world model from a complementary modality by treating the absence of expected co-evidence as evidence of a hidden cause, and instantiates the abductive framework acoustically.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Cong Xu",
    "id": "2128141404",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Ravi Sankar",
    "id": "2265411809",
    "h_index": 4,
    "papers": 10
   }
  ],
  "comment": "7 pages, 3 figures. Working draft prepared for journal submission",
  "topics": [
   "world-models",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.14952v1",
  "pdf_url": "https://arxiv.org/pdf/2608.14952v1",
  "html_url": "https://arxiv.org/html/2608.14952v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.14944",
  "slug": "skillcomposer-learning-reusable-skills-for-natural-language-robot-prog",
  "title": "SkillComposer: Learning Reusable Skills for Natural-Language Robot Programming",
  "abstract": "Natural-language interfaces can lower the barrier to programming robots, but existing systems struggle when users request complex tasks. While large language models (LLMs) perform well with simple commands, they often struggle to generate code for multi-step tasks, decompose high-level instructions, or reuse prior solutions. We present SkillComposer, an interactive natural-language robot programming system for simulation environments that continually learns reusable program abstractions. SkillComposer uses a generate-test architecture in which an LLM iteratively generates and revises robot programs before execution. Successful programs are stored and processed by an online library-learning algorithm that compresses recurring function sequences into reusable macro skills for future tasks. We evaluate SkillComposer through ablation experiments and a user study with 12 participants to determine its effectiveness on manipulation and robot caregiving tasks. The results show that evaluator-guided generation and learned abstractions improve success rates and usability while reducing user effort in natural-language robot programming.",
  "published": "2026-08-14",
  "updated": "2026-08-14",
  "year": "2026",
  "authors": [
   "John Woods",
   "Hasti Seifi"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO",
   "cs.CL",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "Humanoids 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The results show that evaluator-guided generation and learned abstractions improve success rates and usability while reducing user effort in natural-language robot programming.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "J. Woods",
    "id": "2457486269",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Hasti Seifi",
    "id": "35750547",
    "h_index": 18,
    "papers": 96
   }
  ],
  "comment": "8 pages, 6 figures. Submitted to IEEE Humanoids 2026",
  "topics": [
   "humanoids"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.14944v1",
  "pdf_url": "https://arxiv.org/pdf/2608.14944v1",
  "html_url": "https://arxiv.org/html/2608.14944v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.14937",
  "slug": "from-continuous-design-to-delay-aware-discrete-synthesis-guaranteed-hi",
  "title": "From Continuous Design to Delay-Aware Discrete Synthesis: Guaranteed High-Bandwidth Joint Control for PMSM Drives",
  "abstract": "The increasing dynamic demands of modern robotic joints require current controllers to achieve high bandwidth over wide operating ranges of speed, acceleration, and torque, where communication, computation, and discrete-time effects can no longer be neglected. Conventional PMSM current controllers are typically designed in continuous time and subsequently discretized, leaving the sampling frequency and the impact of implementation delays largely to heuristic selection and iterative validation. This paper introduces a task-aware, delay-extended discrete-time joint model that explicitly accounts for physical communication and computation delays and enables direct synthesis of a discrete PI current controller with prescribed bandwidth and delay guarantees throughout the operating envelope. The framework analytically determines the minimum required sampling frequency, controller gains, and DC-link voltage needed to satisfy the specified motor and joint performance. Simulations across a range of dynamic requirements validate the methodology and demonstrate substantially reduced sampling-frequency and DC-link-voltage requirements compared with conventional continuous-time-based design. Experiments on a newly developed custom robotic joint further validate the proposed framework under real embedded implementation conditions.",
  "published": "2026-08-14",
  "updated": "2026-08-14",
  "year": "2026",
  "authors": [
   "Edmundo Pozo Fortuni\u0107",
   "Mehmet C. Yildirim",
   "Sami Haddadin"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A task-aware, delay-extended discrete-time joint model that explicitly accounts for physical communication and computation delays and enables direct synthesis of a discrete PI current controller with prescribed bandwidth and delay guarantees throughout the operating envelope is introduced.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Edmundo Pozo Fortuni\u0107",
    "id": "31805863",
    "h_index": 6,
    "papers": 16
   },
   {
    "name": "M. C. Yildirim",
    "id": "2240521421",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Sami Haddadin",
    "id": "2378835341",
    "h_index": 1,
    "papers": 22
   }
  ],
  "comment": "9 pages, 3 figures",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.14937v1",
  "pdf_url": "https://arxiv.org/pdf/2608.14937v1",
  "html_url": "https://arxiv.org/html/2608.14937v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.14902",
  "slug": "geometry-aware-online-mapping-for-3d-gaussian-splatting-slam",
  "title": "Geometry-Aware Online Mapping for 3D Gaussian Splatting SLAM",
  "abstract": "Recent 3D Gaussian Splatting (3DGS) has enabled efficient photorealistic view synthesis and is rapidly being adopted in simultaneous localization and mapping (SLAM) systems for online mapping. In these systems, a Gaussian map must be expanded and refined incrementally while tracking runs in real time, so initialization and density control directly determine where limited computation and iterations are spent. This contrasts with offline 3DGS reconstruction, where such heuristics can be amortized over long optimization schedules. However, most 3DGS-SLAM pipelines inherit initialization and density-control heuristics from offline reconstruction, which can become brittle under the strict per-keyframe optimization budgets and incremental map growth of online SLAM. In this work, we revisit these heuristics in a decoupled 3DGS-SLAM setting and propose three geometry-aware methods that operate in the mapping thread: transmittance-preserving densification, camera-aware scale initialization from depth and intrinsics, and error-guided densification that focuses new primitives on high-residual regions. Our results show consistent improvements in rendering quality with negligible overhead, highlighting the coupling between photometric residuals and pose uncertainty in online SLAM. We will open-source our code to the community to foster growth and validate reproducibility.",
  "published": "2026-08-14",
  "updated": "2026-08-14",
  "year": "2026",
  "authors": [
   "Thai Luu",
   "Quan Tran",
   "Hieu Phan",
   "Tuan Dang"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work revisits 3D Gaussian Splatting heuristics in a decoupled 3DGS-SLAM setting and proposes three geometry-aware methods that operate in the mapping thread: transmittance-preserving densification, camera-aware scale initialization from depth and intrinsics, and error-guided densification that focuses new primitives on high-residual regions.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Thai Luu",
    "id": "2458034239",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Quan Tran",
    "id": "2397352386",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Hieu Phan",
    "id": "2233584764",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "T. Dang",
    "id": "2284680630",
    "h_index": 3,
    "papers": 12
   }
  ],
  "comment": "",
  "topics": [
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.14902v1",
  "pdf_url": "https://arxiv.org/pdf/2608.14902v1",
  "html_url": "https://arxiv.org/html/2608.14902v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.14868",
  "slug": "beam-wise-statistical-background-subtraction-for-static-roadside-lidar",
  "title": "Beam-Wise Statistical Background Subtraction for Static Roadside LiDAR: A Cross-Sensor Benchmark Study",
  "abstract": "Background subtraction is a key preprocessing step for infrastructure-based LiDAR perception, enabling efficient isolation of dynamic traffic participants without semantic annotations. However, systematic cross-sensor evaluations and reproducible studies for static roadside LiDAR are missing. This paper presents a comparative benchmark of beam-wise statistical background subtraction for statically mounted LiDAR sensors. We formulate background estimation as a per-beam temporal modeling problem and investigate complementary statistical strategies that capture dominant as well as multi-modal background structures, combined with spatial filtering in the angular and 3D domain. To enable reproducible evaluation, we introduce HighwayScene, a new multi-LiDAR dataset recorded in a static roadside setup, and extend the public CoopScenes dataset with static/dynamic point-wise annotations. Across multiple scenes and heterogeneous sensing technologies, we demonstrate that beam-wise statistical modeling provides a robust and transferable solution. Combining lightweight per-beam models with spatial consistency filtering substantially improves precision while maintaining high recall and real-time capability. All datasets, annotations, and implementations are publicly released.",
  "published": "2026-08-14",
  "updated": "2026-08-14",
  "year": "2026",
  "authors": [
   "Alexander Baumann",
   "Marcel Vosshans",
   "Thao Dang"
  ],
  "author_count": 3,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Alex Baumann",
    "id": "2325144274",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Marcel Vosshans",
    "id": "2122052442",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Thao Dang",
    "id": "2310699607",
    "h_index": 2,
    "papers": 3
   }
  ],
  "comment": "Accepted for publication at the 2026 IEEE 29th International Conference on Intelligent Transportation Systems (ITSC), Naples, Italy, September 15-18, 2026",
  "topics": [
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.14868v1",
  "pdf_url": "https://arxiv.org/pdf/2608.14868v1",
  "html_url": "https://arxiv.org/html/2608.14868v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.14865",
  "slug": "real-time-estimator-of-actuator-control-and-health-reach-on-an-eel-ins",
  "title": "Real-time Estimator of Actuator Control and Health (REACH) on an Eel-Inspired Soft Robot",
  "abstract": "An actuator health estimation algorithm for a soft swimming robot that can perform anguilliform swimming is developed. Due to harsh operational environments of underwater robots, and the common degradation of soft robot materials and actuators, accurate estimation of actuator functionality is necessary for robots to perform their missions as well as return to base in the event of actuator degradation and failure. Termed REACH (Real-time Estimator of Actuator Control and Health), the architecture employs a soft robot model, sigma point filter, and a formal statistical hypothesis test to adequately capture the nonlinearities and changes over time. The performance of REACH using three sensor types (GPS, IMU, and Bend Sensor) with one sensor on each actuator is compared, demonstrating that both bend sensor and IMU are adequate choices. Sensor quantity and placement are evaluated for IMU and bend sensor, showing two sensors are sufficient for IMU, whereas three sensors are needed for bend sensor. Three swimming gaits (linear swimming, wide turning, tight turning) are compared, demonstrating that REACH can successfully predict actuator health for all three gaits, with minimal differences in performance. A filter validation method shows the fault estimation algorithm is statistically consistent in finding the correct degradation. The approach is experimentally evaluated using bend sensor data collected from a fish robot, demonstrating that REACH can successfully estimate actuator health with noisy data and variations in manufacturing.",
  "published": "2026-08-14",
  "updated": "2026-08-14",
  "year": "2026",
  "authors": [
   "Zhangjingyi Jiang",
   "Myungsun Park",
   "Michael T. Tolley",
   "Mark Campbell"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "10.1109/robosoft63089.2025.11020927",
  "oa_pdf": "https://doi.org/10.48550/arxiv.2608.14865",
  "s2_authors": [
   {
    "name": "Zhangjingyi Jiang",
    "id": "2327294539",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Myungsun Park",
    "id": "2310775588",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "M. Tolley",
    "id": "48724645",
    "h_index": 41,
    "papers": 144
   },
   {
    "name": "Mark Campbell",
    "id": "2327249496",
    "h_index": 1,
    "papers": 4
   }
  ],
  "comment": "",
  "topics": [
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.14865v1",
  "pdf_url": "https://arxiv.org/pdf/2608.14865v1",
  "html_url": "https://arxiv.org/html/2608.14865v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.14860",
  "slug": "modeling-and-control-of-an-eel-inspired-soft-robot-for-design-optimiza",
  "title": "Modeling and Control of an Eel-Inspired Soft Robot for Design Optimization",
  "abstract": "Anguilliform locomotion is a highly efficient swimming mode; the advent of new materials for soft robots enables the development of an eel-inspired soft robot. This paper presents a simulation model of an eel-inspired soft robot designed for anguilliform swimming. This model can aid in design optimization and the development of model-based estimation, reasoning, and control systems. A Finite Element Method (FEM) model of an elastic rod is used to capture the soft materials of the robotic fish, which makes it particularly amenable to variation over time as the material properties change. The material model is coupled with a hydrodynamic force model to simulate the behavior of a soft, elongated robot in water. The model is used to demonstrate the effectiveness of the proposed control approaches in achieving desired swimming behaviors. It also provides insights into design decisions, including the robustness of different system configurations and the impact of material degradation and failure. The results show that slightly asymmetric designs are advantageous, offering comparable swimming velocities but greater maneuverability. This model can be used to guide future robotic design decisions aimed at optimizing performance for specific tasks.",
  "published": "2026-08-14",
  "updated": "2026-08-14",
  "year": "2026",
  "authors": [
   "Zhangjingyi Jiang",
   "Mark Campbell"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 2,
  "influential_citations": 0,
  "tldr": "",
  "doi": "10.1109/CASE59546.2024.10711519",
  "oa_pdf": "https://doi.org/10.48550/arxiv.2608.14860",
  "s2_authors": [
   {
    "name": "Zhangjingyi Jiang",
    "id": "2327294539",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Mark Campbell",
    "id": "2327249496",
    "h_index": 1,
    "papers": 4
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "hardware-codesign",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.14860v1",
  "pdf_url": "https://arxiv.org/pdf/2608.14860v1",
  "html_url": "https://arxiv.org/html/2608.14860v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.48
 },
 {
  "id": "2608.14822",
  "slug": "imagining-recovery-inference-time-counterfactual-realignment-for-visio",
  "title": "Imagining Recovery: Inference-Time Counterfactual Realignment for Vision-Language-Action Models",
  "abstract": "Vision-language-action (VLA) models have improved the flexibility and generality of robotic manipulation, yet they remain fragile to online disruptions, such as changes in task goal, scene configuration, or robot state. Existing recovery methods often require failure data, policy retraining, or external corrective agents, introducing additional data requirements and execution risks. We propose Counterfactual Realignment (CoRe), a training-free framework that recovers a frozen VLA at inference time without failure data. Upon detecting a deviation, CoRe imagines how the policy would continue toward the current goal from a recent viable state, using synthesized observations in place of physical execution, and then minimally realigns the robot and scene to rejoin this imagined continuation before returning control to the policy. Recovery is therefore planned without physical trial-and-error, preserves completed task progress, and handles both mid-episode instruction changes and physical perturbations in a unified manner. Extensive experiments across multiple simulators, VLA backbones, and real-world settings show that CoRe improves success rates by up to 85.0 percentage points to near-nominal levels while reducing physical restorations by 42.2%, without policy fine-tuning or failure-specific recovery training.",
  "published": "2026-08-14",
  "updated": "2026-08-14",
  "year": "2026",
  "authors": [
   "Yanyan Zhang",
   "Disheng Liu",
   "Kai Ye",
   "Chaoda Song",
   "Xinpeng Li",
   "Mohsen Hariri",
   "Vikash Singh",
   "Yu Yin",
   "Vipin Chaudhary"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Counterfactual Realignment (CoRe), a training-free framework that recovers a frozen VLA at inference time without failure data, is proposed, a training-free framework that recovers a frozen VLA at inference time without policy fine-tuning or failure-specific recovery training.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yanyan Zhang",
    "id": "2325267406",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Disheng Liu",
    "id": "2349335430",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Kai Ye",
    "id": "2335560881",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Chaoda Song",
    "id": "2354112874",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Xinpeng Li",
    "id": "2279666948",
    "h_index": 4,
    "papers": 24
   },
   {
    "name": "Mohsen Hariri",
    "id": "2315812444",
    "h_index": 6,
    "papers": 17
   },
   {
    "name": "Vikash Singh",
    "id": "2363724234",
    "h_index": 4,
    "papers": 15
   },
   {
    "name": "Yu Yin",
    "id": "2365399530",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Vipin Chaudhary",
    "id": "2346129602",
    "h_index": 6,
    "papers": 24
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "sim2real",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.14822v1",
  "pdf_url": "https://arxiv.org/pdf/2608.14822v1",
  "html_url": "https://arxiv.org/html/2608.14822v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.14772",
  "slug": "mistac-a-vision-based-tactile-sensor-for-minimally-invasive-surgery",
  "title": "MISTac: A Vision-Based Tactile Sensor for Minimally Invasive Surgery",
  "abstract": "Minimally invasive and robot-assisted surgery offer many advantages over traditional open surgery, but deprive surgeons of tactile feedback and the ability to palpate tissue with their fingers. To address this lack of tactile feedback, we introduce the MISTac, a high resolution vision-based tactile sensor specifically designed for palpation in MIS. The sensor has a replaceable sensor tip with a diameter of 8 mm which allows it to fit through the trocars used in minimally invasive surgery. Its modular 3D-printed case design allows the use of bulky off-the-shelf illumination and imaging hardware that can easily be exchanged and upgraded. The sensor has an optical resolution of 176.68 $\u03bcm$, a tactile resolution of 250 $\u03bcm$, and can resolve forces as little as 24.3 mN. An in vivo study with the sensor shows its usability in minimally invasive surgery. We trained a machine learning model with the tactile data collected in the trial on a tissue classification task achieving an aggregate accuracy of ~84% in a leave-one-out cross validation. Tactile sensors have the potential to one day aid surgeons during minimally invasive surgery with tasks such as tissue classification or intra-operative tumor localization; MISTac is a small step towards this vision. We open-source MISTac at https://github.com/lasr-lab/mistac",
  "published": "2026-08-14",
  "updated": "2026-08-14",
  "year": "2026",
  "authors": [
   "Robin Koch",
   "Annabella Mascot",
   "Rayan Younis",
   "Martin Wagner",
   "Stefanie Speidel",
   "Mark Cutkosky",
   "Ingo Sieber",
   "Roberto Calandra"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The MISTac is introduced, a high resolution vision-based tactile sensor specifically designed for palpation in MIS, and trained a machine learning model with the tactile data collected in the trial on a tissue classification task achieving an aggregate accuracy of ~84% in a leave-one-out cross validation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "R. Koch",
    "id": "2441020897",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Annabella Mascot",
    "id": "2167787978",
    "h_index": 0,
    "papers": 5
   },
   {
    "name": "R. Younis",
    "id": "2131862806",
    "h_index": 4,
    "papers": 24
   },
   {
    "name": "Martin Wagner",
    "id": "2319825458",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Stefanie Speidel",
    "id": "2290490227",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Mark R. Cutkosky",
    "id": "2329168577",
    "h_index": 4,
    "papers": 21
   },
   {
    "name": "Ingo Sieber",
    "id": "2359911398",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Roberto Calandra",
    "id": "2261268365",
    "h_index": 4,
    "papers": 10
   }
  ],
  "comment": "This work has been submitted to the IEEE for possible publication",
  "topics": [
   "tactile",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.14772v1",
  "pdf_url": "https://arxiv.org/pdf/2608.14772v1",
  "html_url": "https://arxiv.org/html/2608.14772v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.14767",
  "slug": "narrate-a-multimodal-real-world-australian-driving-dataset-for-human-c",
  "title": "NARRATE: A Multimodal Real-World Australian Driving Dataset for Human-Centred Explanations in Automated Driving",
  "abstract": "Automated vehicles must explain their decisions in ways that passengers can understand, monitor, and trust. Existing language-annotated driving datasets are mostly observer-written, post-hoc, simulation-based, or generated from sensor inputs, rather than elicited from the driver performing the action. We introduce NARRATE, a multimodal real-world Australian driving dataset comprising 2,050 annotated events from 35 experienced drivers and driving instructors on public roads. Each event is grounded in synchronised visual, localisation, motion, and LiDAR streams and paired with in-vehicle and/or post-drive free-text explanations. NARRATE provides action labels, scenario-context labels spanning six high-level and 32 fine-grained categories, and span-level Situational Awareness (SA) annotations over driver explanations for Perception, Comprehension and Projection. Four benchmark tasks (SA, scenario-context, driver-action classification, and explanation generation) show that this structure is learnable from driver language, while fine-grained context recognition and explanation generation remain challenging. NARRATE paves a path towards more human-centred and domain-aware explanation models for automated driving.",
  "published": "2026-08-14",
  "updated": "2026-08-14",
  "year": "2026",
  "authors": [
   "Ashkan Yousefi Zadeh",
   "Zishuo Zhu",
   "Xiaomeng Li",
   "Andry Rakotonirainy",
   "Sebastien Glaser",
   "Ronald Schroeter",
   "Patricia Delhomme",
   "Zahra Mehraban"
  ],
  "author_count": 8,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.CL",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "ECCV 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work introduces NARRATE, a multimodal real-world Australian driving dataset comprising 2,050 annotated events from 35 experienced drivers and driving instructors on public roads and paves a path towards more human-centred and domain-aware explanation models for automated driving.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ashkan Y. Zadeh",
    "id": "2043233880",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Zishuo Zhu",
    "id": "2290637823",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Xiaomeng Li",
    "id": "2380413231",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "A. Rakotonirainy",
    "id": "94815327",
    "h_index": 41,
    "papers": 332
   },
   {
    "name": "S\u00e9bastien Glaser",
    "id": "2238774875",
    "h_index": 7,
    "papers": 26
   },
   {
    "name": "Ronald Schroeter",
    "id": "2332585338",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "P. Delhomme",
    "id": "2322969252",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Zahra Mehraban",
    "id": "2311908967",
    "h_index": 3,
    "papers": 11
   }
  ],
  "comment": "Accepted at The 19th European Conference on Computer Vision (ECCV 2026) DriveX Workshop (Foundation Models for Autonomous Driving)",
  "topics": [
   "navigation",
   "foundation-pretraining",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.14767v1",
  "pdf_url": "https://arxiv.org/pdf/2608.14767v1",
  "html_url": "https://arxiv.org/html/2608.14767v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.14531",
  "slug": "spatiotemporal-tube-based-safety-certificate-for-autonomous-navigation",
  "title": "Spatiotemporal Tube-Based Safety-Certificate for Autonomous Navigation of Articulated Vehicles",
  "abstract": "Articulated vehicles are the workhorses of freight transportation, and their autonomous navigation is challenging. Their physical characteristics and motion constraints pose significant challenges in manoeuvring these vehicles on narrow routes. This paper presents a spatiotemporal tube-based approach to plan autonomous navigation of vehicles like tractor semi-trailers, truck/ tractor trailers, towing Automated Guided Vehicles (AGVs), and road trains. This planning approach provides a certified path plan for the truck or tractor, ensuring that the towed series of trailers always remains within the road corridor, limited by permissible corrections. The planning leverages the kinematics of the linked elements along with sway constraints to arrive at a safe tube for the actuated prime mover. We modify the spatiotemporal tube using permissible corrections to provide a route safety certificate to the vehicle for the given route. The proposed planning method is verified on a truck-trailer navigation simulation for a complex route.",
  "published": "2026-08-14",
  "updated": "2026-08-14",
  "year": "2026",
  "authors": [
   "Mohd. Faizuddin Faruqui",
   "Ratnangshu Das",
   "Ravi Kumar L",
   "Pushpak Jagtap"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Mohd. Faizuddin Faruqui",
    "id": "7829984",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Ratnangshu Das",
    "id": "2212754126",
    "h_index": 6,
    "papers": 34
   },
   {
    "name": "Ravi Kumar",
    "id": "2117806105",
    "h_index": 3,
    "papers": 13
   },
   {
    "name": "Pushpak Jagtap",
    "id": "2901163",
    "h_index": 18,
    "papers": 110
   }
  ],
  "comment": "Accepted for presentation at the 2026 IEEE International Conference on Intelligent Transportation Systems (ITSC 2026)",
  "topics": [
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.14531v1",
  "pdf_url": "https://arxiv.org/pdf/2608.14531v1",
  "html_url": "https://arxiv.org/html/2608.14531v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.14481",
  "slug": "ensuring-safe-physical-ai-in-urban-mobility-via-hazard-informed-synthe",
  "title": "Ensuring Safe Physical AI in Urban Mobility via Hazard-Informed Synthesized Envelopes",
  "abstract": "As heterogeneous robotic systems deploy across diverse urban zones, maintaining safety amid complex human-robot interactions remains a critical challenge. We present a unified framework that bridges systematic hazard analysis and runtime enforcement using hazard-informed safety envelopes. Rather than treating safety as a static constraint isolated within individual software modules, we introduce a cross-layer safety transformation process spanning symbolic, spatial, and dynamic world models. We show how this representation naturally interfaces with physical AI runtime harnesses to guarantee safe urban mobility.",
  "published": "2026-08-14",
  "updated": "2026-08-14",
  "year": "2026",
  "authors": [
   "Alexei Odinokov",
   "Rostislav Yavorskiy"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work presents a unified framework that bridges systematic hazard analysis and runtime enforcement using hazard-informed safety envelopes and introduces a cross-layer safety transformation process spanning symbolic, spatial, and dynamic world models.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "A. Odinokov",
    "id": "146148313",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "R. Yavorskiy",
    "id": "144108924",
    "h_index": 6,
    "papers": 39
   }
  ],
  "comment": "The 2026 International Conference on Control, Robotics Engineering and Technology (CRET 2026), https://www.cret.net/",
  "topics": [
   "world-models",
   "hri",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.14481v1",
  "pdf_url": "https://arxiv.org/pdf/2608.14481v1",
  "html_url": "https://arxiv.org/html/2608.14481v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.14466",
  "slug": "expected-free-energy-based-informative-path-planning-for-robotic-mars",
  "title": "Expected Free Energy-based Informative Path Planning for Robotic Mars Exploration",
  "abstract": "An autonomous robot efficiently exploring an unknown environment, such as looking for water sources on Mars, faces two simultaneous demands: building an accurate information map while quickly finding the regions of greatest value, and paying for every meter of travel and the cost of every measurement it takes. Classical information-seeking and reward-seeking criteria address only one of these objectives at a time. Here, we propose Expected Free Energy (EFE), the principled action-selection objective from active inference, as a unifying criterion for budgeted robotic informative path planning. Maintaining a Gaussian-process belief over the information field, our agent plans continuous trajectories that minimize expected free energy under hard path-length constraints. The results from multiple realizations show that EFE-based planning yields accurate posterior maps and locates the highest-value regions simultaneously, outperforming information-theoretic baselines under the same settings. In robotic exploration, these unified, easy-to-tune principled information-gathering strategies facilitate autonomous deployment while enforcing efficiency and resource constraints.",
  "published": "2026-08-14",
  "updated": "2026-08-14",
  "year": "2026",
  "authors": [
   "Ajith Anil Meera",
   "Pablo Lanillos",
   "Wouter Kouw"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.IT",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "A. Meera",
    "id": "145494274",
    "h_index": 7,
    "papers": 26
   },
   {
    "name": "Pablo Lanillos",
    "id": "2258948686",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Wouter M. Kouw",
    "id": "2142600609",
    "h_index": 6,
    "papers": 28
   }
  ],
  "comment": "accepted for IWAI 2026",
  "topics": [
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.14466v1",
  "pdf_url": "https://arxiv.org/pdf/2608.14466v1",
  "html_url": "https://arxiv.org/html/2608.14466v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.14462",
  "slug": "thrive-therapeutic-humanoid-robot-in-virtual-environment",
  "title": "THRIVE: Therapeutic Humanoid Robot In Virtual Environment",
  "abstract": "This paper presents THRIVE (Therapeutic Humanoid Robot In Virtual Environment), an at-home rehabilitation platform that integrates a suite of virtual-reality upper-body rehabilitation games, a real-time camera-based motion-tracking system, and a socially interactive robot therapist. The system is designed for therapy and intervention in children with upper-limb motor impairments, which can be improved through consistent, task-specific practice. THRIVE features a set of newly designed, engaging games that target functional reaching, grasping, and object-manipulation movements through customizable popping, hitting, catching, and grabbing tasks, while the camera-based tracking system captures the child's kinematic performance during play. A robot therapist - deployable either as a physical robotic coach or as a remote-presence virtual agent - delivers adaptive, dynamic feedback to motivate the child and guide their movements toward therapeutic goals. THRIVE decouples the therapeutic games from the robot embodiment, extending the platform to support various embodiments and different robots within one modular system. This robot-agnostic design makes THRIVE affordable, scalable, and readily adaptable for sustained use in the home, offering a practical pathway to more consistent and engaging upper-limb therapy for children with motor function impairments.",
  "published": "2026-08-14",
  "updated": "2026-08-14",
  "year": "2026",
  "authors": [
   "Jin Xu",
   "Yu-Ping Chen",
   "Ayanna Howard"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "An at-home rehabilitation platform that integrates a suite of virtual-reality upper-body rehabilitation games, a real-time camera-based motion-tracking system, and a socially interactive robot therapist for children with upper-limb motor impairments is presented.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jin Xu",
    "id": "2143806389",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Yu-Ping Chen",
    "id": "145848497",
    "h_index": 15,
    "papers": 35
   },
   {
    "name": "Ayanna Howard",
    "id": "2359435094",
    "h_index": 1,
    "papers": 3
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "humanoids"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.14462v1",
  "pdf_url": "https://arxiv.org/pdf/2608.14462v1",
  "html_url": "https://arxiv.org/html/2608.14462v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.14448",
  "slug": "control-informed-constraint-adaptation-in-minimum-time-trajectory-plan",
  "title": "Control-Informed Constraint Adaptation in Minimum-Time Trajectory Planning for Autonomous Racing",
  "abstract": "Autonomous racecars operate at the limits of vehicle dynamics, where small control errors translate into safety-critical behavior and lost performance. Trajectory planners assume perfect tracking and remain blind to execution errors. To guarantee safety, trajectory planners therefore restrict themselves to conservative spatial margins, leaving usable track space untapped. To overcome these issues, we introduce a control-informed online trajectory planning framework that learns from its own execution errors. By measuring systematic tracking deviations during runtime, we dynamically adapt spatial track constraints and iteratively expand the free-space planning area. The planner remains time-optimal while compensating for accumulated execution errors. This method was analyzed in a high-fidelity closed-loop simulation environment with autonomous racecars. The results demonstrate that our approach reduces lap time by 1.8\\,s without increasing computational burden, maintaining a median runtime of 25 ms. Our finding indicates that feeding control-induced deviations back into the planning layer unlocks performance previously inaccessible to modular architectures and enables autonomous vehicles to exploit track limits systematically.",
  "published": "2026-08-14",
  "updated": "2026-08-14",
  "year": "2026",
  "authors": [
   "Ann-Kathrin Schwehn",
   "Alexander Langmann",
   "Mattia Piccinini",
   "Johannes Betz"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work introduces a control-informed online trajectory planning framework that learns from its own execution errors, and demonstrates that feeding control-induced deviations back into the planning layer unlocks performance previously inaccessible to modular architectures and enables autonomous vehicles to exploit track limits systematically.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ann-Kathrin Schwehn",
    "id": "2355647889",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Alexander Langmann",
    "id": "2238799614",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Mattia Piccinini",
    "id": "2049096210",
    "h_index": 11,
    "papers": 47
   },
   {
    "name": "Johannes Betz",
    "id": null,
    "h_index": 0,
    "papers": 0
   }
  ],
  "comment": "Accepted at IEEE ITSC 2026",
  "topics": [
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.14448v1",
  "pdf_url": "https://arxiv.org/pdf/2608.14448v1",
  "html_url": "https://arxiv.org/html/2608.14448v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.14406",
  "slug": "effect-of-twisted-yarn-architecture-on-pressure-and-proximity-sensing",
  "title": "Effect of Twisted-Yarn Architecture on Pressure and Proximity Sensing Characteristics of Textile Capacitive Sensors for Robotic Skin",
  "abstract": "Textile-integrated capacitive sensors offer flexible and conformable tactile sensing for wearable electronics and human-robot interaction; however, the influence of yarn-level architecture on capacitive transduction characteristics remains insufficiently quantified. This work presents a textile capacitive sensing platform based on silver-coated yarns coated with polydimethylsiloxane and assembled into one-, two-, and four-layer twisted configurations. The influence of effective electrode overlap area and inter-fiber separation on the capacitive response is systematically investigated, enabling architecture-dependent tuning of pressure and proximity sensing characteristics. Pressure was calculated using the localized single-fiber contact area, corresponding to stresses of 0.4-3.9 MPa. Increasing the layer number improved mechanical strength and sensing performance: elongation at break increased from 37.5% to 62.5% and 85.0%, while the maximum load increased from 23.3 to 42.7 and 89.7 N. Sensitivity increased with layer number and frequency, reaching 0.1331 MPa$^{-1}$ for the four-layer sensor at 100 kHz. The four-layer configuration also exhibited low hysteresis, minimal thermal drift from 25 to 90 $^\\circ$C, and stable operation over 15,000 cycles. Proximity detection ranges of 60, 50, and 40 mm were obtained for the one-, two-, and four-layer sensors, respectively, revealing an architecture-dependent sensitivity-range trade-off. A 4$\\times$4 textile sensing array enabled spatial contact mapping, while robotic-arm integration demonstrated real-time touch and proximity detection with an end-to-end robotic system latency (from detection to robot reaction) of 403 ms. The results establish yarn architecture as a tunable design parameter governing the measurement characteristics of textile-integrated capacitive sensing systems.",
  "published": "2026-08-14",
  "updated": "2026-08-14",
  "year": "2026",
  "authors": [
   "Ishtia Zahir",
   "Eslam Saleh",
   "Maryam Rezayati",
   "G\u00fcunter Grabher",
   "Gaffar Hossain"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "physics.ins-det"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "I. Zahir",
    "id": "2120587479",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Eslam Saleh",
    "id": "2345035421",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Maryam Rezayati",
    "id": "49770302",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "G\u00fcunter Grabher",
    "id": "2457992991",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Gaffar Hossain",
    "id": "144451366",
    "h_index": 14,
    "papers": 24
   }
  ],
  "comment": "10 pages. Submitted to IEEE Transactions on Instrumentation and Measurement",
  "topics": [
   "tactile",
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.14406v1",
  "pdf_url": "https://arxiv.org/pdf/2608.14406v1",
  "html_url": "https://arxiv.org/html/2608.14406v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.14379",
  "slug": "reflex-enabling-fast-and-predictive-vision-language-action-models-for",
  "title": "Reflex: Enabling Fast and Predictive Vision-Language-Action Models for Reaction-Critical Manipulation",
  "abstract": "Vision-Language-Action (VLA) models have recently achieved promising performance in robotic manipulation. However, existing benchmarks mainly evaluate generalization on static manipulation tasks and largely overlook dynamic interaction scenarios. To address this gap, we present ReflexBench, a benchmark for reaction-critical manipulation. ReflexBench contains six dynamic tasks and introduces an evaluation framework that decouples simulator stepping from robot control while supporting configurable latency under synchronous and asynchronous inference. Building upon ReflexBench, we propose ReflexVLA, an efficient VLA model designed for reaction-critical manipulation without large-scale robot-data pretraining. ReflexVLA enhances temporal reasoning through latent future prediction and multi-frame temporal fusion within the vision backbone, while reducing deployment latency through batched visual encoding and CUDA Graph replay. Experiments show that ReflexVLA consistently improves dynamic manipulation performance while maintaining competitive accuracy on standard static manipulation benchmarks, and real-world experiments further demonstrate its effectiveness under practical deployment conditions. Project website: https://reflexvla.github.io",
  "published": "2026-08-14",
  "updated": "2026-08-14",
  "year": "2026",
  "authors": [
   "Yuxuan Chen",
   "Wanruo Zhang",
   "Xiao Li"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes ReflexVLA, an efficient VLA model designed for reaction-critical manipulation without large-scale robot-data pretraining, which enhances temporal reasoning through latent future prediction and multi-frame temporal fusion within the vision backbone, while reducing deployment latency through batched visual encoding and CUDA Graph replay.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yuxuan Chen",
    "id": "2331571394",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Wanruo Zhang",
    "id": "2215891359",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Xiao Li",
    "id": "2331631591",
    "h_index": 2,
    "papers": 5
   }
  ],
  "comment": "8 pages, 6 pages",
  "topics": [
   "vla",
   "sim2real",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.14379v1",
  "pdf_url": "https://arxiv.org/pdf/2608.14379v1",
  "html_url": "https://arxiv.org/html/2608.14379v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.14332",
  "slug": "coral-curriculum-optimized-reward-adaptation-for-lidar-based-goal-dire",
  "title": "CORAL: Curriculum-Optimized Reward Adaptation for LiDAR-Based Goal-Directed Urban Driving",
  "abstract": "Reinforcement learning is promising for autonomous urban driving, but long-horizon goal-directed navigation asks a policy to acquire several competing behaviors at once--reaching a distant goal, tracking a route, avoiding obstacles, obeying signals--and a fixed objective gives no order in which to learn them. This paper presents CORAL, which advances two schedules together: a five-stage curriculum that progressively lengthens routes and tightens behavioral constraints, and a stage-aware reward whose component weights shift emphasis from mission progress toward route following, safety, smoothness, and rule compliance as the task hardens. The policy is a multi-stream actor-critic network trained with Proximal Policy Optimization (PPO) in CARLA on a compact 99-dimensional state pairing a polar LiDAR histogram with vehicle telemetry, ego-frame route geometry, and traffic-rule indicators--no point-cloud encoder, no bird's-eye-view rasterization. Against two PPO baselines under an identical protocol, CORAL reaches the goal in all twenty evaluation episodes on the longest routes under the full set of behavioral constraints, where the baselines reach 5% and 10%; a factorial ablation shows that neither schedule alone matches their combination: removing either lowers both success and route completion, and disabling both drops success to 55%. Trained in one town, the policy transfers zero-shot to seven unseen towns, succeeding in 68-98% of episodes on routes of the same 100-150 m length, with mean lateral deviation below 0.35 m.",
  "published": "2026-08-14",
  "updated": "2026-08-14",
  "year": "2026",
  "authors": [
   "Anisa Saleem",
   "Duksu Kim"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper presents CORAL, which advances two schedules together: a five-stage curriculum that progressively lengthens routes and tightens behavioral constraints, and a stage-aware reward whose component weights shift emphasis from mission progress toward route following, safety, smoothness, and rule compliance as the task hardens.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Anisa Saleem",
    "id": "2457992510",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Duksu Kim",
    "id": "2346054505",
    "h_index": 2,
    "papers": 5
   }
  ],
  "comment": "13 pages, 6 figures",
  "topics": [
   "rl-control",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.14332v1",
  "pdf_url": "https://arxiv.org/pdf/2608.14332v1",
  "html_url": "https://arxiv.org/html/2608.14332v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.14284",
  "slug": "prm-as-a-judge-1-5-a-toolkit-for-robot-process-assessment",
  "title": "PRM-as-a-Judge 1.5: A Toolkit for Robot Process Assessment",
  "abstract": "Fine-grained robotic evaluation matters for understanding embodied models, going beyond binary success rates and rule-based process scores. We present PRM-as-a-Judge 1.5, a toolkit for robot process assessment that turns rollout videos into dense progress curves and derives multiple fine metrics. PRM-as-a-Judge 1.5 introduces three metrics, building on version 1.0, that characterize failure-side progress, post-drawdown recovery, and success-side execution quality, helping users understand embodied model capability. Based on the rollout videos from benchmarks, we perform a comprehensive assessment of the embodied models, providing some fine-grained metric results and key findings. We further introduce RoboPulse++ to evaluate the reliability of process reward models (PRM), providing evaluators with a more accurate testing platform. Moreover, we release a user-friendly assessment suite, including the benchmark, metric implementation, and visualization tools, to support reproducible manipulation process evaluation. We call on the community to rethink how robots are evaluated and establish transparent, procedural, and reproducible assessment as a foundation for the next generation of embodied intelligence.",
  "published": "2026-08-14",
  "updated": "2026-08-14",
  "year": "2026",
  "authors": [
   "Yuyang Liu",
   "Yanqing Shen",
   "Ruike Chen",
   "Jifan Zhao",
   "Yuxuan Tian",
   "Yichi Zhang",
   "Tianfeng Long",
   "Zixuan Yin",
   "Yipu Wang",
   "Ziheng Qin",
   "Wenxing Tan",
   "Yang Shi",
   "Mingyu Cao",
   "Runze Xiao",
   "Ziqi Wang",
   "Zhixin Yin",
   "Shiwei Chu",
   "Yi-Fan Zhang",
   "Yao Mu",
   "Yuheng Ji",
   "Yihao Wang",
   "Jun Yan",
   "Zhongyuan Wang",
   "Pengwei Wang",
   "Xiaolong Zheng"
  ],
  "author_count": 25,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A toolkit for robot process assessment that turns rollout videos into dense progress curves and derives multiple fine metrics, and introduces RoboPulse++ to evaluate the reliability of process reward models (PRM), providing evaluators with a more accurate testing platform.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yuyang Liu",
    "id": "2375085184",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Yan Shen",
    "id": "2449949169",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Rui Chen",
    "id": "2454433355",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Jifan Zhao",
    "id": "2120471697",
    "h_index": 6,
    "papers": 18
   },
   {
    "name": "Yuxuan Tian",
    "id": "2397153253",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Yichi Zhang",
    "id": "2446302081",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Tianfeng Long",
    "id": "2343410153",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Zixuan Yin",
    "id": "2430712443",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yipu Wang",
    "id": "2375150096",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Ziheng Qin",
    "id": "2397720942",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Wenxing Tan",
    "id": "2313954981",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Yang Shi",
    "id": "2383137537",
    "h_index": 7,
    "papers": 29
   },
   {
    "name": "Mingyu Cao",
    "id": "2361073090",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Runze Xiao",
    "id": "2147189769",
    "h_index": 4,
    "papers": 15
   },
   {
    "name": "Ziqi Wang",
    "id": "2315627736",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Zhixin Yin",
    "id": "2457984598",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Shiwei Chu",
    "id": "2329982090",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Yifan Zhang",
    "id": "2455437850",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yao Mu",
    "id": "2348606790",
    "h_index": 4,
    "papers": 18
   },
   {
    "name": "Yuheng Ji",
    "id": "2297947392",
    "h_index": 10,
    "papers": 29
   },
   {
    "name": "Yihao Wang",
    "id": "2458022768",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Jun Yan",
    "id": "2440917781",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Zhongyuan Wang",
    "id": "2338315894",
    "h_index": 17,
    "papers": 36
   },
   {
    "name": "Pengwei Wang",
    "id": "2338357829",
    "h_index": 17,
    "papers": 44
   },
   {
    "name": "Xiaolong Zheng",
    "id": "2380150862",
    "h_index": 1,
    "papers": 7
   }
  ],
  "comment": "Project page: https://prm-as-a-judge.github.io",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.14284v1",
  "pdf_url": "https://arxiv.org/pdf/2608.14284v1",
  "html_url": "https://arxiv.org/html/2608.14284v1",
  "code_url": "https://prm-as-a-judge.github.io",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2608.14266",
  "slug": "accelerating-large-scale-bundle-adjustment-for-lidar-mapping-via-paral",
  "title": "Accelerating Large-scale Bundle Adjustment for LiDAR Mapping via Parallel Computing",
  "abstract": "LiDAR bundle adjustment is widely utilized in mapping to construct globally consistent point cloud maps. In this paper, we propose the first fully parallel computing framework to accelerate LiDAR bundle adjustment for large-scale mapping, incorporating three key techniques. First, we design an adaptive, asynchronous data loading strategy to efficiently process large-scale point cloud datasets on memory-constrained GPUs. Secondly, we present a novel bottom-up voxelization method for extracting planar features, enabling fully parallelized pre-processing. Thirdly, we build upon a majorization-minimization formulation to accelerate compute-intensive tasks in the optimization via parallel computation, including the computation of residuals, Jacobian and Hessian matrices, and a parallel increment solver. To support our design, we provide both theoretical and experimental analysis of the time complexity of our approach. Extensive benchmarking on large-scale public datasets across various computational platforms validates the robustness and adaptability of our approach, achieving up to a tenfold improvement in computational efficiency while preserving mapping accuracy comparable to state-of-the-art methods. To benefit future research, the implementation code is available on GitHub.",
  "published": "2026-08-14",
  "updated": "2026-08-14",
  "year": "2026",
  "authors": [
   "Yixi Cai",
   "Rundong Li",
   "Yuhan Xie",
   "Qingwen Zhang",
   "Patric Jensfelt",
   "Fu Zhang"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper proposes the first fully parallel computing framework to accelerate LiDAR bundle adjustment for large-scale mapping, incorporating three key techniques, including a novel bottom-up voxelization method for extracting planar features, enabling fully parallelized pre-processing.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yixi Cai",
    "id": "50204612",
    "h_index": 15,
    "papers": 37
   },
   {
    "name": "Rundong Li",
    "id": "2281678106",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Yuhan Xie",
    "id": "2363802599",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Qingwen Zhang",
    "id": "2243332339",
    "h_index": 10,
    "papers": 28
   },
   {
    "name": "P. Jensfelt",
    "id": "1770066",
    "h_index": 55,
    "papers": 238
   },
   {
    "name": "Fu Zhang",
    "id": "2350340097",
    "h_index": 2,
    "papers": 6
   }
  ],
  "comment": "Accepted by IEEE International Conference on Automation Science and Engineering (CASE), 2026",
  "topics": [
   "spatial-3d",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.14266v1",
  "pdf_url": "https://arxiv.org/pdf/2608.14266v1",
  "html_url": "https://arxiv.org/html/2608.14266v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.14244",
  "slug": "vibration-suppression-in-collaborative-flexible-payload-manipulation-u",
  "title": "Vibration Suppression in Collaborative Flexible Payload Manipulation Using Passive Force Control",
  "abstract": "In large and heavy structures, vibrations arise during motion, posing significant challenges for precise manipulation. To accomplish the desired motion, control algorithms must effectively suppress these structural vibrations. In cutting edge projects, such as remote maintenance of future fusion energy reactors (tokamaks), the manipulation of this type of structure is defined as a crucial task. This paper presents a control strategy to suppress transverse vibrations in flexible payloads during motion using a collaborative payload manipulation approach. Two different industrial robot arms are arranged in a leader follower configuration for the manipulation strategy. The leader robot guides the motion with shaped velocity commands, while the follower robot ensures compliance with the estimated external forces applied by the leader on the payload through an admittance controller. Unlike existing methods, the proposed approach enables collaborative manipulation of heavier and larger flexible objects, addressing additional challenges such as vibration suppression and heterogeneous robot specifications. The dynamics of the leader follower payload system are modeled using an equivalent mass spring damper model, and it is shown that, with appropriate admittance parameters, the total energy of the system is passively dissipated. A stability proof is also provided. Numerical simulations validate the proposed method, and experimental results demonstrate its effectiveness.",
  "published": "2026-08-14",
  "updated": "2026-08-14",
  "year": "2026",
  "authors": [
   "Alaa Abderrahim",
   "Antonio Rosales",
   "Ferdinando Milella",
   "Markku Suomalainen",
   "Shuai Li"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Alaa Abderrahim",
    "id": "2375388770",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Antonio Rosales",
    "id": "2303655775",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "F. Milella",
    "id": "97724925",
    "h_index": 7,
    "papers": 15
   },
   {
    "name": "Markku Suomalainen",
    "id": "2365111510",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Shuai Li",
    "id": "2281960644",
    "h_index": 2,
    "papers": 13
   }
  ],
  "comment": "Published in the proceedings of the 2026 European Control Conference (ECC)",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.14244v1",
  "pdf_url": "https://arxiv.org/pdf/2608.14244v1",
  "html_url": "https://arxiv.org/html/2608.14244v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.14239",
  "slug": "a-temporal-barrier-framework-for-collision-avoidance-in-multi-agent-au",
  "title": "A Temporal Barrier Framework for Collision Avoidance in Multi-Agent Autonomous Aerial Vehicles",
  "abstract": "Operating teams of autonomous aircraft in dynamic, uncertain, and potentially adversarial environments requires safety protocols that are reliable yet selective, and allow agents to fly in close proximity while making progress toward mission objectives. We introduce adversarial time-to-collision (aTTC), a risk metric that quantifies, for a given agent, how quickly any surrounding agent could reach it assuming adversarial intent. We embed aTTC into the control barrier function (CBF) framework, defining the barrier directly in time rather than distance or velocity. The resulting aTTC-CBF is inherently anticipatory: agents modulate their own velocity based not on whether a peer is on a collision course, but on how quickly one could reach collision given its dynamical constraints. A differentiable neural-network surrogate makes the aTTC computable in real time within a standard CBF quadratic program. Across long time-horizon simulations of 3D independent-pursuit and formation-flight scenarios, the aTTC-CBF achieves up to twice the waypoint progress at half the collision rate of a higher-order distance-based CBF baseline.",
  "published": "2026-08-14",
  "updated": "2026-08-14",
  "year": "2026",
  "authors": [
   "Benedikt Barthel Sorensen",
   "Mitchell Black",
   "Erfaun Noorani",
   "Themistoklis Sapsis"
  ],
  "author_count": 4,
  "categories": [
   "eess.SY",
   "cs.RO",
   "math.OC"
  ],
  "primary_category": "eess.SY",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Adversarial time-to-collision (aTTC), a risk metric that quantifies, for a given agent, how quickly any surrounding agent could reach it assuming adversarial intent, is introduced.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Benedikt Barthel Sorensen",
    "id": "2287923921",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Mitchell Black",
    "id": "2268673606",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Erfaun Noorani",
    "id": "1598440076",
    "h_index": 7,
    "papers": 35
   },
   {
    "name": "T. Sapsis",
    "id": "2944258",
    "h_index": 41,
    "papers": 240
   }
  ],
  "comment": "",
  "topics": [
   "rl-control",
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.14239v1",
  "pdf_url": "https://arxiv.org/pdf/2608.14239v1",
  "html_url": "https://arxiv.org/html/2608.14239v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.14207",
  "slug": "mmusv-sim-a-perception-oriented-simulation-and-data-generation-platfor",
  "title": "MMUSV-Sim: A Perception-Oriented Simulation and Data-Generation Platform for Multi-USV Cooperative Perception",
  "abstract": "Cooperative perception among multiple unmanned surface vehicles (USVs) combines complementary observations to extend maritime target sensing beyond the view range and field of a single platform. Developing such systems at scale calls for a unified workflow for configurable multi-USV scenarios, multimodal acquisition, and shared annotations. We present MMUSV-Sim, a perception-oriented maritime simulation and data-generation platform built on Unreal Engine 5 and Project AirSim. It provides island, open-sea, and port environments; configurable weather, time of day, and wave conditions; a diverse vessel asset library; and spline-based multi-vessel motion. MMUSV-Sim acquires RGB, depth, semantic, LiDAR, and radar observations across multiple USVs and captures a common world state for per-agent annotation export. Experiments verify that the configured wave settings produce the intended changes in vessel heave, roll, and pitch, and evaluate the geometric consistency between projected annotations and semantic renderings. In LiDAR-based cooperative BEV vessel detection experiments on the generated multi-USV dataset, Early Fusion achieves an AP@0.5 of 72.74, compared with 45.54 using a single USV.",
  "published": "2026-08-14",
  "updated": "2026-08-14",
  "year": "2026",
  "authors": [
   "Ziao Li",
   "Jianxiong Ye",
   "Biao Tang",
   "Leping Zhang",
   "Kun Zuo",
   "Siyu Huang",
   "Chenqiang Gao"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "MMUSV-Sim, a perception-oriented maritime simulation and data-generation platform built on Unreal Engine 5 and Project AirSim, provides island, open-sea, and port environments; configurable weather, time of day, and wave conditions; a diverse vessel asset library; and spline-based multi-vessel motion.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ziao Li",
    "id": "2434075778",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Jianxiong Ye",
    "id": "2254155785",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Biao Tang",
    "id": "2457984203",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Leping Zhang",
    "id": "2314146073",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Kun Zuo",
    "id": "2457977444",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Siyu Huang",
    "id": "2458114077",
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Chenqiang Gao",
    "id": "2262086701",
    "h_index": 12,
    "papers": 63
   }
  ],
  "comment": "",
  "topics": [
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.14207v1",
  "pdf_url": "https://arxiv.org/pdf/2608.14207v1",
  "html_url": "https://arxiv.org/html/2608.14207v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.14160",
  "slug": "occplanner-goal-aware-occupancy-conditioned-diffusion-planner-for-pixe",
  "title": "OccPlanner: Goal-Aware Occupancy-Conditioned Diffusion Planner for Pixel-Goal Navigation",
  "abstract": "Pixel-goal navigation specifies targets directly in the agent's camera view, but a target pixel provides neither metric depth nor traversability, making 3D goal grounding and collision-free continuous planning challenging. We present OccPlanner, a goal-aware occupancy-conditioned diffusion planner that grounds pixel goals in egocentric metric space and sequentially conditions the goal representation on temporal visual context and learned local 3D occupancy features. To provide occupancy supervision at scale, we introduce L3ROcc, which converts monocular RGB navigation videos into robot-centric local 3D occupancy annotations through geometric reconstruction and ray-based visibility reasoning. We train OccPlanner on InternData-N1 and evaluate it in closed-loop simulation across four unseen scene categories from InternScenes and two goal-distance ranges. In the 5-8 m setting, OccPlanner increases the average success rate (SR) over NavDP from 20.81% to 71.55% across the four categories, reaching 86.20% and 84.92% in cluttered-easy and cluttered-hard scenes, respectively. Real-world open-loop experiments on a Unitree Go2 further provide initial evidence of sim-to-real transfer and adaptation with L3ROcc-generated supervision.",
  "published": "2026-08-14",
  "updated": "2026-08-14",
  "year": "2026",
  "authors": [
   "Binling Huang",
   "Nianjin Ye",
   "Xi Yang",
   "Liang Hu",
   "Zhou Huang",
   "Shuang Wei",
   "Longrui Yang",
   "Yanchi Chen",
   "Lanpeng Jia"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work presents OccPlanner, a goal-aware occupancy-conditioned diffusion planner that grounds pixel goals in egocentric metric space and sequentially conditions the goal representation on temporal visual context and learned local 3D occupancy features.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Binling Huang",
    "id": "2458020883",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Nianjin Ye",
    "id": "1454195293",
    "h_index": 9,
    "papers": 13
   },
   {
    "name": "Xi Yang",
    "id": "2458029404",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Liang Hu",
    "id": "2362588463",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Zhou Huang",
    "id": "2458017704",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Shuang Wei",
    "id": "2315662478",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Long Yang",
    "id": "2449934703",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yanchi Chen",
    "id": "2458026112",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Lanpeng Jia",
    "id": "153892954",
    "h_index": 6,
    "papers": 9
   }
  ],
  "comment": "Technical report. 11 pages, 6 figures, and 2 tables",
  "topics": [
   "egocentric-data",
   "sim2real",
   "spatial-3d",
   "navigation",
   "data-teleop"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2608.14160v1",
  "pdf_url": "https://arxiv.org/pdf/2608.14160v1",
  "html_url": "https://arxiv.org/html/2608.14160v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2608.14135",
  "slug": "agilepe-autonomous-uav-pursuit-evasion-via-self-play-reinforcement-lea",
  "title": "AgilePE: Autonomous UAV Pursuit-Evasion via Self-Play Reinforcement Learning",
  "abstract": "Autonomous pursuit-evasion is a fundamental challenge for Unmanned Aerial Vehicles (UAVs), requiring rapid decision-making under tightly coupled dynamics and continuously changing opponent behaviors. Traditional rule-based or differential-game approaches often struggle with high-dimensional aerial interactions and agile maneuvering. We present AgilePE, a complete system for autonomous UAV pursuit-evasion via self-play reinforcement learning. AgilePE integrates agile low-level control, competitive policy optimization, and sim-to-real deployment in a unified framework. The policy directly maps onboard state observations to Collective Thrust and Body Rates (CTBR) commands, enabling end-to-end agile maneuvering without intermediate trajectory planners or waypoint controllers. For training, we use competitive self-play with Prioritized Fictitious Self-Play (PFSP) and a diversified opponent pool, enabling agents to improve against historical policies while stabilizing optimization and reducing policy oscillation. This process leads to the emergence of sophisticated pursuit and evasion strategies. For real-world deployment, we develop a hardware-aligned simulation pipeline that models actuator-response dynamics, communication latency, and domain randomization. The learned policies transfer zero-shot to real quadrotors without task-specific tuning. Real-world experiments reproduce pursuit-evasion tactics observed in simulation, including rapid dodging and flanking, and demonstrate interactive two-agent zero-shot deployment.",
  "published": "2026-08-14",
  "updated": "2026-08-14",
  "year": "2026",
  "authors": [
   "Wenhao Tang",
   "Tianyang Chen",
   "Zhejun Cui",
   "Boyuan An",
   "Jiayu Chen",
   "Ruize Zhang",
   "Huidong Liu",
   "Tianyue Wu",
   "Qingmin Liao",
   "Fei Gao",
   "Yu Wang",
   "Chao Yu"
  ],
  "author_count": 12,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "AgilePE is presented, a complete system for autonomous UAV pursuit-evasion via self-play reinforcement learning that integrates agile low-level control, competitive policy optimization, and sim-to-real deployment in a unified framework.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Wenhao Tang",
    "id": "2323010636",
    "h_index": 5,
    "papers": 18
   },
   {
    "name": "Tianyang Chen",
    "id": "2380083197",
    "h_index": 2,
    "papers": 15
   },
   {
    "name": "Zhejun Cui",
    "id": "2406197775",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Boyuan An",
    "id": "2457978556",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Jiayu Chen",
    "id": "2257135471",
    "h_index": 5,
    "papers": 15
   },
   {
    "name": "Ruize Zhang",
    "id": "2244894838",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Huidong Liu",
    "id": "2354490689",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Tianyue Wu",
    "id": "2158953359",
    "h_index": 7,
    "papers": 18
   },
   {
    "name": "Qingmin Liao",
    "id": "2276430897",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Fei Gao",
    "id": "2349210689",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Yu Wang",
    "id": "2305964414",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Chao Yu",
    "id": "2343795966",
    "h_index": 2,
    "papers": 7
   }
  ],
  "comment": "8 pages, 7 figures. Under review",
  "topics": [
   "sim2real",
   "rl-control",
   "navigation",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.14135v1",
  "pdf_url": "https://arxiv.org/pdf/2608.14135v1",
  "html_url": "https://arxiv.org/html/2608.14135v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.14082",
  "slug": "pilot-privileged-imitation-learning-for-end-to-end-motion-planning-of",
  "title": "PILOT: Privileged Imitation Learning for End-to-End Motion Planning of Autonomous UAVs under Partial Observability",
  "abstract": "Autonomous navigation in cluttered environments is hampered by partial observability and dynamic constraints. This paper presents PILOT, a constraint-aware privileged imitation learning framework for vision-based end-to-end UAV motion planning under partial observability. The framework distills planning strategies from a computationally intensive optimal control expert into a student policy regularized toward safety and dynamic requirements via a dual-objective loss function. To mitigate partial observability, a spatiotemporal perception fusion module using a Temporal Convolutional Network (TCN) is developed to integrate historical depth images and odometry. This module infers task-relevant latent context from historical observations, enhancing spatial awareness beyond the instantaneous FOV without maintaining persistent map memory. A trajectory parameterization layer mapping network outputs to a structured trajectory, while enabling explicit continuity, dynamic-consistency, and obstacle soft penalties during training, encouraging constraint satisfaction for unseen observations without formal guarantees. Simulations on quadrotor and fixed-wing aircraft demonstrate that PILOT achieves performance comparable to the privileged expert while reducing computational overhead by over 80\\%. Successful indoor and outdoor zero-shot deployment confirms the practical feasibility and cross-domain generalization of the planner.",
  "published": "2026-08-14",
  "updated": "2026-08-14",
  "year": "2026",
  "authors": [
   "Qingrui Zhang",
   "Feng Xue",
   "Xiang Zhou",
   "Chenghao Yu"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "PILOT, a constraint-aware privileged imitation learning framework for vision-based end-to-end UAV motion planning under partial observability, is presented, demonstrating the practical feasibility and cross-domain generalization of the planner.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Qingrui Zhang",
    "id": "2257340534",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Feng Xue",
    "id": "2343832646",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Xiang Zhou",
    "id": "2310287243",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Chenghao Yu",
    "id": "2291081992",
    "h_index": 1,
    "papers": 4
   }
  ],
  "comment": "13 Pages, 12 figures",
  "topics": [
   "imitation-diffusion",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.14082v1",
  "pdf_url": "https://arxiv.org/pdf/2608.14082v1",
  "html_url": "https://arxiv.org/html/2608.14082v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.14049",
  "slug": "flatlab-a-unified-methodology-framework-and-simulation-based-benchmark",
  "title": "FlatLab: A Unified Methodology Framework and Simulation-Based Benchmark for Robotic Manipulation of Flat Objects",
  "abstract": "Robotic manipulation of flat objects is challenging due to the ungraspable configurations and strong variations in object geometry and material. Existing methods rely on heuristic pre-manipulation and are often evaluated in closed settings with limited generalization. We propose a unified framework that decouples the manipulation into a strategy generator and an action execution module. The strategy generator predicts appropriate manipulation strategies from object point clouds by learning strategy-centric, object-invariant representations via simulated data transformation and contrastive learning. Conditioned on the predicted strategy, the execution module decomposes long-horizon manipulation into reusable action primitives and dynamically composes them to generate stable trajectories. To enable systematic evaluation, we introduce FlatLab, a comprehensive simulation benchmark for robotic flat object manipulation. FlatLab provides high-fidelity physical simulation of diverse rigid and deformable flat objects, automated multi-modal data collection, and standardized task definitions and evaluation protocols. Experiments conducted in FlatLab demonstrate that our approach generalizes effectively to unseen objects and categories, outperforming existing baselines. The project page and the code are provided at https://flatlab-web.github.io/.",
  "published": "2026-08-14",
  "updated": "2026-08-14",
  "year": "2026",
  "authors": [
   "Xingyu Zhu",
   "Wenshuo Han",
   "Zhouyu Wang",
   "Yuran Wang",
   "Ruihai Wu",
   "Hao Dong",
   "Fan Tang",
   "Hechang Chen",
   "Hyung Jin Chang",
   "Yixing Gao"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "ICML 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A unified framework that decouples the manipulation into a strategy generator and an action execution module, which generalizes effectively to unseen objects and categories, outperforming existing baselines is proposed.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xin-Hai Zhu",
    "id": "2145145096",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Wenshuo Han",
    "id": "2391494474",
    "h_index": 1,
    "papers": 9
   },
   {
    "name": "Zhou Wang",
    "id": "2338315868",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Yuran Wang",
    "id": "2349738743",
    "h_index": 5,
    "papers": 19
   },
   {
    "name": "Ruihai Wu",
    "id": "2382450653",
    "h_index": 3,
    "papers": 13
   },
   {
    "name": "Hao Dong",
    "id": "2239243357",
    "h_index": 9,
    "papers": 23
   },
   {
    "name": "Fan Tang",
    "id": "2318387317",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Hechang Chen",
    "id": "2338514750",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Hyung Jin Chang",
    "id": "2331764742",
    "h_index": 2,
    "papers": 14
   },
   {
    "name": "Yixing Gao",
    "id": "2393152292",
    "h_index": 2,
    "papers": 10
   }
  ],
  "comment": "This paper is accepted to ICML 2026",
  "topics": [
   "dexterous-manipulation",
   "sim2real",
   "spatial-3d",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.14049v1",
  "pdf_url": "https://arxiv.org/pdf/2608.14049v1",
  "html_url": "https://arxiv.org/html/2608.14049v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.14047",
  "slug": "evolve-vision-language-action-model-into-an-agent-with-on-the-fly-tool",
  "title": "Evolve Vision-Language-Action Model into an Agent with On-the-fly Tool-use",
  "abstract": "This paper integrates end-to-end Visual-Language-Action (VLA) models with agentic tool-use to propose Agentic Robot with Tool-use (ART). ART is a tool-injection framework that tunes any VLA model to leverage off-the-shelf tool modules for low-level vision, high-level affordance, and embodiment enhancement. Compared to vanilla VLA models with a whole continuous action solution space, ART reduces the complexity of the action solution space through tool-use, which not only improves generalizability across different tasks but also reduces data dependency. To demonstrate the advantages (high generalizability and low data dependency) of this framework, we first built a dataset of 30K tool-use trajectories and action demonstrations, which is much smaller than those used by baseline methods. We then designed a training regimen for long-trajectory tool-use reasoning in challenging environments. Experiments show that ART achieves a 20% higher success rate than mainstream baselines on simulation and real-world tasks, such as pick-and-place in the dark at novel viewpoints. Empirical results highlight the benefits of an agent-based approach: modular tool utilization enables more efficient training, lightweight deployment, and scalable integration of new tools. This design fosters robustness, adaptability, and extensibility, paving the way for the practical deployment of VLA systems in complex real-world scenarios.",
  "published": "2026-08-14",
  "updated": "2026-08-14",
  "year": "2026",
  "authors": [
   "Yi Ding",
   "Yanzhao Yu",
   "Xili Dai",
   "Xianbiao Qi",
   "Peiwen Sun",
   "Xueqian Wang",
   "Xiangyu Yue",
   "Jianan Wang"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "CVPR",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "ART is a tool-injection framework that tunes any VLA model to leverage off-the-shelf tool modules for low-level vision, high-level affordance, and embodiment enhancement, and ART reduces the complexity of the action solution space through tool-use, which improves generalizability across different tasks but also reduces data dependency.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yi Ding",
    "id": "2385454191",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Yanzhao Yu",
    "id": "2374089960",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Xili Dai",
    "id": "2327939064",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Xianbiao Qi",
    "id": "2259005278",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Peiwen Sun",
    "id": "2298400700",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Xueqian Wang",
    "id": "2337851514",
    "h_index": 1,
    "papers": 18
   },
   {
    "name": "Xiangyu Yue",
    "id": "2364030943",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Jianan Wang",
    "id": "2396383245",
    "h_index": 4,
    "papers": 15
   }
  ],
  "comment": "12 pages, 4 figures, Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern (CVPR) Findings",
  "topics": [
   "vla",
   "spatial-3d",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.14047v1",
  "pdf_url": "https://arxiv.org/pdf/2608.14047v1",
  "html_url": "https://arxiv.org/html/2608.14047v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.14031",
  "slug": "demonstration-of-space-robot-teleoperation-over-a-lossy-and-delayed-ne",
  "title": "Demonstration of Space Robot Teleoperation over a Lossy and Delayed Network using ATMOS",
  "abstract": "We present a demonstration showcasing the Autonomy Testbed for Multi-purpose Orbiting Systems (ATMOS), a planar spacecraft-analog robot designed for hardware-in-the-loop evaluation of guidance and control strategies in microgravity-like conditions. Using ATMOS as the physical test platform, we investigate the design, analysis, and performance evaluation of control architectures for remotely operated spacecraft under round-trip communication delays. In this work, we develop and experimentally validate a control strategy that combines state prediction and trajectory tracking control to perform a docking maneuver, accounting for time-varying random communication latency between ground operators and the ATMOS system. The demonstration includes a long-distance remote control experiment between Seoul and Stockholm, introducing realistic intercontinental delays and variability. The results highlight the capability of ATMOS to support rapid, reliable, and cost-effective testing of spacecraft teleoperation concepts, establishing a first step toward robust validation of on-orbit operations in microgravity-like environments.",
  "published": "2026-08-14",
  "updated": "2026-08-14",
  "year": "2026",
  "authors": [
   "Inkyu Jang",
   "Gregorio Marchesini",
   "Nicola De Carli",
   "Byeongjun Kim",
   "Sunwoo Hwang",
   "Dabin Kim",
   "Elias Krantz",
   "Youngkyoung Kong",
   "Frank J. Jiang",
   "Annika Wong",
   "Pedro Roque",
   "Prasetyo W. L. Sanjaya",
   "Nicola Bastianello",
   "Mani H. Dhullipalla",
   "Karl H. Johansson",
   "Hyungbo Shim",
   "Dimos V. Dimarogonas",
   "H. Jin Kim"
  ],
  "author_count": 18,
  "categories": [
   "cs.RO",
   "eess.SY",
   "math.OC"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Inkyu Jang",
    "id": "30566259",
    "h_index": 12,
    "papers": 36
   },
   {
    "name": "Gregorio Marchesini",
    "id": "2279925700",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "N. D. Carli",
    "id": "2145193862",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Byeongjun Kim",
    "id": "2220652656",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Sunwoo Hwang",
    "id": "2297047787",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Dabin Kim",
    "id": "1853578028",
    "h_index": 6,
    "papers": 18
   },
   {
    "name": "Elias Krantz",
    "id": "2342503487",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Youngkyoung Kong",
    "id": "2406218838",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Frank J. Jiang",
    "id": "144797863",
    "h_index": 9,
    "papers": 20
   },
   {
    "name": "Annika Wong",
    "id": "2374194368",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Pedro Roque",
    "id": "47232594",
    "h_index": 7,
    "papers": 24
   },
   {
    "name": "Prasetyo W. L. Sanjaya",
    "id": "2065763160",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Nicola Bastianello",
    "id": "2279914854",
    "h_index": 5,
    "papers": 31
   },
   {
    "name": "Mani H. Dhullipalla",
    "id": "35725924",
    "h_index": 5,
    "papers": 20
   },
   {
    "name": "K. H. Johansson",
    "id": "2267967270",
    "h_index": 4,
    "papers": 17
   },
   {
    "name": "Hyun-Seung Shim",
    "id": "121284699",
    "h_index": 6,
    "papers": 47
   },
   {
    "name": "Dimos V. Dimarogonas",
    "id": "1722151",
    "h_index": 65,
    "papers": 593
   },
   {
    "name": "H. Kim",
    "id": "49717577",
    "h_index": 27,
    "papers": 158
   }
  ],
  "comment": "(c) 2026 the authors. This work has been accepted to IFAC for publication under a Creative Commons License CC-BY-NC-ND. 6 pages, 8 figures. Inkyu Jang and Gregorio Marchesini contributed equally to this work",
  "topics": [
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.14031v1",
  "pdf_url": "https://arxiv.org/pdf/2608.14031v1",
  "html_url": "https://arxiv.org/html/2608.14031v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.14028",
  "slug": "advdex-learning-dexterous-manipulation-from-human-demonstrations-via-j",
  "title": "AdvDex: Learning Dexterous Manipulation from Human Demonstrations via Joint-Aligned Actions and Adversarial Learning",
  "abstract": "Dexterous manipulation is a fundamental capability for embodied intelligence, but scaling it remains difficult because robot demonstrations are expensive to collect and action spaces vary across embodiments. Policies trained on heterogeneous data can also entangle task-relevant visual cues with embodiment-specific appearance, limiting cross-embodiment generalization. We present AdvDex, a unified Vision-Language-Action framework for learning dexterous manipulation from human and robot demonstrations. First, we introduce OmniShare, a large-scale multimodal dataset of human manipulation demonstrations that provides high-quality kinematic supervision and tactile measurements while reducing reliance on robot teleoperation. Second, we propose the Joint-Aligned Action Space (JAAS), a canonical action representation comprising an $\\mathrm{SE}(3)$ wrist pose and 15 finger joints, thereby functionally aligning human hands, dexterous robot hands, and parallel grippers. Finally, we use domain-adversarial learning to reduce embodiment-specific information in the learned visual representation. Experiments on hand-action prediction and real-world dexterous manipulation show consistent improvements over baselines, effective zero-shot human-to-robot skill transfer, generalization to unseen objects and environments, and data-efficient few-shot adaptation.",
  "published": "2026-08-14",
  "updated": "2026-08-14",
  "year": "2026",
  "authors": [
   "Zhiyue Zhao",
   "Jingyi Wu",
   "Hairuo Liu",
   "Mingyu Liu",
   "Liyang Li",
   "Hengdi Zhang",
   "Tong He",
   "Zhengxue Cheng"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "AdvDex is presented, a unified Vision-Language-Action framework for learning dexterous manipulation from human and robot demonstrations and the Joint-Aligned Action Space (JAAS) is proposed, a canonical action representation comprising an $\\mathrm{SE}(3)$ wrist pose and 15 finger joints, thereby functionally aligning human hands, dexterous robot hands, and parallel grippers.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhiyue Zhao",
    "id": "2291027362",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Jing Wu",
    "id": "2456549735",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Hairuo Liu",
    "id": "2364015850",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Mingyu Liu",
    "id": "2363320659",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Liyang Li",
    "id": "2342997738",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Hengdi Zhang",
    "id": "2363321787",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Tong He",
    "id": "2325202989",
    "h_index": 7,
    "papers": 12
   },
   {
    "name": "Zhengxue Cheng",
    "id": "2310296090",
    "h_index": 9,
    "papers": 63
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "dexterous-manipulation",
   "egocentric-data",
   "tactile",
   "foundation-pretraining",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.14028v1",
  "pdf_url": "https://arxiv.org/pdf/2608.14028v1",
  "html_url": "https://arxiv.org/html/2608.14028v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.13924",
  "slug": "bicpo-vla-behavior-identified-continuation-preference-optimization-for",
  "title": "BICPO-VLA: Behavior-Identified Continuation Preference Optimization for Smooth Asynchronous Vision-Language-Action Control",
  "abstract": "The request-to-handoff gap has three coupled sources: ambiguity about the behavior intended at request time, physical-state drift accumulated during action generation, and residual incompatibility when the new action finally assumes control. BICPO-VLA addresses them in sequence. First, an instruction-aware causal history encoder identifies the behavior supported by the command and current task progress. Second, sequential Haar subspace generation decomposes each action chunk into complementary pairwise scaffold and residual coefficients, enabling two specialized generation stages followed by exact reconstruction. By reducing iterative refinement in the original action space, it shortens the interval over which the robot continues moving before the new chunk becomes available. Finally, BICPO rolls the known outgoing actions to the actual handoff state and applies reference-relative Flow-DPO among behaviorally matched candidates, adapting the generated chunk to the remaining request-to-handoff mismatch without changing its intended behavior.",
  "published": "2026-08-14",
  "updated": "2026-08-14",
  "year": "2026",
  "authors": [
   "Ming Shang",
   "Yuchen Huang",
   "Jiaoyang Chen",
   "Haoyuan Hu",
   "Han Yu",
   "Liping Song",
   "Luyun Feng",
   "Shuo Bao",
   "Wei Dong",
   "Xinzhou Wang",
   "Fuchun Sun"
  ],
  "author_count": 11,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "BICPO-VLA addresses the request-to-handoff gap by reducing iterative refinement in the original action space and applies reference-relative Flow-DPO among behaviorally matched candidates, adapting the generated chunk to the remaining request-to-handoff mismatch without changing its intended behavior.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ming Shang",
    "id": "2456858887",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Yuchen Huang",
    "id": "2456309998",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Jiaoyang Chen",
    "id": "2458021115",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Haoyuan Hu",
    "id": "2458019663",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Hangzheng Yu",
    "id": "2453956532",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Liping Song",
    "id": "2458027860",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Luyun Feng",
    "id": "2458019995",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Shuo Bao",
    "id": "2456859199",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Wei Dong",
    "id": "2290852686",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Xinzhou Wang",
    "id": "2196924058",
    "h_index": 11,
    "papers": 21
   },
   {
    "name": "Fuchun Sun",
    "id": "2242091789",
    "h_index": 9,
    "papers": 26
   }
  ],
  "comment": "9 pages,4 figures",
  "topics": [
   "vla"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.13924v1",
  "pdf_url": "https://arxiv.org/pdf/2608.13924v1",
  "html_url": "https://arxiv.org/html/2608.13924v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.13923",
  "slug": "openbelief-nav-evidence-preserving-object-memory-for-open-vocabulary-l",
  "title": "OpenBelief-Nav: Evidence-Preserving Object Memory for Open-Vocabulary Language-Guided Navigation",
  "abstract": "Open-vocabulary 3D scene graphs provide compact semantic memory for language-guided navigation, but mapped objects are often exposed through a single fused feature or committed semantic label. Such commitment can remove minority yet task-relevant hypotheses from the task-time interface. We present OpenBelief-Nav, an evidence-preserving object memory that retains observation-level phrases, reliability cues, and frame-mask provenance while maintaining separate aggregate geometric and visual representations. Semantically related phrases are consolidated into a vocabulary-independent object belief from which task-specific readouts perform fixed-vocabulary projection or free-form retrieval. On five ScanNet200 and eight Replica scenes, full-belief projection achieves mIoU scores of 0.2742 and 0.2912, compared with 0.2393 and 0.2701 for a matched early-commit readout. Across 78 HM3D-YCB navigation trials, consensus and early-commit retrieval each achieve 60/78 successes, compared with 58/78 for belief-weighted retrieval and 55/78 for DualMap. Across 20 Unitree G1 runs organized as 10 matched evaluation cases, a correction policy permitting at most two verified candidate attempts improves target-confirmation success from 6/10 to 8/10 relative to top-1-only execution. Code will be released upon acceptance at https://openbelief-nav.github.io/.",
  "published": "2026-08-14",
  "updated": "2026-08-14",
  "year": "2026",
  "authors": [
   "Dinh Tuan Nguyen",
   "Anh Dao",
   "Phuong Nam Dang",
   "Quan-Dung Pham",
   "Tuyen P. Le",
   "Truong Nguyen",
   "Quan Nguyen"
  ],
  "author_count": 7,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "OpenBelief-Nav is presented, an evidence-preserving object memory that retains observation-level phrases, reliability cues, and frame-mask provenance while maintaining separate aggregate geometric and visual representations.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Dinh Tuan Nguyen",
    "id": "2454104313",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Anh Dao",
    "id": "2345925754",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Phuong Nam Dang",
    "id": "2457312846",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Q. Ph\u1ea1m",
    "id": "2287378142",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "T. P. Le",
    "id": "51304815",
    "h_index": 6,
    "papers": 14
   },
   {
    "name": "Truong Q. Nguyen",
    "id": "2332465538",
    "h_index": 3,
    "papers": 15
   },
   {
    "name": "Quan Nguyen",
    "id": "2313283640",
    "h_index": 2,
    "papers": 6
   }
  ],
  "comment": "",
  "topics": [
   "navigation"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2608.13923v1",
  "pdf_url": "https://arxiv.org/pdf/2608.13923v1",
  "html_url": "https://arxiv.org/html/2608.13923v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2608.13904",
  "slug": "communication-in-modular-robotic-motor-control-bilateral-controllers-u",
  "title": "Communication in modular robotic motor control: Bilateral controllers under realistic constraints",
  "abstract": "Robotic motor control in musculoskeletal systems requires fast, accurate movement and robust postural stabilization under signal-dependent noise (where motor command variance scales with command magnitude) and energetic cost. Modular controllers can distribute these competing demands across interacting submodules, but it remains unclear whether they outperform monolithic architectures under realistic constraints, and how inter-module communication shapes the resulting strategy. Inspired by the bilateral hemispheric organization of the brain, we introduce a recurrent controller of two GRU-based modules connected by a learnable, delayed inter-hemispheric channel, trained end-to-end in a differentiable two-arm musculoskeletal simulator. Across reaching and holding tasks, the modular architecture substantially outperforms a capacity-matched monolithic baseline. Compared to a matched modular controller without communication, learned inter-hemispheric communication reshapes the solution: improved endpoint precision, lower energetic cost in non-zero-delay regimes, and reduced muscle co-contraction. Our findings show that for robotics, biologically inspired modular controllers offer a practical route to robust movement under noise and energetic constraints, with inter-module communication providing a mechanism to tune trade-offs between precision, stability, and actuation cost.",
  "published": "2026-08-14",
  "updated": "2026-08-14",
  "year": "2026",
  "authors": [
   "Jingwen Li",
   "Levin Kuhlmann",
   "Jason Friedman",
   "Gideon Kowadlo"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work introduces a recurrent controller of two GRU-based modules connected by a learnable, delayed inter-hemispheric channel, trained end-to-end in a differentiable two-arm musculoskeletal simulator, and shows that the modular architecture substantially outperforms a capacity-matched monolithic baseline.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jingwen Li",
    "id": "2457051796",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "L. Kuhlmann",
    "id": "89617810",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "J. Friedman",
    "id": "143697294",
    "h_index": 19,
    "papers": 80
   },
   {
    "name": "Gideon Kowadlo",
    "id": "2545985",
    "h_index": 8,
    "papers": 36
   }
  ],
  "comment": "15 pages, 8 figures",
  "topics": [
   "sim2real"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.13904v1",
  "pdf_url": "https://arxiv.org/pdf/2608.13904v1",
  "html_url": "https://arxiv.org/html/2608.13904v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.13901",
  "slug": "ontology-grounded-world-models-for-failure-diagnosis-and-closed-loop-r",
  "title": "Ontology-Grounded World Models for Failure Diagnosis and Closed-Loop Repair in Physical AI Systems",
  "abstract": "EV-WM represents candidate quality with feature and event scores, but these scores do not explicitly record an unmet task predicate, a route label for an available correction mechanism, or a post-correction acceptance result. We present Onto-EV-WM, an ontology-grounded diagnosis and verification-gated correction interface layered above EV-WM rather than a replacement world-model architecture. The implemented task-local TBox defines entity types, predicate signatures, and constraints; source-specific grounding maps predicted or simulator-observed states to task ABoxes; and deterministic rules retain each missing predicate and its arguments when assigning a route label. Learned or heuristic proposers remain separate from this symbolic interface; native task predicates determine acceptance, and the bounded protocol determines whether a failed verification is retried. In the aligned PointMaze evaluation, EV-WM and Onto-EV-WM both report 94% success, with mean final-state distances of 0.90573 and 0.61177, respectively; the separately budgeted search reaches 100% success. On LIBERO-Goal, the ontology represents failed task conditions as typed records, retains their predicate arguments, and associates them with the declared source/joint correction route and predicate-gated acceptance; the complete configuration reports 93.8% corrected-window success on seed 0 and 94.05 +- 0.30% across four evaluation-sampling seeds. On the fixed 10,030-task LIBERO-Plus registry, Onto-EV-WM succeeds on 8,526 tasks (85.00%), with suite-level success rates of 65.98% for LIBERO-10, 91.39% for LIBERO-Goal, and 91.38% for both LIBERO-Object and LIBERO-Spatial. These numbers report the performance of the complete ontology-grounded configurations under the tested simulator protocols; an ontology-only causal share is not measured separately, and real-robot recovery is not evaluated.",
  "published": "2026-08-14",
  "updated": "2026-08-14",
  "year": "2026",
  "authors": [
   "Kailin Wang",
   "Haoxiang Jie",
   "Yaoyuan Yan",
   "Jiacheng Zhou",
   "Zhiyou Heng"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Onto-EV-WM is presented, an ontology-grounded diagnosis and verification-gated correction interface layered above EV-WM rather than a replacement world-model architecture, an ontology-grounded diagnosis and verification-gated correction interface layered above EV-WM.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Kailin Wang",
    "id": "2307181803",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Haoxiang Jie",
    "id": "2375917830",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Yaoyuan Yan",
    "id": "2376120413",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Jiachen Zhou",
    "id": "2332594891",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Zhiyou Heng",
    "id": "2430642793",
    "h_index": 0,
    "papers": 3
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "sim2real",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.13901v1",
  "pdf_url": "https://arxiv.org/pdf/2608.13901v1",
  "html_url": "https://arxiv.org/html/2608.13901v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.13878",
  "slug": "knowledge-data-dual-driven-reinforcement-learning-for-autonomous-vehic",
  "title": "Knowledge-Data-Dual-Driven Reinforcement Learning for Autonomous Vehicle Control in Mixed Traffic",
  "abstract": "In mixed traffic, decision-making for autonomous vehicles (AVs) confronts three interrelated challenges. First, physics-based priors incorporated into reinforcement learning (RL) models fail to capture latent interactive vehicle intentions and diverse driver behaviors, limiting the proactive reasoning capabilities. Second, abrupt maneuvers by surrounding vehicles cause non-stationarity, leaving long-tail safety events under-explored. Third, hybrid action spaces destabilize unified RL training due to the different temporal scales of continuous car-following and discrete lane-changing maneuvers. To address these issues, we propose Knowledge-Data Dual-driven Reinforcement Learning (KDDRL). First, a conditional deep generative model synthesizes intention-aware future trajectories, converting passive perception into proactive predictive states. Second, a knowledge-data dual-driven paradigm operates on these predictive states, fusing probabilistic data-driven insights with physical constraints to guide safe exploration through safety-critical scenarios. Third, a coupling module compresses both intention-aware trajectories and physical constraints into compact shared embeddings. This unified representation enables asynchronous multi-timescale optimization of continuous car-following and discrete lane-changing while preserving mutual information. Evaluations on dataset-calibrated simulations demonstrate that KDDRL effectively handles intention uncertainty, accelerates training convergence, and outperforms conventional baseline methods in terms of safety, efficiency, and comfort.",
  "published": "2026-08-14",
  "updated": "2026-08-14",
  "year": "2026",
  "authors": [
   "Jie Fang",
   "Wei Zheng",
   "Mengyun Xu",
   "Eui-Jin Kim"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes Knowledge-Data Dual-driven Reinforcement Learning (KDDRL), a conditional deep generative model that effectively handles intention uncertainty, accelerates training convergence, and outperforms conventional baseline methods in terms of safety, efficiency, and comfort.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jie Fang",
    "id": "2150470772",
    "h_index": 7,
    "papers": 25
   },
   {
    "name": "Wei Zheng",
    "id": "2453426924",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Mengyun Xu",
    "id": "151142511",
    "h_index": 12,
    "papers": 33
   },
   {
    "name": "Eui-Jin Kim",
    "id": "2180042231",
    "h_index": 8,
    "papers": 28
   }
  ],
  "comment": "16 pages, 17 figures",
  "topics": [
   "rl-control",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.13878v1",
  "pdf_url": "https://arxiv.org/pdf/2608.13878v1",
  "html_url": "https://arxiv.org/html/2608.13878v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.13767",
  "slug": "simulation-aware-in-context-policy-improvement-for-llm-aided-analog-la",
  "title": "Simulation-Aware In-Context Policy Improvement for LLM-Aided Analog Layout Refinement",
  "abstract": "Analog IC layout design remains a labor-intensive iterative process dominated by simulation-driven refinement. Although end-to-end layout generators accelerate initial placement and routing, they still require experts to manually tune layout optimization parameters with repeated post-layout simulations for stringent design specifications. While Bayesian Optimization (BO) is widely adopted for parameter tuning in analog IC design, at the layout level it typically requires hundreds to thousands of evaluations, each involving costly parasitic extraction and post-layout simulation, which makes it impractical. Recently, Large Language Models (LLMs) have demonstrated potential in improving the sample efficiency of such simulation-driven tuning. However, their restricted access to geometric layout context and design-specific heuristics limits their ability to manipulate the layout optimization process. In this paper, we propose a simulation-aware LLM multi-agent framework that performs in-context policy improvement (ICPI) by iteratively updating layout optimization parameters exposed by an analog layout generator through an act-observe-reflect loop on compact structured layout representations. Experiments on real-world analog circuits show that, with only tens of post-layout simulations, our approach improves post-layout performance over the generator's built-in heuristics and BO-based tuning method.",
  "published": "2026-08-13",
  "updated": "2026-08-13",
  "year": "2026",
  "authors": [
   "Bingyang Liu",
   "Ziming Wei",
   "Xiaohan Gao",
   "David Z. Pan"
  ],
  "author_count": 4,
  "categories": [
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.AI",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper proposes a simulation-aware LLM multi-agent framework that performs in-context policy improvement (ICPI) by iteratively updating layout optimization parameters exposed by an analog layout generator through an act-observe-reflect loop on compact structured layout representations.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Bingyang Liu",
    "id": "2308487124",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Zi-Yu Wei",
    "id": "2381314790",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Xiaohan Gao",
    "id": "2149395383",
    "h_index": 6,
    "papers": 24
   },
   {
    "name": "David Z. Pan",
    "id": "2261280159",
    "h_index": 2,
    "papers": 5
   }
  ],
  "comment": "7 pages, 3 figures. To appear in the Proceedings of the 2026 International Conference on LLM-Aided Design (ICLAD 2026)",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.13767v1",
  "pdf_url": "https://arxiv.org/pdf/2608.13767v1",
  "html_url": "https://arxiv.org/html/2608.13767v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.13723",
  "slug": "graph-mambanav-spatial-temporal-graph-mamba-leveraging-object-relation",
  "title": "Graph-MambaNav: Spatial-Temporal Graph Mamba Leveraging Object-Relation Knowledge for Object-Goal Navigation",
  "abstract": "Object-goal navigation requires an agent to reason over object relationships and prioritize target-relevant objects for efficient decision making in unseen environments. While existing graph-based methods incorporate target-awareness at the feature or attention level, they remain permutation-invariant and lack an explicit mechanism to control information propagation order, limiting their ability to model target-dependent importance and long-range dependencies. In contrast, Graph-Mamba highlights that node prioritization through sequence ordering is critical for effective global reasoning. In this work, we investigate the node prioritization mechanism in Graph-Mamba and study its role in object navigation. We propose Graph-MambaNav, a target-aware spatial-temporal graph encoding framework that introduces a heuristic ordering over objects based on their relevance to the target, allowing more informative objects to be processed later to aggregate richer context. Both node ordering and edge weights are initialized from LLM-derived commonsense object relationships, providing a unified prior for structured reasoning. A spatial module integrates local message passing with global GraphMamba-based selective scanning, while a temporal module applies Mamba-based sequence modeling over object-wise temporal orders, allowing selective aggregation of historical context for long-range temporal reasoning. Experiments on AI2-THOR and RoboTHOR demonstrate improved navigation performance with generalization, and additional real-world robot deployment further validates the effectiveness of our proposed approach.",
  "published": "2026-08-13",
  "updated": "2026-08-13",
  "year": "2026",
  "authors": [
   "Leyuan Sun",
   "Genxin Chen",
   "Linwei Ye",
   "Yan Zhang",
   "Xi Kan",
   "Yanfei Sun"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA 2027",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes Graph-MambaNav, a target-aware spatial-temporal graph encoding framework that introduces a heuristic ordering over objects based on their relevance to the target, allowing more informative objects to be processed later to aggregate richer context.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Leyuan Sun",
    "id": "2242489741",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Genxin Chen",
    "id": "2119120309",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Linwei Ye",
    "id": "2457984828",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yan Zhang",
    "id": "2455002389",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Xi Kan",
    "id": "2457976330",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yanfei Sun",
    "id": "2275899481",
    "h_index": 5,
    "papers": 24
   }
  ],
  "comment": "Accepted by IEEE Robotics and Automation Letters (IEEE RA-L), will transfer to 2027 IEEE International Conference on Robotics & Automation (ICRA)",
  "topics": [
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.13723v1",
  "pdf_url": "https://arxiv.org/pdf/2608.13723v1",
  "html_url": "https://arxiv.org/html/2608.13723v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.13719",
  "slug": "coverage-aware-active-evaluation-for-failure-discovery-with-paired-sys",
  "title": "Coverage Aware Active Evaluation for Failure Discovery with Paired Systems",
  "abstract": "Autonomous systems can fail in rare and heterogeneous ways, making real-world failure discovery difficult under limited testing budgets. Although cheaper proxies such as simulators, lower-fidelity systems, or related policies can be sampled extensively to find failures, proxy failures often do not transfer to the real world due to sim-to-real and system-to-system gaps. The key challenge is therefore to effectively leverage proxy system information for accurate prediction of severe target system failures. We propose an adaptive failure discovery method that combines proxy evaluations with limited target system results to guide scenario selection for target system testing. Our method learns a local predictor of target risk by correcting proxy failure signals using control-variate-inspired residual modeling. To find failures that are both likely and diverse, we combine this predictor with a support-aware mutual-information objective that favors realistic, well-supported regions while expanding coverage across failure modes. Across autonomous driving, manipulation, and quadruped velocity-tracking tasks, our method discovers up to 2$\\times$ as many failures as random sampling and active-learning baselines, including severe and diverse failures missed by competing methods.",
  "published": "2026-08-13",
  "updated": "2026-08-13",
  "year": "2026",
  "authors": [
   "Anjali Parashar",
   "Rachel Luo",
   "Apoorva Sharma",
   "Sushant Veer",
   "Edward Schmerling",
   "Carson Sobolewski",
   "Mingxin Yu",
   "Chuchu Fan",
   "Marco Pavone"
  ],
  "author_count": 9,
  "categories": [
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.AI",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes an adaptive failure discovery method that combines proxy evaluations with limited target system results to guide scenario selection for target system testing, and learns a local predictor of target risk by correcting proxy failure signals using control-variate-inspired residual modeling.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Anjali Parashar",
    "id": "2294872048",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Rachel Luo",
    "id": "2371071202",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Apoorva Sharma",
    "id": "2109540240",
    "h_index": 12,
    "papers": 40
   },
   {
    "name": "Sushant Veer",
    "id": "1491176908",
    "h_index": 11,
    "papers": 35
   },
   {
    "name": "E. Schmerling",
    "id": "1868195",
    "h_index": 25,
    "papers": 65
   },
   {
    "name": "Carson Sobolewski",
    "id": "2233087939",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Mingxin Yu",
    "id": "2294504091",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Chuchu Fan",
    "id": "2409631965",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Marco Pavone",
    "id": "2257966119",
    "h_index": 3,
    "papers": 10
   }
  ],
  "comment": "9 main pages followed by Appendix, total 21 pages, 12 figures",
  "topics": [
   "humanoids",
   "sim2real",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.13719v1",
  "pdf_url": "https://arxiv.org/pdf/2608.13719v1",
  "html_url": "https://arxiv.org/html/2608.13719v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.13678",
  "slug": "hint-2-hierarchical-world-models-for-inference-time-temporal-logic-gui",
  "title": "hint$^2$: Hierarchical World Models for Inference-Time Temporal Logic Guidance",
  "abstract": "A central goal of robot learning is to enable robots to execute rich instructions specified at runtime. Large-scale language-conditioned policies have made substantial progress toward this goal, yet still struggle with temporal structure and safety constraints. Linear Temporal Logic (LTL) provides a powerful language to express complex, non-Markovian instructions. However, guiding learned manipulation policies toward LTL satisfaction remains challenging because modern policies generate short-horizon action chunks and replan in closed loop, while almost all LTL specifications are evaluated over long-horizon trajectories. In this paper, we introduce hint$^2$, a method for guiding short-horizon policies toward satisfying complex LTL specifications at inference time using hierarchical world models. Our key idea is to derive two separate guidance objectives using each world model's abstraction level. A high-level model predicts future action-induced transitions in task-relevant atomic propositions to guide progress through the LTL automaton, while a low-level dynamics model predicts immediate state evolution for accurate local safety guidance. Our results show that hint$^2$ overcomes the limitations of current LTL-guided diffusion methods, outperforms existing inference-time steering methods in CALVIN, and successfully completes instructions with complex liveness and safety constraints more elegantly than language-conditioned alternatives. Finally, we demonstrate that hint$^2$ can handle complex instructions on a real UR5e manipulator.",
  "published": "2026-08-13",
  "updated": "2026-08-13",
  "year": "2026",
  "authors": [
   "Moritz Zoellner",
   "Anastasios Manganaris",
   "Ahmed H. Qureshi",
   "Rohan Paleja"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper introduces hint, a method for guiding short-horizon policies toward satisfying complex LTL specifications at inference time using hierarchical world models, and shows that hint$^2$ can handle complex instructions on a real UR5e manipulator.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Moritz Zoellner",
    "id": "77800408",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Anastasios Manganaris",
    "id": "2384129065",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "A. H. Qureshi",
    "id": "51014775",
    "h_index": 13,
    "papers": 55
   },
   {
    "name": "R. Paleja",
    "id": "83862731",
    "h_index": 13,
    "papers": 49
   }
  ],
  "comment": "Videos available on our project page: https://anonymous-hint2.github.io/",
  "topics": [
   "world-models",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.13678v1",
  "pdf_url": "https://arxiv.org/pdf/2608.13678v1",
  "html_url": "https://arxiv.org/html/2608.13678v1",
  "code_url": "https://anonymous-hint2.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2608.13625",
  "slug": "reward-machines-for-signal-temporal-logic",
  "title": "Reward Machines for Signal Temporal Logic",
  "abstract": "Signal temporal logic (STL) provides a formal language for specifying real-time properties of real-valued observations, along with a quantitative robustness score for monitoring satisfaction. Control synthesis from STL specifications is of interest since manual controller design becomes infeasible as real-world systems grow in complexity. Moreover, many modern autonomous and AI-enabled systems lack accurate and complete system models, which makes optimization-based synthesis approaches unsuitable and motivates learning-based control. Prior work uses STL robustness scores as rewards in reinforcement learning (RL) to obtain control policies satisfying given specifications; however, robustness depends on execution history, leading to intractable state space expansion for general long-horizon specifications with arbitrarily nested temporal operators. This work introduces a novel automata-based approach that provides an efficient memory mechanism and associated Markovian rewards suitable for RL frameworks. Our approach constructs a timed alternating automaton from the given STL specifications, augments the state space with automaton locations and clock valuations, and derives rewards from the automaton acceptance condition. We empirically demonstrate that our approach learns policies that achieve higher robustness scores and satisfaction rates than those learned by existing approaches using robustness-based rewards.",
  "published": "2026-08-13",
  "updated": "2026-08-13",
  "year": "2026",
  "authors": [
   "Alper Kamil Bozkurt",
   "Shangtong Zhang",
   "Yuichi Motai"
  ],
  "author_count": 3,
  "categories": [
   "cs.AI",
   "cs.LG",
   "cs.RO"
  ],
  "primary_category": "cs.AI",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work introduces a novel automata-based approach that provides an efficient memory mechanism and associated Markovian rewards suitable for RL frameworks and empirically demonstrates that this approach learns policies that achieve higher robustness scores and satisfaction rates than those learned by existing approaches using robustness-based rewards.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "A. Bozkurt",
    "id": "144229259",
    "h_index": 6,
    "papers": 20
   },
   {
    "name": "Shangtong Zhang",
    "id": "2350229245",
    "h_index": 4,
    "papers": 15
   },
   {
    "name": "Yuichi Motai",
    "id": "1731507",
    "h_index": 22,
    "papers": 103
   }
  ],
  "comment": "",
  "topics": [
   "rl-control",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.13625v1",
  "pdf_url": "https://arxiv.org/pdf/2608.13625v1",
  "html_url": "https://arxiv.org/html/2608.13625v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.13555",
  "slug": "humantracker-towards-comprehensive-and-human-aligned-motion-tracking-b",
  "title": "HumanTracker: Towards Comprehensive and Human-Aligned Motion Tracking Benchmark",
  "abstract": "Humanoid motion tracking is central to teleoperation and whole-body imitation, yet evaluation often disagrees with what people perceive in videos. Kinematic errors average per-frame pose differences but miss the physical artifacts that matter most, particularly unstable support and incorrect contacts such as foot skating and mistimed touch-downs. Meanwhile, widely used test suites are small and lack the diversity needed to stress contact-rich, long-horizon behaviors. We introduce HumanTracker to make humanoid tracking evaluation both perceptually aligned and scalable. The HumanTracker benchmark contains approximately 153 hours of optical motion trajectories from multiple professional performers, organized into four motion families with text labels for fine-grained diagnosis. We further propose HumanScore, a preference-aligned metric trained on 12K motion pairs containing 24K motions. Across representative state-of-the-art trackers, HumanScore better predicts human preferences and reveals contact and stability failures that kinematic metrics often miss.",
  "published": "2026-08-13",
  "updated": "2026-08-13",
  "year": "2026",
  "authors": [
   "Dairu Liu",
   "Zekun Qi",
   "Jiayu Zeng",
   "Ruixi Yu",
   "Yu Guan",
   "Yintianrun Zhang",
   "Xuchuan Chen",
   "Sikai Liang",
   "Zekai Li",
   "Chenghuai Lin",
   "Xinqiang Yu",
   "Wenyao Zhang",
   "He Wang",
   "Li Yi"
  ],
  "author_count": 14,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "ECCV 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work introduces HumanTracker, a preference-aligned metric trained on 12K motion pairs containing 24K motions that better predicts human preferences and reveals contact and stability failures that kinematic metrics often miss.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Dai-En Liu",
    "id": "2301193049",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Zekun Qi",
    "id": "2385150703",
    "h_index": 4,
    "papers": 14
   },
   {
    "name": "Jiayu Zeng",
    "id": "2323331422",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Ruixin Yu",
    "id": "2446733099",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yuxuan Guan",
    "id": "2346789227",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Yintianrun Zhang",
    "id": "2440685250",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Xu-Chuan Chen",
    "id": "2265920777",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Sikai Liang",
    "id": "2362195622",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Zekai Li",
    "id": "2457303785",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Chenghua Lin",
    "id": "2402902166",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Xinqiang Yu",
    "id": "2328936584",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Wenyao Zhang",
    "id": "2282545418",
    "h_index": 10,
    "papers": 26
   },
   {
    "name": "He Wang",
    "id": "2336953381",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Li Yi",
    "id": "2242612318",
    "h_index": 6,
    "papers": 9
   }
  ],
  "comment": "Accepted to ECCV 2026",
  "topics": [
   "humanoids",
   "tactile",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.13555v1",
  "pdf_url": "https://arxiv.org/pdf/2608.13555v1",
  "html_url": "https://arxiv.org/html/2608.13555v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.13511",
  "slug": "a-browser-native-digital-test-range-for-benchmarking-4d-ocean-glider-p",
  "title": "A Browser-Native Digital Test Range for Benchmarking 4D Ocean-Glider Planning Algorithms",
  "abstract": "Repeated in-situ evaluation of ocean-glider planners requires scarce vehicles, operators, deployment and recovery resources, and ocean conditions that cannot be reset for competing algorithms. We present a guided, installation-free browser-native digital test range that transforms a selected region into a reproducible four-dimensional experiment. The system leads users from regional domain selection through mission-scoped bathymetry, time/depth forcing, science objectives, optional task decomposition, route specification, current-advected execution, observation generation, and scoring. Its primary contribution is a common plan-to-observation contract unifying vehicle, sensing, and evaluator assumptions across manual routes, transparent built-in algorithms, and imported classical or learned-planner outputs, while exported artifacts form dataset-ready records. A controlled Observing System Simulation Experiment (OSSE) evaluates five classical planners in two episodes, three deterministic seeds, and a calibrated 60-hour horizon. All 54 missions completed and recovered without hard violations, while planner rankings and dive-policy effects revealed operational-scientific tradeoffs. An authentic public deployment supplied a field-referenced audit to scope current kinematic boundaries. Separately, source-locked GliderFlight 1.2.0 achieved native-to-browser parity through Pyodide/WebAssembly, establishing a pathway for high-fidelity multi-tier simulation. The resulting operational space is scientifically traceable and component-qualified for mission-scale pre-deployment experimentation.",
  "published": "2026-08-13",
  "updated": "2026-08-13",
  "year": "2026",
  "authors": [
   "Edward Holmberg",
   "Elias Ioup",
   "Mahdi Abdelguerfi"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A guided, installation-free browser-native digital test range that transforms a selected region into a reproducible four-dimensional experiment that is scientifically traceable and component-qualified for mission-scale pre-deployment experimentation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "E. Holmberg",
    "id": "153460612",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Elias Ioup",
    "id": "2456955",
    "h_index": 11,
    "papers": 67
   },
   {
    "name": "Mahdi Abdelguerfi",
    "id": "2273493200",
    "h_index": 2,
    "papers": 12
   }
  ],
  "comment": "6 pages, 5 figures",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.13511v1",
  "pdf_url": "https://arxiv.org/pdf/2608.13511v1",
  "html_url": "https://arxiv.org/html/2608.13511v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.13489",
  "slug": "dreamx-phi-1-0-action-conditioned-video-world-model-for-robotic-manipu",
  "title": "DreamX-Phi 1.0: Action-Conditioned Video World Model for Robotic Manipulation",
  "abstract": "We present \\textbf{DreamX-Phi 1.0}, an action-conditioned video world model for robotic manipulation that, given an observed frame, a language instruction, and a prescribed action sequence comprising end-effector poses and gripper states, predicts the resulting future observations. Yet realism alone does not guarantee faithfulness: a convincing rollout can still move the wrong arm or lose the manipulated object. To ensure the prediction respects each arm's commanded path, we inject per-arm $\\mathrm{SE}(3)$ transformations into attention via \\textbf{PRoPE-style geometric encoding}, preserving arm identity and rigid-motion structure. Action control alone does not fully constrain scene geometry or the evolution of small manipulated objects. We therefore add a lightweight \\textbf{depth branch} for scene-level geometry and use \\textbf{SAM3 masks} with a frozen \\textbf{V-JEPA teacher} to maintain object consistency throughout grasping. We further distill the multi-step generator into a few-step student via distribution-matching distillation for efficient deployment. At the time of writing, \\model{} achieves first place on Track~1 and second place on Track~2 of the WorldArena~2.0 Challenge. Our model and code will be publicly available.",
  "published": "2026-08-13",
  "updated": "2026-08-13",
  "year": "2026",
  "authors": [
   " DreamX Team",
   "Rui Chen",
   "Xiangxiang Chu",
   "Geng Li",
   "Jifan Li",
   "Qingfeng Shi",
   "Datao Tang",
   "Jing Tang",
   "Jun Wang",
   "Pengfei Zhang"
  ],
  "author_count": 10,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "An action-conditioned video world model for robotic manipulation that, given an observed frame, a language instruction, and a prescribed action sequence comprising end-effector poses and gripper states, predicts the resulting future observations is presented.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Dream Team",
    "id": "108490083",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Rui Chen",
    "id": "2355395424",
    "h_index": 6,
    "papers": 16
   },
   {
    "name": "Xiangxiang Chu",
    "id": "27628828",
    "h_index": 34,
    "papers": 142
   },
   {
    "name": "Ge Li",
    "id": "2449361148",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Jifa Li",
    "id": "2430086554",
    "h_index": 0,
    "papers": 8
   },
   {
    "name": "Qingfeng Shi",
    "id": "2359010719",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Datao Tang",
    "id": "2292198559",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Jing Tang",
    "id": "2355902327",
    "h_index": 5,
    "papers": 18
   },
   {
    "name": "Jun Wang",
    "id": "2340512740",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Pengfei Zhang",
    "id": "2335548116",
    "h_index": 3,
    "papers": 13
   }
  ],
  "comment": "Code: https://github.com/AMAP-ML/DreamX-Phi",
  "topics": [
   "world-models",
   "dexterous-manipulation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.13489v1",
  "pdf_url": "https://arxiv.org/pdf/2608.13489v1",
  "html_url": "https://arxiv.org/html/2608.13489v1",
  "code_url": "https://github.com/AMAP-ML/DreamX-Phi",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2608.13474",
  "slug": "decoding-task-progress-from-vla-representations",
  "title": "Decoding Task Progress from VLA Representations",
  "abstract": "Vision-language-action models (VLAs) are moving rapidly towards deployment as general-purpose manipulation policies, but we currently lack basic tools for understanding what these models represent internally or for monitoring them at runtime. Leveraging ideas from mechanistic interpretability, we probe the residual stream of $\u03c0_{0.5}$ and find that task progress, the normalized time remaining in a trajectory, is linearly readable from the activations. We find that this signal is present in the pretrained PaliGemma backbone prior to training on any robot-specific data. A single linear probe generalizes to unseen tasks and varies under language counterfactuals when trained on multi-prompt data, but does not enable meaningful steering of the policy. These properties make the signal directly useful for instrumenting deployed VLAs. We use the probe as a simple label-free OOD detector, which detects stalled task progress, and find it competitive with state-of-the-art methods. Our results suggest that VLAs have rich, linearly readable internal representations of semantic quantities like task progress, and that learning to read these signals offers a lightweight, interpretable path toward monitoring deployed visuomotor policies.",
  "published": "2026-08-13",
  "updated": "2026-08-13",
  "year": "2026",
  "authors": [
   "Atiksh Bhardwaj",
   "Edward Weiyi Duan",
   "Prithwish Dan",
   "Wei-Chiu Ma",
   "Preston Culbertson"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The results suggest that VLAs have rich, linearly readable internal representations of semantic quantities like task progress, and that learning to read these signals offers a lightweight, interpretable path toward monitoring deployed visuomotor policies.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Atiksh Bhardwaj",
    "id": "2261084253",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "E. W. Duan",
    "id": "2360358263",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Prithwish Dan",
    "id": "2228665381",
    "h_index": 7,
    "papers": 9
   },
   {
    "name": "Wei-Chiu Ma",
    "id": "2367353706",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Preston Culbertson",
    "id": "2386019755",
    "h_index": 1,
    "papers": 7
   }
  ],
  "comment": "",
  "topics": [
   "vla"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.13474v1",
  "pdf_url": "https://arxiv.org/pdf/2608.13474v1",
  "html_url": "https://arxiv.org/html/2608.13474v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.13448",
  "slug": "mind-the-context-continual-learning-of-socially-appropriate-robot-acti",
  "title": "Mind the Context: Continual Learning of Socially Appropriate Robot Actions via Environmental-Social Disentanglement",
  "abstract": "Social robots are expected to operate across diverse environments, where similar arrangements can imply different socially appropriate actions, e.g., starting a conversation may be acceptable in a crowded home but disruptive in an office meeting. Because such norms and environments cannot all be anticipated in advance, robots require continual learning (CL) to adapt from sequential experience while retaining previously acquired knowledge. Prior work has studied CL for generating socially appropriate robot actions, but it has not addressed domain-incremental settings in which the robot incrementally encounters diverse contexts (e.g., living room, meeting room, office, hallway), where both environmental (e.g., whether the space is open or cluttered with furniture) and social cues (e.g., how people or other agents are positioned around the robot) jointly shape the appropriateness of robot actions. We address this gap with the Explicit Disentanglement Dual-Branch (EDD) framework. EDD explicitly separates environmental and social-agent related knowledge and uses replay-based rehearsal to mitigate forgetting while learning the appropriateness of robot actions (e.g., cleaning, serving, starting a conversation) across several indoor domains. Experiments show that EDD outperforms several state-of-the-art baselines, and ablation studies further evaluate different disentanglement strategies and the sensitivity to domain ordering. Our code is publicly available at https://github.com/Cambridge-AFAR/Mind-the-Context.git.",
  "published": "2026-08-13",
  "updated": "2026-08-13",
  "year": "2026",
  "authors": [
   "Rafal Robert Karpinski",
   "Fethiye Irmak Dogan",
   "Nikhil Churamani",
   "Yiming Luo",
   "Maartje M. A. de Graaf",
   "Davide Dell'Anna",
   "Hatice Gunes"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Experiments show that EDD outperforms several state-of-the-art baselines, and ablation studies further evaluate different disentanglement strategies and the sensitivity to domain ordering.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Rafa\u0142 Karpi\u0144ski",
    "id": "2067165511",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Fethiye Irmak Do\u011fan",
    "id": "9074985",
    "h_index": 9,
    "papers": 42
   },
   {
    "name": "Nikhil Churamani",
    "id": "19175266",
    "h_index": 16,
    "papers": 37
   },
   {
    "name": "Yiming Luo",
    "id": "2453964682",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "M. D. Graaf",
    "id": "144232743",
    "h_index": 20,
    "papers": 46
   },
   {
    "name": "Davide Dell\u2019Anna",
    "id": "1405720262",
    "h_index": 9,
    "papers": 24
   },
   {
    "name": "Hatice Gunes",
    "id": "2290916149",
    "h_index": 3,
    "papers": 4
   }
  ],
  "comment": "Extended version of the paper accepted at the 2026 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS 2026)",
  "topics": [
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.13448v1",
  "pdf_url": "https://arxiv.org/pdf/2608.13448v1",
  "html_url": "https://arxiv.org/html/2608.13448v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.13438",
  "slug": "contactguard-pre-contact-execution-monitoring-with-action-conditioned",
  "title": "ContactGuard: Pre-Contact Execution Monitoring with Action-Conditioned Latent World Models",
  "abstract": "Contact-rich manipulation failures are often detected only after the robot has committed to contact. This is especially limiting in wrist-camera setups: close gripper--object views help observe contact, but a poor approach may already push, miss, slip, or disturb the object before conventional detectors react. We introduce \\emph{ContactGuard}, a pre-contact execution monitor for chunked visuomotor policies. Given the policy's planned action chunk, ContactGuard predicts its short-horizon consequence in latent visual space and aborts if the predicted future latent indicates likely failure. Its latent world model is trained from unlabelled robot trajectories to predict compact multi-view visual embeddings under planned actions, avoiding pixel-level video prediction. A lightweight failure probe is then trained from a small labelled set of pre-contact clips. At deployment, ContactGuard anchors prediction before an imminent contact event, rolls the model forward under the policy's own actions, and verifies the predicted post-contact latent. Across real-world contact-rich manipulation tasks, ContactGuard predicts failure more accurately than direct and corrupted-action ablations, and transfers to live robot as a pre-contact abort signal without modifying the underlying policy.",
  "published": "2026-08-13",
  "updated": "2026-08-13",
  "year": "2026",
  "authors": [
   "Gehan Zheng",
   "Matthew Johnson-Roberson",
   "Weiming Zhi"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Across real-world contact-rich manipulation tasks, ContactGuard predicts failure more accurately than direct and corrupted-action ablations, and transfers to live robot as a pre-contact abort signal without modifying the underlying policy.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Gehan Zheng",
    "id": "2404435747",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Matthew Johnson-Roberson",
    "id": "2278448710",
    "h_index": 9,
    "papers": 36
   },
   {
    "name": "Weiming Zhi",
    "id": "2238209684",
    "h_index": 9,
    "papers": 37
   }
  ],
  "comment": "14 pages, 5 figures, 8 tables",
  "topics": [
   "world-models",
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.13438v1",
  "pdf_url": "https://arxiv.org/pdf/2608.13438v1",
  "html_url": "https://arxiv.org/html/2608.13438v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.13422",
  "slug": "attention-from-action-for-action-emergent-visual-bottlenecks-for-polic",
  "title": "Attention from Action, for Action: Emergent Visual Bottlenecks for Policy Learning",
  "abstract": "Visual bottlenecks that focus policy inputs on regions of interest (ROIs) can improve data-efficient visuomotor learning by separating where to look from how to act. Many ROI interfaces rely on external spatial labels, such as gaze, object classes, or affordance annotations. Label-free alternatives often derive crops from trajectories by detecting gripper or motion events and centering a fixed crop at the projected end-effector. Such action-derived crops are useful spatial priors that require no additional labels, but they encode fixed choices about event timing, proxy points, and crop scale. When the visual evidence needed for control lies away from the end-effector or changes continuously with task progress, these crops can become misaligned. We propose Seeker, a task- and state-conditioned readout that learns attention from action. Starting from frozen DINOv3 features, Seeker iteratively updates a query with gathered visual evidence, producing progression-aware ROIs solely from action supervision. The learned ROI serves as a spatial interface for RGB cropping, mask-guided background augmentation, and point-cloud filtering. In simulation and the real world, Seeker improves data efficiency and robustness over no-crop, augmentation, and action-derived crop baselines. On real robots, Seeker raises average in-domain success from the best baseline's 48.3% to 76.7% and success under lighting/background shifts from 20.0% to 60.0%.",
  "published": "2026-08-13",
  "updated": "2026-08-13",
  "year": "2026",
  "authors": [
   "Zheyu Zhuang",
   "Ruiyu Wang",
   "Nick Heppert",
   "Johannes Fabian Hahn",
   "Abhinav Valada",
   "Florian T. Pokorny",
   "Danica Kragic"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Seeker is proposed, a task- and state-conditioned readout that learns attention from action, producing progression-aware ROIs solely from action supervision and serves as a spatial interface for RGB cropping, mask-guided background augmentation, and point-cloud filtering.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zheyu Zhuang",
    "id": "2323552853",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Ruiyu Wang",
    "id": "2255293781",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Nick Heppert",
    "id": "2164381280",
    "h_index": 7,
    "papers": 15
   },
   {
    "name": "Johannes Fabian Hahn",
    "id": "2457353378",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "A. Valada",
    "id": "2131132945",
    "h_index": 17,
    "papers": 86
   },
   {
    "name": "Florian T. Pokorny",
    "id": "1881469",
    "h_index": 25,
    "papers": 132
   },
   {
    "name": "Danica Kragic",
    "id": "2283846373",
    "h_index": 3,
    "papers": 15
   }
  ],
  "comment": "Code: https://github.com/zheyu-zhuang/seeker",
  "topics": [
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.13422v1",
  "pdf_url": "https://arxiv.org/pdf/2608.13422v1",
  "html_url": "https://arxiv.org/html/2608.13422v1",
  "code_url": "https://github.com/zheyu-zhuang/seeker",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2608.13415",
  "slug": "deliberate-practice-learning-robot-skills-under-a-budget",
  "title": "Deliberate Practice: Learning Robot Skills under a Budget",
  "abstract": "We consider the problem of autonomously learning robot skills under a limited practice budget for sequential tasks. We propose an active skill learning algorithm, \\emph{Deliberate Practice (DP)}, that computes a provably \\emph{budget-optimal} allocation---practicing skills that maximize expected cumulative reward while being learnable within the budget. DP estimates both the time needed to master skills and the cumulative reward of the task plans that the skills unlock. Computing a budget-optimal allocation is challenging as it requires reasoning about combinatorially many skill plans over a large practice budget. Our key contribution is a bilinear program that can compute this exactly using off-the-shelf solvers. Through simulated and real-world experiments on long-horizon manipulation tasks, we show that our approach allows robots to optimally use limited practice time to acquire useful policies and improve long-horizon planning.",
  "published": "2026-08-13",
  "updated": "2026-08-13",
  "year": "2026",
  "authors": [
   "Shivam Vats",
   "Sudarshan Harithas",
   "Mete Tuluhan Akbulut",
   "Arvind Raghunathan",
   "George Konidaris"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Through simulated and real-world experiments on long-horizon manipulation tasks, it is shown that the proposed active skill learning algorithm allows robots to optimally use limited practice time to acquire useful policies and improve long-horizon planning.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Shivam Vats",
    "id": "2139955295",
    "h_index": 6,
    "papers": 30
   },
   {
    "name": "Sudarshan S. Harithas",
    "id": "1703181521",
    "h_index": 4,
    "papers": 15
   },
   {
    "name": "M. Akbulut",
    "id": "2007758652",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "A. Raghunathan",
    "id": "144936495",
    "h_index": 30,
    "papers": 129
   },
   {
    "name": "G. Konidaris",
    "id": "1765407",
    "h_index": 49,
    "papers": 268
   }
  ],
  "comment": "16 pages including appendices",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.13415v1",
  "pdf_url": "https://arxiv.org/pdf/2608.13415v1",
  "html_url": "https://arxiv.org/html/2608.13415v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.13396",
  "slug": "capstan-driven-continuum-surgical-robot-design-modeling-and-perception",
  "title": "Capstan-driven Continuum Surgical Robot: Design, Modeling, and Perception",
  "abstract": "Shape and force sensing have long been critical bottlenecks in the development of compact capstan-driven continuum surgical robots, primarily due to the difficulty of obtaining cable tension information within the confined capstan assembly. To overcome these challenges, this paper presents an integrated design-modeling-sensing approach based on the concept of actuation-perception co-design. A compliant element is introduced into the motor mounting bracket of the drive system, enabling micro-deformation under the cable reaction force and thereby allowing real-time cable tension measurement without occupying the compact capstan space. To address the modeling complexity arising from unconventional joint configurations introduced by the spatial cable routing strategy, a parallel computation framework based on a multibody short-thick-beam model is proposed, which captures shear effects in short beam segments and synergistic multi-cable interactions while achieving real-time performance. Building on this framework, stable shape and force sensing is achieved by incorporating a proximal multi-axis force/torque sensor as an additional measurement anchor. Following this design-modeling-sensing framework, capstan-driven continuum surgical robots with single- and dual-segment configurations are developed. Experimental results validate the proposed framework in both single- and dual-segment continuum robots, demonstrating real-time tip pose estimation together with contact force and location perception. By enabling cable tension feedback without compromising the compact capstan architecture, the proposed framework makes integrated perception feasible for capstan-driven continuum surgical robots.",
  "published": "2026-08-13",
  "updated": "2026-08-13",
  "year": "2026",
  "authors": [
   "Gang Zhang",
   "Yufu Qiu",
   "Junyan Yan",
   "Wenhui Zeng",
   "Wenlong Lu",
   "Shing Shin Cheng"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "An integrated design-modeling-sensing approach based on the concept of actuation-perception co-design is presented, which makes integrated perception feasible for capstan-driven continuum surgical robots.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Gang Zhang",
    "id": "2160687184",
    "h_index": 9,
    "papers": 25
   },
   {
    "name": "Yufu Qiu",
    "id": "2218486941",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Junyan Yan",
    "id": "2041774793",
    "h_index": 10,
    "papers": 20
   },
   {
    "name": "Wenhui Zeng",
    "id": "2041791507",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Wenlong Lu",
    "id": "2304443272",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "S. Cheng",
    "id": "21594385",
    "h_index": 23,
    "papers": 88
   }
  ],
  "comment": "",
  "topics": [
   "tactile",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.13396v1",
  "pdf_url": "https://arxiv.org/pdf/2608.13396v1",
  "html_url": "https://arxiv.org/html/2608.13396v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.13395",
  "slug": "fire-vla-failure-informed-self-evolution-for-vision-language-action-mo",
  "title": "FIRE-VLA: Failure-Informed Self-Evolution for Vision-Language-Action Models in Autonomous Driving",
  "abstract": "Reinforcement learning improves autonomous-driving vision-language-action (VLA) models by evaluating trajectories sampled from the current policy. Group relative policy optimization (GRPO) learns from reward differences within each rollout group. When all sampled trajectories are poor, this relative signal can rank failures without identifying behavior outside the failed region. We introduce FIRE-VLA, a failure-informed self-evolution framework that converts such unresolved failures into privileged supervision for the next policy. Low-reward, low-diversity groups trigger self-distillation from a frozen round-start copy of the same model. Teacher and student have the same parameter scale, but only the teacher observes the hidden future trajectory. Supervision follows the student's generated prefix and is restricted to answer tokens, while GRPO remains active for every group. The updated policy supplies the teacher for the next round, allowing the routed failure distribution to change with the policy without requiring a larger external teacher. Starting from the same Qwen2.5-VL-3B SFT checkpoint, the comparison matches student rollout and policy-update counts. On 6,019 examples from 150 held-out nuScenes scenes, FIRE-VLA retains comparable single-sample planning, reduces G=4 mean L2 from 1.848 to 1.500 m, and lowers evaluation-persistent failure prevalence from 13.03% to 11.20%. The reduction in mean error arises mainly from rare severe rollouts rather than uniform improvement across ordinary trajectories.",
  "published": "2026-08-13",
  "updated": "2026-08-13",
  "year": "2026",
  "authors": [
   "Hao Dou"
  ],
  "author_count": 1,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Fire-VLA is introduced, a failure-informed self-evolution framework that converts unresolved failures into privileged supervision for the next policy in autonomous-driving vision-language-action models.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hao Dou",
    "id": "2298908307",
    "h_index": 4,
    "papers": 13
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "rl-control",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.13395v1",
  "pdf_url": "https://arxiv.org/pdf/2608.13395v1",
  "html_url": "https://arxiv.org/html/2608.13395v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.13362",
  "slug": "nestdex-nested-policy-learning-with-copilot-assisted-teleoperation-for",
  "title": "NestDex: Nested Policy Learning with Copilot Assisted Teleoperation for Dexterous Manipulation",
  "abstract": "Dexterous manipulation promises substantially richer robot interaction with the physical world, but learning these behaviours remains constrained by the difficulty of collecting consistent, complete-task demonstrations. Unlike parallel-jaw manipulation, dexterous tasks require the operator to coordinate arm motion with precise, contact-rich finger behaviour throughout the task. We introduce NestDex, a nested policy-learning framework that reduces this burden by using learned hand skills to assist demonstration collection. The operator controls the arm and regulates the active hand skill through a single-DoF clutch, rather than directly specifying the full finger trajectory. The inner hand policy adapts its motion from the latest proprioceptive history, while a vision-language selector activates the appropriate skill for each task stage. The resulting demonstrations train a separate outer visuomotor policy that controls both the arm and hand without the inner policies at deployment. A hand-action variational autoencoder provides compact hand-action targets while retaining arm commands in joint space. Across real-world dexterous manipulation experiments, NestDex improves demonstration reliability and efficiency, and the resulting empirical evaluations support effective autonomous policy learning. Video Demo are available at project website https://aus.bot/research/nestdex.",
  "published": "2026-08-13",
  "updated": "2026-08-13",
  "year": "2026",
  "authors": [
   "James Zhao",
   "Jinhe Tang",
   "Mingyuan Ba",
   "Weiming Zhi"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "NestDex, a nested policy-learning framework that reduces this burden by using learned hand skills to assist demonstration collection, improves demonstration reliability and efficiency, and the resulting empirical evaluations support effective autonomous policy learning.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "James Zhao",
    "id": "2454064283",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Jinhe Tang",
    "id": "2456604735",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Mingyuan Ba",
    "id": "2453997493",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Weiming Zhi",
    "id": "32184659",
    "h_index": 8,
    "papers": 28
   }
  ],
  "comment": "9 pages, 11 figures, 3 tables. Project website: https://aus.bot/research/nestdex",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.13362v1",
  "pdf_url": "https://arxiv.org/pdf/2608.13362v1",
  "html_url": "https://arxiv.org/html/2608.13362v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.13284",
  "slug": "predictive-relative-velocity-steering-for-safe-robotic-manipulator-tel",
  "title": "Predictive Relative-Velocity Steering for Safe Robotic Manipulator Teleoperation in Dynamic Environments",
  "abstract": "Recent advances in teleoperation have enabled robotic manipulators to perform dexterous, human-arm-like motions. However, human operators may fail to avoid suddenly appearing obstacles promptly and effectively, particularly under network latency or limited attention, thereby creating safety risks. To address this issue, we propose a lightweight and modular framework for proactive collision avoidance, operating directly at the end-effector velocity-command level. After preprocessing the point cloud, the framework first predicts potential collisions based on time-to-collision (TTC) with integrated overshoot protection, and subsequently rotates the relative-velocity vector using Rodrigues' rotation formula. The deflection changes only the direction of the relative velocity while preserving its magnitude, thereby mitigating the deadlock problem commonly encountered by conventional artificial potential field (APF) methods. The prediction module compensates for point-cloud processing latency introduced by complex teleoperation pipelines, while the lightweight design enables the high-frequency control required for teleoperation. Simulations across diverse scenarios show that the proposed method achieves a higher end-effector collision avoidance rate than the baseline methods. Experiments on a physical robotic system further validate its collision-avoidance effectiveness.",
  "published": "2026-08-13",
  "updated": "2026-08-13",
  "year": "2026",
  "authors": [
   "Changhao Hu",
   "Zeyi Liu",
   "Songqiao Hu",
   "Shuang Liu",
   "Zihan Meng",
   "Xiao He"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes a lightweight and modular framework for proactive collision avoidance, operating directly at the end-effector velocity-command level, and shows that the proposed method achieves a higher end-effector collision avoidance rate than the baseline methods.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Changhao Hu",
    "id": "2322832156",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Zeyi Liu",
    "id": "2142151507",
    "h_index": 13,
    "papers": 40
   },
   {
    "name": "Song Hu",
    "id": "2115190679",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Shuang Liu",
    "id": "2399074356",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Zihan Meng",
    "id": "2400494715",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Xiao He",
    "id": "2338540150",
    "h_index": 3,
    "papers": 14
   }
  ],
  "comment": "8 pages, 8 figures",
  "topics": [
   "dexterous-manipulation",
   "spatial-3d",
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.13284v1",
  "pdf_url": "https://arxiv.org/pdf/2608.13284v1",
  "html_url": "https://arxiv.org/html/2608.13284v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.13233",
  "slug": "manufacturing-complex-airtight-soft-pneumatic-actuators-for-soft-robot",
  "title": "Manufacturing Complex Airtight Soft Pneumatic Actuators for Soft Robotics: Process Evaluation and Optimization",
  "abstract": "Manufacturing complex soft pneumatic actuators remains challenging because geometric fidelity, compliance, structural integrity, and airtightness must be achieved simultaneously. This study presents a manufacturing-focused evaluation of several fabrication routes for complex pneumatic structures, including heat-shrink forming, silicone casting, powder- and liquid-based additive manufacturing, and fused deposition modeling (FDM). The processes were assessed through process screening, baseline fabrication, failure analysis, and process improvement to distinguish inherent process limitations from correctable manufacturing defects. Heat-shrink forming was limited by geometric conformity, casting by mold accessibility and bonded interfaces, powder-based methods by residual material trapped within enclosed passages, and digital light processing by the material properties and post-processing requirements of the investigated system. FDM provided the most adaptable route because its dominant defects could be progressively reduced through process optimization. The results further showed that airtightness depends not only on nominal wall thickness but also on extrusion-path architecture, while support-free geometry is important when access for internal post-processing is limited. These findings establish a practical design-for-manufacturing approach in which process selection is guided by the compatibility between actuator architecture and manufacturing constraints. The proposed approach provides practical guidance for developing complex, flexible, and airtight soft pneumatic actuators for soft robotic applications",
  "published": "2026-08-13",
  "updated": "2026-08-13",
  "year": "2026",
  "authors": [
   "Mohammed Abboodi"
  ],
  "author_count": 1,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Mohammed Abboodi",
    "id": "2291239323",
    "h_index": 2,
    "papers": 10
   }
  ],
  "comment": "",
  "topics": [
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.13233v1",
  "pdf_url": "https://arxiv.org/pdf/2608.13233v1",
  "html_url": "https://arxiv.org/html/2608.13233v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.13220",
  "slug": "fam-dq-a-dual-quadrotor-based-fully-actuated-aerial-manipulator-for-hi",
  "title": "FAM-DQ: A Dual-Quadrotor-Based Fully Actuated Aerial Manipulator for High-Torque Interaction",
  "abstract": "Aerial physical interaction requires aerial manipulation platforms to generate large interaction forces and torques while maintaining precise end-effector control. However, conventional underactuated aerial manipulators suffer from strong position-attitude coupling, whereas fully actuated platform designs often face structural complexity, limited payload capacity, and insufficient torque output. This paper presents FAM-DQ, a dual-quadrotor based fully actuated aerial manipulator designed for high-torque physical interaction tasks. By mounting two quadrotor propulsion modules at the ends of a central frame through passive joints, while using a gear-driven servo to regulate the pointing direction, FAM-DQ achieves decoupled $6$-DoF end-effector control with omnidirectional manipulation capability and enhanced torque output. Experiments including trajectory tracking, attitude tracking, static torque measurement, and screw driving validate the proposed design. FAM-DQ achieves a maximum torque of $1.019~\\mathrm{N}\\cdot\\mathrm{m}$ with a total mass of $0.447~\\mathrm{kg}$, corresponding to a torque-to-mass ratio of $2.28~\\mathrm{N}\\cdot\\mathrm{m/kg}$.",
  "published": "2026-08-13",
  "updated": "2026-08-13",
  "year": "2026",
  "authors": [
   "Xuwei Yang",
   "Ruoyu Ren",
   "Ziqian Guo"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xuwei Yang",
    "id": "2386875920",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Ruoyu Ren",
    "id": "2373971355",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Ziqian Guo",
    "id": "2311307279",
    "h_index": 1,
    "papers": 2
   }
  ],
  "comment": "7 pages, 7 figures",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.13220v1",
  "pdf_url": "https://arxiv.org/pdf/2608.13220v1",
  "html_url": "https://arxiv.org/html/2608.13220v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.13103",
  "slug": "s2-hwm-sparse-event-structured-hierarchical-world-model-for-long-horiz",
  "title": "S2-HWM: Sparse Event-Structured Hierarchical World Model for Long-Horizon Surgical Robot Manipulation",
  "abstract": "Long-horizon surgical robot manipulation is challenging because task rewards are sparse, while meaningful interaction changes occur at irregular intervals. Existing world-model agents typically imagine at primitive-step resolution, leaving variable-duration task progress implicit. Manually specified stages can provide intermediate structure, but their task specific boundaries are difficult to align with state-dependent interaction transitions. We propose S2-HWM, a Sparse Event-Structured Hierarchical World Model that learns sparse event evidence from primitive latent trajectories to coordinate an event-level manager and a primitive-step worker. The event evidence schedules manager goal updates, and each selected latent goal conditions the worker's primitive actions until the next update. The learned event evidence also forms variable-duration segments for an Event Transition Model (ETM), which predicts the next?boundary stochastic state, segment duration, and accumulated segment reward. Chaining these event-level predictions provides a variable-duration continuation beyond the primitive imagination horizon for manager learning, while the worker retains primitive-step actor-critic learning. On a SurRoL-based PegTransfer task, S2-HWM achieves a success rate of 98.7%, outperforming the flat GAS DreamerV3 baseline by 22.7 percentage points.",
  "published": "2026-08-13",
  "updated": "2026-08-13",
  "year": "2026",
  "authors": [
   "Shuzhe Zhang",
   "Xin Zhu",
   "Yinling Qian",
   "Qiong Wang"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "S2-HWM is proposed, a Sparse Event-Structured Hierarchical World Model that learns sparse event evidence from primitive latent trajectories to coordinate an event-level manager and a primitive-step worker and achieves a success rate of 98.7%, outperforming the flat GAS DreamerV3 baseline by 22.7 percentage points.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Shuzhe Zhang",
    "id": "2344609753",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Xin Zhu",
    "id": "2158762689",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Yinling Qian",
    "id": "2278482",
    "h_index": 10,
    "papers": 35
   },
   {
    "name": "Qiong Wang",
    "id": "2302475954",
    "h_index": 1,
    "papers": 7
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.13103v1",
  "pdf_url": "https://arxiv.org/pdf/2608.13103v1",
  "html_url": "https://arxiv.org/html/2608.13103v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.13095",
  "slug": "semantic-radiance-fields-as-simulators-for-spatial-reasoning-in-real-w",
  "title": "Semantic Radiance Fields as Simulators for Spatial Reasoning in Real-World Scenes",
  "abstract": "Training and evaluating spatial reasoning in embodied agents requires diverse environments that are both geometrically faithful and semantically queryable. Synthetic simulators offer ground truth semantics but sacrifice realism; simulators based on reconstructions of real-world environments have realistic appearance but lack ground truth semantics by default. We propose using Semantic Radiance Fields (SRF) as simulators for spatial reasoning agents. SRFs are a representation that unifies these requirements by lifting 2D semantic segmentations from pretrained vision models into a 3D radiance field that jointly encodes geometry, appearance, and per-class semantic identity. The resulting fields are reconstructed from posed RGB captures of real scenes and support novel-view synthesis, semantic and free-space queries within a single grounded representation. This enables the efficient generation of diverse real-world environments to train and evaluate spatial reasoning models. As an example application, we outline an SRF-driven simulator for an orchard apple-reaching task, in which the radiance field supplies camera rendering, semantic ground truth, and occupancy queries to a physics engine.",
  "published": "2026-08-13",
  "updated": "2026-08-13",
  "year": "2026",
  "authors": [
   "Nico Heider",
   "Micha\u0142 Jan W\u0142odarczyk",
   "Katarzyna Wasielewska-Michniewska",
   "Przemys\u0142aw Ho\u0142da",
   "Martin Schieck",
   "Marcin Paprzycki",
   "Maria Ganzha",
   "Bogdan Franczyk"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Semantic Radiance Fields (SRF) are proposed as simulators for spatial reasoning agents by lifting 2D semantic segmentations from pretrained vision models into a 3D radiance field that jointly encodes geometry, appearance, and per-class semantic identity.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Nico Heider",
    "id": "2346981378",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Michal Jan Wlodarczyk",
    "id": "2367733159",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Katarzyna Wasielewska-Michniewska",
    "id": "2126081290",
    "h_index": 4,
    "papers": 27
   },
   {
    "name": "Przemyslaw Holda",
    "id": "2188266566",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Martin Schieck",
    "id": "1994480692",
    "h_index": 5,
    "papers": 15
   },
   {
    "name": "M. Paprzycki",
    "id": "1702211",
    "h_index": 31,
    "papers": 517
   },
   {
    "name": "M. Ganzha",
    "id": "1829313",
    "h_index": 22,
    "papers": 342
   },
   {
    "name": "Bogdan Franczyk",
    "id": "2287924423",
    "h_index": 1,
    "papers": 10
   }
  ],
  "comment": "Accepted at the IJCAI 2026 Workshop on Spatio-Temporal Reasoning and Learning (STRL), oral presentation",
  "topics": [
   "sim2real",
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.13095v1",
  "pdf_url": "https://arxiv.org/pdf/2608.13095v1",
  "html_url": "https://arxiv.org/html/2608.13095v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.13049",
  "slug": "h2r-bench-benchmarking-human-to-robot-manipulation-video-generation-in",
  "title": "H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models",
  "abstract": "Large-scale manipulation data is essential for robot learning, yet collecting robot demonstrations remains expensive and difficult to scale. Meanwhile, abundant egocentric human manipulation videos provide rich behavioral experiences, but transferring them across embodiments remains challenging due to differences between human hands and robotic end-effectors. Recent advances in video world models offer a promising pathway to synthesize robot-centric manipulation videos from human observations, while their cross-embodiment transfer capability remains largely unexplored. Therefore, we introduce H2R-Bench, a benchmark for evaluating cross-embodiment human-to-robot manipulation video generation, where models transform egocentric human demonstrations into robot manipulation videos under specified embodiments. Each benchmark instance contains a human demonstration video, target embodiment constraints, and source-grounded annotations covering task goals, action events, functional contacts, and object responses. H2R-Bench evaluates generated videos through five dimensions, including goal-state completion, action-event completion, functional contact transfer, embodiment correctness, and general video quality. We benchmark eleven state-of-the-art video generation models across six manipulation families and two robot embodiments. Our evaluation reveals that current video world models remain limited in human-to-robot manipulation transfer: even leading models often fail in embodiment consistency, functional interaction, and task execution. H2R-Bench provides a systematic diagnostic framework for evaluating whether video world models can bridge the human-to-robot embodiment gap and convert human manipulation observations into robot-centric training resources.",
  "published": "2026-08-13",
  "updated": "2026-08-13",
  "year": "2026",
  "authors": [
   "Dingyi Rong",
   "Yue Shi",
   "Chaofan Ma",
   "Jiezhang Cao",
   "Zongrui Wang",
   "Zeyu Zhang",
   "Yao Mu",
   "Guangtao Zhai",
   "Ning Liu"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "H2R-Bench provides a systematic diagnostic framework for evaluating whether video world models can bridge the human-to-robot embodiment gap and convert human manipulation observations into robot-centric training resources.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Dingyi Rong",
    "id": "2141237089",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Yue Shi",
    "id": "2424484434",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Chaofan Ma",
    "id": "2124034538",
    "h_index": 12,
    "papers": 28
   },
   {
    "name": "Jiezhang Cao",
    "id": "2331576735",
    "h_index": 4,
    "papers": 21
   },
   {
    "name": "Zongrui Wang",
    "id": "2258669883",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Zeyu Zhang",
    "id": "2383312645",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Yao Mu",
    "id": "2301922237",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Guangtao Zhai",
    "id": "2333365277",
    "h_index": 5,
    "papers": 27
   },
   {
    "name": "Ning Liu",
    "id": "2315813494",
    "h_index": 6,
    "papers": 11
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "egocentric-data",
   "foundation-pretraining",
   "data-teleop",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.13049v1",
  "pdf_url": "https://arxiv.org/pdf/2608.13049v1",
  "html_url": "https://arxiv.org/html/2608.13049v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2608.13028",
  "slug": "rgb-d-video-generation-for-improving-human-to-robot-object-handover-pr",
  "title": "RGB-D Video Generation for Improving Human-to-Robot Object Handover Prediction",
  "abstract": "Human-to-robot (H2R) object handover is a fundamental capability for human-robot collaboration, yet progress is hindered by the scarcity of large-scale, human-centric datasets and the significant sim-to-real gap. To address these challenges, we introduce Hand2Bot, an RGB-D video dataset that provides rich contextual information such as body posture and facial expressions, specifically collected for handover scenarios with real-world noise patterns. We further propose PassGen, a generative pipeline that leverages stable video diffusion and an Intention-Aware Temporal Face Encoder to synthesize realistic handover sequences while ensuring hand-object consistency. To bridge the sim-to-real gap, we implement a morphology-based depth editing strategy that replicates realistic sensor noise found in physical depth maps. Experimental evaluations demonstrate that our framework achieves high intention identification accuracy and low false trigger rates in both ablation studies and real-world deployment on a physical robot platform. Our results confirm that training on PassGen allows for robust zero-shot transfer and earlier intention anticipation compared to traditional hand-centric baselines, effectively enabling socially aware robotic behavior in shared workspaces.",
  "published": "2026-08-13",
  "updated": "2026-08-13",
  "year": "2026",
  "authors": [
   "Tianyu Sun",
   "Zhoujie Fu",
   "Zihui Gao",
   "Bang Zhang",
   "Guosheng Lin"
  ],
  "author_count": 5,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work introduces Hand2Bot, an RGB-D video dataset that provides rich contextual information such as body posture and facial expressions, specifically collected for handover scenarios with real-world noise patterns, and proposes PassGen, a generative pipeline that leverages stable video diffusion and an Intention-Aware Temporal Face Encoder to synthesize realistic handover sequences while ensuring hand-object consistency.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Tianyu Sun",
    "id": "2364093224",
    "h_index": 4,
    "papers": 22
   },
   {
    "name": "Zhoujie Fu",
    "id": "2157046216",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Zihui Gao",
    "id": "2335520158",
    "h_index": 2,
    "papers": 11
   },
   {
    "name": "Bang Zhang",
    "id": "2155711503",
    "h_index": 9,
    "papers": 13
   },
   {
    "name": "Guosheng Lin",
    "id": "2268646778",
    "h_index": 8,
    "papers": 18
   }
  ],
  "comment": "",
  "topics": [
   "sim2real",
   "hardware-codesign",
   "hri",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.13028v1",
  "pdf_url": "https://arxiv.org/pdf/2608.13028v1",
  "html_url": "https://arxiv.org/html/2608.13028v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.13026",
  "slug": "temporal-grpo-beyond-trajectory-level-credit-in-vision-language-action",
  "title": "Temporal GRPO: Beyond Trajectory-Level Credit in Vision-Language-Action Reinforcement Learning",
  "abstract": "Outcome-driven reinforcement learning offers a scalable way to post-train vision-language-action (VLA) policies from sparse task-success feedback. In common GRPO-based VLA post-training, one rollout-level advantage is applied to every action in the trajectory. A rollout that completes several valid stages but fails later can therefore penalize the actions that produced its earlier progress. We call this trajectory-level credit aliasing. Temporal GRPO addresses this problem by constructing detectable task stages, aligning each rollout with stage-specific action intervals, and comparing only rollouts that have entered the same stage. The resulting stage advantages are applied to their corresponding intervals in a single policy update. On RoboTwin 2.0, Temporal GRPO improves task success and sample efficiency, with consistent gains across task horizons. Controlled updates on LIBERO-Long preserve shared prerequisite stages and concentrate improvement at the first stage where rollout outcomes diverge.",
  "published": "2026-08-13",
  "updated": "2026-08-13",
  "year": "2026",
  "authors": [
   "Yao Zhou",
   "Hang Gao",
   "Fengge Wu",
   "Changwen Zheng",
   "Wenwen Qiang"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Temporal GRPO addresses the problem of trajectory-level credit aliasing in post-train VLA policies by constructing detectable task stages, aligning each rollout with stage-specific action intervals, and comparing only rollouts that have entered the same stage.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yao Zhou",
    "id": "2331946343",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Hang Gao",
    "id": "2168532506",
    "h_index": 3,
    "papers": 14
   },
   {
    "name": "Fengge Wu",
    "id": "2257430420",
    "h_index": 4,
    "papers": 32
   },
   {
    "name": "Changwen Zheng",
    "id": "2153619515",
    "h_index": 14,
    "papers": 54
   },
   {
    "name": "Wenwen Qiang",
    "id": "2059455684",
    "h_index": 15,
    "papers": 99
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.13026v1",
  "pdf_url": "https://arxiv.org/pdf/2608.13026v1",
  "html_url": "https://arxiv.org/html/2608.13026v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.13014",
  "slug": "egophi-estimating-contact-and-force-from-egocentric-vision",
  "title": "EgoPHI: Estimating Contact and Force from Egocentric Vision",
  "abstract": "Understanding hand-object interaction from egocentric vision is essential for modeling how people physically engage with the surrounding world. Yet reasoning about physically grounded interaction requires estimating the forces acting on hands and objects, beyond localizing contact. We present EgoPHI, the first method that jointly estimates dense contact maps and 3D force distributions on hand and object meshes from a single monocular RGB image and object geometry. To address the lack of scalable ground-truth force annotations, we introduce a physics-based simulation pipeline that augments existing hand-object datasets with dense per-vertex force supervision. EgoPHI then learns dense 3D contact and force on interacting hand and articulated object meshes, extending vision-based force estimation beyond image-space or planar settings. Our evaluation on in-distribution and out-of-distribution benchmarks shows that EgoPHI improves force estimation over existing approaches while generalizing to unseen datasets. To evaluate sim-to-real transfer, we constructed two physical objects that capture dense object contact and force magnitude and used them to record a dataset of interactions from eight participants across diverse touch and grasp types. Our results demonstrate that EgoPHI recovers meaningful 3D contact and force distributions in simulated, out-of-distribution, and real-world settings, advancing egocentric hand-object understanding from contact localization toward physically grounded interaction reasoning.",
  "published": "2026-08-13",
  "updated": "2026-08-13",
  "year": "2026",
  "authors": [
   "Andela Ilic",
   "Rachel Schuchert",
   "Yijing Jiang",
   "Christian Holz"
  ],
  "author_count": 4,
  "categories": [
   "cs.CV",
   "cs.GR",
   "cs.HC",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "ECCV 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work presents EgoPHI, the first method that jointly estimates dense contact maps and 3D force distributions on hand and object meshes from a single monocular RGB image and object geometry, and demonstrates that EgoPHI improves force estimation over existing approaches while generalizing to unseen datasets.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Andela Ilic",
    "id": "2367528625",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Rachel Schuchert",
    "id": "2450251189",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yijing Jiang",
    "id": "2314822075",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Christian Holz",
    "id": "2319602671",
    "h_index": 6,
    "papers": 19
   }
  ],
  "comment": "Accepted by ECCV 2026",
  "topics": [
   "dexterous-manipulation",
   "egocentric-data",
   "sim2real",
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.13014v1",
  "pdf_url": "https://arxiv.org/pdf/2608.13014v1",
  "html_url": "https://arxiv.org/html/2608.13014v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.12917",
  "slug": "towards-socially-compliant-navigation-in-deep-reinforcement-learning-v",
  "title": "Towards Socially Compliant Navigation in Deep Reinforcement Learning via Proxemics-Based Reward Modeling",
  "abstract": "Developing effective robot navigation methods in crowded environments is essential for real-world applications. Although recent deep reinforcement learning (DRL) methods have improved navigation performance in crowded environments, they often focus primarily on task-centric objectives and underrepresent social compliance objectives. In this paper, we introduce a novel proxemics-based reward formulation for DRL social navigation that provides a dense, interpretable social learning signal while maintaining navigation efficiency. Our approach models each human's personal space as a radial Gaussian-mixture field derived from Hall's proxemics theory and computes a robot-centric local cost over the robot's field of view. We integrate the proposed reward into established DRL navigation methods and evaluate it in simulation across multiple crowd scenarios, reward baselines, and crowd densities using both navigation metrics and social metrics. Results show that the proposed reward consistently improves social metrics in simulation while maintaining competitive navigation performance relative to the compared reward models.",
  "published": "2026-08-13",
  "updated": "2026-08-13",
  "year": "2026",
  "authors": [
   "Takieddine Soualhi",
   "Jacques Saraydaryan",
   "Laetitia Matignon"
  ],
  "author_count": 3,
  "categories": [
   "cs.LG",
   "cs.RO"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A novel proxemics-based reward formulation for DRL social navigation that provides a dense, interpretable social learning signal while maintaining navigation efficiency while maintaining competitive navigation performance relative to the compared reward models is introduced.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Takieddine Soualhi",
    "id": "2313115465",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Jacques Saraydaryan",
    "id": "2206281",
    "h_index": 7,
    "papers": 30
   },
   {
    "name": "L. Matignon",
    "id": "2335305",
    "h_index": 16,
    "papers": 57
   }
  ],
  "comment": "",
  "topics": [
   "rl-control",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.12917v1",
  "pdf_url": "https://arxiv.org/pdf/2608.12917v1",
  "html_url": "https://arxiv.org/html/2608.12917v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.12866",
  "slug": "amr-pose-an-active-led-marker-based-relative-pose-estimation-framework",
  "title": "AMR-Pose: An Active LED Marker-Based Relative Pose Estimation Framework With Probabilistic Switching PnP for Cooperative AUVs",
  "abstract": "Reliable relative pose estimation between autonomous underwater vehicles (AUVs) is critical for cooperative ocean exploration, sampling, and multi-robot coordination. However, achieving robust vision-based relative localization in underwater environments remains challenging due to severe optical degradation, including turbidity, illumination variations, reflections, and intermittent feature occlusions. This paper presents AMR-Pose, an active LED marker-based relative pose estimation framework for cooperative AUVs. A compact marker module consisting of one red central LED and three blue peripheral LEDs is developed and integrated onto the leader AUV to provide distinctive visual features under complex underwater conditions. Building upon the detected marker observations, a probabilistic switching Perspective-n-Point estimator (PSwPnP) is developed by combining Lie-group pose propagation on $SE(3)$, probabilistic marker association, and visibility-adaptive measurement fusion for robust six-degree-of-freedom relative pose estimation. The proposed framework dynamically adapts the estimation process according to marker visibility, maintaining geometric consistency and temporal stability during partial observations and visibility transitions. Extensive water-tank experiments with motion-capture ground truth validate that AMR-Pose achieves accurate, smooth, and robust relative pose estimation under challenging underwater conditions. Closed-loop leader-follower experiments further demonstrate its feasibility for real-time relative pose feedback in cooperative underwater robotics.",
  "published": "2026-08-13",
  "updated": "2026-08-13",
  "year": "2026",
  "authors": [
   "Zeyu Sha",
   "Xiaorui Wang",
   "Mingyang Yang",
   "Feitian Zhang"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zeyu Sha",
    "id": "2219274329",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Xiaorui Wang",
    "id": "2299281668",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Mingyang Yang",
    "id": "2278970077",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Feitian Zhang",
    "id": "2278836182",
    "h_index": 3,
    "papers": 7
   }
  ],
  "comment": "",
  "topics": [
   "navigation",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.12866v1",
  "pdf_url": "https://arxiv.org/pdf/2608.12866v1",
  "html_url": "https://arxiv.org/html/2608.12866v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.12860",
  "slug": "humanoidvln-a-physics-grounded-simulator-and-benchmark-for-vision-lang",
  "title": "HumanoidVLN: A Physics-Grounded Simulator and Benchmark for Vision-Language Navigation Across Diverse Humanoid Embodiments",
  "abstract": "Vision-Language Navigation (VLN) for humanoid robots poses challenges existing benchmarks fail to address: bipedal locomotion imposes physical constraints absent from wheeled agents, humanoid morphologies vary across platforms, and egocentric observations are distorted by locomotion-induced camera dynamics. We present HumanoidVLN, a physics-grounded simulator and benchmark for VLN across diverse humanoid embodiments. Built on NVIDIA Isaac Sim, our platform supports an extensible set of humanoid configurations, demonstrated on four robots (Unitree G1, Unitree H1, Internal-A, Internal-B) spanning 10-12 lower-body DoF and heights from 1.17m to 1.80m, via a hierarchical control stack combining a reinforcement learning locomotion policy with interchangeable PD or MPC path trackers. New robots and VLN models integrate with minimal effort; we demonstrate compatibility with NaVILA, DualVLN, StreamVLN, and JanusVLN. Environments are drawn from artist-designed scenes and 3D Gaussian Splatting reconstructions, filtered for navigable areas exceeding 100 square meters. Instructions are generated by a dual generator-reviewer plus paraphraser multi-agent pipeline with human-in-the-loop verification, yielding 933 collision-aware reference episodes, each paired with one fine-grained instruction and three coarse-grained stylistic variants (formal, natural, casual). Across four models and four embodiments, JanusVLN achieves the highest mean success rate of 43.55% and nDTW of 48.38. In a 20-episode sim-to-real pilot with DualVLN and the Unitree G1, navigation errors correlate strongly (r=0.935), with a mean absolute difference of 0.68m and mean trajectory similarity of 0.782 (+/-0.188) nDTW. These results highlight the interaction between VLN models, controllers, and humanoid embodiments under physical execution. Code, benchmark, and data will be released upon acceptance at https://humanoid-vln.github.io/.",
  "published": "2026-08-13",
  "updated": "2026-08-13",
  "year": "2026",
  "authors": [
   "Quan-Dung Pham",
   "Anh Dao",
   "The-Anh Nguyen",
   "Minh Nguyen-Dinh",
   "Phuong Nam Dang",
   "Tri Pham",
   "Hung Tran",
   "Bach Dao",
   "Tuyen P. Le",
   "Truong Nguyen",
   "Quan Nguyen"
  ],
  "author_count": 11,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "HumanoidVLN, a physics-grounded simulator and benchmark for VLN across diverse humanoid embodiments, and compatibility with NaVILA, DualVLN, StreamVLN, and JanusVLN are presented.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Q. Ph\u1ea1m",
    "id": "2287378142",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Anh Dao",
    "id": "2345925754",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "T. Nguy\u1ec5n",
    "id": "2363880617",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Minh Nguyen-Dinh",
    "id": "2457316783",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Phuong Nam Dang",
    "id": "2457312846",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "T. Ph\u1ea1m",
    "id": "2265361003",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Hung Tran",
    "id": "1657559403",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Bach H Dao",
    "id": "2176846323",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "T. P. Le",
    "id": "51304815",
    "h_index": 6,
    "papers": 14
   },
   {
    "name": "Truong Q. Nguyen",
    "id": "2332465538",
    "h_index": 3,
    "papers": 15
   },
   {
    "name": "Quan Nguyen",
    "id": "2313283640",
    "h_index": 2,
    "papers": 6
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "egocentric-data",
   "sim2real",
   "rl-control",
   "spatial-3d",
   "navigation",
   "hri",
   "safety-eval"
  ],
  "orgs": [
   "NVIDIA",
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2608.12860v1",
  "pdf_url": "https://arxiv.org/pdf/2608.12860v1",
  "html_url": "https://arxiv.org/html/2608.12860v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2608.12854",
  "slug": "brainwam-action-space-coordination-of-semantic-priors-and-predictive-d",
  "title": "BrainWAM: Action-Space Coordination of Semantic Priors and Predictive Dynamics for Autonomous Driving",
  "abstract": "Autonomous driving requires planning under both semantic constraints and predictive dynamics. Existing end-to-end driving approaches, however, typically emphasize only one side of this requirement: Vision-Language-Action (VLA) models exploit VLM priors for semantic reasoning, while World Action Models (WAMs) provide future-aware prediction through generative world modeling. This naturally motivates a unified planner that can leverage both semantic priors and predictive dynamics. However, we find that a naive combination through joint token-level attention suffers from an attention-allocation mismatch, where semantic shortcuts dominate the shared attention space and suppress predictive dynamics. Inspired by neuroscience evidence that complex behavior arises from coordination among functionally specialized systems, we propose BrainWAM, a structured action-space coordination framework that converts semantic reasoning and predictive world modeling into two specialized action-oriented pathways, and aligns them at the level of compact action representations. We further introduce an asynchronous rectified-flow inference strategy with decoupled video and action denoising, which shortens inference latency while preserving planning-relevant predictive context. BrainWAM reaches state-of-the-art performance on both NAVSIM v1 (89.5 PDMS) and NAVSIM v2 (89.6 EPDMS), consistently outperforming VLA-only or WAM-only methods, highlighting BrainWAM as a practical and promising direction for autonomous driving systems.",
  "published": "2026-08-13",
  "updated": "2026-08-19",
  "year": "2026",
  "authors": [
   "Bing Zhan",
   "Shuyao Shang",
   "Shuo Lu",
   "Yuan Xu",
   "Zhao Wang",
   "Yida Wang",
   "Xueyang Zhang",
   "Kun Zhan",
   "Jiahao Gu"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes BrainWAM, a structured action-space coordination framework that converts semantic reasoning and predictive world modeling into two specialized action-oriented pathways, and aligns them at the level of compact action representations, and introduces an asynchronous rectified-flow inference strategy with decoupled video and action denoising.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Bing Zhan",
    "id": "2382765965",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Shuyao Shang",
    "id": "2381363193",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Shuo Lu",
    "id": "2383879812",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Yuan Xu",
    "id": "2291076906",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Zhao Wang",
    "id": "2372486280",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Yida Wang",
    "id": "2326986535",
    "h_index": 6,
    "papers": 14
   },
   {
    "name": "Xueyang Zhang",
    "id": "2326990384",
    "h_index": 7,
    "papers": 20
   },
   {
    "name": "Kun Zhan",
    "id": "2277448471",
    "h_index": 16,
    "papers": 79
   },
   {
    "name": "Jiahao Gu",
    "id": "2457482671",
    "h_index": 0,
    "papers": 1
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "vla",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.12854v2",
  "pdf_url": "https://arxiv.org/pdf/2608.12854v2",
  "html_url": "https://arxiv.org/html/2608.12854v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.12840",
  "slug": "aspire-vins-adaptive-spline-based-visual-inertial-navigation-system-wi",
  "title": "ASPIRE-VINS: Adaptive Spline-based Visual-inertial Navigation System With Robust 3D Measurement Residuals",
  "abstract": "Visual-inertial navigation systems estimate six-degree-of-freedom motion by fusing visual and inertial data. Modern discrete-time methods with IMU preintegration provide strong accuracy and efficiency, but keyframe-based representations can be less flexible when residuals must be evaluated at arbitrary timestamps or when motion-dependent temporal resolution is needed. Continuous-time splines address this issue by representing the trajectory as a smooth temporal function, but uniformly spaced knots can under-represent rapid dynamics or over-parameterize static intervals. This letter proposes ASPIRE-VINS, a continuous-time VINS framework that combines adaptive knot placement (AKP), multi-resolution splines (MRS), and 3D measurement-space residuals (3D-MSR). AKP allocates knots according to local motion variation, MRS adds bounded local refinement in tangent space, and 3D-MSR provides bearing consistency by aligning transformed features with calibrated observation rays in 3D measurement space. Experiments show that ASPIRE-VINS achieves competitive or lower trajectory errors than the compared baselines, demonstrating the effectiveness of motion-adaptive continuous-time trajectory modeling under diverse motion and sensing conditions.",
  "published": "2026-08-13",
  "updated": "2026-08-13",
  "year": "2026",
  "authors": [
   "Kwangyik Jung",
   "Eungchang Mason Lee",
   "Taekjun Oh",
   "Hyun Myung"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "RA-L",
  "venue_source": "semantic-scholar",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Experiments show that ASPIRE-VINS achieves competitive or lower trajectory errors than the compared baselines, demonstrating the effectiveness of motion-adaptive continuous-time trajectory modeling under diverse motion and sensing conditions.",
  "doi": "10.1109/LRA.2026.3711842",
  "oa_pdf": "https://arxiv.org/pdf/2608.12840",
  "s2_authors": [
   {
    "name": "Kwangyik Jung",
    "id": "2105821",
    "h_index": 6,
    "papers": 19
   },
   {
    "name": "E. Lee",
    "id": "80883785",
    "h_index": 6,
    "papers": 30
   },
   {
    "name": "Taekjun Oh",
    "id": "1782760",
    "h_index": 6,
    "papers": 24
   },
   {
    "name": "Hyun Myung",
    "id": "2352435750",
    "h_index": 2,
    "papers": 4
   }
  ],
  "comment": "8 pages, 5 figures. Accepted for publication in IEEE Robotics and Automation Letters (RA-L), June 2026",
  "topics": [
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.12840v1",
  "pdf_url": "https://arxiv.org/pdf/2608.12840v1",
  "html_url": "https://arxiv.org/html/2608.12840v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.12835",
  "slug": "airforesight-current-to-future-spatial-map-imagination-with-cross-spac",
  "title": "AirForesight: Current-to-Future Spatial Map Imagination with Cross-Space Planning Consistency for UAV-VLN",
  "abstract": "Unmanned Aerial Vehicle Vision-Language Navigation (UAV-VLN) requires agents to follow language instructions, infer spatial structure from sparse multi-view observations, and execute feasible 3D motion in complex outdoor environments. Despite recent progress with large language models, most existing methods still map vision-language inputs directly to actions, providing limited explicit scene grounding and future-aware spatial reasoning. We propose AirForesight, a current-to-future spatial map imagination framework for UAV-VLN. AirForesight first learns a structured current-map representation from multi-view observations. This representation is jointly supervised by current-map reconstruction and future-trajectory prediction, encouraging it to encode both present scene structure and future motion intent. Under structured causal attention, the current spatial knowledge is propagated to future-map reasoning, and the resulting current and future representations are aggregated to predict the next 3D waypoint. To make spatial imagination more relevant to navigation, we introduce a cross-space planning consistency loss that encourages directional agreement between the predicted map-space trajectory and the expert action direction derived from the ground-truth waypoint displacement. Experiments on OpenUAV and AerialVLN-S, together with extensive ablations, demonstrate strong performance and support the effectiveness and stability of the proposed framework.",
  "published": "2026-08-13",
  "updated": "2026-08-13",
  "year": "2026",
  "authors": [
   "Yutong Liu",
   "Xiaojie Li",
   "Mingzhu Xu",
   "Jianlong Wu"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "To make spatial imagination more relevant to navigation, this work introduces a cross-space planning consistency loss that encourages directional agreement between the predicted map-space trajectory and the expert action direction derived from the ground-truth waypoint displacement.",
  "doi": "10.1145/3767308.3836065",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yutong Liu",
    "id": "2315112966",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Xiaojie Li",
    "id": "2261904581",
    "h_index": 7,
    "papers": 14
   },
   {
    "name": "Mingzhu Xu",
    "id": "2153557026",
    "h_index": 10,
    "papers": 23
   },
   {
    "name": "Jianlong Wu",
    "id": "2375391450",
    "h_index": 3,
    "papers": 6
   }
  ],
  "comment": "Accepted by ACM Multimedia 2026",
  "topics": [
   "spatial-3d",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.12835v1",
  "pdf_url": "https://arxiv.org/pdf/2608.12835v1",
  "html_url": "https://arxiv.org/html/2608.12835v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.12760",
  "slug": "genetic-fuzzy-system-based-control-for-final-approach-of-spacecraft-re",
  "title": "Genetic Fuzzy System-based Control for Final Approach of Spacecraft Rendezvous and Proximity Operations",
  "abstract": "In-space servicing has been receiving great attention to extend the operation of spacecraft with defective components. This requires rendezvous and proximity operations for a chaser to provide service to a target. This work constructs a fuzzy inference system-based controller for the chaser to reach the cooperative target on a circular orbit in the final approach phase while minimizing the energy consumption of the chaser. The offline training process performed by a genetic algorithm deals with multiple initial relative positions of the chaser, and the trained controller is validated using a testing environment with disturbances, which differs from the training scenarios.",
  "published": "2026-08-13",
  "updated": "2026-08-13",
  "year": "2026",
  "authors": [
   "Daegyun Choi",
   "Donghoon Kim",
   "Henzeh Leeghim"
  ],
  "author_count": 3,
  "categories": [
   "physics.space-ph",
   "cs.RO",
   "math.OC"
  ],
  "primary_category": "physics.space-ph",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Daegyun Choi",
    "id": "9416065",
    "h_index": 7,
    "papers": 40
   },
   {
    "name": "Donghoon Kim",
    "id": "2243658166",
    "h_index": 4,
    "papers": 32
   },
   {
    "name": "H. Leeghim",
    "id": "1868259",
    "h_index": 14,
    "papers": 79
   }
  ],
  "comment": "10 pages, 6 figures, 2023 33rd AAS/AIAA Space Flight Mechanics Meeting",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.12760v1",
  "pdf_url": "https://arxiv.org/pdf/2608.12760v1",
  "html_url": "https://arxiv.org/html/2608.12760v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.12755",
  "slug": "genetic-fuzzy-system-based-multi-robot-coordination-for-planetary-miss",
  "title": "Genetic Fuzzy System-Based Multi-Robot Coordination for Planetary Missions",
  "abstract": "This paper proposes a decentralized approach for a multi-robot system (MRS) using a genetic fuzzy system to perform a collaborative object transportation task that minimizes the total path length of the MRS in unstructured environment while avoiding obstacles. For an environment given by an elevation map, terrain traversability analysis with respect to the slope is performed to reduce the dimension and identify non-traversable areas that can be considered as obstacles, and the given map is converted into a traversability map in two dimensional space. In the training process, proposed fuzzy inference systems (FISs) to generate the MRS's velocity for transporting an object to a target position are optimized by a genetic algorithm with several scenarios, such as a local minima, a target that is close to an obstacle, and a cluttered environment. The trained FIS models are applied to the testing environment, which is the converted traversability map, and validated using multiple scenarios.",
  "published": "2026-08-13",
  "updated": "2026-08-13",
  "year": "2026",
  "authors": [
   "Daegyun Choi",
   "Donghoon Kim"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Daegyun Choi",
    "id": "9416065",
    "h_index": 7,
    "papers": 40
   },
   {
    "name": "Donghoon Kim",
    "id": "2243658166",
    "h_index": 4,
    "papers": 32
   }
  ],
  "comment": "14 pages, 14 figures, 2021 31st AAS/AIAA Space Flight Mechanics Meeting",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.12755v1",
  "pdf_url": "https://arxiv.org/pdf/2608.12755v1",
  "html_url": "https://arxiv.org/html/2608.12755v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.12707",
  "slug": "sap-nav-spatial-semantic-representation-meets-active-perception-for-hi",
  "title": "SAP-Nav: Spatial Semantic Representation Meets Active Perception for Hierarchical Open-Vocabulary Object Navigation",
  "abstract": "Hierarchical open-vocabulary object navigation (OVON) requires agents to follow free-form instructions that may specify targets through scene-, room-, region-, and instance-level cues in unseen environments. Although recent work LangMap has formalized this setting, reliably solving it under partial observations remains challenging: spatial grounding requires persistent environment-level evidence, whereas target verification requires clear and discriminative candidate views. We present SAP-Nav, a fully online, zero-shot framework that addresses both requirements through active perception. SAP-Nav incrementally constructs a Queryable Spatial-Semantic Representation from actively acquired room views, enabling spatial semantic queries from any explored location. It further employs Active Viewpoint Verification to assess whether the current observation provides sufficient evidence and, when necessary, reposition the agent to a more informative viewpoint before verifying candidates against category and attribute constraints. Although designed for hierarchical OVON, SAP-Nav supports both hierarchical and standard category-level OVON without task-specific training or precomputed scene maps. Experiments on LangMap and HM3D-OVON show that SAP-Nav achieves the overall best performance, including a 12.2% improvement in SR over training-based methods on region-level navigation. Real-world robot experiments further demonstrate its practical feasibility. Code will be made publicly available upon acceptance.",
  "published": "2026-08-13",
  "updated": "2026-08-13",
  "year": "2026",
  "authors": [
   "Xuetong Pei",
   "Jian Liu",
   "Vidura Munasinghe",
   "Bo Miao",
   "U-Xuan Tan",
   "Wenrui Ding",
   "Na Zhao"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "SAP-Nav is presented, a fully online, zero-shot framework that supports both hierarchical and standard category-level OVON without task-specific training or precomputed scene maps and is designed for hierarchical OVON without task-specific training or precomputed scene maps.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xuetong Pei",
    "id": "2383188988",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Jian Liu",
    "id": "2342875529",
    "h_index": 4,
    "papers": 20
   },
   {
    "name": "Vidura Munasinghe",
    "id": "2094257682",
    "h_index": 3,
    "papers": 18
   },
   {
    "name": "Bo Miao",
    "id": "2368121411",
    "h_index": 0,
    "papers": 5
   },
   {
    "name": "U-Xuan Tan",
    "id": "2335668731",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Wenrui Ding",
    "id": "2238952970",
    "h_index": 7,
    "papers": 53
   },
   {
    "name": "Na Zhao",
    "id": "2303407737",
    "h_index": 3,
    "papers": 13
   }
  ],
  "comment": "",
  "topics": [
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.12707v1",
  "pdf_url": "https://arxiv.org/pdf/2608.12707v1",
  "html_url": "https://arxiv.org/html/2608.12707v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.12683",
  "slug": "fuse-active-functional-affordance-grounding-through-adaptive-semantic",
  "title": "FUSE: Active Functional Affordance Grounding through Adaptive Semantic-Geometric Evidence Acquisition",
  "abstract": "Embodied agents must often identify and interact with objects based on their function rather than their identity, requiring them to actively acquire observations that reveal discriminative functional evidence. Existing affordance grounding methods operate from fixed viewpoints and lack mechanisms for deciding where to look when functional cues are occluded or incomplete. We introduce Active Functional Affordance Grounding, a new task in which an agent sequentially explores a scene to identify and spatially ground an object satisfying a functional query. To address this problem, we propose FUSE, an adaptive semantic-geometric evidence acquisition framework that combines explicit uncertainty-driven exploration with a learned amortized planner to efficiently select informative viewpoints. We further introduce a Habitat-based benchmark for evaluating active functional grounding. Experiments show that FUSE achieves the highest observed non-oracle grounding performance while reducing computation by 1.33x relative to fully explicit exploration, and remains effective across multiple affordance knowledge sources.",
  "published": "2026-08-13",
  "updated": "2026-08-13",
  "year": "2026",
  "authors": [
   "Zhou Chen",
   "Sathyanarayanan N. Aakur"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "FUSE is proposed, an adaptive semantic-geometric evidence acquisition framework that combines explicit uncertainty-driven exploration with a learned amortized planner to efficiently select informative viewpoints to addressEmbodied agents must often identify and interact with objects based on their function rather than their identity.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhou Chen",
    "id": "2369645972",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Sathyanarayanan N. Aakur",
    "id": "24057502",
    "h_index": 9,
    "papers": 65
   }
  ],
  "comment": "Under review. 15 Pages. 9 tables, 3 Figures",
  "topics": [
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.12683v1",
  "pdf_url": "https://arxiv.org/pdf/2608.12683v1",
  "html_url": "https://arxiv.org/html/2608.12683v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.13616",
  "slug": "adjacency-based-spectral-proxy-control-of-mobile-communication-agents",
  "title": "Adjacency-Based Spectral Proxy Control of Mobile Communication Agents",
  "abstract": "We consider a heterogeneous mobile-agent network composed of uncontrolled task agents and controllable communication agents. The objective is to reposition communication agents online as task agents move. Since throughput-based objectives are generally unsuitable for real-time control, spectral graph metrics such as algebraic connectivity are commonly adopted as surrogate objectives. However, controlling algebraic connectivity relies on the eigenvector corresponding to the second-smallest eigenvalue of a graph's Laplacian matrix (i.e., the Fiedler vector), whose distributed estimation requires an unbounded number of communication rounds to converge. In this work, we identify a structural decomposition of this Fiedler-gradient controller into a local interaction rule and a graph embedding component, suggesting the use of alternative embeddings that are easier to estimate distributively than the Fiedler vector. As a particular instance, we propose A-Fiedler, which replaces the Fiedler embedding with the dominant eigenvector of the adjacency matrix, commonly used as a graph embedding of nodes into a latent geometry. This representation is more naturally suited for distributed implementation under local communication constraints. We evaluate A-Fiedler against the classical Fiedler-gradient controller. Results show comparable network performance in the absence of communication constraints and improved robustness under distributed estimation. For instance, under the same number of communication rounds, the Fielder-gradient may even converge to disconnected configurations whereas our proposition maintains performance. We believe our contribution provides a simpler path toward distributed network control.",
  "published": "2026-08-12",
  "updated": "2026-08-12",
  "year": "2026",
  "authors": [
   "Mariana del Castillo",
   "Federico Larroca"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO",
   "cs.LG",
   "cs.MA",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A structural decomposition of this Fiedler-gradient controller is identified, suggesting the use of alternative embeddings that are easier to estimate distributively than the Fiedler vector, and A-Fiedler is proposed, which replaces the Fiedler embedding with the dominant eigenvector of the adjacency matrix, commonly used as a graph embedding of nodes into a latent geometry.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Mariana del Castillo",
    "id": "143775110",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "F. Larroca",
    "id": "2333433438",
    "h_index": 2,
    "papers": 11
   }
  ],
  "comment": "",
  "topics": [
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.13616v1",
  "pdf_url": "https://arxiv.org/pdf/2608.13616v1",
  "html_url": "https://arxiv.org/html/2608.13616v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.12650",
  "slug": "attune-a-self-annotation-tool-for-understanding-robot-operator-attenti",
  "title": "Attune: A Self-Annotation Tool for Understanding Robot Operator Attention Profiles",
  "abstract": "Deploying robot fleets in complex, real-world environments requires human operators to supervise multiple robots simultaneously. Managing operator attention is a fundamental challenge of designing multi-robot supervision interfaces, encompassing both feed layout and feed content (i.e., robot behavior design). Thus far, designers lack empirical guidance on the latter-how to change a robot's behavior to capture, sustain, or relinquish operator attention during multi-robot supervision. In our vision of the future, designers should be able to use this guidance to calibrate robot behavior to different operator attention profiles. Treating operator eye gaze as a robot behavior design clue, we created a pre-deployment elicitation tool called Attune. Attune automatically identifies when meaningful gaze shifts occur, provides AI assistance for annotating why shifts occurred, and outputs a summary of operator gaze patterns for operator review. We evaluated Attune through a user study in which participants annotated the visual triggers that drew their attention. Our findings unveil variation in observed gaze patterns and reveal how Attune helps characterize operator attention.",
  "published": "2026-08-12",
  "updated": "2026-08-12",
  "year": "2026",
  "authors": [
   "Puqi Zhou",
   "Sungsoo Ray Hong",
   "David Porfirio"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.HC"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work created a pre-deployment elicitation tool called Attune, which automatically identifies when meaningful gaze shifts occur, provides AI assistance for annotating why shifts occurred, and outputs a summary of operator gaze patterns for operator review.",
  "doi": "10.1145/3830398.3830514",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Puqi Zhou",
    "id": "2408915108",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Sungsoo Ray Hong",
    "id": "2290236747",
    "h_index": 1,
    "papers": 9
   },
   {
    "name": "David J. Porfirio",
    "id": "1666625708",
    "h_index": 9,
    "papers": 33
   }
  ],
  "comment": "13 pages, 9 figures. To appear in the Proceedings of the 39th Annual ACM Symposium on User Interface Software and Technology (UIST '26)",
  "topics": [
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.12650v1",
  "pdf_url": "https://arxiv.org/pdf/2608.12650v1",
  "html_url": "https://arxiv.org/html/2608.12650v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.12547",
  "slug": "do-llms-beat-nash-testing-decentralized-coordination-in-self-play-mult",
  "title": "Do LLMs Beat Nash? Testing Decentralized Coordination in Self-Play Multi-Agent Games",
  "abstract": "Large language model agents deployed without a central controller are often assumed to require communication to coordinate their actions. We ask what remains possible without it: when independent instances of the same model cannot communicate, can they still reason about their counterparts well enough to exceed the standard game-theoretic baseline for uncoordinated play? We introduce a benchmark of one-shot, no-communication games in which each of thirteen language models is told only that its counterparts are running the same model and is evaluated against the Nash equilibrium of the underlying game. In two-player matrix games spanning seven archetypes and two to ten actions per player, two frontier-hosted models consistently exceed their Nash benchmark, approaching the optimal joint outcome in several archetypes, while most open-weight models achieve only partial gains that vary sharply by game structure. Performance degrades substantially in team-based games with four or more interchangeable agents, particularly as the action space grows, suggesting that whatever capability drives self-play gains in dyadic games does not transfer to larger multi-agent teams.",
  "published": "2026-08-12",
  "updated": "2026-08-12",
  "year": "2026",
  "authors": [
   "Deborah Sinishaw",
   "Qile Zhu",
   "Edwin Meriaux",
   "Gregory Dudek"
  ],
  "author_count": 4,
  "categories": [
   "cs.MA",
   "cs.RO"
  ],
  "primary_category": "cs.MA",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A benchmark of one-shot, no-communication games in which each of thirteen language models is told only that its counterparts are running the same model and is evaluated against the Nash equilibrium of the underlying game.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Deborah Sinishaw",
    "id": "2457315855",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Qile Zhu",
    "id": "2457518270",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Edwin Meriaux",
    "id": "2293375812",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Gregory Dudek",
    "id": "2359452012",
    "h_index": 1,
    "papers": 7
   }
  ],
  "comment": "5 pages, 5 figures. Submitted to the 2026 IEEE MIT Undergraduate Research Technology Conference (URTC)",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.12547v1",
  "pdf_url": "https://arxiv.org/pdf/2608.12547v1",
  "html_url": "https://arxiv.org/html/2608.12547v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.12534",
  "slug": "entropy-augmented-multi-objective-policy-optimization-in-multiagent-sy",
  "title": "Entropy-Augmented Multi-Objective Policy Optimization in Multiagent Systems",
  "abstract": "Autonomous agent teams deployed in settings such as marine and extraterrestrial outposts must coordinate actions to achieve optimal outcomes across multiple competing objectives. Multi-objective evolutionary algorithms such as NSGA-II optimize for diversity in the objective space, but neglect diversity in the behavior space, possibly leading to premature convergence and a collapse in behaviors that may differentiate policies in different external conditions. To address this, we introduce an entropy-augmented policy evaluation strategy that incorporates an entropy bonus into agent fitness scores, discouraging behavioral homogeneity across the evolving population. By augmenting policy evaluation with a behavior-space diversity signal while preserving the underlying Pareto optimization framework, our method is designed to encourage exploration of behaviorally distinct policies in multiagent domains. We evaluate our approach across rover-domain experiments with qualitatively distinct reward structures and observe hypervolume improvements of up to 48% relative to the NSGA-II baseline, suggesting that behavioral diversity is a promising and underexplored direction for improving multi-objective multiagent evolutionary optimization.",
  "published": "2026-08-12",
  "updated": "2026-08-12",
  "year": "2026",
  "authors": [
   "Jamie Santos",
   "Ayhan Alp Aydeniz",
   "Raghav Thakar",
   "Kagan Tumer"
  ],
  "author_count": 4,
  "categories": [
   "cs.MA",
   "cs.RO"
  ],
  "primary_category": "cs.MA",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work introduces an entropy-augmented policy evaluation strategy that incorporates an entropy bonus into agent fitness scores, discouraging behavioral homogeneity across the evolving population.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jamie Santos",
    "id": "2265516455",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Ayhan Alp Aydeniz",
    "id": "2118303861",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Raghav Thakar",
    "id": "2274107384",
    "h_index": 0,
    "papers": 7
   },
   {
    "name": "Kagan Tumer",
    "id": "2257270056",
    "h_index": 6,
    "papers": 39
   }
  ],
  "comment": "",
  "topics": [
   "rl-control",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.12534v1",
  "pdf_url": "https://arxiv.org/pdf/2608.12534v1",
  "html_url": "https://arxiv.org/html/2608.12534v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.12528",
  "slug": "excitation-supervised-closed-loop-self-calibration-and-target-seeking",
  "title": "Excitation-Supervised Closed-Loop Self-Calibration and Target Seeking for an Unknown-Pose Range-Bearing Relay",
  "abstract": "A vehicle seeking a hidden target through a range-bearing relay of unknown position and yaw must decide, online, whether its own motion has already made the relay calibration trustworthy, and what to do when it has not. Two distinct vehicle-relative observations are known to remove the calibration gauge and make the target's relay-local packet globally actionable (arXiv:2608.09464), but that statement is static: it classifies a stored window only after the fact. This paper supplies the closed-loop layer: we show that the trajectory-spread margin $S_v$ that governs identifiability is simultaneously a finite-noise seed-accuracy bound, a local-vector variance decomposition, and a circle-geometry excitation budget, and we use it to supervise an excitation-reset controller. An excitation-supervised algorithm retriggers exploratory motion whenever the spread certificate is insufficient, projecting the target-seeking input away from the excitation's push, and otherwise proceeds to unrestricted target seeking. Under explicit sampling assumptions the supervision rule provably acquires any required excitation in finite time; in the noiseless local regime with positive excitation decay, estimator convergence yields target-seeking convergence after certification; and the threshold is selected from a desired calibration-accuracy level rather than chosen heuristically. Closed-loop simulation, paired Monte Carlo comparisons, a spread-threshold ablation, and a ROS 2/Gazebo software-in-the-loop experiment with sensing delay validate the approach. A decay-rate sweep shows that supervision matters when a fixed schedule's decay outruns the unknown time-to-adequate-excitation: over 100 paired trials the fixed baseline's yaw RMSE rises from 0.010 to 0.065 rad and success falls to 56%, while target-tracking error remains insensitive; supervision keeps yaw RMSE between 0.0095 and 0.0191 rad with 100% success.",
  "published": "2026-08-12",
  "updated": "2026-08-12",
  "year": "2026",
  "authors": [
   "Yash Bagla"
  ],
  "author_count": 1,
  "categories": [
   "eess.SY",
   "cs.RO"
  ],
  "primary_category": "eess.SY",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "It is shown that the trajectory-spread margin that governs identifiability is simultaneously a finite-noise seed-accuracy bound, a local-vector variance decomposition, and a circle-geometry excitation budget, and it is used to supervise an excitation-reset controller.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Y. Bagla",
    "id": "1580586149",
    "h_index": 0,
    "papers": 7
   }
  ],
  "comment": "12 pages, 6 figures, 5 tables. Code and data: https://github.com/yashbagla321/excitation-supervised-closed-loop (archived at https://doi.org/10.5281/zenodo.21892671)",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.12528v1",
  "pdf_url": "https://arxiv.org/pdf/2608.12528v1",
  "html_url": "https://arxiv.org/html/2608.12528v1",
  "code_url": "https://github.com/yashbagla321/excitation-supervised-closed-loop",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.3
 },
 {
  "id": "2608.12515",
  "slug": "can-vision-language-models-assess-proxemic-risk-from-egocentric-robot",
  "title": "Can Vision-Language Models Assess Proxemic Risk from Egocentric Robot Images?",
  "abstract": "Assessing proxemic danger from a robot's egocentric perspective is critical for safe embodied navigation in human environments and requires both visual and contextual reasoning. We evaluate three opensource vision-language models (VLMs) (\\textit{InternVL}, \\textit{Qwen-VL}, and \\textit{SmolVLM}) on the classification of egocentric robot images into four danger levels, comparing three prompting strategies and two rounds of QLoRA fine-tuning against a stratified random baseline. Without fine-tuning, all models perform near the baseline, while fine-tuning yields only modest overall improvements. However, \\textit{Qwen-VL} with an advanced prompt achieves substantially higher recall for high-danger cases than the other models. An analysis of person localization further shows that correct danger classification does not correspond to better spatial grounding, indicating that a model may produce a useful safety label without attending to the relevant region of the scene. These results show that current VLMs remain limited in fine-grained proxemic reasoning and spatial grounding, although targeted prompting and fine-tuning can improve high-danger detection in selected models.",
  "published": "2026-08-12",
  "updated": "2026-08-12",
  "year": "2026",
  "authors": [
   "Vladyslava Rudas",
   "Dmytro Kuzmenko"
  ],
  "author_count": 2,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "ECCV 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work evaluates three opensource vision-language models on the classification of egocentric robot images into four danger levels, comparing three prompting strategies and two rounds of QLoRA fine-tuning against a stratified random baseline.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Vladyslava Rudas",
    "id": "2457316897",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Dmytro Kuzmenko",
    "id": "2221013619",
    "h_index": 2,
    "papers": 10
   }
  ],
  "comment": "Accepted at the EMR 2026 workshop at ECCV 2026 (non-archival)",
  "topics": [
   "egocentric-data",
   "navigation",
   "video-generation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.12515v1",
  "pdf_url": "https://arxiv.org/pdf/2608.12515v1",
  "html_url": "https://arxiv.org/html/2608.12515v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.12416",
  "slug": "robosynchallenge-mastering-real-world-dexterity-via-generalizing-synth",
  "title": "RoboSynChallenge: Mastering Real-World Dexterity via Generalizing Synthesized Manipulation Skills",
  "abstract": "Achieving generalizable robotic manipulation remains a central challenge in embodied intelligence. Despite rapid advances in model architectures and learning algorithms, progress is often limited by the scarcity and narrow diversity of real-world data. The RoboSynChallenge competition introduces a unified benchmark to evaluate and advance the generalizability of manipulation policies across a spectrum of tasks, environments, and difficulty levels. To alleviate the shortage of realistic data, the challenge integrates large-scale synthetic data generation with standardized real-world robotic evaluation. Participants are encouraged to leverage synthesized state-action trials to improve general-purpose policy learning, while final assessments are conducted exclusively on unseen real-world manipulation environments. Baseline implementations, including Transformer-, Diffusion-, Vision-Language-Action, and World-Action-Model-based policies, are provided to ensure reproducibility and comparability. By coupling scalable simulation-based training with rigorous real-world validation, RoboSynChallenge aims to foster the development of broadly capable, data-efficient, and adaptable manipulation systems, thereby paving the way toward truly general robotic intelligence.",
  "published": "2026-08-12",
  "updated": "2026-08-12",
  "year": "2026",
  "authors": [
   "Runyi Zhao",
   "Ruixin Wu",
   "Chengkun Li",
   "Hongrui Zhang",
   "Ang Li",
   "Ruixing Jin",
   "Yueci Deng",
   "Yingying Guo",
   "Lihe Ding",
   "Shaocong Dong",
   "Tianfan Xue",
   "Yanjun Gao",
   "Yudong Luo",
   "Pascal Poupart",
   "Simo Wu",
   "Kui Jia",
   "Wei-shi Zheng",
   "Guiliang Liu"
  ],
  "author_count": 18,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "NeurIPS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The RoboSynChallenge competition introduces a unified benchmark to evaluate and advance the generalizability of manipulation policies across a spectrum of tasks, environments, and difficulty levels and aims to foster the development of broadly capable, data-efficient, and adaptable manipulation systems, thereby paving the way toward truly general robotic intelligence.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Runyi Zhao",
    "id": "2359793561",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Ruixin Wu",
    "id": "2457351942",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Chengkun Li",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Hongrui Zhang",
    "id": "2457323703",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Ang Li",
    "id": "2325491830",
    "h_index": 6,
    "papers": 18
   },
   {
    "name": "Ruixing Jin",
    "id": "2425461078",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Yueci Deng",
    "id": "2342396453",
    "h_index": 2,
    "papers": 11
   },
   {
    "name": "Yingying Guo",
    "id": "2456853468",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Lihe Ding",
    "id": "2268759828",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "Shaocong Dong",
    "id": "2185739943",
    "h_index": 10,
    "papers": 13
   },
   {
    "name": "Tianfan Xue",
    "id": "2276425959",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Yan Gao",
    "id": "2456548205",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yudong Luo",
    "id": "1825697681",
    "h_index": 7,
    "papers": 14
   },
   {
    "name": "Pascal Poupart",
    "id": "2245466516",
    "h_index": 4,
    "papers": 16
   },
   {
    "name": "Si Wu",
    "id": "2355121952",
    "h_index": 3,
    "papers": 21
   },
   {
    "name": "Kui Jia",
    "id": "2370507",
    "h_index": 63,
    "papers": 249
   },
   {
    "name": "Wei-Shi Zheng",
    "id": "2298568547",
    "h_index": 7,
    "papers": 25
   },
   {
    "name": "Guiliang Liu",
    "id": "2319185234",
    "h_index": 3,
    "papers": 18
   }
  ],
  "comment": "NeurIPS 2026 Competition Track",
  "topics": [
   "vla",
   "dexterous-manipulation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.12416v1",
  "pdf_url": "https://arxiv.org/pdf/2608.12416v1",
  "html_url": "https://arxiv.org/html/2608.12416v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.12198",
  "slug": "learning-based-behavior-planning-for-automated-driving-real-world-inte",
  "title": "Learning-Based Behavior Planning for Automated Driving: Real-World Integration and Deployment",
  "abstract": "Recent research in machine and deep learning has shown the potential of learningbased motion planning approaches to improve the driving behavior of automated vehicles, especially in complex environments. However, their complex nature and lack of transparency can hinder explainability and trustworthiness and complicate safety assurance. Motivated by these challenges, we propose a hybrid planning architecture that combines the advantages of machine learning with the verifiability and the determinism of classical approaches. Specifically, we developed a deep neural network to interpret complex traffic scenes and propose driving behavior, while an optimization-based supervision layer validates this proposal and enforces explicit drivability and safety constraints. We evaluate the learned planner's driving behavior in open-loop studies on real-world urban data, discuss system integration aspects for stable closed-loop operation, and report results from real-world deployment on our research vehicle karl..",
  "published": "2026-08-12",
  "updated": "2026-08-12",
  "year": "2026",
  "authors": [
   "Jean-Pierre Busch",
   "Guido Linden",
   "Jan Bergmann",
   "Lutz Eckstein"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A deep neural network is developed to interpret complex traffic scenes and propose driving behavior, while an optimization-based supervision layer validates this proposal and enforces explicit drivability and safety constraints.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jean-Pierre Busch",
    "id": "2239201697",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Guido Linden",
    "id": "2409826372",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Jan Bergmann",
    "id": "2457026113",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Lutz Eckstein",
    "id": "2276203235",
    "h_index": 4,
    "papers": 11
   }
  ],
  "comment": "17 pages; Accepted to be published as part of the 17. Uni-DAS e.V. Workshop \"Fahrerassistenz und automatisiertes Fahren\", September 29-30, 2026",
  "topics": [
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.12198v1",
  "pdf_url": "https://arxiv.org/pdf/2608.12198v1",
  "html_url": "https://arxiv.org/html/2608.12198v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.12122",
  "slug": "handedit-a-unified-benchmark-for-egocentric-human-to-robot-dexterous-h",
  "title": "HandEdit: A Unified Benchmark for Egocentric Human-to-Robot Dexterous Hand Image Editing",
  "abstract": "Robotic manipulation with dexterous hands is a cornerstone of Embodied AI, yet its progress is stifled by the high cost of collecting embodiment-aware teleoperation data. While abundant egocentric videos of human hands offer a scalable alternative, the profound discrepancies in appearance, articulation, and camera viewpoints between human and robotic data raise significant challenges for co-training. Though existing general image-editing models demonstrate strong capabilities, they lack necessary embodiment-specific priors to fully bridge this gap. In this work, we present HandEdit, a unified large-scale embodiment-aware image-editing dataset and benchmark specifically designed to transform human hands and arms into various dexterous robotic embodiments within egocentric frames. HandEdit comprises over 200M editing instances derived from five diverse source datasets, covering 26 distinct URDFs, including 13 hand-only and 13 hand-arm configurations. Alongside the dataset, we establish a unified benchmark protocol with two tracks: Hand-only and Hand-Arm, supporting URDF-conditioned evaluation. We conduct extensive evaluations of 11 representative image-editing baselines using a multi-dimensional metric suite, including generic similarity metrics, VLM-based judgment, and embodiment-aware metrics. HandEdit serves as a critical resource at the intersection of image editing and robotics: it advances embodiment-aware editing models while enabling scalable dexterous robotic learning from abundant human video data, paving the way for more generalizable Embodied AI.",
  "published": "2026-08-12",
  "updated": "2026-08-12",
  "year": "2026",
  "authors": [
   "Zhenjie Yang",
   "Xingyu Jiao",
   "Guopeng Zhong",
   "Shuzhe Yang",
   "Shi Che",
   "Chao Wu",
   "Chenyu Jiang",
   "Dongjie Zhang",
   "Yideng Zhang",
   "Zheng Zhang",
   "Muyun Jiang",
   "Haisheng Su",
   "Shuang Jin",
   "Donghang Zhang",
   "Chao Yang",
   "Li Chen",
   "Hongyang Li",
   "Zuxuan Wu",
   "Yu-Gang Jiang",
   "Xiaosong Jia",
   "Junchi Yan"
  ],
  "author_count": 21,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "HandEdit serves as a critical resource at the intersection of image editing and robotics: it advances embodiment-aware editing models while enabling scalable dexterous robotic learning from abundant human video data, paving the way for more generalizable Embodied AI.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhenjie Yang",
    "id": "2257389790",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Xingyu Jiao",
    "id": "2457027366",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Guopeng Zhong",
    "id": "2457028104",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Shuzhe Yang",
    "id": "2449432093",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Shi Che",
    "id": "2457030411",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Chao Wu",
    "id": "2385840798",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Chenyu Jiang",
    "id": "2457202948",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Dongjie Zhang",
    "id": "2457134770",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yideng Zhang",
    "id": "2457299282",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Zheng Zhang",
    "id": "2457359748",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Muyun Jiang",
    "id": "2299110237",
    "h_index": 5,
    "papers": 19
   },
   {
    "name": "Haisheng Su",
    "id": "2363482540",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Shuang Jin",
    "id": "2457308132",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Donghang Zhang",
    "id": "2457134172",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Chao Yang",
    "id": "2447734256",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Li Chen",
    "id": "2357896168",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Hongyang Li",
    "id": "2290243003",
    "h_index": 11,
    "papers": 19
   },
   {
    "name": "Zuxuan Wu",
    "id": "2384164233",
    "h_index": 2,
    "papers": 13
   },
   {
    "name": "Yu-Gang Jiang",
    "id": "1717861",
    "h_index": 47,
    "papers": 141
   },
   {
    "name": "Xiaosong Jia",
    "id": "1958998899",
    "h_index": 23,
    "papers": 53
   },
   {
    "name": "Junchi Yan",
    "id": "2262560236",
    "h_index": 5,
    "papers": 10
   }
  ],
  "comment": "Technical Report. Project Page: https://handedit.github.io/",
  "topics": [
   "dexterous-manipulation",
   "egocentric-data",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.12122v1",
  "pdf_url": "https://arxiv.org/pdf/2608.12122v1",
  "html_url": "https://arxiv.org/html/2608.12122v1",
  "code_url": "https://handedit.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2608.12063",
  "slug": "learning-loco-manipulation-from-smpc-demonstrations-with-sparse-offlin",
  "title": "Learning Loco-Manipulation From SMPC Demonstrations With Sparse Offline-to-Online RL",
  "abstract": "Integrating locomotion and manipulation is essential for robot autonomy, but scaling standard Reinforcement Learning (RL) to complex tasks is severely bottlenecked by the slow, manual process of dense reward shaping. To bypass this limitation, we leverage Sample-based Model Predictive Control (SMPC) entirely in simulation as an automated, rapidly tunable expert to generate massive offline datasets. Because this data solves the fundamental exploration problem, we can train an off-policy RL agent using purely sparse task rewards, drastically reducing the time required to learn new skills and eliminating the need for manual tuning. Integrating this high-level agent with a low-level dynamic stability controller yields more optimal behaviors that strictly align with true task objectives, ultimately allowing the learned policies to surpass the original optimal control teacher. We validate the robustness of this sim-to-real framework by successfully deploying complex loco-manipulation skills across different morphologies, including an arm-equipped Spot quadruped and a G1 humanoid.",
  "published": "2026-08-12",
  "updated": "2026-08-12",
  "year": "2026",
  "authors": [
   "Martin Schuck",
   "Maks Sorokin",
   "Simone Manni",
   "Duy Ta",
   "Angela P. Schoellig",
   "Marco Hutter",
   "Simon Le Cleac'H",
   "Jan Br\u00fcdigam"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work uses Sample-based Model Predictive Control entirely in simulation as an automated, rapidly tunable expert to generate massive offline datasets and validate the robustness of this sim-to-real framework by successfully deploying complex loco-manipulation skills across different morphologies.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Martin Schuck",
    "id": "2333661564",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Maks Sorokin",
    "id": "2139985531",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "S. Manni",
    "id": "2440458562",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "D. Ta",
    "id": "2330244421",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Angela P. Schoellig",
    "id": "2321572233",
    "h_index": 4,
    "papers": 20
   },
   {
    "name": "Marco Hutter",
    "id": "2340685198",
    "h_index": 6,
    "papers": 20
   },
   {
    "name": "Simon Le Cleac'h",
    "id": "1412717129",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Jan Br\u00fcdigam",
    "id": "1518628970",
    "h_index": 6,
    "papers": 13
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "sim2real",
   "rl-control",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.12063v1",
  "pdf_url": "https://arxiv.org/pdf/2608.12063v1",
  "html_url": "https://arxiv.org/html/2608.12063v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.11901",
  "slug": "davinci-a-dataset-towards-outdoor-vision-and-language-navigation-with",
  "title": "DaViNCi: A Dataset Towards Outdoor Vision-and-Language Navigation with Continuous Actions and Dynamic Elements",
  "abstract": "Vision-and-Language Navigation (VLN) has progressively expanded from indoor to outdoor environments. However, existing outdoor VLN datasets still rely on fixed discrete topological graphs for construction. It fails to align with the rapidly changing real-world outdoor environments and impedes the sim-to-real transfer of VLN agents. To address this limitation, we propose DaViNCi (\\textbf{D}yn\\textbf{a}mic \\textbf{Vi}sion-and-Language \\textbf{N}avigation in \\textbf{C}ont\\textbf{i}nuous Environment), the first outdoor VLN dataset that simultaneously introduces both continuous and dynamic factors. The agent not only moves in the outdoor environment using continuous actions but is also required to handle unpredictable dynamic elements. The dataset encompasses six distinct maps with a total of 6,933 trajectories. Through comprehensive comparative experiments, we find that the success rate on DaViNCi decreased by more than 10\\% in discrete environments compared to previous datasets. And there is an even greater decline in continuous settings, demonstrating the challenge of DaViNCi. Furthermore, we clarify the impact of action granularity and dynamic elements. These results demonstrate the practical value of DaViNCi in advancing outdoor VLN toward more realistic environments. The website is https://xzh0312.github.io/DaViNCi/.",
  "published": "2026-08-12",
  "updated": "2026-08-12",
  "year": "2026",
  "authors": [
   "Zihao Xie",
   "Pingrui Lai",
   "Yitong Wu",
   "Hua Yang"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The proposed DaViNCi is the first outdoor VLN dataset that simultaneously introduces both continuous and dynamic factors, and the practical value of DaViNCi in advancing outdoor VLN toward more realistic environments is demonstrated.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zihao Xie",
    "id": "2300249571",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Pingrui Lai",
    "id": "2295941820",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Yitong Wu",
    "id": "2455652074",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Hua Yang",
    "id": "2348292913",
    "h_index": 4,
    "papers": 12
   }
  ],
  "comment": "",
  "topics": [
   "sim2real",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.11901v1",
  "pdf_url": "https://arxiv.org/pdf/2608.11901v1",
  "html_url": "https://arxiv.org/html/2608.11901v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.11895",
  "slug": "scalable-multi-agent-maze-traversal-with-local-communication",
  "title": "Scalable Multi-Agent Maze Traversal with Local Communication",
  "abstract": "Cave networks, pipe systems, and similar maze-like environments pose significant challenges for multi-agent navigation in unknown settings with limited communication. We propose a distributed algorithm that enables agents to collectively traverse an unknown, possibly cyclic graph. Agents enter sequentially at a designated start node and are tasked to localize and reach an undisclosed goal while avoiding collisions. They coordinate via local communication using leader-follower relationships and leader switching. At any moment in time, exploration is performed by only one of the agents, which runs a single-agent maze solver. We prove that the algorithm is complete, that its makespan is asymptotically equivalent (in the number of agents) to that of an optimal full-knowledge strategy, and derive its time and space complexity. Simulations with up to $625$ agents show a decreasing average sum-of-fuels as the number of agents increases and demonstrate that the proposed approach outperforms a na\u00efve baseline in which all agents independently execute the single-agent solver.",
  "published": "2026-08-12",
  "updated": "2026-08-12",
  "year": "2026",
  "authors": [
   "Julian Rau",
   "Jahir Argote-Gerald",
   "Grace McFassel",
   "Genki Miyauchi",
   "Paul Trodden",
   "Roderich Gro\u00df"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.MA"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes a distributed algorithm that enables agents to collectively traverse an unknown, possibly cyclic graph and proves that the algorithm is complete, that its makespan is asymptotically equivalent to that of an optimal full-knowledge strategy, and derive its time and space complexity.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Julian Rau",
    "id": "2390202686",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Jahir Argote-Gerald",
    "id": "2326842728",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Grace McFassel",
    "id": "72305241",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Genki Miyauchi",
    "id": "1603562610",
    "h_index": 3,
    "papers": 13
   },
   {
    "name": "Paul A. Trodden",
    "id": "2267959573",
    "h_index": 2,
    "papers": 11
   },
   {
    "name": "Roderich Gro\u00df",
    "id": "2323368438",
    "h_index": 1,
    "papers": 10
   }
  ],
  "comment": "This manuscript has been accepted for publication in the proceedings of the World Symposium on the Algorithmic Foundations of Robotics (WAFR 2026), to be published by Springer in the Springer Proceedings in Advanced Robotics (SPAR) series",
  "topics": [
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.11895v1",
  "pdf_url": "https://arxiv.org/pdf/2608.11895v1",
  "html_url": "https://arxiv.org/html/2608.11895v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.11876",
  "slug": "d3d-gen-robot-aware-domain-grounded-interactive-3d-world-generation-fo",
  "title": "D3D-GEN: Robot-Aware Domain-Grounded Interactive 3D World Generation for Social Robotics",
  "abstract": "Training and validation of Embodied AI for social navigation critically depends on realistic simulation environments, yet many current approaches fail to find a balance between realism and simulability. We propose D3D-GEN, a novel world generation system that combines a domain agent with a retrieval-augmented generation (RAG) pipeline grounded in that domain. Our system enables users to rapidly generate domain-grounded, fully interactive 3D worlds by automating both the collection of domain knowledge and the synthesis of realistic floorplans and object placements, without dependence on any fixed 3D model database. Given a domain description prompt, the research agent collects publicly accessible domain-specific data and constructs a persistent domain database. Using this database, our RAG pipeline generates plausible floorplans and object placements by dynamically querying a user-provided semantic database, which can be easily extended or modified. The output is a fully interactive 3D world loadable by the popular simulators Isaac Sim and Gazebo. With our approach, we have built databases for several common domains (indoor residential, hospital, office) and generated dozens of distinct, plausible simulation environments for each domain. We present D3D-GEN with a local web frontend that facilitates rapid, interactive world generation for robot simulation.",
  "published": "2026-08-12",
  "updated": "2026-08-12",
  "year": "2026",
  "authors": [
   "Anh Duc Do",
   "Volodymyr Scherbyna",
   "Tai Duc Nguyen",
   "Spaarsh Thakkar",
   "Zhengcheng Shen",
   "Teham Buiyan",
   "Archan Misra",
   "Linh K\u00e4stner"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO",
   "cs.HC"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "D3D-GEN, a novel world generation system that combines a domain agent with a retrieval-augmented generation (RAG) pipeline grounded in that domain, is proposed and presented with a local web frontend that facilitates rapid, interactive world generation for robot simulation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "A. Do",
    "id": "2456996169",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "V. Scherbyna",
    "id": "2289467004",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Tai Duc Nguyen",
    "id": "2457504965",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Spaarsh Thakkar",
    "id": "2423276399",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Zhengcheng Shen",
    "id": "2111640114",
    "h_index": 6,
    "papers": 18
   },
   {
    "name": "Teham Buiyan",
    "id": "2067799194",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Archan Misra",
    "id": "2150552689",
    "h_index": 41,
    "papers": 304
   },
   {
    "name": "Linh K\u00e4stner",
    "id": "1474235731",
    "h_index": 12,
    "papers": 47
   }
  ],
  "comment": "8 pages, 5 figures, and 5 tables. Accepted at the 2026 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS 2026)",
  "topics": [
   "sim2real",
   "navigation",
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.11876v1",
  "pdf_url": "https://arxiv.org/pdf/2608.11876v1",
  "html_url": "https://arxiv.org/html/2608.11876v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.11870",
  "slug": "enhancing-visual-domain-robustness-in-behaviour-cloning-via-saliency-g",
  "title": "Enhancing Visual Domain Robustness in Behaviour Cloning via Saliency-Guided Augmentation",
  "abstract": "In vision-based behavior cloning (BC), conventional image augmentations such as Random Crop and Color Jitter often fall short under substantial visual domain shifts, including changes in shadows, distractors, and backgrounds. Superimposition-based augmentations, which blend in-domain and out-of-domain images, have shown promise for improving generalization in computer vision, but their suitability for BC remains uncertain because task-critical semantics, spatiotemporal relationships, and agent-target interactions must be preserved. To address this, we introduce RoboSaGA, a Saliency-Guided Augmentation method within the superimposition family tailored for vision-based BC. RoboSaGA dynamically adjusts augmentation intensity at the pixel level using policy-driven saliency, enabling aggressive augmentation in task-irrelevant regions while preserving task-critical information. It integrates seamlessly into existing architectures without requiring structural modifications or additional learning objectives. Experiments in both simulated and real-world settings show that RoboSaGA preserves in-domain performance while substantially improving robustness to visual domain shifts, including distractor and background changes, as well as lighting and shadow variations. Code is available at https://github.com/Zheyu-Zhuang/RoboSaGA.",
  "published": "2026-08-12",
  "updated": "2026-08-12",
  "year": "2026",
  "authors": [
   "Zheyu Zhuang",
   "Ruiyu Wang",
   "Nils Ingelhag",
   "Ville Kyrki",
   "Danica Kragic"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 11,
  "influential_citations": 1,
  "tldr": "Experiments show that RoboSaGA preserves in-domain performance while substantially improving robustness to visual domain shifts, including distractor and background changes, as well as lighting and shadow variations.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zheyu Zhuang",
    "id": "2323552853",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Ruiyu Wang",
    "id": "2255293781",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Nils Ingelhag",
    "id": "2293312071",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Ville Kyrki",
    "id": "2717428",
    "h_index": 39,
    "papers": 249
   },
   {
    "name": "Danica Kragic",
    "id": "2153721470",
    "h_index": 13,
    "papers": 102
   }
  ],
  "comment": "Accepted at the Conference on Robot Learning (CoRL) 2024",
  "topics": [
   "imitation-diffusion",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.11870v1",
  "pdf_url": "https://arxiv.org/pdf/2608.11870v1",
  "html_url": "https://arxiv.org/html/2608.11870v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.58
 },
 {
  "id": "2608.11769",
  "slug": "policy-induced-hand-priors-in-humanoid-dual-arm-manipulation-diagnosin",
  "title": "Policy-Induced Hand Priors in Humanoid Dual-Arm Manipulation: Diagnosing and Mitigating Initial-Pose Dependence",
  "abstract": "Vision-language-action (VLA) policies are expected to operate robustly across variations in the robot's initial configuration, yet aggregate task success can conceal pose-specific failures and inappropriate hand selection. This work investigates initial-pose dependence in VLA-based humanoid dual-arm manipulation. We characterize the initial-condition-dependent early hand preference as a policy-induced hand prior and quantify it using HandPriorScore, residual hand bias, and target responsiveness. Evaluations across multiple policies and 17 initial configurations reveal strong initial-pose--policy interactions: the same pose produces substantially different success rates across policies, while a single policy exhibits large performance variation across poses. Specific initial arm configurations can suppress or induce an asymmetric hand preference, with the resulting effect varying in direction and strength across policies. Wrist-camera observations also influence hand selection and task performance. Expanding initial-pose coverage in the training dataset substantially improves robustness, while targeted augmentation around a low-performing configuration increases its success rate. Comparisons across training configurations show that sufficient exposure to the target simulation task is beneficial, whereas the effect of real or auxiliary data depends on pose coverage, simulation ratio, and observation availability. These findings characterize a pose-conditioned hand prior, identify a localized initial arm configuration as a causal handle on hand-selection behavior, and demonstrate how data coverage and training composition affect initial-pose robustness.",
  "published": "2026-08-12",
  "updated": "2026-08-12",
  "year": "2026",
  "authors": [
   "Chaeyeon Jung",
   "Juyoun Park"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Investigation of initial-pose dependence in VLA-based humanoid dual-arm manipulation identifies a pose-conditioned hand prior, identifies a localized initial arm configuration as a causal handle on hand-selection behavior, and demonstrates how data coverage and training composition affect initial-pose robustness.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Chaeyeon Jung",
    "id": "2150179879",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Juyoun Park",
    "id": "2331454212",
    "h_index": 2,
    "papers": 7
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "humanoids",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.11769v1",
  "pdf_url": "https://arxiv.org/pdf/2608.11769v1",
  "html_url": "https://arxiv.org/html/2608.11769v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.11739",
  "slug": "g0-5-one-autoregressive-stream-for-robot-reasoning-and-action",
  "title": "G0.5: One Autoregressive Stream for Robot Reasoning and Action",
  "abstract": "The prevailing recipe for Vision-Language-Action (VLA) models couples a pretrained VLM with a separately trained flow-matching action expert. This makes the VLM a context encoder rather than a decision-maker. We introduce G0.5, a pretrained autoregressive VLA in which a single transformer decoder emits reasoning and action tokens under a single objective. Three components make this tractable at foundation-model scale: a learnable cross-embodiment action tokenizer that maps heterogeneous robot actions into a shared vocabulary; a native chain-of-thought stream interleaving task decomposition, object grounding, and action hints with action tokens; and a visual memory module that injects multi-second history through the vision encoder. Because reasoning and action share a single set of weights, the pretrained VLM's capabilities carry over to physical behavior: the model follows instructions closely, and prompts directly steer action granularity, task horizon, and out-of-distribution scene handling without further training. Pretrained on a large collection of robot datasets together with VQA samples, G0.5 surpasses state-of-the-art models across 7 independent regimes: real-world fine-tuning on R1lite and R1pro robots (76.7\\% vs.\\ 53.3\\% for $\u03c0_{0.5}$ and 24.4\\% for GR00T-N1.7), the 2025 BEHAVIOR Challenge on 50 long-horizon household mobile manipulation tasks using a generalist policy (31.4\\% vs.\\ 26.3\\% for $\u03c0_{0.5}$ and 26.1\\% for the challenge winner), DROID post-training followed by zero-shot transfer to an unseen environment and objects (82.5\\%), a language-following Pick-and-Place benchmark, LIBERO (98.9\\%), RoboTwin 2.0 (93.3\\%), and SimplerEnv-Bridge (87.3\\%).",
  "published": "2026-08-12",
  "updated": "2026-08-12",
  "year": "2026",
  "authors": [
   "Yicheng Liu",
   "Zibin Dong",
   "Baijun Ye",
   "Tianyuan Yuan",
   "Tao Jiang",
   "Anqi Yang",
   "Shicheng Cao",
   "Haonan Liu",
   "Yue Sun",
   "Zihan Guo",
   "Xiao Liu",
   "Dong Ke",
   "Changxun Pan",
   "Chenru Wu",
   "Tailai Cheng",
   "Xiaoshu Ren",
   "Xinlei Zhang",
   "Jianning Cui",
   "Zijie Zhao",
   "Haoyu Zhang",
   "Kaiming Xu",
   "Haodong Yang",
   "Bowen Zhang",
   "Jiahui Niu",
   "Shaoting Zhu",
   "Shiduo Zhang",
   "Hang Zhao"
  ],
  "author_count": 27,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "G0.5 is introduced, a pretrained autoregressive VLA in which a single transformer decoder emits reasoning and action tokens under a single objective, which exceeds state-of-the-art models across 7 independent regimes.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yicheng Liu",
    "id": "2284726857",
    "h_index": 7,
    "papers": 17
   },
   {
    "name": "Zibin Dong",
    "id": "2256465607",
    "h_index": 9,
    "papers": 21
   },
   {
    "name": "Baijun Ye",
    "id": "2299156070",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "Tianyuan Yuan",
    "id": "2214583235",
    "h_index": 11,
    "papers": 15
   },
   {
    "name": "Tao Jiang",
    "id": "2384816862",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Anqi Yang",
    "id": "2457025691",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Shicheng Cao",
    "id": "2457025762",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Haonan Liu",
    "id": "2457277619",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Yue Sun",
    "id": "2317121008",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Zihan Guo",
    "id": "2447371109",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Xiao Liu",
    "id": "2354276733",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Dong Ke",
    "id": "2453157736",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Changxun Pan",
    "id": "2457033300",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Chenru Wu",
    "id": "2456994827",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Tailai Cheng",
    "id": "2371151628",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Xiaoshu Ren",
    "id": "2450276227",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Xinlei Zhang",
    "id": "2456612572",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Ji Cui",
    "id": "2087017690",
    "h_index": 15,
    "papers": 34
   },
   {
    "name": "Zijie Zhao",
    "id": "2315439477",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Haoyu Zhang",
    "id": "2276656133",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Kaiming Xu",
    "id": "2455654337",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Hao Yang",
    "id": "2456307000",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Bowen Zhang",
    "id": "2456259738",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Jiahui Niu",
    "id": "2320584365",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Shaoting Zhu",
    "id": "2216251632",
    "h_index": 8,
    "papers": 16
   },
   {
    "name": "Shiduo Zhang",
    "id": "2302704871",
    "h_index": 8,
    "papers": 14
   },
   {
    "name": "Hang Zhao",
    "id": "2378973882",
    "h_index": 4,
    "papers": 8
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "navigation",
   "foundation-pretraining",
   "video-generation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.11739v1",
  "pdf_url": "https://arxiv.org/pdf/2608.11739v1",
  "html_url": "https://arxiv.org/html/2608.11739v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.11731",
  "slug": "contactipm-a-structure-exploiting-interior-point-solver-for-contact-im",
  "title": "ContactIPM: A Structure-Exploiting Interior-Point Solver for Contact-Implicit Trajectory Optimization",
  "abstract": "Contact-implicit trajectory optimization avoids prescribing contact sequences, but yields mathematical programs with complementarity constraints (MPCCs) whose degeneracy challenges conventional primal--dual solvers. Existing contact-specific methods improve robustness to this degeneracy but do not leverage a stagewise optimal-control factorization and primal--dual consistency, while structure-exploiting optimal-control solvers are not designed for complementarity constraints. We show that these capabilities can be combined in a single primal--dual method. ContactIPM identifies complementary inequality pairs, embeds them through a barrier-coupled elastic interior relaxation, eliminates slack and dual variables stagewise, and solves the reduced Newton system using a Riccati recursion. A fixed multi-phase MPCC recovery schedule provides four continuation and restart attempts from naive initializations, while termination is gated by the unrelaxed physical complementarity residual. We compare ContactIPM with two contact-specific MPCC solvers, CRISP and IMPACT, using matched benchmark conditions and common post-solve acceptance criteria. On four fixed CRISP benchmark cases, ContactIPM is $2.17$--$8.87\\times$ faster over 20 paired timing repetitions per case and achieves higher success on the Push Box and Push-T robustness suites. Against IMPACT, ContactIPM is \\(2.96\\times\\) faster on Push T and \\(4.91\\times\\) faster on Cart Transport, but \\(4.46\\times\\) slower on Push Box. In 50 closed-loop Push Box rollouts spanning model mismatch, measurement noise, initial-pose errors, and state resets,",
  "published": "2026-08-12",
  "updated": "2026-08-12",
  "year": "2026",
  "authors": [
   "Yucheng Chen"
  ],
  "author_count": 1,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work identifies complementary inequality pairs, embeds them through a barrier-coupled elastic interior relaxation, eliminates slack and dual variables stagewise, and solves the reduced Newton system using a Riccati recursion in a single primal--dual method.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yucheng Chen",
    "id": "2109364060",
    "h_index": 3,
    "papers": 9
   }
  ],
  "comment": "",
  "topics": [
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.11731v1",
  "pdf_url": "https://arxiv.org/pdf/2608.11731v1",
  "html_url": "https://arxiv.org/html/2608.11731v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.11671",
  "slug": "stellavla-in-context-structured-demonstration-for-generalizable-vision",
  "title": "StellaVLA: In-Context Structured Demonstration for Generalizable Vision-Language-Action Models",
  "abstract": "Vision-Language-Action (VLA) models can follow instructions and manipulate objects, but their performance often collapses out of distribution (OOD), when the scene, viewpoint, or object differs from training. Adapting to each new situation typically requires collecting more data and fine-tuning. We present StellaVLA, a framework that instead adapts at test time by conditioning on a single retrieved demonstration. The key idea is to move beyond imitating what an expert did and instead convey why: an automated offline pipeline converts each raw trajectory into a structured demonstration, e.g., a task plan, sub-goal descriptions, and verbalized 3D motion, at zero human-annotation cost. Provided as in-context guidance, this structured demonstration lets the policy reason about the task rather than mimic a pixel trajectory, which also makes it transferable across embodiments (real-robot, human-hand, or XR demonstrations). A parallel dual-training design internalizes this reasoning during training through a joint action-and-language objective, while inference uses the action expert alone, preserving real-time, high-frequency control with no added latency. On the VLA-Arena leaderboard(Aug 1, 2026), StellaVLA ranks first with an overall score of 0.63, versus 0.44 and 0.22 for the strong prior models ($\u03c0_{0.5}$ and LingBot-VLA), and it further leads on LIBERO with 98.8% average success rate and LIBERO-Plus with 85.1% success rate. Our real-robot benchmark demonstrates that StellaVLA can use both human/robot demos and human-to-robot (XR) demos as in-context structured demonstration to help VLA model adapt to OOD tasks.",
  "published": "2026-08-12",
  "updated": "2026-08-12",
  "year": "2026",
  "authors": [
   "Siyu Xu",
   "Yunke Wang",
   "Zijian Wang",
   "Dihao Zhu",
   "Chenghao Xia",
   "Chengbin Du",
   "Daochang Liu",
   "Tao Huang",
   "Chang Xu"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The real-robot benchmark demonstrates that StellaVLA can use both human/robot demos and human-to-robot (XR) demos as in-context structured demonstration to help VLA model adapt to OOD tasks.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Siyu Xu",
    "id": "2292087285",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Yunke Wang",
    "id": "2119215768",
    "h_index": 10,
    "papers": 43
   },
   {
    "name": "Zijian Wang",
    "id": "2259065741",
    "h_index": 12,
    "papers": 53
   },
   {
    "name": "Di Zhu",
    "id": "2375301481",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Chenghao Xia",
    "id": "2343742679",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Chengbin Du",
    "id": "2205535308",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Daochang Liu",
    "id": "51023221",
    "h_index": 11,
    "papers": 51
   },
   {
    "name": "Tao Huang",
    "id": "2265957484",
    "h_index": 7,
    "papers": 25
   },
   {
    "name": "Chang Xu",
    "id": "2292018438",
    "h_index": 6,
    "papers": 20
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.11671v1",
  "pdf_url": "https://arxiv.org/pdf/2608.11671v1",
  "html_url": "https://arxiv.org/html/2608.11671v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.11641",
  "slug": "energy-aware-wind-resilient-routing-for-truck-assisted-multi-uav-deliv",
  "title": "Energy-Aware Wind-Resilient Routing for Truck-Assisted Multi-UAV Delivery under Wind Uncertainty",
  "abstract": "Energy feasibility under wind uncertainty is a critical safety issue for low-altitude air-ground delivery. In truck-UAV systems, UAVs complete assigned deliveries and safely return to a mobile truck or depot, while wind-induced propulsion costs vary online and are only partially observable. Existing routing methods often rely on static or deterministic energy models, which may underestimate headwind, crosswind, battery-voltage, and return-feasibility risks. This paper proposes Energy-Aware Wind-Resilient Routing (EWR), an online risk-sensitive planning framework for wind-aware and energy-safe UAV routing. The delivery environment is represented as a time-dependent directed energy graph whose edge costs are updated using delayed noisy wind estimates, payload states, and conservative uncertainty margins. Experiments using synthetic delivery graphs with replayed wind logs from a public truck-UAV delivery dataset show that EWR improves mission success rates and reduces wind-induced return failures.",
  "published": "2026-08-12",
  "updated": "2026-08-12",
  "year": "2026",
  "authors": [
   "Tianshun Li",
   "Yanggang Sheng",
   "Hongliang Lu",
   "Zhongzhen Wang",
   "Haoang Li",
   "Xinhu Zheng"
  ],
  "author_count": 6,
  "categories": [
   "eess.SY",
   "cs.RO"
  ],
  "primary_category": "eess.SY",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Tianshun Li",
    "id": "2238301682",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Yanggang Sheng",
    "id": "2429659221",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Hongliang Lu",
    "id": "2255592398",
    "h_index": 5,
    "papers": 15
   },
   {
    "name": "Zhongzheng Wang",
    "id": "2344268319",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Haoang Li",
    "id": "2384363611",
    "h_index": 9,
    "papers": 38
   },
   {
    "name": "Xinhu Zheng",
    "id": "2367145350",
    "h_index": 4,
    "papers": 10
   }
  ],
  "comment": "",
  "topics": [
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.11641v1",
  "pdf_url": "https://arxiv.org/pdf/2608.11641v1",
  "html_url": "https://arxiv.org/html/2608.11641v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.11597",
  "slug": "iot-enabled-autonomous-maritime-navigation-in-smart-ports-a-curriculum",
  "title": "IoT-Enabled Autonomous Maritime Navigation in Smart Ports: A Curriculum-Guided Shared Policy Learning Framework",
  "abstract": "As smart port infrastructures increasingly rely on autonomous maritime devices enabled by the Internet of Things (IoT), ensuring reliable onboard navigation intelligence has become a critical challenge for safe and scalable operations in congested waterways. This paper investigates onboard autonomous navigation for such IoT devices under partial observability and dense traffic conditions. A curriculum-guided reinforcement learning framework with a shared recurrent policy is developed to enhance temporal reasoning, deployment scalability, and robustness of edge-level decision-making. Centralized training is adopted as an offline design-time strategy, while all navigation actions are executed fully onboard, consistent with IoT edge intelligence paradigms. Extensive simulations in multiple realistic port environments demonstrate that the proposed approach improves navigation reliability, collision avoidance, and training stability compared with standard baseline methods, and generalizes effectively to previously unseen high-density scenarios. The results indicate that curriculum-guided shared learning provides a practical solution for scalable deployment of IoT-enabled autonomous maritime devices in smart port operations.",
  "published": "2026-08-12",
  "updated": "2026-08-12",
  "year": "2026",
  "authors": [
   "Yuqing Lin",
   "Rangya Zhang",
   "Kum Fai Yuen"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Investigation of onboard autonomous navigation for IoT-enabled autonomous maritime devices under partial observability and dense traffic conditions suggests that curriculum-guided shared learning provides a practical solution for scalable deployment of IoT-enabled autonomous maritime devices in smart port operations.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yuqing Lin",
    "id": "2337084291",
    "h_index": 3,
    "papers": 13
   },
   {
    "name": "Rangya Zhang",
    "id": "2273464408",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Kum Fai Yuen",
    "id": "40095565",
    "h_index": 61,
    "papers": 314
   }
  ],
  "comment": "",
  "topics": [
   "rl-control",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.11597v1",
  "pdf_url": "https://arxiv.org/pdf/2608.11597v1",
  "html_url": "https://arxiv.org/html/2608.11597v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.11592",
  "slug": "video2track-from-real-world-interaction-videos-to-steerable-adversaria",
  "title": "Video2Track: From Real-World Interaction Videos to Steerable Adversarial Closed-Track Testing for Automated Driving Systems",
  "abstract": "Closed-track testing plays a fundamental role in the verification and validation of automated driving systems (ADS), particularly for safety-critical scenarios, by enabling reproducible evaluation under controlled conditions. However, most existing approaches still rely on standardized protocols or predefined trajectories, leading to overly scripted interactions and limited ability to reproduce the natural complexity of public-road traffic. To address this limitation, we propose Video2Track, a framework that transfers real-world interactive driving scenarios from videos into steerable adversarial closed-track testing. The framework consists of two tightly coupled modules. The first is a scenario semantic mapping module, which extracts structured semantics from driving videos using a vision-language model and grounds them onto a closed-track topology library via retrieval-augmented generation, thereby identifying compatible map segments and interaction anchors. The second is a dynamic interactive testing module, which conditions on the grounded topology and anchors to generate diverse multi-agent trajectories through a conditional diffusion model, while regulating interaction intensity via a Stackelberg game with a parameterized adversarial objective. Closed-track experiments demonstrate that the proposed framework can faithfully reproduce representative real-world interaction scenarios and generate executable scenario variants with controllable risk levels and interaction styles, providing a scalable approach for realistic and steerable ADS validation.",
  "published": "2026-08-12",
  "updated": "2026-08-12",
  "year": "2026",
  "authors": [
   "Mengjie Tian",
   "Xinrui Zhang",
   "Tianyu Li",
   "Peizhi Zhang",
   "Guirong Zhou",
   "Haojie Feng",
   "Junpeng Huang",
   "Qixiang Zhang",
   "Lu Xiong"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "Video2Track is proposed, a framework that transfers real-world interactive driving scenarios from videos into steerable adversarial closed-track testing, and can faithfully reproduce representative real-world interaction scenarios and generate executable scenario variants with controllable risk levels and interaction styles.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Mengjie Tian",
    "id": "2375168161",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Xinrui Zhang",
    "id": "2311568440",
    "h_index": 4,
    "papers": 17
   },
   {
    "name": "Tianyu Li",
    "id": "2118911129",
    "h_index": 14,
    "papers": 30
   },
   {
    "name": "Peizhi Zhang",
    "id": "152487584",
    "h_index": 9,
    "papers": 41
   },
   {
    "name": "Guirong Zhou",
    "id": "2457007879",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Haojie Feng",
    "id": "2355804851",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Junpeng Huang",
    "id": "2346842135",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Qixiang Zhang",
    "id": "2457296319",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Lu Xiong",
    "id": "2346330882",
    "h_index": 2,
    "papers": 8
   }
  ],
  "comment": "",
  "topics": [
   "video-generation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.11592v1",
  "pdf_url": "https://arxiv.org/pdf/2608.11592v1",
  "html_url": "https://arxiv.org/html/2608.11592v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2608.11580",
  "slug": "roadweaver-large-scale-lane-level-hd-map-generation-from-scratch-for-a",
  "title": "RoadWeaver: Large-Scale Lane-Level HD Map Generation from Scratch for Autonomous Driving Simulation",
  "abstract": "Autonomous driving simulation requires diverse and scalable lane-level HD maps to support long-horizon evaluation across complex road networks. Existing approaches either rely on handcrafted or reconstructed real-world maps, which limits scalability, or generate only local road structures rather than complete HD maps. We present RoadWeaver, a coarse-to-fine framework for from-scratch generation of diverse, large-scale HD maps. RoadWeaver first synthesizes a global road layout, expands it into a connected road network, and then constructs lane-level geometry with topologically consistent lane connectivity. Experimental results show that RoadWeaver achieves a 99.8\\% reachability, a 10.7\\% dead-end ratio, and an endpoint alignment error of 0.24 m. Compared with SOTA generation methods, it reduces endpoint alignment error by 94.4\\% while generating complete HD maps in 1.39--3.50 s. The generated maps can be directly deployed in driving simulators, providing scalable simulation environments for future closed-loop evaluation of autonomous driving systems. The training code and an out-of-the-box implementation of RoadWeaver will be released upon acceptance.",
  "published": "2026-08-12",
  "updated": "2026-08-12",
  "year": "2026",
  "authors": [
   "Yueyuan Li",
   "Zexi Chen",
   "Weijie Xi",
   "Mingyang Jiang",
   "Songan Zhang",
   "Hanyang Zhuang",
   "Ming Yang"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "RoadWeaver is presented, a coarse-to-fine framework for from-scratch generation of diverse, large-scale HD maps, which first synthesizes a global road layout, expands it into a connected road network, and then constructs lane-level geometry with topologically consistent lane connectivity.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yueyuan Li",
    "id": "2267448151",
    "h_index": 3,
    "papers": 13
   },
   {
    "name": "Zexi Chen",
    "id": "2457217337",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Weijie Xi",
    "id": "2457000511",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Mingyang Jiang",
    "id": "2267335524",
    "h_index": 3,
    "papers": 13
   },
   {
    "name": "Songan Zhang",
    "id": "2267428852",
    "h_index": 5,
    "papers": 16
   },
   {
    "name": "Hanyang Zhuang",
    "id": "91124478",
    "h_index": 11,
    "papers": 60
   },
   {
    "name": "Ming Yang",
    "id": "2267843422",
    "h_index": 4,
    "papers": 19
   }
  ],
  "comment": "8 pages, 6 figures, 2 tables",
  "topics": [
   "sim2real",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.11580v1",
  "pdf_url": "https://arxiv.org/pdf/2608.11580v1",
  "html_url": "https://arxiv.org/html/2608.11580v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.11564",
  "slug": "repurposing-rgb-based-foundation-model-for-depth-estimation-on-thermal",
  "title": "Repurposing RGB-based Foundation Model for Depth Estimation on Thermal Images Using Hierarchical Supervision",
  "abstract": "Depth estimation from thermal images is highly valuable for robotic applications in adverse conditions, such as nighttime and rainy weather. Recent studies have sought to transfer knowledge from RGB-based foundation models to thermal modalities, yet the rich hierarchical representations these models encode remain underutilized. To address this limitation, we propose RGB-HS, a novel framework for thermal-image depth estimation that leverages hierarchical supervision from an RGB-based foundation model. Specifically, we first replace the baseline thermal encoder with a foundational model and introduce a parallel RGB branch that also employs a foundational model as an encoder of the same architecture, taking RGB images as input. The alignment is then performed across multiple levels between the tokens of the two encoders, allowing the thermal student branch to capture both structural precision and semantic abstraction from the RGB teacher branch. Furthermore, we introduce verification to refine the alignment process by weighting tokens from the RGB branch based on RGB image quality. Extensive experiments on the popular benchmark demonstrate that RGB-HS achieves competitive performance and more effectively exploits the representational capacity of RGB-based foundation models for depth estimation on thermal images.",
  "published": "2026-08-12",
  "updated": "2026-08-12",
  "year": "2026",
  "authors": [
   "Jie Hong",
   "Tingtian Li",
   "Xuesong Li",
   "Xiao Li"
  ],
  "author_count": 4,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "RGB-HS is proposed, a novel framework for thermal-image depth estimation that leverages hierarchical supervision from an RGB-based foundation model and introduces a parallel RGB branch that also employs a foundational model as an encoder of the same architecture, taking RGB images as input.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jie Hong",
    "id": "2295867000",
    "h_index": 2,
    "papers": 14
   },
   {
    "name": "Tingtian Li",
    "id": "143626433",
    "h_index": 7,
    "papers": 18
   },
   {
    "name": "Xuesong Li",
    "id": "2295679522",
    "h_index": 3,
    "papers": 15
   },
   {
    "name": "Xiao Li",
    "id": "2445565075",
    "h_index": 0,
    "papers": 3
   }
  ],
  "comment": "Accepted in IROS 2026",
  "topics": [
   "spatial-3d",
   "foundation-pretraining",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.11564v1",
  "pdf_url": "https://arxiv.org/pdf/2608.11564v1",
  "html_url": "https://arxiv.org/html/2608.11564v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.14713",
  "slug": "spotlessgs-relightable-3d-gaussian-splatting-under-dynamic-illuminatio",
  "title": "SpotlessGS: Relightable 3D Gaussian Splatting under Dynamic Illumination for Robotic Perception",
  "abstract": "Robots operating in dark or poorly lit environments rely on onboard lights, which often produce uneven illumination that degrades downstream perception tasks. Prior approaches based on 2D image enhancement lack reliable supervision and fail to preserve multi-view geometric consistency. To address these limitations, we extend Dark Gaussian Splatting (DarkGS) toward a more accurate and flexible relightable 3D reconstruction framework. First, we eliminate the need for explicit light parameter calibration by jointly optimizing lighting parameters within the Gaussian Splatting framework. Second, we introduce a low-frequency illumination model based on spherical harmonics (SH) to capture spatially varying residual and ambient lighting effects. Third, we incorporate an MLP-based Bidirectional Reflectance Distribution Function (BRDF) to model non-Lambertian reflectance. Experiments on synthetic and real-world datasets demonstrate that our method effectively mitigates illumination artifacts while improving rendering quality and quantitative performance over prior approaches. We further validate its benefits for robotic perception through a downstream task.",
  "published": "2026-08-11",
  "updated": "2026-08-11",
  "year": "2026",
  "authors": [
   "Liang Hong",
   "Jiaxin Wei",
   "Simon Schaefer",
   "Stefan Leutenegger",
   "Jaehyung Jung"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Dark Gaussian Splatting is extended toward a more accurate and flexible relightable 3D reconstruction framework and an MLP-based Bidirectional Reflectance Distribution Function (BRDF) is incorporated to model non-Lambertian reflectance.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Liang Hong",
    "id": "2458048815",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Jiaxin Wei",
    "id": "2317035934",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Simon Schaefer",
    "id": "2237801279",
    "h_index": 7,
    "papers": 17
   },
   {
    "name": "Stefan Leutenegger",
    "id": "2268758886",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Jaehyung Jung",
    "id": "2321686937",
    "h_index": 3,
    "papers": 3
   }
  ],
  "comment": "Accepted to the 2026 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS 2026)",
  "topics": [
   "spatial-3d",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.14713v1",
  "pdf_url": "https://arxiv.org/pdf/2608.14713v1",
  "html_url": "https://arxiv.org/html/2608.14713v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.11451",
  "slug": "herding-end-to-end-autonomous-driving-via-neuro-symbolic-safety-guards",
  "title": "Herding End-to-End Autonomous Driving via Neuro-Symbolic Safety Guards",
  "abstract": "Modern end-to-end driving agents can achieve high average performance yet still violate basic traffic rules that a human driver would never miss. The reason is structural: they learn statistical patterns rather than the physical conditions that guarantee safe driving, leaving their decision-making process opaque and safety constraints unenforced. We introduce a neuro-symbolic safety guard, a lightweight module that attaches to the final command interface of an already-trained agent. Immediately before a command reaches the vehicle, it checks the command against explicit safety rules and, only when necessary, replaces it with the nearest safe alternative. Each intervention is directly executable and traceable to the rule that triggered it, while the guard itself requires no retraining and adds no learned component. Evaluated on the long-tail benchmarks Fail2Drive and Bench2Drive using the state-of-the-art TransFuser v6 (TFv6) as a case study, the guard improves Success Rate by 15% and reduces safety-critical collisions by up to 53%, while preserving the original Driving Score.",
  "published": "2026-08-11",
  "updated": "2026-08-11",
  "year": "2026",
  "authors": [
   "Sim\u00f3n Pati\u00f1o Idarraga",
   "Erick Silva",
   "Rehana Yasmin",
   "Ali Shoker"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.AI",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A neuro-symbolic safety guard is introduced, a lightweight module that attaches to the final command interface of an already-trained agent and checks the command against explicit safety rules and, only when necessary, replaces it with the nearest safe alternative.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Sim\u00f3n Pati\u00f1o Idarraga",
    "id": "2456997029",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Erick Silva",
    "id": "2351715296",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "R. Yasmin",
    "id": "50014441",
    "h_index": 9,
    "papers": 45
   },
   {
    "name": "Ali Shoker",
    "id": "35269115",
    "h_index": 10,
    "papers": 59
   }
  ],
  "comment": "",
  "topics": [
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.11451v1",
  "pdf_url": "https://arxiv.org/pdf/2608.11451v1",
  "html_url": "https://arxiv.org/html/2608.11451v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.11407",
  "slug": "top-down-traffic-scenario-generation-via-joint-initial-goal-diffusion",
  "title": "Top-down Traffic Scenario Generation via Joint Initial-Goal Diffusion and Trajectory Infilling",
  "abstract": "Robust traffic simulators are crucial for developing and testing autonomous vehicles to reduce the costly, labor-intensive real-world data collection process and the need for physical presence on the road. However, existing simulators require agents' initial states to generate trajectories, which limits scalability and diversity due to restrictions on the given initial states. While data-driven agent initialization has been widely studied, the generated initial states are not interpretable in terms of why the agents are initialized at those specific locations. Given known initial states, trajectory generation is also a challenging problem, as the model must learn the variability of the destination and how agents should reach it over time. In this paper, we propose TrafficDiffuser, a top-down traffic scenario generation framework that generates high-level traffic scenarios, defined by initial and goal state pairs, by jointly modeling them. The high-level scenario generation makes initial states better interpretable and reduces trajectory generation into as simple as an infilling problem. We demonstrate how the generated high-level traffic scenarios can be used, including constraining based on different trajectory modes and integrating them with existing trajectory generation models. We conduct extensive experiments on the Argoverse 2 motion prediction dataset to evaluate how well the generated outputs capture real-world distributions. In addition to generating goal states, TrafficDiffuser outperforms the next-best approach for agent initialization, reducing speed distribution distance by 55.3% and the off-road rate by 2.8%.",
  "published": "2026-08-11",
  "updated": "2026-08-11",
  "year": "2026",
  "authors": [
   "Da Saem Lee",
   "Yash Vardhan Pant",
   "Sebastian Fischmeister"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.MA"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "TrafficDiffuser is proposed, a top-down traffic scenario generation framework that generates high-level traffic scenarios, defined by initial and goal state pairs, by jointly modeling them and outperforms the next-best approach for agent initialization.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Daeun Lee",
    "id": "2284334007",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Yash Pant",
    "id": "2354226782",
    "h_index": 1,
    "papers": 10
   },
   {
    "name": "Sebastian Fischmeister",
    "id": "2354226865",
    "h_index": 1,
    "papers": 5
   }
  ],
  "comment": "Accepted for publication at the IEEE International Conference on Intelligent Transportation Systems (ITSC), 2026",
  "topics": [
   "sim2real",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.11407v1",
  "pdf_url": "https://arxiv.org/pdf/2608.11407v1",
  "html_url": "https://arxiv.org/html/2608.11407v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.11363",
  "slug": "adaptation-of-generalist-robot-policies-with-minimal-data",
  "title": "Adaptation of Generalist Robot Policies with Minimal Data",
  "abstract": "A central goal in robot learning is to move beyond task-specific human data collection toward robots that improve through autonomous interaction. Yet fully autonomous learning remains difficult with current policies: sparse rewards and weak zero-shot exploration make it unlikely that a robot will discover successful behavior from scratch. We study minimal-data adaptation, a regime in which a pre-trained robot policy must learn a new task from as little as one demonstration followed by autonomous online interaction. This setting serves as the closest tractable proxy for fully autonomous improvement, allowing us to study whether minimal human guidance can bootstrap autonomous learning and what algorithmic ingredients make it feasible. We build MiDAS, a simple offline-to-online RL recipe that first anchors a pre-trained VLA to the target task with behavior cloning on single/few demonstrations, then improves it through value-based online RL on a residual policy parameterization. Across LIBERO and RoboCasa, MiDAS recovers strong task performance from as little as one demonstration, substantially outperforming baselines and generalizing beyond demonstrated conditions. We further evaluate MiDAS on a bimanual YAM platform. Starting from a fragile low-success policy obtained from a single demonstration, MiDAS improves its robustness and learns new successful behaviors over ~6 hours of online interaction. To the best of our knowledge, this is the first demonstration of reliable robot policy adaptation from a single task demonstration.",
  "published": "2026-08-11",
  "updated": "2026-08-11",
  "year": "2026",
  "authors": [
   "Shreyas Kowshik",
   "Sreyas Venkataraman",
   "Leo Wang",
   "Niharika Pant",
   "Max Simchowitz",
   "Aviral Kumar"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work builds MiDAS, a simple offline-to-online RL recipe that first anchors a pre-trained VLA to the target task with behavior cloning on single/few demonstrations, then improves it through value-based online RL on a residual policy parameterization.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Shreyas Kowshik",
    "id": "2456995544",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Sreyas Venkataraman",
    "id": "2279751373",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Leo Wang",
    "id": "2411922000",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Niharika Pant",
    "id": "2456995542",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Max Simchowitz",
    "id": "2395672175",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Aviral Kumar",
    "id": "1488785534",
    "h_index": 46,
    "papers": 83
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "imitation-diffusion",
   "navigation",
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.11363v1",
  "pdf_url": "https://arxiv.org/pdf/2608.11363v1",
  "html_url": "https://arxiv.org/html/2608.11363v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.11350",
  "slug": "self-evolving-embodied-agents-via-skill-harness-evolution",
  "title": "Self-Evolving Embodied Agents via Skill-Harness Evolution",
  "abstract": "Embodied agents are increasingly built as systems around foundation models, where performance depends not only on model weights but also on the skills, context, action interfaces, and execution harness surrounding the model. While supervised fine-tuning and reinforcement learning can adapt agents to new environments, they require additional data, rewards, and training runs; meanwhile, many train-free code-centric approaches rely on programmable robot APIs that may be unavailable in fixed-interface settings. We propose SHAPER, a self-evolving framework for train-free embodied adaptation that keeps model parameters frozen and improves the non-parametric agent system by evolving reusable skills and a context-code harness through target-environment rollouts. In SHAPER, the same frozen model can serve as both planner and optimizer, refining its external skills and context-code harness without parameter updates. We evaluate SHAPER on VLABench and ESI-Bench, covering embodied agents with different low-level action interfaces, and compare against pure execution, supervised fine-tuning, and test-time-scaling baselines such as verifier-free selection and voting. Our results suggest that skill-and-harness optimization is a practical route to self-evolving embodied agents when model training is expensive, unavailable, or undesirable.",
  "published": "2026-08-11",
  "updated": "2026-08-11",
  "year": "2026",
  "authors": [
   "Peidong Wang",
   "Zhiming Ma",
   "Ying Chang",
   "Xufang Luo",
   "Xiaocui Yang",
   "Shi Feng",
   "Yuqing Yang",
   "Dongsheng Li"
  ],
  "author_count": 8,
  "categories": [
   "cs.CL",
   "cs.RO"
  ],
  "primary_category": "cs.CL",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes SHAPER, a self-evolving framework for train-free embodied adaptation that keeps model parameters frozen and improves the non-parametric agent system by evolving reusable skills and a context-code harness through target-environment rollouts.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Peidong Wang",
    "id": "2282963978",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Zhiming Ma",
    "id": "2336442658",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Ying Chang",
    "id": "2337351729",
    "h_index": 0,
    "papers": 5
   },
   {
    "name": "Xufang Luo",
    "id": "13289447",
    "h_index": 21,
    "papers": 63
   },
   {
    "name": "Xiaocui Yang",
    "id": "2135971356",
    "h_index": 13,
    "papers": 73
   },
   {
    "name": "Shi Feng",
    "id": "2347923365",
    "h_index": 4,
    "papers": 27
   },
   {
    "name": "Yuqing Yang",
    "id": "2125051198",
    "h_index": 30,
    "papers": 123
   },
   {
    "name": "Dongsheng Li",
    "id": "2305587638",
    "h_index": 9,
    "papers": 19
   }
  ],
  "comment": "",
  "topics": [
   "rl-control",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.11350v1",
  "pdf_url": "https://arxiv.org/pdf/2608.11350v1",
  "html_url": "https://arxiv.org/html/2608.11350v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.11204",
  "slug": "surgical-wam-a-world-action-model-for-data-efficient-surgical-robot-le",
  "title": "Surgical WAM: A World-Action Model for Data-Efficient Surgical Robot Learning",
  "abstract": "Learning reliable surgical manipulation policies is bottlenecked by the scarcity of action-labeled demonstrations: teleoperated surgical robot (e.g., dVRK) trajectories with synchronized kinematics are costly to collect, while surgical tasks demand precise contact handling, long-horizon reasoning, and bimanual coordination. Endoscopic video is comparatively inexpensive and abundant relative to synchronized video--kinematics trajectories, and a natural way to exploit it is to learn world models of surgical scenes. However, existing surgical world models use video primarily for simulation or policy evaluation, and rarely translate the learned dynamics into closed-loop control. This gap raises our central question: under a fixed budget of action-labeled demonstrations, does action-free video pretraining improve closed-loop surgical manipulation? To answer it, we introduce the Surgical World-Action Model (Surgical WAM), a unified generative model built on Cosmos Policy that jointly predicts future endoscopic observations and executable surgical robot action chunks. Surgical WAM first learns surgical visual dynamics from action-free video and is then fine-tuned on the fixed action-labeled budget; at deployment, it acts as a closed-loop, receding-horizon controller that executes a short prefix of each predicted action chunk and replans from the resulting observation. On a suite of four simulated surgical manipulation tasks, video pretraining improves the average success rate from 63.5% to 77.8%, including an absolute gain of 20 percentage points on PegTransfer, with the largest improvements on contact-rich and bimanual tasks. These results demonstrate that action-free video provides transferable visual dynamics priors for learning surgical robot control with limited action supervision, positioning data-efficient video pretraining as a practical path toward scaling up surgical robot learning.",
  "published": "2026-08-11",
  "updated": "2026-08-11",
  "year": "2026",
  "authors": [
   "Wenrui Bao",
   "Tianyun Jiang",
   "Zhiben Chen",
   "Ser-Nam Lim",
   "Peter D. Peng",
   "Yuzhang Shang"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The Surgical World-Action Model is introduced, a unified generative model built on Cosmos Policy that jointly predicts future endoscopic observations and executable surgical robot action chunks and acts as a closed-loop, receding-horizon controller that executes a short prefix of each predicted action chunk and replans from the resulting observation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Wenrui Bao",
    "id": "2382928127",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Tianyun Jiang",
    "id": "2456965960",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Zhiben Chen",
    "id": "2382929205",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Ser-Nam Lim",
    "id": "2456973276",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Peter D. Peng",
    "id": "2456895079",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yuzhang Shang",
    "id": "2380471827",
    "h_index": 2,
    "papers": 5
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "tactile",
   "foundation-pretraining"
  ],
  "orgs": [
   "NVIDIA"
  ],
  "abs_url": "https://arxiv.org/abs/2608.11204v1",
  "pdf_url": "https://arxiv.org/pdf/2608.11204v1",
  "html_url": "https://arxiv.org/html/2608.11204v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2608.11174",
  "slug": "viscore-diagnosing-planning-relevant-quality-in-latent-world-models",
  "title": "VIScore: Diagnosing Planning-Relevant Quality in Latent World Models",
  "abstract": "Regulating the latent space to an isotropic Gaussian distribution provides a stable and information-maximized landscape for world model planning. However, the latent space property and successful planning remain disconnected. We first study this by comparing SIGReg and VISReg, two regularization loss functions with the same distribution target but different properties. Compared with SIGReg, VISReg has more flexibility in controlling the weights of center, scale, and shape regularization, and a larger batch size brings a finer distribution approximation. We find that the former, despite being beneficial in self-supervised learning (SSL), does not help the planning, whereas the latter improves the planning success on out-of-domain (OOD) datasets. This motivates a deep understanding of the factors that correlate with the success rate. Unlike the previous metrics focusing on the encoded latent only, we propose the Veracity-Influence-Sobriety score (VIScore), a metric that quantifies the reachability and capacity of a predictor given the encoded feature, and the hallucination of the searching-based planner. Compared with straightness, physical-state probing, and empowerment, we show that, with the measurement covering encoder, predictor, and planner, VIScore explains the success rate better than the others, as reflected by a strong Spearman correlation. Specifically, VIScore consistently achieves a Spearman correlation over 0.75 on both seen and unseen models and datasets on the cross-task success rate pool. Moreover, VIScore is the only metric that has a calibration error below the constant fit across all testing scenarios, showcasing the importance of these three aspects in planning success. We hope this metric can help future studies on world model design and diagnosis.",
  "published": "2026-08-11",
  "updated": "2026-08-12",
  "year": "2026",
  "authors": [
   "Haiyu Wu",
   "Randall Balestriero",
   "Morgan Levine"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The Veracity-Influence-Sobriety score (VIScore), a metric that quantifies the reachability and capacity of a predictor given the encoded feature, and the hallucination of the searching-based planner, is proposed, showcasing the importance of these three aspects in planning success.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Haiyu Wu",
    "id": "103543652",
    "h_index": 8,
    "papers": 37
   },
   {
    "name": "Randall Balestriero",
    "id": "2378151189",
    "h_index": 3,
    "papers": 16
   },
   {
    "name": "Morgan E. Levine",
    "id": "2249974543",
    "h_index": 2,
    "papers": 7
   }
  ],
  "comment": "",
  "topics": [
   "world-models"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.11174v2",
  "pdf_url": "https://arxiv.org/pdf/2608.11174v2",
  "html_url": "https://arxiv.org/html/2608.11174v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.10997",
  "slug": "seeing-above-the-waves-a-modular-sensing-framework-for-data-acquisitio",
  "title": "Seeing above the waves: A modular sensing framework for data acquisition at sea",
  "abstract": "Advancing autonomy for surface vessels requires systematic evaluation of their sensing and perception subsystems. Yet, maritime environments impose unique challenges: sensor installation is constrained by vessel layout, environmental conditions such as fog or sea clutter are difficult to reproduce, and long-duration missions complicate data collection. This work addresses the question: How can we design a modular and reproducible sensor platform for maritime autonomy? We present a comprehensive design blueprint that incorporates diverse modalities - RADAR, LiDAR, IMU, GNSS, AIS, RGB and LWIR cameras, and weather sensors - to enhance environmental awareness and vessel proprioception. Supported by a dedicated ROS2-based software framework for data management, our modular platform enables long-term data collection, hardware-in-the-loop testing, and integration with existing sensors and algorithms. By unifying hardware design and data capture methodology, the platform enhances reproducibility and comparability across vessels and research projects. The proposed framework bridges engineering implementation and research methodology, providing the foundation for standardized, verifiable datasets essential to advancing situational awareness and autonomous maritime navigation.",
  "published": "2026-08-11",
  "updated": "2026-08-11",
  "year": "2026",
  "authors": [
   "Jonathan E. Schmidt",
   "Julius Wirbel",
   "P. Nicholas Hansen",
   "Morgan Lou\u00e9dec",
   "Christian L. H. Westerdahl",
   "Dimitrios Dagdilelis",
   "Roberto Galeazzi"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jonathan E. Schmidt",
    "id": "2386494625",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Julius Wirbel",
    "id": "2383172401",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "P. N. Hansen",
    "id": "153628398",
    "h_index": 8,
    "papers": 20
   },
   {
    "name": "Morgan Lou\u00e9dec",
    "id": "2386221903",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Christian Westerdahl",
    "id": "2397182395",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Dimitrios Dagdilelis",
    "id": "2148501409",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Roberto Galeazzi",
    "id": "2386213005",
    "h_index": 1,
    "papers": 6
   }
  ],
  "comment": "Submitted and accepted to the IFAC WC 2026 as an invited session paper for track 7.2 Transportation and Vehicle Systems - Marine Systems",
  "topics": [
   "navigation",
   "data-teleop",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.10997v1",
  "pdf_url": "https://arxiv.org/pdf/2608.10997v1",
  "html_url": "https://arxiv.org/html/2608.10997v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.10872",
  "slug": "robust-safety-filtering-for-input-constrained-underactuated-linear-sys",
  "title": "Robust Safety Filtering for Input-Constrained Underactuated Linear Systems",
  "abstract": "We present a robust safety-filtering framework for input-constrained underactuated linear systems subject to unknown disturbances. A baseline H-$\\infty$ input is derived from a zero-sum differential game, while a disturbance observer supplies an estimate and a transient error bound. The baseline input is adjusted using the disturbance estimate, while the estimate and its error bound are used to define robust high-order control barrier function constraints; forward invariance holds as long as the admissible-input set remains nonempty. For scalar-input systems, pointwise feasibility is determined from an exact input interval, and the interval width defines the feasibility margin. A finite-horizon H-$\\infty$ performance balance accounts for the accumulated deviation of the applied input from the baseline H-$\\infty$ policy. Simulations on a linearized two-wheeled balancing robot show how position and body-pitch constraints compete for the same bounded wheel-torque input.",
  "published": "2026-08-11",
  "updated": "2026-08-11",
  "year": "2026",
  "authors": [
   "Muhamad Rausyan Fikri"
  ],
  "author_count": 1,
  "categories": [
   "eess.SY",
   "cs.RO"
  ],
  "primary_category": "eess.SY",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Muhamad Rausyan Fikri",
    "id": "153939751",
    "h_index": 2,
    "papers": 11
   }
  ],
  "comment": "",
  "topics": [
   "rl-control",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.10872v1",
  "pdf_url": "https://arxiv.org/pdf/2608.10872v1",
  "html_url": "https://arxiv.org/html/2608.10872v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.10847",
  "slug": "enabling-scalable-kinesthetic-teaching-via-observer-based-hand-guiding",
  "title": "Enabling Scalable Kinesthetic Teaching via Observer-based Hand-guiding with Active Support",
  "abstract": "Kinesthetic teaching through robot hand-guiding provides a natural interface for collecting demonstrations in imitation learning and programming-by-demonstration. However, extended sessions cause operator fatigue, reducing demonstration quality and limiting scalability. Current industrial hand-guiding approaches typically provide no active assistance, and alternatives require costly wrist-mounted force-torque sensors or rely on learned motion priors unavailable for new tasks. We propose RHOAS, a hand-guiding scheme that actively supports operator-intended motions using model-based force estimation without additional hardware. Our approach considers robot hand-guiding as an actively controlled interaction by the human operator, rather than an interaction with a passive environment. Standard methods used for hand-guiding typically rely on general passivity-based compliant control architectures that unnecessarily increase operator effort and limit the range of demonstrable motions without providing the intended stability guarantees in active interaction. Instead, our design utilizes model-based external torque estimation, internal joint torque sensing, and redundant robot kinematics to actively support human physical input within the human interaction frequency bandwidth. We address practical challenges of relying on observer-based force estimation, including suppression of unmodeled joint elastic dynamic effects and measurement noise in the feedback path, reduced estimate accuracy close to kinematic singularities, and static gravity compensation errors. In a user study with 16 participants on a KUKA LWR iiwa we demonstrate statistically significant reductions in physical effort, improved maneuverability for both precise and agile tasks, and clear user preference.",
  "published": "2026-08-11",
  "updated": "2026-08-11",
  "year": "2026",
  "authors": [
   "Anna Tuma",
   "Giuseppe Monetti",
   "Jochen J. Steil",
   "Niels Dehio"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "RHOAS is proposed, a hand-guiding scheme that actively supports operator-intended motions using model-based force estimation without additional hardware and considers robot hand-guiding as an actively controlled interaction by the human operator, rather than an interaction with a passive environment.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Anna Tuma",
    "id": "2456892866",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "G. Monetti",
    "id": "47893688",
    "h_index": 4,
    "papers": 25
   },
   {
    "name": "Jochen J. Steil",
    "id": "2238802185",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Niels Dehio",
    "id": "2427869",
    "h_index": 11,
    "papers": 25
   }
  ],
  "comment": "",
  "topics": [
   "imitation-diffusion"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.10847v1",
  "pdf_url": "https://arxiv.org/pdf/2608.10847v1",
  "html_url": "https://arxiv.org/html/2608.10847v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.10824",
  "slug": "neural-introspection-gating-for-adaptive-kv-cache-reuse-in-vision-lang",
  "title": "Neural Introspection Gating for Adaptive KV-Cache Reuse in Vision-Language-Action Models",
  "abstract": "Vision-Language-Action(VLA) models map camera images and language instructions directly to motor commands through a single autoregressive transformer. In real-time control, they still spend substantial compute recomputing key-value(KV) representations for visual tokens that barely change across neighboring frames. Recent work such as VLA-Cache reduces that cost by reusing KV states for visually static patches, but its policy relies only on observation-space heuristics and does not account for the model's own uncertainty. We propose Gated VLA-Cache, a lightweight, training-free extension that augments visual-similarity caching with neural introspection. The method monitors the logit margin between the top two predicted action tokens, a zero-cost confidence signal available during decoding. When the margin drops below a threshold, the cache is invalidated and a full recompute is triggered. Evaluated on four LIBERO benchmark suites with both OpenVLA and OpenVLA-OFT, Gated VLA-Cache improves reliability when blind caching hurts. On LIBERO-Goal and LIBERO-Long, it recovers over 100% of the lost accuracy while retaining 80% of the compute savings.",
  "published": "2026-08-11",
  "updated": "2026-08-11",
  "year": "2026",
  "authors": [
   "Zhijie Wu",
   "Kento Kawaharazuka",
   "Kei Okada"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Gated VLA-Cache is proposed, a lightweight, training-free extension that augments visual-similarity caching with neural introspection that improves reliability when blind caching hurts.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhijie Wu",
    "id": "2146254998",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "Kento Kawaharazuka",
    "id": "8308607",
    "h_index": 17,
    "papers": 220
   },
   {
    "name": "Kei Okada",
    "id": "2248244895",
    "h_index": 5,
    "papers": 68
   }
  ],
  "comment": "6 pages, 5 figures, Accepted in IROS 2026. Project Page: https://zjw4321.github.io/neural-introspection-gating-page/",
  "topics": [
   "vla"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.10824v1",
  "pdf_url": "https://arxiv.org/pdf/2608.10824v1",
  "html_url": "https://arxiv.org/html/2608.10824v1",
  "code_url": "https://zjw4321.github.io/neural-introspection-gating-page/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.5
 },
 {
  "id": "2608.10817",
  "slug": "aecnav-active-evidence-consolidation-for-efficient-zero-shot-open-voca",
  "title": "AECNav: Active Evidence Consolidation for Efficient Zero-Shot Open-Vocabulary Object Navigation",
  "abstract": "Zero-shot object-goal navigation (ZSON) in open-vocabulary scenarios is challenging, as it requires a robot to locate an arbitrarily specified object in an unseen environment without task-specific training. Currently, the task still suffers from high latency and limited accuracy due to redundant perception pipelines and insufficient evidence for reliable target confirmation. In this letter, we reframe ZSON as an evidence-driven perception-to-decision problem and present AECNav, a training-free pipeline built on three components: i) Evidence-gated perception, which utilizes a shared encoding across all reasoning stages to establish a unified semantic basis and eliminate redundant computations; ii) Evidence consolidation, which aggregates detections into cluster-level log-odds beliefs. This explicitly separates genuine target support from the false confidence of visually similar distractors, while treating the absence of expected detections as negative evidence; and iii) Active evidence acquisition, which sustains productive exploration under weak semantic cues by selecting frontiers that maximize information gain at minimal traversal cost. As a result, AECNav significantly outperforms previous methods and achieves state-of-the-art success rates of 84.7%, 57.3%, and 51.3% on HM3D-v2, HM3D-OVON, and MP3D, respectively, with substantially lower inference overhead, and attains 95% success across 40 trials on a physical quadruped robot at roughly 5Hz. Code will be made publicly available upon acceptance.",
  "published": "2026-08-11",
  "updated": "2026-08-11",
  "year": "2026",
  "authors": [
   "Guanlin Liu",
   "Shaobin Ling",
   "Renyuan Liu",
   "Zeying Gong",
   "Junjie Hu"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This letter reframe ZSON as an evidence-driven perception-to-decision problem and presents AECNav, a training-free pipeline built on three components: i) Evidence-gated perception, which utilizes a shared encoding across all reasoning stages to establish a unified semantic basis and eliminate redundant computations.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Guanlin Liu",
    "id": "2456980753",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Shaobin Ling",
    "id": "2382924647",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Renyuan Liu",
    "id": "2456973860",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Zeying Gong",
    "id": "2249762165",
    "h_index": 6,
    "papers": 21
   },
   {
    "name": "Junjie Hu",
    "id": "1409846329",
    "h_index": 17,
    "papers": 51
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.10817v1",
  "pdf_url": "https://arxiv.org/pdf/2608.10817v1",
  "html_url": "https://arxiv.org/html/2608.10817v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.10791",
  "slug": "dual-stress-runtime-safety-monitoring-for-safety-constrained-mpc-navig",
  "title": "Dual Stress: Runtime Safety Monitoring for Safety-Constrained MPC Navigation",
  "abstract": "Runtime hazard monitors for autonomous naviga- tion are conventionally built from geometric quantities: predicted clearance, time to collision, and required deceleration. A model-predictive controller that enforces safety through explicit con- straints computes, as a by-product of every control step, a second information channel that such monitors ignore: the Karush-Kuhn-Tucker multipliers of its constrained optimization, which measure the marginal control effort spent to maintain safety against each obstacle. This paper evaluates whether a horizon-weighted sum of those multipliers, a dual stress signal, provides a hazard monitor complementary to the geometric warnings the same state already supports. We compare it against a battery of fifteen geometric detectors tuned to a matched false-alarm budget, on preregistered held-out crossing scenarios driven through a physics simulator. The stress alarm actionably flags 4.7 times as many collisions missed by the entire geometric battery as the geometric battery flags in return (85 versus 18); combined, the two channels warn of three quarters of the collisions for which braking remained feasible, against under half for the geometric battery alone.",
  "published": "2026-08-11",
  "updated": "2026-08-11",
  "year": "2026",
  "authors": [
   "Jamil Chahine",
   "Wenqi Cai",
   "John Abanes",
   "Anthony Tzes"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper evaluates whether a horizon-weighted sum of those multipliers, a dual stress signal, provides a hazard monitor complementary to the geometric warnings the same state already supports, against a battery of fifteen geometric detectors tuned to a matched false-alarm budget.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jamil\u00e9 Chahine",
    "id": "2243781985",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Wenqi Cai",
    "id": "2249303211",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "John Abanes",
    "id": "2332535283",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Anthony Tzes",
    "id": "2282297404",
    "h_index": 6,
    "papers": 58
   }
  ],
  "comment": "6 pages, 4 figures, 2 tables, submitted to the 13th International Conference on Automation, Robotics and Applications (ICARA 2027)",
  "topics": [
   "sim2real",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.10791v1",
  "pdf_url": "https://arxiv.org/pdf/2608.10791v1",
  "html_url": "https://arxiv.org/html/2608.10791v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.10756",
  "slug": "embodied-multimodal-grounding-for-open-vocabulary-mobile-manipulation",
  "title": "Embodied Multimodal Grounding for Open-Vocabulary Mobile Manipulation via Semantic 3D Gaussian Splatting",
  "abstract": "Embodied mobile manipulation requires language, visual observations, three-dimensional scene structure, and action feasibility to be aligned before execution. We study open-vocabulary target grounding with few-shot manipulation in local household workspaces and present an embodied multimodal grounding framework that integrates active multi-view Semantic 3D Gaussian Splatting (Semantic-3DGS), reachability-aware base positioning, and a diffusion-based vision-language-action policy. A task-driven local Semantic-3DGS serves as a shared interface across active sensing, language-conditioned 3D localization, obstacle-aware scene reasoning, base preparation, and semantic conditioning of the action model. To preserve pretrained action priors, the 3D semantic cues are injected only into the late action-expert blocks. In expanded 50-trial real-robot evaluations against representative vision-language-action (VLA) approaches, the full system achieves 60% long-horizon success compared with 40% for PointVLA and 28% for DexVLA, and reaches 74% success in heavily cluttered manipulation compared with 52% for the single-view variant and 46% for PointVLA. It also maintains 75% success under a 75 cm height shift and eliminates photo-induced false grasps. These results indicate that explicit, refreshable 3D semantic grounding can improve robustness under clutter, occlusion, viewpoint variation, and embodiment constraints.",
  "published": "2026-08-11",
  "updated": "2026-08-11",
  "year": "2026",
  "authors": [
   "Huosen Ou",
   "Dongni Song",
   "Yuncong Wang",
   "Tao Zhou",
   "Yiding Ji"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Results indicate that explicit, refreshable 3D semantic grounding can improve robustness under clutter, occlusion, viewpoint variation, and embodiment constraints.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Huosen Ou",
    "id": "2179264982",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Dongni Song",
    "id": "2456939306",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yuncong Wang",
    "id": "2447715305",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Tao Zhou",
    "id": "2453307563",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yiding Ji",
    "id": "2299294479",
    "h_index": 3,
    "papers": 17
   }
  ],
  "comment": "9 pages, 11 figures. Accepted to ACM Multimedia 2026 (MM '26)",
  "topics": [
   "vla",
   "dexterous-manipulation",
   "spatial-3d",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.10756v1",
  "pdf_url": "https://arxiv.org/pdf/2608.10756v1",
  "html_url": "https://arxiv.org/html/2608.10756v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.10718",
  "slug": "tcam-for-autonomous-deformable-manipulation-the-rmc2-champion-system-f",
  "title": "TCAM for Autonomous Deformable Manipulation: The RMC2 Champion System for WBCD 2026 Track 4",
  "abstract": "This technical report describes the RMC2 Team's champion solution for the WBCD 2026 Track 4: Deformable Manipulation Challenge. The task requires a robot to pick a single T-shirt from a stack, load it onto a printing pallet, align the collar with a target area, and smooth the printing region, a sequence that involves single-layer separation, deformable transport, precise placement, and contact-rich surface adjustment. The competition strongly incentivizes fully autonomous execution, motivating the development of an autonomous solution. We built a fully autonomous system around the TCAM (TermiBrain Causal Action Model) framework, with the design principle that hardware, perception, data, and learning should jointly reduce the physical interaction complexity the policy must handle. A custom 3D-printed gripper designed for single-layer fabric separation improves picking reliability on a dual-arm ARX X5 platform. A wrist-centric four-camera setup pairs upper fisheye cameras for task-level context with lower RGB cameras for close-range gripper-cloth contact observation. We combine portable UMI-style demonstrations with real-robot demonstrations collected on the deployable platform to provide both broad manipulation priors and deployment-specific dynamics. TCAM ties these components into a closed loop: each trajectory is analyzed to identify the physical factors contributing to its outcome, driving targeted data recollection and policy fine-tuning. The policy outputs 30-step end-effector delta-pose action chunks from a multi-view VLA backbone. In the final competition, our system loaded 25 T-shirts at an average of approximately 23 seconds per attempt, with 22 achieving the required surface smoothness, securing first place in Track 4.",
  "published": "2026-08-11",
  "updated": "2026-08-11",
  "year": "2026",
  "authors": [
   "Guangrui Shen",
   "Zhili He",
   "Shigang Wang",
   "Yuanjun Sun",
   "Qing Yu"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This technical report describes the RMC2 Team's champion solution for the WBCD 2026 Track 4: Deformable Manipulation Challenge, and builds a fully autonomous system around the TCAM framework, with the design principle that hardware, perception, data, and learning should jointly reduce the physical interaction complexity the policy must handle.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Guangrui Shen",
    "id": "2456893361",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Zhiling He",
    "id": "2456974046",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Shigang Wang",
    "id": "2456970432",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yuanjun Sun",
    "id": "2457147242",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Qing Yu",
    "id": "2456534746",
    "h_index": 0,
    "papers": 2
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "tactile",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.10718v1",
  "pdf_url": "https://arxiv.org/pdf/2608.10718v1",
  "html_url": "https://arxiv.org/html/2608.10718v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.10651",
  "slug": "oaa-three-phases-of-vocal-guidance-in-human-drone-teleoperation",
  "title": "OAA: Three Phases of Vocal Guidance in Human-Drone Teleoperation",
  "abstract": "Voice-guided teleoperation requires systems that adapt to the evolving dynamics of human guidance. Yet most voice-controlled robot systems treat spoken commands as a stationary stream, ignoring how the guide's communicative behavior changes as the task progresses. Using motion capture and speech data from two experimental configurations, humanhuman guidance (finger pointing, N =10 dyads) and humandrone teleoperation (gamepad control, N =29 dyads), we show that spontaneous vocal guidance consistently organizes into three kinematically and linguistically distinct phases: Orientation, Approach, and Adjustment. These phases are identified automatically via change point detection on 3D trajectory signals, and validated statistically (Kruskal-Wallis, p<.001). Three lexical families replicate across configurations: rotation vocabulary marks Orientation, translation vocabulary is scarce there, and attenuators accumulate toward Adjustment. Together with inter-utterance silence, these cues mark the Orientation boundary that speech rate alone leaves unmarked. The same three-phase structure emerges in both configurations despite radically different motor interfaces, suggesting it is an intrinsic property of human spatial guidance rather than an artifact of the experimental setup. We discuss implications for OAA-aware adaptive control in voice-guided teleoperation.",
  "published": "2026-08-11",
  "updated": "2026-08-11",
  "year": "2026",
  "authors": [
   "Allan Henry",
   "Christian Graff",
   "Solange Rossato",
   "Jos\u00e9-Ernesto Gomez-Balderas",
   "Sylvain Huet"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "It is shown that spontaneous vocal guidance consistently organizes into three kinematically and linguistically distinct phases: Orientation, Approach, and Adjustment, suggesting it is an intrinsic property of human spatial guidance rather than an artifact of the experimental setup.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "A. Henry",
    "id": "2387922628",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Christian Graff",
    "id": "2289878323",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Solange Rossato",
    "id": "2354850347",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "J. Gomez-Balderas",
    "id": "1401832461",
    "h_index": 5,
    "papers": 20
   },
   {
    "name": "Sylvain Huet",
    "id": "2263309601",
    "h_index": 2,
    "papers": 10
   }
  ],
  "comment": "",
  "topics": [
   "egocentric-data",
   "navigation",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.10651v1",
  "pdf_url": "https://arxiv.org/pdf/2608.10651v1",
  "html_url": "https://arxiv.org/html/2608.10651v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.10618",
  "slug": "toward-the-cognitive-physical-limits-of-embodied-intelligence-through",
  "title": "Toward the Cognitive--Physical Limits of Embodied Intelligence through a World-Model-Centric Autonomous Racing Agent",
  "abstract": "Embodied artificial intelligence aims to develop agents that perceive, reason, and act through continuous interaction with the physical world. However, most embodied systems are still evaluated within conservative safety margins or moderate interaction regimes, leaving their capability boundaries under extreme conditions insufficiently understood. Autonomous racing provides a stringent testbed by combining high-frequency localization and perception, adversarial interaction, near-saturated vehicle dynamics, and strict safety constraints. Existing systems push high-speed performance but rarely model and refine cognitive and physical limits jointly. Here we show that a world-model-centric autonomous racing agent provides a concrete step toward exploring these coupled limits. The framework learns predictive world models from near-limit successes and failures to capture interaction evolution, ego dynamics, and feasible-motion boundaries, coupling world-state construction, future-aware reasoning, and near-limit control in a closed-loop refinement process. Training data were collected from real-vehicle autonomous racing, where the onboard system maintained robust localization and perception at speeds up to 256.3 km/h and peak lateral acceleration of 26.8 m/s$^2$. In full-scale simulated racing, the well trained world-model-centric agent achieves an 88.3% interaction success rate across various challenging simulated racing scenarios. Closed-loop refinement of the world model and policy further improved utilization of cognitive-physical limits, recovery from failure modes, and generalization across varying conditions and unseen circuits. These results suggest a boundary-aware methodology in which world models help embodied agents represent, predict, and continually refine their capability boundaries for safer real-world deployment.",
  "published": "2026-08-11",
  "updated": "2026-08-11",
  "year": "2026",
  "authors": [
   "Zitong Shan",
   "Baichuan Lou",
   "Yanxin Zhou",
   "Shuge Wu",
   "Xianqi He",
   "Bolin Zhao",
   "Sheng Zhao",
   "Zhouheng Li",
   "Chee Kiong Ong",
   "King Ho Holden Li",
   "Chen Lv"
  ],
  "author_count": 11,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A boundary-aware methodology in which world models help embodied agents represent, predict, and continually refine their capability boundaries for safer real-world deployment is suggested.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zitong Shan",
    "id": "2164442430",
    "h_index": 6,
    "papers": 14
   },
   {
    "name": "Baichuan Lou",
    "id": "2095709284",
    "h_index": 5,
    "papers": 18
   },
   {
    "name": "Yanxin Zhou",
    "id": "2240495378",
    "h_index": 6,
    "papers": 17
   },
   {
    "name": "Shuge Wu",
    "id": "2315888879",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Xianqi He",
    "id": "2456976383",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Bolin Zhao",
    "id": "2179641325",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Sheng Zhao",
    "id": "2311074187",
    "h_index": 2,
    "papers": 14
   },
   {
    "name": "Zhouheng Li",
    "id": "2311856227",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "Chee Kiong Ong",
    "id": "2456894694",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "King Ho Holden Li",
    "id": "2219132492",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "C. L. S. O. Mechanical",
    "id": "2251681079",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "A. Engineering",
    "id": "88738504",
    "h_index": 7,
    "papers": 20
   },
   {
    "name": "Nanyang Technological University",
    "id": "88740224",
    "h_index": 3,
    "papers": 18
   },
   {
    "name": "Singapore",
    "id": "2066220022",
    "h_index": 5,
    "papers": 26
   },
   {
    "name": "Kelly Holding",
    "id": "2202416398",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "L.L.C",
    "id": "2446396741",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Abu Dhabi",
    "id": "52557015",
    "h_index": 15,
    "papers": 165
   },
   {
    "name": "U. Emirates",
    "id": "145045818",
    "h_index": 25,
    "papers": 456
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.10618v1",
  "pdf_url": "https://arxiv.org/pdf/2608.10618v1",
  "html_url": "https://arxiv.org/html/2608.10618v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.10484",
  "slug": "lost-in-reconstruction-aligning-action-representations-with-language-i",
  "title": "Lost in Reconstruction: Aligning Action Representations with Language in Vision-Language-Action Models",
  "abstract": "Action verbs describe not only the physical outcomes of actions, but also how those actions are performed. Yet action representations in vision-language-action models (VLAs) are typically optimized for reconstruction under L1/L2 losses in raw action space, where numerical proximity need not reflect linguistically meaningful distinctions. On BridgeV2, we show that action trajectories contain verb-grounding information beyond visual state changes, and that reconstruction-only discrete tokenization systematically erodes this information. To address this problem, we introduce SALT, a Semantically ALigned action Tokenizer that augments a VQ-VAE-style tokenizer with an auxiliary objective requiring a frozen vision-language model to recover the episode instruction from quantized action latents. Policies trained with SALT achieve 71.9% average success in SimplerEnv, compared with 42.7% for a reconstruction-only VQ-VAE tokenizer and 31.2% for FAST. SALT also develops verb-specialized codes while maintaining reconstruction fidelity. These results show that robot action trajectories provide a source of language grounding and that preserving this structure in action representations can substantially improve language-conditioned control.",
  "published": "2026-08-11",
  "updated": "2026-08-11",
  "year": "2026",
  "authors": [
   "Li Wenjie",
   "Yash Jangir",
   "Ignacy Stepka",
   "Yash Agarwal",
   "Marion Kipsang",
   "Yonatan Bisk"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CL"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "SALT is introduced, a Semantically ALigned action Tokenizer that augments a VQ-VAE-style tokenizer with an auxiliary objective requiring a frozen vision-language model to recover the episode instruction from quantized action latents to substantially improve language-conditioned control.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Wenjie Li",
    "id": "2455823364",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yash Jangir",
    "id": "2130181270",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Ignacy St\u0119pka",
    "id": "2295622370",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Yash Agarwal",
    "id": "2456705661",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Marion Kipsang",
    "id": "2399621134",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Yonatan Bisk",
    "id": "2372760417",
    "h_index": 4,
    "papers": 18
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.10484v1",
  "pdf_url": "https://arxiv.org/pdf/2608.10484v1",
  "html_url": "https://arxiv.org/html/2608.10484v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.10449",
  "slug": "pbd-ag-persistent-baseline-delta-active-graphs-with-uncertainty-aware",
  "title": "PBD-AG: Persistent Baseline-Delta Active Graphs with Uncertainty-Aware Inspection for Long-Horizon Service Robots",
  "abstract": "Long-horizon service robots require persistent world models that can be built autonomously in unseen environments and revised as task-relevant objects change. Existing methods rely on online mapping, which accumulates localization and observation errors, static scene representations that cannot capture persistent object changes, or holistic vision-language predictions that lack verifiable 3D geometric evidence. We present PBD-AG, a persistent baseline-delta active graph framework that decouples robot-verified stable fixtures from revisable dynamic object events. Under our framework, the robot autonomously bootstraps the structural baseline from onboard exploration and inspects discovered fixtures to ground hierarchical object beliefs. PBD-AG maintains reliability-weighted object states over geometry, semantics, identity, existence, and support relations, utilizing a geometric visibility gate to mitigate false deletions under occlusion. Inspection viewpoints are selected by a graph-conditioned policy that balances target coverage, travel cost, collision risk, and redundant observation. Simulation experiments in multiple environments and under controlled dynamic evaluation show higher aggregate coarse-fixture F1 than capability-matched controls, as well as stronger identity continuity and event recall. A qualitative physical-robot demonstration further illustrates integration with onboard sensing, providing a traceable world model for long-horizon robotic perception. The project page of PBD-AG is available at https://shuobao214.github.io/PBD-AG/",
  "published": "2026-08-11",
  "updated": "2026-08-12",
  "year": "2026",
  "authors": [
   "Shuo Bao",
   "Wei Dong",
   "Shuyue Zhang",
   "Ming Shang",
   "Yuchen Huang",
   "Han Yu",
   "Chengjie Xu",
   "Yiheng Bi",
   "Kai Sun",
   "Fuchun Sun",
   "Xinzhou Wang"
  ],
  "author_count": 11,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "PBD-AG is presented, a persistent baseline-delta active graph framework that decouples robot-verified stable fixtures from revisable dynamic object events, and maintains reliability-weighted object states over geometry, semantics, identity, existence, and support relations.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Shuo Bao",
    "id": "2456859199",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Wei Dong",
    "id": "2290852686",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Shuyue Zhang",
    "id": "2456984927",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Ming Shang",
    "id": "2456858887",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Yuchen Huang",
    "id": "2456309998",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Hangzheng Yu",
    "id": "2453956532",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Chengjie Xu",
    "id": "2445260400",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yiheng Bi",
    "id": "2456858917",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Kai Sun",
    "id": "2273285163",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Fuchun Sun",
    "id": "2242091789",
    "h_index": 9,
    "papers": 26
   },
   {
    "name": "Xinzhou Wang",
    "id": "2196924058",
    "h_index": 11,
    "papers": 21
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "spatial-3d",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.10449v2",
  "pdf_url": "https://arxiv.org/pdf/2608.10449v2",
  "html_url": "https://arxiv.org/html/2608.10449v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.10393",
  "slug": "hidden-in-plain-sight-diffusion-based-unrestricted-robotic-attacks-on",
  "title": "Hidden in Plain Sight: Diffusion-Based Unrestricted Robotic Attacks on Vision-Language-Action Models",
  "abstract": "Vision-Language-Action (VLA) models have shown strong capabilities in controlling robots across diverse manipulation tasks. However, their adversarial robustness remains largely underexplored, and exploiting this weakness can lead to physical-world harm. Existing attacks on VLA models often rely on pixel-space perturbations or white-box access, resulting in noticeable artifacts and limited deployability in real-world robotic systems. In this work, we propose DURA, a diffusion-based unrestricted robotic attack that generates visually natural adversarial patches for VLA models. DURA supports both white-box and black-box attack settings, where the black-box setting requires only the predicted actions of the victim model. By optimizing along the latent trajectory of a pretrained diffusion model, DURA generates visually natural patches while steering the robot toward attacker-specified target actions. Extensive experiments in both simulation and the real physical world show that DURA consistently outperforms existing methods. Our findings expose a safety risk for physically deployed VLA models and call for stronger defenses.",
  "published": "2026-08-11",
  "updated": "2026-08-11",
  "year": "2026",
  "authors": [
   "Jiahui Han",
   "Yuhui Yao",
   "Xin Wang",
   "Jiafei Cao",
   "Mingxuan Zhang",
   "Danfeng Shan",
   "Huiqi Deng",
   "Guanchu Wang",
   "Xia Hu"
  ],
  "author_count": 9,
  "categories": [
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.AI",
  "venue": "",
  "venue_source": "",
  "citations": null,
  "influential_citations": null,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [],
  "comment": "",
  "topics": [
   "vla",
   "video-generation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.10393v1",
  "pdf_url": "https://arxiv.org/pdf/2608.10393v1",
  "html_url": "https://arxiv.org/html/2608.10393v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.10386",
  "slug": "dreamer-sac-off-policy-learning-in-latent-world-models-for-sample-effi",
  "title": "Dreamer-SAC: Off-Policy Learning in Latent World Models for Sample-Efficient Autonomous Driving",
  "abstract": "Sample-efficient reinforcement learning for autonomous driving is often limited by the trade-off between data efficiency and model bias. While world models reduce the reliance on costly environment interactions, policy optimization over learned dynamics remains sensitive to prediction errors. This paper proposes the Dreamer-SAC framework, which integrates a recurrent state-space world model with an off-policy soft actor-critic algorithm trained directly in latent space. The framework uses a combination of real interactions and short-horizon generated trajectories with n-step target estimation and multi-objective supervision. Evaluated in autonomous driving scenarios with objectives encompassing driving efficiency and safety, the proposed framework consistently outperforms representative reinforcement learning baselines, including DreamerV3, SAC, and PPO, while achieving improved performance with substantially fewer real environment interactions. Experiments reveal an inverted-U relationship between rollout horizon and policy performance, where short-horizon latent rollouts achieve the best trade-off between additional training signals and accumulated model bias. Furthermore, n-step target estimation demonstrates more effectiveness over one-step temporal-difference targets in exploiting predicted experience for value learning.",
  "published": "2026-08-11",
  "updated": "2026-08-11",
  "year": "2026",
  "authors": [
   "Jiazhuo Li",
   "Linjiang Cao",
   "Qi Liu",
   "Xi Xiong"
  ],
  "author_count": 4,
  "categories": [
   "cs.LG",
   "cs.RO"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The Dreamer-SAC framework, which integrates a recurrent state-space world model with an off-policy soft actor-critic algorithm trained directly in latent space, uses a combination of real interactions and short-horizon generated trajectories with n-step target estimation and multi-objective supervision.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jiazhuo Li",
    "id": "2292294220",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Linjiang Cao",
    "id": "2422956496",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Qi Liu",
    "id": "2332691115",
    "h_index": 7,
    "papers": 29
   },
   {
    "name": "Xi Xiong",
    "id": "2272594813",
    "h_index": 4,
    "papers": 8
   }
  ],
  "comment": "13 pages, 6 figures",
  "topics": [
   "world-models",
   "rl-control",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.10386v1",
  "pdf_url": "https://arxiv.org/pdf/2608.10386v1",
  "html_url": "https://arxiv.org/html/2608.10386v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.10367",
  "slug": "a-neural-network-based-teleoperation-for-remote-controlled-vehicles",
  "title": "A Neural Network Based Teleoperation for Remote Controlled Vehicles",
  "abstract": "Direct teleoperation of vehicles faces critical technical bottlenecks: communication latency and the operator's inability to physically perceive unmodeled environmental disturbances (e.g., aerodynamic drag, bank angles) coupled with highly nonlinear tire-road dynamics. To address these challenges, we propose a tailored unilateral teleoperation framework. The system integrates the Wave Variable (WV) approach to passively guarantee stability under stochastic delays, and an adaptive Radial Basis Function Network (RBFN) to actively compensate for vehicle-specific uncertainties. Unlike existing WV-neural network architectures designed for bilateral robotic arms, our framework features decoupled adaptive laws specifically designed for vehicle longitudinal and lateral dynamics. Furthermore, compared to model-heavy predictive controllers, the model-free RBFN offers rapid online adaptation without heavy computational overhead. Building upon our preliminary theoretical formulation, this brief paper presents comprehensive comparative analyses and real-world hardware validations. Simulation benchmarks against PID, LQR, MPC, and NMPC demonstrate that the RBFN achieves superior robustness against unmodeled disturbances while requiring orders of magnitude less execution time than MPC and NMPC, making it ideal for resource-constrained vehicle edge computing. Finally, hardware-in-the-loop experiments using a 1/10th scale vehicle over a 4G network validate the system's practical feasibility, safety, and robust trajectory tracking under physical road uncertainties.",
  "published": "2026-08-11",
  "updated": "2026-08-11",
  "year": "2026",
  "authors": [
   "Ning Ding",
   "Azim Eskandarian"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO",
   "cs.HC"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Simulation benchmarks against PID, LQR, MPC, and NMPC demonstrate that the RBFN achieves superior robustness against unmodeled disturbances while requiring orders of magnitude less execution time than MPC and NMPC, making it ideal for resource-constrained vehicle edge computing.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ning Ding",
    "id": "2212909522",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "A. Eskandarian",
    "id": "2323305",
    "h_index": 34,
    "papers": 234
   }
  ],
  "comment": "",
  "topics": [
   "sim2real",
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.10367v1",
  "pdf_url": "https://arxiv.org/pdf/2608.10367v1",
  "html_url": "https://arxiv.org/html/2608.10367v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.10232",
  "slug": "fact-failure-aware-causal-training-for-world-action-models",
  "title": "FACT: Failure-Aware Causal Training for World-Action Models",
  "abstract": "Recent world-action models (WAMs) show that co-training policies with future prediction can provide physical priors for action generation. Building on the future-prediction ability of video models, many WAMs generate future videos and recover actions with inverse-dynamics models, or use these predicted videos as goal conditions for action generation. In both cases, the world model is trained mostly on successful demonstrations and has little reason to predict the consequences of bad actions. We introduce FACT, a causal World-Action Model that predicts future video and task progress conditioned on the executed action. This action-conditioned interface allows failure rollouts to supervise action consequences, turning bad actions into valid future targets rather than being discarded. Failure-aware training makes the progress predictor aware of both successful and failed action outcomes, which can optionally be used to score sampled action candidates at inference. Extensive experiments on simulation and real-world bimanual manipulation tasks show that FACT outperforms many existing baselines, improves as failure data are incorporated into training, and reduces success-biased future hallucination under bad actions. See more details at https://fact-wam.github.io/",
  "published": "2026-08-10",
  "updated": "2026-08-10",
  "year": "2026",
  "authors": [
   "Quanquan Peng",
   "Yutong Liang",
   "Rui Yan",
   "Nicklas Hansen",
   "Xiaolong Wang"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "FACT is introduced, a causal World-Action Model that predicts future video and task progress conditioned on the executed action, and allows failure rollouts to supervise action consequences, turning bad actions into valid future targets rather than being discarded.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Quanquan Peng",
    "id": "2409184845",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Yutong Liang",
    "id": "2358611288",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Rui Yan",
    "id": "2395719044",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Nicklas Hansen",
    "id": "1491707104",
    "h_index": 19,
    "papers": 39
   },
   {
    "name": "Xiaolong Wang",
    "id": "2255251113",
    "h_index": 10,
    "papers": 17
   }
  ],
  "comment": "",
  "topics": [
   "world-models"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.10232v1",
  "pdf_url": "https://arxiv.org/pdf/2608.10232v1",
  "html_url": "https://arxiv.org/html/2608.10232v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.10220",
  "slug": "whole-body-planning-for-humanoids-navigating-confined-spaces-via-self",
  "title": "Whole-Body Planning for Humanoids Navigating Confined Spaces via Self-Collision Avoidance References",
  "abstract": "Humanoid locomotion in highly confined environments requires navigating dense environmental obstacles and complex self-collision bounds while maintaining multi-contact dynamic feasibility. Traditional trajectory optimizers frequently struggle in these restricted spaces, as navigating the large collision space with splines on particle abstractions is insufficient and leads to poor local minima. To address this, we propose a three-stage whole-body planning framework that formulates kinematic path planning directly over kinematically reachable rigid-body volumes. By integrating differentiable collision avoidance into a reachability-constrained formulation, our framework synthesizes volume-informed guides that reliably guide a full-order trajectory optimizer over long horizons. We show that these optimized plans serve as high-quality references to train a residual reinforcement learning policy for robust online execution. We validate our approach on the Unitree G1 humanoid across three benchmark testbeds exceeding NIST emergency response standards, achieving restricted confinement ratios ($C_r < 1.5$). Our framework generates feasible trajectories across 12-to-18-second tasks with complex foot and hand contacts where standard baselines fail, while the learned policy successfully tracks these plans under extensive domain randomization in physics simulation.",
  "published": "2026-08-10",
  "updated": "2026-08-10",
  "year": "2026",
  "authors": [
   "Carlos Gonzalez",
   "Luis Sentis"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes a three-stage whole-body planning framework that formulates kinematic path planning directly over kinematically reachable rigid-body volumes and shows that these optimized plans serve as high-quality references to train a residual reinforcement learning policy for robust online execution.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Carlos Gonzalez",
    "id": "2237956971",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Luis Sentis",
    "id": "2237810419",
    "h_index": 6,
    "papers": 18
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "sim2real",
   "rl-control",
   "navigation"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2608.10220v1",
  "pdf_url": "https://arxiv.org/pdf/2608.10220v1",
  "html_url": "https://arxiv.org/html/2608.10220v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2608.10056",
  "slug": "navigating-the-proximity-safety-balance-constraint-decomposition-for-h",
  "title": "Navigating the Proximity-Safety Balance: Constraint Decomposition for Human Following in Pedestrian Crowds",
  "abstract": "Following a target human in crowded environments involves an inherent conflict between staying close to the target and navigating safely among surrounding pedestrians and obstacles. This conflict becomes more severe in dense scenarios, where aggressive following risks collisions and conservative margins lead to target loss, especially when pedestrian behaviors are unfamiliar or unpredictable. Existing reinforcement learning (RL) methods typically encode these competing objectives into a single dense reward, but the resulting proximity-safety balance is implicit and difficult to adjust across conditions. To address this, we decompose the human-following task into a sparse task reward and independent cost constraints within a multi-constraint RL formulation, where each constraint is managed through cost thresholds with direct behavioral meaning rather than implicit reward weight ratios, allowing explicit and tunable control over the trade-off. We further quantify the prediction uncertainty of human motions and integrate these estimates into the RL costs to enhance safety under unpredictable conditions. Extensive experiments across both in-distribution and out-of-distribution settings demonstrate that our method achieves an effective proximity-safety balance compared to baselines. Real-robot deployment further validates the feasibility of our method in real-world scenarios. More details are available on our project page: https://nav-ps-balance.github.io/.",
  "published": "2026-08-10",
  "updated": "2026-08-10",
  "year": "2026",
  "authors": [
   "Shiting Gong",
   "Jianpeng Yao",
   "Jinfeng Wang",
   "Marco Pavone",
   "Jiachen Li"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work decomposes the human-following task into a sparse task reward and independent cost constraints within a multi-constraint RL formulation, where each constraint is managed through cost thresholds with direct behavioral meaning rather than implicit reward weight ratios, allowing explicit and tunable control over the trade-off.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Shiting Gong",
    "id": "2364973688",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Jianpeng Yao",
    "id": "2312892469",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Jinfeng Wang",
    "id": "2456978828",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Marco Pavone",
    "id": "2237790577",
    "h_index": 26,
    "papers": 65
   },
   {
    "name": "Jiachen Li",
    "id": "2290852097",
    "h_index": 2,
    "papers": 6
   }
  ],
  "comment": "IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS 2026); Project Website: https://nav-ps-balance.github.io/",
  "topics": [
   "rl-control",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.10056v1",
  "pdf_url": "https://arxiv.org/pdf/2608.10056v1",
  "html_url": "https://arxiv.org/html/2608.10056v1",
  "code_url": "https://nav-ps-balance.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.5
 },
 {
  "id": "2608.09891",
  "slug": "rose-a-robotic-soft-esophagus-for-endoprosthetic-stent-testing",
  "title": "RoSE: A Robotic Soft Esophagus for Endoprosthetic Stent Testing",
  "abstract": "Soft robotic systems are well suited for developing devices for biomedical applications. A bio-mimicking robotic soft esophagus (RoSE) is developed as an in vitro testing device of endoprosthetic stents for dysphagia management. Endoprosthetic stent placement is an immediate and cost-effective therapy for dysphagia caused by malignant esophageal strictures from esophageal cancer. However, later stage complications, like stent migration, could weaken swallow efficacy in the esophagus. The stent radial force (RF) on the esophageal wall is pivotal in avoiding stent migration. Due to limited randomized controlled trials in patients, stent design and stenting guidelines remain incomplete. To address this knowledge deficit, we investigate RoSE by implanting two stents (A and B) of different radial stiffness characteristics, to measure stent RF and its effect on migration. Endoscopic manometry under peristalsis is also performed to study the impact of stenting and stent dysfunction on intra-bolus pressure signatures (IBPSs) and swallowing efficacy. Each implanted stent undergoes experiments with varied peristalsis velocity, wavelength, and bolus concentrations. The results show that stiffer stent B has a higher RF, whereas stent A maintains a lower RF profile due to lesser stiffness. High RF is necessary to minimize migration under prolonged peristaltic contractions in RoSE. For manometry, stent A slightly increases IBPS, but stiffer stent B significantly decreases IBPS, especially for higher-concentration boluses. If a stiffer stent buckles, it can reduce swallow efficacy and cause recurrent dysphagia. RoSE is therefore an innovative soft robotic platform for testing endoprosthetic stents and addressing clinical challenges in stent evaluation.",
  "published": "2026-08-10",
  "updated": "2026-08-10",
  "year": "2026",
  "authors": [
   "Dipankar Bhattacharya",
   "Sherine Jesna V. A.",
   "Leo K. Cheng",
   "Weiliang Xu"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "Soft Robotics",
  "venue_source": "semantic-scholar",
  "citations": 28,
  "influential_citations": 1,
  "tldr": "RoSE is an innovative soft robotic platform that is capable of testing various endoprosthetic stents, thereby offering a solution to many existing clinical challenges in the area of stent testing.",
  "doi": "10.1089/soro.2019.0205",
  "oa_pdf": "https://arxiv.org/pdf/2608.09891",
  "s2_authors": [
   {
    "name": "Dipankar Bhattacharya",
    "id": "2052797407",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "S. J. Ali",
    "id": "1741616163",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Leo K. Cheng",
    "id": "144787481",
    "h_index": 37,
    "papers": 315
   },
   {
    "name": "Weiliang Xu",
    "id": "41156497",
    "h_index": 37,
    "papers": 242
   }
  ],
  "comment": "Author accepted manuscript. 31 pages, 13 figures, 3 tables. Published in Soft Robotics (2021). Project page: https://bhattner143.github.io/rose-stent.github.io/",
  "topics": [
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.09891v1",
  "pdf_url": "https://arxiv.org/pdf/2608.09891v1",
  "html_url": "https://arxiv.org/html/2608.09891v1",
  "code_url": "https://bhattner143.github.io/rose-stent.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.96
 },
 {
  "id": "2608.09876",
  "slug": "energy-structured-latent-world-models-with-neural-time-fields-for-phys",
  "title": "Energy-Structured Latent World Models with Neural Time Fields for Physically Constistent Open-World Motion Planning",
  "abstract": "Physically consistent motion planning remains a fundamental challenge in embodied AI, as generated trajectories must strictly conform to real-world execution dynamics. While latent world models offer a promising approach by predicting these dynamics, existing methods learn unconstrained future representations where absorbed physics remains implicit. Therefore, they fail to form reusable physical knowledge, which compromises reliability in unpredictable open-world navigation. To address this, we propose a novel Energy-Structured Latent World Model (ELWM). Our key idea is to structure the ELWM latent state to explicitly carry energy and momentum, ensuring strictly causal transitions via dissipation and control ports. Trained on multimodal RGB-D and inertial interaction histories, our model guarantees physically consistent predictions. We further implement this for motion planning by constructing Physics-Conditioned Neural Time Fields (PC-NTF), a key technical cornerstone that integrates ELWM into an arrival time field via the Eikonal equation to yield a physically-informed navigation policy. Across held-out scenes, our evaluation reveals significant improvements. Compared to generic latent models, PC-NTF reduces 0.8-s motion-prediction NRMSE from 0.36 to 0.29. Against Active Neural Time Fields, it improves navigation success from 81.3% to 89.7% and SPL from 0.64 to 0.73, while cutting the physical collision rate from 12.1% to 5.8% and the Eikonal residual from 0.083 to 0.031. Beyond these targeted gains, our results demonstrate that embedding explicit physical structures into latent spaces intrinsically bridges the gap between predictive world models and safe, dynamically feasible motion planning.",
  "published": "2026-08-10",
  "updated": "2026-08-10",
  "year": "2026",
  "authors": [
   "Yapeng Liu",
   "Yuanzhao Zhai",
   "Bo Ding",
   "Huaimin Wang",
   "Lin Wang"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A novel Energy-Structured Latent Latent World Model to structure the ELWM latent state to explicitly carry energy and momentum, ensuring strictly causal transitions via dissipation and control ports, and demonstrates that embedding explicit physical structures into latent spaces intrinsically bridges the gap between predictive world models and safe, dynamically feasible motion planning.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yapeng Liu",
    "id": "2456988850",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yuanzhao Zhai",
    "id": "1931511592",
    "h_index": 7,
    "papers": 37
   },
   {
    "name": "Bo Ding",
    "id": "152854523",
    "h_index": 7,
    "papers": 25
   },
   {
    "name": "Huaimin Wang",
    "id": "2113255325",
    "h_index": 11,
    "papers": 62
   },
   {
    "name": "Lin Wang",
    "id": "2375286831",
    "h_index": 3,
    "papers": 10
   }
  ],
  "comment": "9 pages, 5 figures",
  "topics": [
   "world-models",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.09876v1",
  "pdf_url": "https://arxiv.org/pdf/2608.09876v1",
  "html_url": "https://arxiv.org/html/2608.09876v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.09860",
  "slug": "entanglement-free-trajectory-planning-for-tethered-mobile-robots-with",
  "title": "Entanglement-Free Trajectory Planning for Tethered Mobile Robots with a Slack Tether",
  "abstract": "In motion planning algorithms for tethered mobile robots, the entanglement state of the tether is a critical aspect to consider during the planning phase. This is particularly important in case of a slack tether, where the shape of the tether is not determined solely by the geometry of the environment and the location of the obstacles, but also by the dynamics of the tether, by the trajectory followed by the robot, and possibly by exogenous forces. In this scenario, preventing entanglement requires planning a robot trajectory that accounts for the entanglement definition and for the dynamics of the robot and of the tether. In this work, we propose a motion planning algorithm for tethered mobile robots with a slack tether that computes dynamically feasible entanglement-free trajectories to navigate through an environment with static obstacles. By considering the entanglement state during all the stages of the planning pipeline, we are able to compute safer trajectories that avoid entanglement during the motion of the robot. We achieve this through a three-step pipeline, which includes (i) the construction of a topological model of the entanglement-free configuration space of the tethered robot, (ii) the generation of a set of candidate paths using this model, and (iii) the computation of a dynamically feasible entanglement-free trajectory by solving a homotopy-constrained trajectory generation problem. The resulting trajectory can then be executed to lead the robot to its target location, while maintaining the tether in an entanglement-free configuration. We demonstrate the benefits of this algorithm in simulations, where we show how the planning algorithm avoids violations of the entanglement constraints, resulting in safer and more reliable trajectories.",
  "published": "2026-08-10",
  "updated": "2026-08-10",
  "year": "2026",
  "authors": [
   "Gianpietro Battocletti",
   "Dimitris Boskos",
   "Bart De Schutter"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes a motion planning algorithm for tethered mobile robots with a slack tether that computes dynamically feasible entanglement-free trajectories to navigate through an environment with static obstacles and demonstrates the benefits of this algorithm in simulations.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Gianpietro Battocletti",
    "id": "2128723506",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Dimitris Boskos",
    "id": "3174480",
    "h_index": 10,
    "papers": 46
   },
   {
    "name": "B. D. Schutter",
    "id": "2246170679",
    "h_index": 11,
    "papers": 86
   }
  ],
  "comment": "19 pages, 13 figures",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.09860v1",
  "pdf_url": "https://arxiv.org/pdf/2608.09860v1",
  "html_url": "https://arxiv.org/html/2608.09860v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.09853",
  "slug": "rynnvalue-scaling-robotic-value-foundation-models-with-temporal-distan",
  "title": "RynnValue: Scaling Robotic Value Foundation Models with Temporal Distance",
  "abstract": "General-purpose reward models are increasingly the bottleneck for scaling robot learning, yet the recipe for learning value-related capabilities from large-scale heterogeneous corpora remains underexplored. Existing approaches tie supervision to task-internal anchors such as preferences or normalized progress, none of which transfer cleanly across embodiments and data sources. We introduce RynnValue, an open-source value foundation model for robotic manipulation that replaces these anchors with temporal distance, the directed cost-to-go from an observation to the language-specified goal. Because temporal-distance labels can be derived directly from timestamps, RynnValue scales to over 7,000 hours and roughly 3M instruction-conditioned clips without preference or progress annotations. To make temporal-value learning reliable at scale, we combine random temporal sampling, temporal-order shuffling, and value-isolation attention, suppressing shortcuts that would leave predictions insensitive to failures and regressions. Trained without preference labels, RynnValue attains an average Kendall's tau_a of 0.675 on RBM-EVAL-OOD, surpassing the fully preference-supervised state of the art (0.655) and more than doubling a progress-only counterpart (0.292), while generalizing zero-shot to unseen tasks, embodiments, and viewpoints. Converted into dense rewards via potential-based shaping, it raises real-world policy success from 52.5% to 72.5% online and from 63.8% to 82.5% offline. These results establish temporal distance as a scalable supervision target and practical reward interface for generalist robot policies.",
  "published": "2026-08-10",
  "updated": "2026-08-10",
  "year": "2026",
  "authors": [
   "Dongchi Huang",
   "Hongyin Zhang",
   "Bohan Hou",
   "Siteng Huang",
   "Zhian Su",
   "Hang Guo",
   "Tong Lu",
   "Zhaofeng Xu",
   "Jiahao Tang",
   "Jianfei Yang",
   "Donglin Wang",
   "Peixi Peng",
   "Mingxiu Chen",
   "Deli Zhao",
   "Xin Li"
  ],
  "author_count": 15,
  "categories": [
   "cs.RO",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "RynnValue, an open-source value foundation model for robotic manipulation that replaces task-internal anchors with temporal distance, the directed cost-to-go from an observation to the language-specified goal, establishes temporal distance as a scalable supervision target and practical reward interface for generalist robot policies.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Dongchi Huang",
    "id": "2310399585",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Hongyin Zhang",
    "id": "2155343887",
    "h_index": 11,
    "papers": 29
   },
   {
    "name": "Bohan Hou",
    "id": "2335000929",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Siteng Huang",
    "id": "2371084328",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Zhian Su",
    "id": "2371108777",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Hang Guo",
    "id": "2324074676",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Tong Lu",
    "id": "2440105974",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Zhaofeng Xu",
    "id": "2456844745",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Jiahao Tang",
    "id": "2115856143",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Jianfei Yang",
    "id": "2398616773",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Donglin Wang",
    "id": "2275032226",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Peixi Peng",
    "id": "2363498743",
    "h_index": 3,
    "papers": 13
   },
   {
    "name": "Mingxiu Chen",
    "id": "2381065591",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Deli Zhao",
    "id": "2303980061",
    "h_index": 13,
    "papers": 24
   },
   {
    "name": "Xin Li",
    "id": "2376124596",
    "h_index": 5,
    "papers": 11
   }
  ],
  "comment": "23 pages, 5 figures",
  "topics": [
   "rl-control",
   "foundation-pretraining",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.09853v1",
  "pdf_url": "https://arxiv.org/pdf/2608.09853v1",
  "html_url": "https://arxiv.org/html/2608.09853v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2608.09816",
  "slug": "hierarchical-fast-slow-react-agent-for-zero-shot-object-goal-navigatio",
  "title": "Hierarchical Fast-Slow ReAct Agent for Zero-Shot Object-Goal Navigation",
  "abstract": "Zero-shot object-goal navigation (ZSON) requires a robot to find a named object category in a building it has never entered. The prevailing approach scores frontiers with a vision-language value map: every decision is another argmax over the map as it currently stands, and the evidence behind that score is discarded the moment it is taken. Systems that place a large vision-language model inside the perception-action loop typically query it on a fixed schedule from the current view alone; a room the robot walked through minutes earlier is never reconsidered, and a failed call has no defined fallback. We turn what the robot has already seen into the object of deliberation. Our hierarchical fast-slow agent leaves the value-map controller running at every step and writes a coordinate-anchored memory as it moves: a semantic grid of room types and confirmed object instances, together with a bounded store of pose-tagged keyframes. A VLM screens each candidate detection before it is written. A deliberative layer reads this memory in a bounded reason-retrieve-act loop. It wakes on structural events the reactive layer computes, reasons first over text, and recalls a first-person view only for candidates that text alone cannot separate. Per-invocation and per-run caps bound its calls, a call-free first tier resolves the most frequent stall, and any failure returns control to the reactive controller. Our system reaches 68.75% SR on HM3D v1 val and 47.29% on MP3D val, the highest success rate among the zero-shot methods compared here. Choosing among far frontiers by argmax instead of deliberating costs 3.40 SR points in a paired comparison over all 2000 HM3D episodes (95% CI [1.70, 5.05]); deliberating over every frontier does not recover them.",
  "published": "2026-08-10",
  "updated": "2026-08-11",
  "year": "2026",
  "authors": [
   "Zhaochen Lan",
   "Zhi Yang",
   "Yuxiang Fu",
   "Mengxiang Lin"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper presents a hierarchical fast-slow agent that turns what the robot has already seen into the object of deliberation in zero-shot object-goal navigation, and reaches the highest success rate among the zero-shot methods compared here.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhaochen Lan",
    "id": "2456705079",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Zhi Yang",
    "id": "2457305218",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yuxiang Fu",
    "id": "2158322569",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Mengxiang Lin",
    "id": "2253977788",
    "h_index": 3,
    "papers": 7
   }
  ],
  "comment": "11 pages, 6 figures",
  "topics": [
   "egocentric-data",
   "navigation",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.09816v2",
  "pdf_url": "https://arxiv.org/pdf/2608.09816v2",
  "html_url": "https://arxiv.org/html/2608.09816v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.09771",
  "slug": "slim-0-5b-learning-action-grounded-predictive-latents-for-robot-manipu",
  "title": "SLIM-0.5B: Learning Action-Grounded Predictive Latents for Robot Manipulation",
  "abstract": "Vision-language-action policies rely on large multimodal backbones to jointly perform perception, language conditioning, and action generation at every control step. Much of this capacity supports open-domain semantics, whereas continuous robot manipulation primarily requires compact representations of observations, actions, and the transitions induced by actions. Pixel-level world models provide another route, but predicting visual details irrelevant to control can be unnecessarily expensive. We propose SLIM (Self-supervised Latent Interaction Model), a compact 0.5B-parameter latent interaction policy. SLIM learns action-grounded predictive latents that capture both action-conditioned future transitions and the actions that explain observed changes. SLIM learns these representations through self-supervised masked trajectory prediction, combining action reconstruction with future-latent prediction. A compact Mixture-of-Transformers (MoT) backbone models interactions between observation latents and action tokens. The resulting policy is trained with flow matching for language-conditioned action generation. Across simulation benchmarks and real-world evaluation, SLIM matches or exceeds representative large-scale VLA and world-action-model baselines with fewer parameters, no additional embodied pretraining, lower inference latency, and substantially lower GPU memory usage.",
  "published": "2026-08-10",
  "updated": "2026-08-10",
  "year": "2026",
  "authors": [
   "Jingkai Wang",
   "Zihan Tang",
   "Gu Zhang",
   "Mingyu Cao",
   "Jiapeng Chen",
   "Jingjiao Zhao",
   "Xiansheng Chen",
   "Pengwei Wang",
   "Lemao Liu",
   "Dejing Dou"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "SLIM (Self-supervised Latent Interaction Model), a compact 0.5B-parameter latent interaction policy, which matches or exceeds representative large-scale VLA and world-action-model baselines with fewer parameters, no additional embodied pretraining, lower inference latency, and substantially lower GPU memory usage.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jingkai Wang",
    "id": "2456978661",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Zihan Tang",
    "id": "2456831609",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Gu Zhang",
    "id": "2220861421",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Mingyu Cao",
    "id": "2361073090",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Jiapeng Chen",
    "id": "2349540633",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Jingjiao Zhao",
    "id": "2456890583",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Xiansheng Chen",
    "id": "2384353683",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Pengwei Wang",
    "id": "2338357829",
    "h_index": 17,
    "papers": 44
   },
   {
    "name": "Lemao Liu",
    "id": "2273767663",
    "h_index": 4,
    "papers": 15
   },
   {
    "name": "Dejing Dou",
    "id": "2336954849",
    "h_index": 3,
    "papers": 11
   }
  ],
  "comment": "18 pages, 11 figures. Project page: https://kzz1031.github.io/slim-project-page/",
  "topics": [
   "world-models",
   "vla",
   "sim2real",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.09771v1",
  "pdf_url": "https://arxiv.org/pdf/2608.09771v1",
  "html_url": "https://arxiv.org/html/2608.09771v1",
  "code_url": "https://kzz1031.github.io/slim-project-page/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2608.09762",
  "slug": "efficient-real-world-online-reinforcement-learning-for-robot-manipulat",
  "title": "Efficient Real-World Online Reinforcement Learning for Robot Manipulation via Centralized Training and Critic Decomposition",
  "abstract": "Real-world online reinforcement learning (RL) provides a promising approach for training robotic manipulation policies directly in the physical world, avoiding the sim-to-real gap and enabling continuous policy refinement through human-in-the-loop interaction. Recent methods have demonstrated sample-efficient learning through human intervention but remain limited to small randomization ranges and encounter challenges with the non-stationarity induced by concurrently training multiple agents. To address these limitations, we introduce a unified framework that combines centralized training with decentralized execution (CTDE) and a Hybrid Reward Architecture (HRA). This enables multiple actors to share a centralized multi-head critic. The critic is decomposed into task and grasp heads, corresponding to the sparse task reward and a potential-based grasping reward, respectively. We accordingly reformulate the critic and actor objectives to exploit the decomposed Q-values while explicitly accounting for the categorical action distribution of the discrete gripper policy. Experimental results demonstrate that the proposed framework substantially improves both sample efficiency and policy performance. We validate our approach on two robotic arms and a simulated humanoid robot across tennis ball and banana pick-and-place, pot reset, and simulated block relocation tasks under dimension-wise domain randomization, approximately 5-25x larger than those considered in prior work. Compared with a state-of-the-art baseline, our method improves the success rate from 60% to 80% on tennis ball pick-and-place, from 60% to 90% on banana pick-and-place, and from 25% to 95% on simulated block relocation, while also successfully accomplishing a task where the baseline consistently fails. Videos and more details are available at our project website: https://hil-harc.github.io/.",
  "published": "2026-08-10",
  "updated": "2026-08-10",
  "year": "2026",
  "authors": [
   "Changhao Li",
   "Yifang Zhang",
   "Heng Zhang",
   "Davide Torielli",
   "Damiano Gasperini",
   "Arturo Laurenzi",
   "Luca Muratore",
   "Arash Ajoudani",
   "Nikos Tsagarakis"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A unified framework that combines centralized training with decentralized execution (CTDE) and a Hybrid Reward Architecture (HRA) is introduced that enables multiple actors to share a centralized multi-head critic and substantially improves both sample efficiency and policy performance.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Changhao Li",
    "id": "2452996881",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yifang Zhang",
    "id": "2145062560",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Heng Zhang",
    "id": "2453893121",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Davide Torielli",
    "id": "2148648865",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Damiano Gasperini",
    "id": "2333663361",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Arturo Laurenzi",
    "id": "20815291",
    "h_index": 18,
    "papers": 69
   },
   {
    "name": "L. Muratore",
    "id": "48421415",
    "h_index": 17,
    "papers": 63
   },
   {
    "name": "Arash Ajoudani",
    "id": "2349803262",
    "h_index": 7,
    "papers": 34
   },
   {
    "name": "Nikos G. Tsagarakis",
    "id": "2307917489",
    "h_index": 2,
    "papers": 11
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "humanoids",
   "sim2real",
   "rl-control",
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.09762v1",
  "pdf_url": "https://arxiv.org/pdf/2608.09762v1",
  "html_url": "https://arxiv.org/html/2608.09762v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.09731",
  "slug": "tams-task-aware-multi-view-adaptive-streaming-for-wireless-telerobotic",
  "title": "TAMS: Task-Aware Multi-View Adaptive Streaming for Wireless Telerobotic Manipulation",
  "abstract": "Wireless telerobotic manipulation relies on timely multi-view video feedback, but the available uplink bandwidth is often limited and dynamic. This paper presents Task-Aware Multi-View Adaptive Streaming (TAMS), a system that allocates video bitrate according to the current manipulation phase. TAMS infers task phase from lightweight robot-side signals and prioritizes the camera view most relevant to the operator while preserving baseline visibility for secondary views. Experiments on a six-degree-of-freedom (6-DoF) teleoperation testbed under three constrained network conditions show that TAMS improves primary view Structural Similarity Index (SSIM), reduces task completion time, and increases trial success rate compared with equal and static allocation baselines. Under the most constrained bandwidth condition, TAMS reduces mean completion time from 68.9 s to 43.9 s relative to equal allocation and increases trial success rate from 48% to 71%. Code is available at: https://github.com/Dzxx623/TAMS.",
  "published": "2026-08-10",
  "updated": "2026-08-10",
  "year": "2026",
  "authors": [
   "Zexin Deng",
   "Zhenhui Yuan",
   "Lu Tian",
   "Subhash Lakshminarayana",
   "Longhao Zou"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.MM"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Experiments on a six-degree-of-freedom teleoperation testbed under three constrained network conditions show that TAMS improves primary view Structural Similarity Index (SSIM), reduces task completion time, and increases trial success rate compared with equal and static allocation baselines.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zexin Deng",
    "id": "2408350069",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Zhenhui Yuan",
    "id": "2286129065",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Lu Tian",
    "id": "2450970793",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Subhash Lakshminarayana",
    "id": "2456707829",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Longhao Zou",
    "id": "2310969952",
    "h_index": 5,
    "papers": 24
   }
  ],
  "comment": "6 pages, 5 figures, 2 tables. Code available at: https://github.com/Dzxx623/TAMS",
  "topics": [
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.09731v1",
  "pdf_url": "https://arxiv.org/pdf/2608.09731v1",
  "html_url": "https://arxiv.org/html/2608.09731v1",
  "code_url": "https://github.com/Dzxx623/TAMS",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2608.09730",
  "slug": "world-tokens-enhancing-embodied-policies-with-training-time-world-mode",
  "title": "World Tokens: Enhancing Embodied Policies with Training-Time World Modeling",
  "abstract": "Vision-language-action (VLA) models are a widely adopted paradigm for embodied policies. They excel at efficient closed-loop control but do not explicitly model how physical scenes evolve as a task unfolds. Recently emerging world-action models (WAMs) leverage pretrained video world models to capture spatiotemporal evolution, yet retaining future generation or a large video backbone in the control loop substantially increases inference cost. We introduce World Tokens, an embodied policy architecture built around a World Adapter that bridges visual-language understanding, world-dynamics modeling, and action generation. It uses world modeling during training to enhance the action policy while preserving efficient deployment. Specifically, the World Adapter transforms VLM features into a fixed set of world tokens, which condition a jointly fine-tuned future-video denoiser and simultaneously serve as the action expert's sole visual-language context. This shared conditioning allows gradients from future-video denoising to directly shape the representation used for action prediction, while exclusive routing prevents the policy from bypassing that representation. At deployment, the world-model branch is removed, leaving only the VLM, World Adapter, and action expert, with no online video-model inference. With a 2B backbone and no embodied action pretraining, World Tokens is highly competitive on LIBERO, attains the best reported averages on SIMPLER, substantially improves real-world R1 Pro success over a matched action-only baseline, and generates each action chunk at VLA-level latency.",
  "published": "2026-08-10",
  "updated": "2026-08-10",
  "year": "2026",
  "authors": [
   "Qu Tang",
   "Benhui Zhuang",
   "Bo Yuan",
   "Xue Yu",
   "Longteng Guo",
   "Junlan Feng"
  ],
  "author_count": 6,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "World Tokens is an embodied policy architecture built around a World Adapter that bridges visual-language understanding, world-dynamics modeling, and action generation and is highly competitive on LIBERO, attains the best reported averages on SIMPLER, and substantially improves real-world R1 Pro success over a matched action-only baseline.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Qu Tang",
    "id": "2337855594",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Benhui Zhuang",
    "id": "28459413",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Bo Yuan",
    "id": "2385322730",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Xue Yu",
    "id": "2451217468",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Longteng Guo",
    "id": "26982950",
    "h_index": 18,
    "papers": 73
   },
   {
    "name": "Junlan Feng",
    "id": "2363504956",
    "h_index": 1,
    "papers": 9
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "vla",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.09730v1",
  "pdf_url": "https://arxiv.org/pdf/2608.09730v1",
  "html_url": "https://arxiv.org/html/2608.09730v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.09628",
  "slug": "satellite-trajectory-optimization-via-proximal-policy-optimization-for",
  "title": "Satellite Trajectory Optimization via Proximal Policy Optimization for Space Debris Avoidance",
  "abstract": "Collision avoidance systems are commonly used to avoid fragmentation events occurring in Low-Earth Orbit (LEO) and Geosynchronous Equatorial Orbit (GEO). However, these events have been growing in frequency as orbital congestion worsens with the launch of megaconstellations. Consequently, conjunction alerts and collision risks are becoming increasingly common. Current practices, which are commonly manual or rule-based, have difficulty scaling to these worsening dynamic environments. To address this intensifying situation, we propose a reinforcement-learning policy for autonomous collision avoidance, trained via Proximal Policy Optimization (PPO) along with an open-source, high-fidelity astrodynamics simulator for training and evaluation. In 1,000 deterministic GEO episodes, our agent achieves a 97.5% collision avoidance success rate, outperforming traditional controllers such as a rule-based baseline (20.7% success) and an impulsive delta-v planner baseline (27.5% success). To achieve these results, we designed a simulator to train and evaluate our agent, using real-world and simulated debris. We simulate Newtonian two-body dynamics using Sun/Moon third-body perturbations, fuel-dependent thrust, and configurable debris fields. The agent is trained with curriculum learning and shaped rewards oriented toward encouraging survival, adequate projected miss distance, and delta-v conservation. Finally, our evaluation consisted of a fully deterministic pipeline, including shared seeds, per-episode logs, and telemetry exports. Our work is a publicly available framework at https://purl.org/sat-trajectory-avoidance",
  "published": "2026-08-10",
  "updated": "2026-08-10",
  "year": "2026",
  "authors": [
   "Logan Luna",
   "Juan Ortiz Couder",
   "Raul Alejandro Vargas-Acosta"
  ],
  "author_count": 3,
  "categories": [
   "cs.LG",
   "cs.RO"
  ],
  "primary_category": "cs.LG",
  "venue": "IEEE Access",
  "venue_source": "semantic-scholar",
  "citations": 2,
  "influential_citations": 0,
  "tldr": "",
  "doi": "10.1109/ACCESS.2026.3655237",
  "oa_pdf": "https://doi.org/10.1109/access.2026.3655237",
  "s2_authors": [
   {
    "name": "Logan Luna",
    "id": "2357290560",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Juan Ortiz Couder",
    "id": "2312492773",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Raul Alejandro Vargas-Acosta",
    "id": "2408939157",
    "h_index": 1,
    "papers": 1
   }
  ],
  "comment": "18 pages, 16 figures. Published in IEEE Access, vol. 14, pp. 18138-18154, 2026",
  "topics": [
   "sim2real",
   "rl-control",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.09628v1",
  "pdf_url": "https://arxiv.org/pdf/2608.09628v1",
  "html_url": "https://arxiv.org/html/2608.09628v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.98
 },
 {
  "id": "2608.09602",
  "slug": "nonlinear-model-predictive-control-of-a-robotic-soft-esophagus",
  "title": "Nonlinear Model Predictive Control of a Robotic Soft Esophagus",
  "abstract": "Strictures caused by esophageal cancer can narrow down the esophageal lumen, leading to dysphagia. Palliation of dysphagia has driven the development of a Robotic Soft Esophagus (RoSE), which provides a novel in vitro platform for esophageal stent testing and food viscosity studies. In RoSE, peristaltic wave generation and control were done in an open-loop manner since the conduit lacked visibility and embedded sensing capability. Hence, in this work, RoSE version 2.0 (RoSEv2.0) is designed with embedded Time Of Flight (TOF) and pressure sensors to measure conduit displacement and air pressure, respectively, for modeling and control. Model Predictive Control (MPC) of RoSEv2.0 is implemented to govern the peristalsis and air pressure profile autonomously. The implemented MPC used Sparse Identification Nonlinear Dynamics with Control (SINDYC) models to estimate the future states of ROSEv2.0. The dynamic models are discovered from the TOF and pressure sensor data. Peristalsis waves of speed 20 mm/s, wavelength 75 mm, and amplitudes 5, 7.5, and 10 mm were successfully generated by the MPC. Additionally, RoSEv2.0 with the MPC was employed to perform stent migration testing with various food bolus consistencies. The major contribution claimed in this paper is the application of SINDYC-based MPC to solve the closed-loop control problem of RoSE for achieving desired peristaltic waves.",
  "published": "2026-08-10",
  "updated": "2026-08-10",
  "year": "2026",
  "authors": [
   "Dipankar Bhattacharya",
   "Ryman Hashem",
   "Leo K. Cheng",
   "Weiliang Xu"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 10,
  "influential_citations": 0,
  "tldr": "The major contribution claimed in this article is the application of SINDYC-based MPC to solve the closed-loop control problem of RoSE for achieving desired peristaltic waves.",
  "doi": "10.1109/TIE.2021.3121755",
  "oa_pdf": "https://arxiv.org/pdf/2608.09602",
  "s2_authors": [
   {
    "name": "Dipankar Bhattacharya",
    "id": "2052797407",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Ryman Hashem",
    "id": "31328086",
    "h_index": 8,
    "papers": 29
   },
   {
    "name": "Leo K. Cheng",
    "id": "144787481",
    "h_index": 37,
    "papers": 315
   },
   {
    "name": "Weiliang Xu",
    "id": "2155782739",
    "h_index": 8,
    "papers": 26
   }
  ],
  "comment": "Accepted manuscript. 12 pages. Published in IEEE Transactions on Industrial Electronics. Project page: https://bhattner143.github.io/rosev2-dtsindyc.github.io/ Code: https://github.com/bhattner143/SINDYc_MPC_RoSE_symmetric_peristaltic",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.09602v1",
  "pdf_url": "https://arxiv.org/pdf/2608.09602v1",
  "html_url": "https://arxiv.org/html/2608.09602v1",
  "code_url": "https://bhattner143.github.io/rosev2-dtsindyc.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.04
 },
 {
  "id": "2608.09547",
  "slug": "model-based-systems-engineering-framework-for-sysml-driven-design-of-a",
  "title": "Model-Based Systems Engineering Framework for SysML-Driven Design of Autonomous UAVs",
  "abstract": "Autonomous Unmanned Aerial Vehicles (UAVs) are complex cyber-physical systems that require the coordinated integration of flight control, navigation, perception, communication, power management, and mission-level decision-making under safety, timing, and reliability constraints. However, many autonomous UAV development workflows still rely on document-centric requirements, separated architectural descriptions, and software implementation artifacts, which can lead to ambiguity, interface inconsistencies, and weak traceability during early design. This paper presents a Model-Based Systems Engineering (MBSE) design framework for the SysML-driven development of autonomous UAVs. The proposed framework uses the Systems Modeling Language (SysML) as a formal design backbone to structure UAV development across four connected layers: stakeholder requirements, functional decomposition, logical architecture, and physical/software allocation. SysML requirement diagrams, activity diagrams, block definition diagrams, internal block diagrams, state machine diagrams, and parametric diagrams are used to capture the functional, structural, behavioral, interface, and performance aspects of the UAV system. The logical architecture is then systematically mapped to a Robot Operating System 2 (ROS 2) software architecture by relating SysML blocks to ROS 2 nodes, flow ports and connectors to topics, request-response interactions to services, and goal-oriented behaviors to actions. The framework is illustrated at the design level using representative autonomous UAV mission scenarios, including autonomous take-off, waypoint navigation, hover stabilization, obstacle avoidance, return-to-home, and emergency handling. The resulting model supports requirement allocation, interface definition, subsystem responsibility assignment, and verification planning before simulation or physical deployment.",
  "published": "2026-08-10",
  "updated": "2026-08-10",
  "year": "2026",
  "authors": [
   "Deekshitha Angadi",
   "Naveena Budda",
   "Vikas Agarwal",
   "Mohamed Samshad",
   "Bharath Kumar Suryadevara",
   "Narsimlu Kemsaram"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper presents a Model-Based Systems Engineering (MBSE) design framework for the SysML-driven development of autonomous UAVs that supports requirement allocation, interface definition, subsystem responsibility assignment, and verification planning before simulation or physical deployment.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Deekshitha Angadi",
    "id": "2400587158",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Naveena Budda",
    "id": "2347371454",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Vikas Agarwal",
    "id": "2347358827",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Mohamed Samshad",
    "id": "2329186284",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "B. Suryadevara",
    "id": "2003396908",
    "h_index": 0,
    "papers": 5
   },
   {
    "name": "Narsimlu Kemsaram",
    "id": "101882219",
    "h_index": 6,
    "papers": 22
   }
  ],
  "comment": "Accepted for presentation at the 2026 International Conference on Autonomous Aerial Vehicles (ICAAV-2026), 20-21 Aug 2026, Bengaluru, India",
  "topics": [
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.09547v1",
  "pdf_url": "https://arxiv.org/pdf/2608.09547v1",
  "html_url": "https://arxiv.org/html/2608.09547v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.09484",
  "slug": "graph-guided-safe-diffuser-topological-graph-guidance-for-safe-diffusi",
  "title": "Graph-Guided Safe Diffuser: Topological Graph Guidance for Safe Diffusion Planning",
  "abstract": "Many diffusion-based planners enforce safety through inference-time guidance, but such interleaved trajectory deformations often degrade kinematic feasibility due to manifold rupture. We propose Graph-Guided Safe Diffuser (G2SD), a hierarchical framework that leverages a high-level topological graph planner to guide a low-level diffusion model. G2SD enforces safety at a structural level by abstracting the data manifold into a learned latent graph, on which high-level planning is performed. Continuous trajectories are generated by diffusion planners, which are conditioned on the graph node representations selected by the high-level planner. Theoretical analyses demonstrate conditions under which manifold rupture occurs in diffusion planners, and show that G2SD improves safety by reducing the constraint violation probability as the number of segments increases. Experiments demonstrate that G2SD substantially outperforms baselines, increasing goal-reaching rate without any collision from 40-50% to 98% in Maze2D navigation and also achieving superior task scores in locomotion.",
  "published": "2026-08-10",
  "updated": "2026-08-10",
  "year": "2026",
  "authors": [
   "Nakgyu Yang",
   "KwangBin Lee",
   "SooJean Han"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Graph-Guided Safe Diffuser (G2SD), a hierarchical framework that leverages a high-level topological graph planner to guide a low-level diffusion model, is proposed, which enforces safety at a structural level by abstracting the data manifold into a learned latent graph, on which high-level planning is performed.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Nakgyu Yang",
    "id": "2456783889",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "KwangBin Lee",
    "id": "2374458817",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "SooJean Han",
    "id": "2373591361",
    "h_index": 1,
    "papers": 9
   }
  ],
  "comment": "18 pages, 2 figures",
  "topics": [
   "humanoids",
   "navigation",
   "video-generation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.09484v1",
  "pdf_url": "https://arxiv.org/pdf/2608.09484v1",
  "html_url": "https://arxiv.org/html/2608.09484v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.09448",
  "slug": "vane-reliable-test-time-training-for-vision-language-action-models-via",
  "title": "VANE: Reliable Test-Time Training for Vision-Language-Action Models via Future Visual Representation Prediction",
  "abstract": "Test-time training (TTT) offers a lightweight way to adapt vision--language--action (VLA) policies from unlabeled deployment streams, but it remains difficult to use reliably in closed-loop manipulation. A shared adaptation space can mix incompatible task corrections, while an online update can alter subsequent actions before its consequences are known. We introduce a reliable TTT framework for VLA policies (VANE). VANE conditions prompt adaptation on the current vision--language context and learns from the future visual consequences of executed actions. Candidate updates are isolated from the live policy, evaluated on subsequent observations, and committed only when supported by future evidence, making adaptation selective and reversible. On SimplerEnv WidowX, VANE improves average success by $3.2$ percentage points over the corresponding TTT baseline. Results on Google Robot further show that deployment-time gains remain task- and embodiment-dependent. Together, these results demonstrate a constrained, evidence-based approach to adapting VLA policies during interaction.",
  "published": "2026-08-10",
  "updated": "2026-08-12",
  "year": "2026",
  "authors": [
   "Hongjin Ji",
   "Guoyang Xia",
   "Luoyang Sun",
   "Fangxiang Feng",
   "Lei Ren"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A reliable TTT framework for VLA policies (VANE), where candidate updates are isolated from the live policy, evaluated on subsequent observations, and committed only when supported by future evidence, making adaptation selective and reversible.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "H. Ji",
    "id": "2310197335",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Guoyang Xia",
    "id": "2366066587",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Luoyang Sun",
    "id": "2282957446",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Fangxiang Feng",
    "id": "39825530",
    "h_index": 16,
    "papers": 64
   },
   {
    "name": "Lei Ren",
    "id": "2357293781",
    "h_index": 3,
    "papers": 14
   }
  ],
  "comment": "",
  "topics": [
   "vla"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.09448v2",
  "pdf_url": "https://arxiv.org/pdf/2608.09448v2",
  "html_url": "https://arxiv.org/html/2608.09448v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.09410",
  "slug": "skills-in-weights-memory-in-code-hybrid-learning-for-memory-dependent",
  "title": "Skills in Weights, Memory in Code: Hybrid Learning for Memory-Dependent Robot Manipulation",
  "abstract": "Modern vision-language-action (VLA) policies have acquired broad manipulation skills, but typically generate each action chunk from the current observation or a short fixed-length history. However, real-world manipulation is often non-Markovian, requiring robots to retain and reason over task-relevant information from long-horizon interaction histories to determine the next action. To address this challenge, we propose HyMeS, a hybrid learning framework that leverages the reasoning and memory-management capabilities of coding agents to steer a Markovian VLA for memory-dependent manipulation. Specifically, HyMeS learns low-level motor skills through gradient-based imitation learning, while a coding agent acquires high-level memory-management strategies through heuristic learning by iteratively updating an executable heuristic system from rollout feedback. Furthermore, we close the loop between steering and execution through multimodal stage-completion verification, which updates memory using proprioceptive signals and multi-frame VLM judgments. Compared with end-to-end memory-augmented VLAs, HyMeS requires demonstrations only for reusable motor skills rather than for every history-dependent task configuration, enabling data-efficient compositional generalization. On RoboMemArena, HyMeS improves mean cumulative success from 52.5% to 66.2% and mean task success from 41.3% to 60.1% over pi0.5, while outperforming PrediMem by 4.5 points in cumulative success and 14.5 points in task success.",
  "published": "2026-08-10",
  "updated": "2026-08-10",
  "year": "2026",
  "authors": [
   "Yunhao Zhao",
   "Zhenyang Ni",
   "Haoyang Chen",
   "Ruohan Zhang",
   "Qi Zhu"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "HyMeS, a hybrid learning framework that leverages the reasoning and memory-management capabilities of coding agents to steer a Markovian VLA for memory-dependent manipulation, is proposed, enabling data-efficient compositional generalization.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yunhao Zhao",
    "id": "2451000278",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Zhenyang Ni",
    "id": "2162837304",
    "h_index": 7,
    "papers": 12
   },
   {
    "name": "Haoyang Chen",
    "id": "2446891600",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Ruohan Zhang",
    "id": "2285397162",
    "h_index": 8,
    "papers": 17
   },
   {
    "name": "Qi Zhu",
    "id": "2269768951",
    "h_index": 5,
    "papers": 16
   }
  ],
  "comment": "9 pages, 4 figures, and 3 tables",
  "topics": [
   "vla",
   "imitation-diffusion",
   "safety-eval"
  ],
  "orgs": [
   "Physical Intelligence"
  ],
  "abs_url": "https://arxiv.org/abs/2608.09410v1",
  "pdf_url": "https://arxiv.org/pdf/2608.09410v1",
  "html_url": "https://arxiv.org/html/2608.09410v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2608.09381",
  "slug": "jepa-wam-learning-vision-language-action-policies-with-joint-embedding",
  "title": "JEPA-WAM: Learning Vision-Language-Action Policies with Joint-Embedding World Modeling",
  "abstract": "Robust robot control benefits from explicitly modeling state transitions, but video-generation world action models (WAMs) introduce substantial deployment cost. Existing latent WAMs avoid explicit future generation, but often compress predictive representations or separate predictive modeling from the representations used for action generation. We introduce JEPA-WAM, a latent WAM built in a pretrained V-JEPA space, which couples latent transition prediction with continuous action generation through a shared predictor. JEPA-WAM predicts a spatially structured joint current-future target that captures task-shared visual temporal structure between current and future observations, while preserving dense patch-level correspondence. Through the shared predictor, transition supervision directly shapes the backbone, from which dedicated representations are extracted for action prediction. The same design can also be instantiated in pretrained VLA policies while preserving their original perception and action pathways. On LIBERO-Plus, JEPA-WAM achieves 79.2%, the best result without large-scale robot-policy pretraining, while its pretrained $\u03c0_{0.5}$ instantiation reaches 86.3%, achieving the best overall performance. Experiments on RoboTwin 2.0 and real-world bimanual manipulation further demonstrate strong generalization under visual and spatial shifts.",
  "published": "2026-08-10",
  "updated": "2026-08-10",
  "year": "2026",
  "authors": [
   "Yihan Lin",
   "Jiawei He",
   "Shifeng Bao",
   "Chen Zhao",
   "Yang Li",
   "Xiaobo Wang",
   "Yan Wang",
   "Cheng Chi",
   "Jing Zhang"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "JEPA-WAM, a latent WAM built in a pretrained V-JEPA space, which couples latent transition prediction with continuous action generation through a shared predictor, predicts a spatially structured joint current-future target that captures task-shared visual temporal structure between current and future observations, while preserving dense patch-level correspondence.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yihan Lin",
    "id": "2303858887",
    "h_index": 3,
    "papers": 15
   },
   {
    "name": "Jiawei He",
    "id": "2153103015",
    "h_index": 21,
    "papers": 56
   },
   {
    "name": "Shifeng Bao",
    "id": "2376405150",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Chen Zhao",
    "id": "2279771742",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Yang Li",
    "id": "2371140693",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Xiaobo Wang",
    "id": "2455478486",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Yan Wang",
    "id": "2386705642",
    "h_index": 3,
    "papers": 17
   },
   {
    "name": "Cheng Chi",
    "id": "2384366428",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Jing Zhang",
    "id": "2268783318",
    "h_index": 6,
    "papers": 11
   }
  ],
  "comment": "22 pages, 7 figures. Project page: https://spritewithoutice.github.io/JEPA_WAM/",
  "topics": [
   "world-models",
   "vla",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.09381v1",
  "pdf_url": "https://arxiv.org/pdf/2608.09381v1",
  "html_url": "https://arxiv.org/html/2608.09381v1",
  "code_url": "https://spritewithoutice.github.io/JEPA_WAM/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2608.09303",
  "slug": "safe-chem-uncertainty-aware-policy-switching-for-robust-robotic-chemis",
  "title": "SAFE-CHEM: Uncertainty-Aware Policy Switching for Robust Robotic Chemistry",
  "abstract": "The deployment of autonomous robotic systems in chemistry laboratories is accelerating experimental workflows and providing the foundational data for AI-driven scientific discovery. However, despite the success of data-driven methods in acquiring dexterous skills, safety remains a primary barrier to their deployment in high-risk domains, such as early-stage materials chemistry experiments. Specifically, learning-based policies frequently struggle to distinguish between safe and unsafe actions, leading to overconfident extrapolation and potentially catastrophic failures. To mitigate these safety risks, we propose SAFE-CHEM, an uncertainty-aware framework designed for robust, learning-based robotic chemists. Our approach leverages an ensemble of recurrent neural network-based imitation learning policies to quantify epistemic uncertainty online through the variance of action predictions. By characterising the success-conditioned density of this variance using kernel density estimation, we introduce a hybrid control architecture that autonomously switches from the learned policy to a deterministic, rule-based backup controller when uncertainty exceeds a calibrated safety threshold. We evaluate SAFE-CHEM across three fundamental laboratory manipulation tasks, where our empirical results demonstrate that this hybrid strategy improves overall task success rates and reduces critical safety violations compared to traditional single-policy baselines. Finally, we demonstrate the practical viability of the framework through zero-shot sim-to-real transfer onto a physical Franka Production 3 robot manipulator.",
  "published": "2026-08-10",
  "updated": "2026-08-10",
  "year": "2026",
  "authors": [
   "Laura Jones",
   "Shazil Shahzad",
   "Ayesha Sana",
   "Gabriella Pizzuto"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes SAFE-CHEM, an uncertainty-aware framework designed for robust, learning-based robotic chemists that introduces a hybrid control architecture that autonomously switches from the learned policy to a deterministic, rule-based backup controller when uncertainty exceeds a calibrated safety threshold.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Laura Jones",
    "id": "2456670699",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Shazil Shahzad",
    "id": "2456706317",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Ayesha Sana",
    "id": "2079449644",
    "h_index": 3,
    "papers": 34
   },
   {
    "name": "Gabriella Pizzuto",
    "id": "70305781",
    "h_index": 10,
    "papers": 29
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "sim2real",
   "imitation-diffusion",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.09303v1",
  "pdf_url": "https://arxiv.org/pdf/2608.09303v1",
  "html_url": "https://arxiv.org/html/2608.09303v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.09298",
  "slug": "worldsimprobe-diagnosing-simulator-faithfulness-in-action-conditioned",
  "title": "WorldSimProbe: Diagnosing Simulator Faithfulness in Action-Conditioned World Models for Embodied Manipulation",
  "abstract": "Action-conditioned world models (ACWMs) promise to provide embodied AI with scalable predictive simulators for planning, policy evaluation, and data generation. Realizing this promise requires precise action-conditioned transitions rather than merely plausible outputs. Yet their applicability remains difficult to establish because prevailing evaluations emphasize visual quality, task outcomes, or coarse rollout-level responsiveness without directly testing simulator fidelity. To address this gap, we evaluate ACWMs through the observable capabilities expected of physical simulators. Accordingly, we formalize Observable Simulator Contract, a minimal contract that any action-conditioned physical simulator should satisfy: supplied actions must induce corresponding agent motion, and environment responses must be grounded in that realized motion. To operationalize this contract, we introduce WorldSimProbe, comprising five controlled suites spanning local control sensitivity, global trajectory variation, source-diverse actions, interaction grounding, and dynamics. Suite-specific evaluators assess simulator-relative calibration, dense action-to-motion correspondence, false-interaction grounding, and primitive-level dynamics. We evaluate six open-source ACWMs on more than 18,000 instances across RoboTwin, ManiSkill, and LIBERO. World-SimProbe reveals systematic action-realization degradation across control variation, structured failures in interaction grounding and dynamics, and benchmark signals consistent with human judgments and downstream outcomes. Together, this capability-based framework provides a transparent, and standardized paradigm for diagnosing ACWM simulator fidelity beyond coarse, task-directed evaluation.",
  "published": "2026-08-10",
  "updated": "2026-08-10",
  "year": "2026",
  "authors": [
   "Peterson Co",
   "Sicheng Hu",
   "Chunxuan Jiao",
   "Hongyang Cheng",
   "Yulin Luo",
   "Yijie Xu",
   "Sixiang Chen",
   "Zhongxia Zhao",
   "Zihao Wang",
   "DaFeng Chi",
   "Peidong Liu",
   "YuTong Chen",
   "Henghua Liu",
   "Zhihao Yuan",
   "Huizhu Jia",
   "Yuzheng Zhuang",
   "Tianle Zhang",
   "Liang Lin",
   "Huajie Tan",
   "Shanghang Zhang"
  ],
  "author_count": 20,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work formalizes Observable Simulator Contract, a minimal contract that any action-conditioned physical simulator should satisfy: supplied actions must induce corresponding agent motion, and environment responses must be grounded in that realized motion.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "P. Co",
    "id": "2183781780",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Sichen Hu",
    "id": "2403530766",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Chunxuan Jiao",
    "id": "2313376420",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Hongyang Cheng",
    "id": "2372562573",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Yulin Luo",
    "id": "2276754886",
    "h_index": 9,
    "papers": 17
   },
   {
    "name": "Yijie Xu",
    "id": "2456838237",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Sixiang Chen",
    "id": "2303462098",
    "h_index": 9,
    "papers": 12
   },
   {
    "name": "Zhongxia Zhao",
    "id": "2386784478",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Zihao Wang",
    "id": "2457120952",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Dafeng Chi",
    "id": "2203793260",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Peidong Liu",
    "id": "2268467993",
    "h_index": 3,
    "papers": 15
   },
   {
    "name": "Yutong Chen",
    "id": "2455108045",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Henghua Liu",
    "id": "2455666182",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Zhihao Yuan",
    "id": "2040839502",
    "h_index": 9,
    "papers": 23
   },
   {
    "name": "Huizhu Jia",
    "id": "2292897290",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Yuzheng Zhuang",
    "id": "8773733",
    "h_index": 13,
    "papers": 49
   },
   {
    "name": "Tianle Zhang",
    "id": "2304081264",
    "h_index": 5,
    "papers": 15
   },
   {
    "name": "Liang Lin",
    "id": "2294180866",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Huajie Tan",
    "id": "2348888831",
    "h_index": 8,
    "papers": 15
   },
   {
    "name": "Shanghang Zhang",
    "id": "2346116279",
    "h_index": 16,
    "papers": 51
   }
  ],
  "comment": "20 pages, 18 figures, and 10 tables, including supplementary material. Code and data: https://evophys.com/WorldSimProbe/",
  "topics": [
   "world-models",
   "sim2real"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.09298v1",
  "pdf_url": "https://arxiv.org/pdf/2608.09298v1",
  "html_url": "https://arxiv.org/html/2608.09298v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.09272",
  "slug": "real-time-nonlinear-mpc-via-sequential-quadratic-programming-with-stru",
  "title": "Real-Time Nonlinear MPC via Sequential Quadratic Programming with Structure-Exploiting ADMM and Interior-Point Methods for Underactuated Double-Pendulum Swing-Up",
  "abstract": "The 4th \"AI Olympics with RealAIGym\" competition, to be held at IJCAI-ECAI 2026 in Bremen, challenges participants to develop a global control policy for swinging up and stabilizing an underactuated two-link system in its upright position. In contrast to previous editions, participants develop and evaluate their control strategies directly on remotely accessible CloudPendulum hardware, with limited interaction time and without prior knowledge of the system's model parameters. This paper presents an optimal-control-based approach employing real-time nonlinear model predictive control implemented using sequential quadratic programming. The results demonstrate that the proposed SQP-based MPC controller achieves reliable swing-up and stabilization performance, while maintaining robustness against disturbances.",
  "published": "2026-08-10",
  "updated": "2026-08-10",
  "year": "2026",
  "authors": [
   "Nick Karydakis",
   "Konstantinos Chatzilygeroudis"
  ],
  "author_count": 2,
  "categories": [
   "eess.SY",
   "cs.RO"
  ],
  "primary_category": "eess.SY",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "An optimal-control-based approach employing real-time nonlinear model predictive control implemented using sequential quadratic programming is presented, which achieves reliable swing-up and stabilization performance, while maintaining robustness against disturbances.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Nick Karydakis",
    "id": "2456702474",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Konstantinos I. Chatzilygeroudis",
    "id": "2265653690",
    "h_index": 2,
    "papers": 18
   }
  ],
  "comment": "6 pages, 4 figures, 1 table, finalist in the 4th AI Olympics with RealAIGym (https://ai-olympics.dfki-bremen.de)",
  "topics": [
   "rl-control",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.09272v1",
  "pdf_url": "https://arxiv.org/pdf/2608.09272v1",
  "html_url": "https://arxiv.org/html/2608.09272v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.09258",
  "slug": "task-oriented-formation-decision-via-reinforcement-learning-herding-an",
  "title": "Task-Oriented Formation Decision via Reinforcement Learning: Herding an Attacking Swarm",
  "abstract": "Multi-robot systems can accomplish tasks that are difficult for a single robot by organizing into task-specific formations. Different from existing studies on multi-robot shape formation, we here study the task-oriented formation decision problem, with a focus on the herding task. This task is challenging due to the attackers' superior maneuverability and their unknown strategies. To address these challenges, we propose the following novel results. First, we encode the formation shape using a low-dimensional parameter vector. This parametric representation reformulates the formation decision as a parameter optimization problem, thereby resolving the limited flexibility of predefined shapes. By optimizing these formation parameters, the defenders' maneuverability disadvantage is mitigated through a formation shape that continuously adapts to task requirements. Second, we develop a reinforcement learning-based policy to regulate the formation parameters. Trained offline in simulations covering diverse attacking strategies, the learned policy can effectively handle adversarial unpredictability during online deployment. Comparative simulations against three baselines demonstrate that our method can successfully accomplish challenging herding tasks. Additional scalability simulations further verify its applicability to simulated scenarios involving dozens of robots. We also validate the practical feasibility of our method on a physical robotic platform with 3 attackers and 7 defenders.",
  "published": "2026-08-10",
  "updated": "2026-08-10",
  "year": "2026",
  "authors": [
   "Zhaozong Wang",
   "Guibin Sun",
   "Jinyong Chen",
   "Rui Zhou"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work studies the task-oriented formation decision problem, with a focus on the herding task, and develops a reinforcement learning-based policy to regulate the formation parameters.",
  "doi": "10.1109/tase.2026.3723454",
  "oa_pdf": "https://arxiv.org/pdf/2608.09258",
  "s2_authors": [
   {
    "name": "Zhaozong Wang",
    "id": "2328653679",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Guibin Sun",
    "id": "5657203",
    "h_index": 8,
    "papers": 22
   },
   {
    "name": "Jinyong Chen",
    "id": "2289412004",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Rui Zhou",
    "id": "145976284",
    "h_index": 2,
    "papers": 7
   }
  ],
  "comment": "Accepted for publication in IEEE Transactions on Automation Science and Engineering",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.09258v1",
  "pdf_url": "https://arxiv.org/pdf/2608.09258v1",
  "html_url": "https://arxiv.org/html/2608.09258v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.09196",
  "slug": "sain-structure-aware-interactive-navigation-with-active-dialogue-groun",
  "title": "SAIN: Structure-Aware Interactive Navigation with Active Dialogue Grounding for Mobile Robot",
  "abstract": "Most existing vision-language navigation tasks assume that instructions are complete and unambiguous. However, real-world robots often encounter natural human instructions that are ambiguous, underspecified, or incomplete, requiring them to resolve such uncertainties through active questioning. Interactive Instance Goal Navigation (IIGN) requires an embodied agent to find the specific instance under an ambiguous category-level instruction through active dialogue. However, existing dialogue-enabled methods often consume oracle answers as transient textual context for immediate decisions, rather than persistent spatial or object-centric structured state. We present SAIN, a zero-shot framework that turns active dialogue into persistent navigation state. Instead of consuming oracle answers as one-step text hints, SAIN compiles them into target evidence, route-level corridor memory, and object-candidate labels. These states are stored in structured value, room, graph, and object memories, then consumed by a unified policy for frontier ranking and final target approach. On the VL-LN IIGN benchmark, SAIN improves SR from 20.2 to 25.4 and SPL from 13.07 to 14.17 over the strongest reported dialogue-enabled baseline, while requiring no task-specific policy training. The results support dialogue-to-state conversion as an effective zero-shot mechanism for long-horizon interactive instance navigation. Project website: https://zorattc.github.io/SAIN/",
  "published": "2026-08-10",
  "updated": "2026-08-10",
  "year": "2026",
  "authors": [
   "Yuhao Cao",
   "Xiao Liu",
   "Yang Xie",
   "Lu Liu",
   "Haoyao Chen"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "SAIN is presented, a zero-shot framework that turns active dialogue into persistent navigation state and supports dialogue-to-state conversion as an effective zero-shot mechanism for long-horizon interactive instance navigation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yuhao Cao",
    "id": "2317913290",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Xiao Liu",
    "id": "2315892830",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Yang Xie",
    "id": "2366285306",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Luoyang Liu",
    "id": "2455824637",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Haoyao Chen",
    "id": "2345254637",
    "h_index": 3,
    "papers": 9
   }
  ],
  "comment": "8 pages, 4 figures, and 4 tables",
  "topics": [
   "navigation",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.09196v1",
  "pdf_url": "https://arxiv.org/pdf/2608.09196v1",
  "html_url": "https://arxiv.org/html/2608.09196v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.09177",
  "slug": "intuitive-directional-sense-presentation-to-the-torso-using-mckibben-b",
  "title": "Intuitive Directional Sense Presentation to the Torso Using McKibben-Based Surface Haptic Sensation in Immersive Space",
  "abstract": "In recent years, systems that utilize immersive space have been developed in various fields. Immersive spaces often contain considerable amounts of visual information; therefore, users often fail to obtain their desired information. Therefore, various methods have been developed to guide users toward haptic sensations. However, many of these methods have limitations in terms of the intuitive perception of haptic sensation and require practice for familiarization with haptic sensation. Fabric actuators are wearable haptic devices that combine fabric and McKibben artificial muscles to provide wearers with surface haptic sensation. These sensations can be provided to a wide area of the body with intuitive perception, instead of only to a part of the body. This paper presents a novel air pressure adjustment method for whole-body motion guidance using surface haptic sensations provided by a wearable fabric actuator. The proposed system can provide users with a directional sense without visual information in an immersive space. The effectiveness of the proposed system was evaluated through subject experiments and statistical data analysis. Finally, a directional sense presentation was conducted for users performing micromanipulations in a mixed-reality space to demonstrate the applicability of the proposed system for teleoperation.",
  "published": "2026-08-10",
  "updated": "2026-08-10",
  "year": "2026",
  "authors": [
   "Kenta Yokoe",
   "Tadayoshi Aoyama",
   "Yuki Funabora",
   "Masaru Takeuchi",
   "Yasuhisa Hasegawa"
  ],
  "author_count": 5,
  "categories": [
   "cs.HC",
   "cs.RO"
  ],
  "primary_category": "cs.HC",
  "venue": "",
  "venue_source": "",
  "citations": 2,
  "influential_citations": 0,
  "tldr": "",
  "doi": "10.1109/TOH.2024.3522897",
  "oa_pdf": "https://arxiv.org/pdf/2608.09177",
  "s2_authors": [
   {
    "name": "Kenta Yokoe",
    "id": "2147156163",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "T. Aoyama",
    "id": "1752849",
    "h_index": 21,
    "papers": 282
   },
   {
    "name": "Yuki Funabora",
    "id": "2377060",
    "h_index": 8,
    "papers": 154
   },
   {
    "name": "Masaru Takeuchi",
    "id": "2283502281",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Yasuhisa Hasegawa",
    "id": "2237520520",
    "h_index": 10,
    "papers": 63
   }
  ],
  "comment": "This is the accepted version of an article published in IEEE Transactions on Haptics 18(1), 244-254 (2025). DOI: 10.1109/TOH.2024.3522897. Open Access under CC BY-NC-ND 4.0",
  "topics": [
   "humanoids",
   "data-teleop",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.09177v1",
  "pdf_url": "https://arxiv.org/pdf/2608.09177v1",
  "html_url": "https://arxiv.org/html/2608.09177v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.48
 },
 {
  "id": "2608.09167",
  "slug": "intuitive-hand-positional-guidance-using-mckibben-based-surface-tactil",
  "title": "Intuitive Hand Positional Guidance Using McKibben-Based Surface Tactile Sensations to Shoulder and Elbow",
  "abstract": "Hand positional guidance with intuitive perception is crucial for enhancing user interaction and task performance in immersive environments. However, conventional hand positional guidance methods, relying on tactile sensations, lack intuitiveness. Consequently, users require instruction on the relationship between the tactile sensation and target position of the guidance before using these methods. Additionally, the user needs training to become familiar with tactile sensations. This study presents a hand positional guidance system with intuitive perception that leverages McKibben-based surface tactile sensations directed to the shoulder and elbow. We developed a wearable fabric actuator that provides McKibben-based surface tactile sensations to induce six specific movements: elbow flexion, extension, shoulder abduction, adduction, horizontal abduction, and horizontal adduction. The effectiveness of the actuator was experimentally validated, demonstrating its high accuracy in intuitively inducing six movements. An algorithm based on the equilibrium point hypothesis and Weber-Fechner law was implemented to regulate the intensity of the tactile sensations for hand positional guidance. Furthermore, the accuracy and speed of the proposed system were compared with that of conventional guidance methods utilizing synthesized speech and vibrotactile guidance.",
  "published": "2026-08-10",
  "updated": "2026-08-10",
  "year": "2026",
  "authors": [
   "Kenta Yokoe",
   "Yuki Funabora",
   "Tadayoshi Aoyama"
  ],
  "author_count": 3,
  "categories": [
   "cs.HC",
   "cs.RO"
  ],
  "primary_category": "cs.HC",
  "venue": "RA-L",
  "venue_source": "semantic-scholar",
  "citations": 3,
  "influential_citations": 0,
  "tldr": "A wearable fabric actuator is developed that provides McKibben-based surface tactile sensations to induce six specific movements to induce hand positional guidance, demonstrating its high accuracy in intuitively inducing six movements.",
  "doi": "10.1109/LRA.2025.3540579",
  "oa_pdf": "https://arxiv.org/pdf/2608.09167",
  "s2_authors": [
   {
    "name": "Kenta Yokoe",
    "id": "2147156163",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Yuki Funabora",
    "id": "2377060",
    "h_index": 8,
    "papers": 154
   },
   {
    "name": "T. Aoyama",
    "id": "1752849",
    "h_index": 21,
    "papers": 282
   }
  ],
  "comment": "This is the accepted version of an article published in IEEE Robotics and Automation Letters 10(4), 3254-3261 (2025). DOI: 10.1109/LRA.2025.3540579. Open Access under CC BY-NC-ND 4.0",
  "topics": [
   "tactile",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.09167v1",
  "pdf_url": "https://arxiv.org/pdf/2608.09167v1",
  "html_url": "https://arxiv.org/html/2608.09167v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.1
 },
 {
  "id": "2608.09166",
  "slug": "particle-based-conformal-prediction-for-contact-aware-uncertainty-cali",
  "title": "Particle-Based Conformal Prediction for Contact-Aware Uncertainty Calibration in Stratified Configuration Spaces",
  "abstract": "Reliable uncertainty representation is essential for deploying autonomous systems that interact with their environment, as robots must reason about how uncertainty arising from both stochasticity and model mismatch is impacted by contacts with obstacles (e.g., when navigating through a cluttered environment or inserting a part into an assembly). We propose Calibrated Particle-sets for Trans-dimensional Uncertainty Representation (CaPTURe), a geometry-aware, conformal prediction-based algorithm that generates probabilistically valid prediction regions of the unknown future system configuration using particle-based models of arbitrary fidelity. While calibrated uncertainty predictions are essential for safe and efficient planning, analytical or learned motion models are often inaccurate - due to limited data, simplifying assumptions, unmodeled effects, etc. - which can lead to unsafe executions or task failure. Additionally, when a robot contacts an obstacle, the distribution of its future configurations can become multimodal or disjoint, or lie along manifolds of lower intrinsic dimension than the space of possible robot configurations. Our method uses a calibration dataset of system transitions to locally calibrate motion uncertainty estimates, constructing regions guaranteed to contain the future robot configuration at a user-set probability. Our calibration procedure captures how motion uncertainty varies between contact-rich and contactless motions, leading to sufficient coverage in both cases. We evaluate our method on two simulated planning tasks: controlling a marble around a labyrinth and performing tight-tolerance peg-in-hole insertion with a manipulator. Compared to relevant baselines, CaPTURe achieves the user-specified coverage requirement both in and out of contact and achieves up to a 30% absolute improvement in task success rate over the best baseline.",
  "published": "2026-08-10",
  "updated": "2026-08-10",
  "year": "2026",
  "authors": [
   "Lu\u00eds Marques",
   "Kristian Popov",
   "Dmitry Berenson"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.LG",
   "stat.ME"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Calibrated Particle-sets for Trans-dimensional Uncertainty Representation (CaPTURe), a geometry-aware, conformal prediction-based algorithm that generates probabilistically valid prediction regions of the unknown future system configuration using particle-based models of arbitrary fidelity is proposed.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Lu\u00eds Marques",
    "id": "2320807524",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "K. Popov",
    "id": "2070931553",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Dmitry Berenson",
    "id": "2320801045",
    "h_index": 2,
    "papers": 4
   }
  ],
  "comment": "31 pages, 10 figures, 5 tables. Accepted at COPA 2026 (Conformal and Probabilistic Prediction with Applications). Project page: https://um-arm-lab.github.io/capture/",
  "topics": [
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.09166v1",
  "pdf_url": "https://arxiv.org/pdf/2608.09166v1",
  "html_url": "https://arxiv.org/html/2608.09166v1",
  "code_url": "https://um-arm-lab.github.io/capture/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2608.09138",
  "slug": "speedtuning-speeding-up-policy-execution-with-lightweight-reinforcemen",
  "title": "SpeedTuning: Speeding Up Policy Execution with Lightweight Reinforcement Learning",
  "abstract": "While learned robotic policies hold promise for advancing generalizable manipulation, their practical deployment is often hindered by suboptimal execution speeds. Imitation learning policies are inherently limited by hardware constraints and the speed of the operator during data collection. In addition, there are no established methods for accelerating policies learned via imitation, and the empirical relationship between execution speed and task success remains underexplored. To address these issues, we introduce SpeedTuning, a reinforcement learning framework specifically designed to enhance the speed of manipulation policies. SpeedTuning learns to predict the optimal execution speed for actions, thereby complementing a base policy without necessitating additional data collection. We provide empirical evidence that SpeedTuning achieves substantial improvements in execution speed, exceeding 2.4x speed-up, while preserving an adequate success rate compared to both the original task policy and straightforward speed-up methods such as linear interpolation at a fixed speed. We evaluate our approach across a diverse set of dynamic and precise tasks, including pouring, throwing, and picking, demonstrating its effectiveness and robustness in enhancing real-world robotic manipulation. Videos and code are available at https://daivdyuan.github.io/speed-tuning/",
  "published": "2026-08-10",
  "updated": "2026-08-11",
  "year": "2026",
  "authors": [
   "David D. Yuan",
   "Tony Z. Zhao",
   "Kaylee Burns",
   "Chelsea Finn"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA",
  "venue_source": "semantic-scholar",
  "citations": 4,
  "influential_citations": 1,
  "tldr": "Speed Tuning is introduced, a reinforcement learning framework specifically designed to enhance the speed of manipulation policies, which learns to predict the optimal execution speed for actions, thereby complementing a base policy without necessitating additional data collection.",
  "doi": "10.1109/ICRA55743.2025.11128753",
  "oa_pdf": "https://arxiv.org/pdf/2608.09138",
  "s2_authors": [
   {
    "name": "David Yuan",
    "id": "2329095988",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Tony Z. Zhao",
    "id": "2327844315",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Kaylee Burns",
    "id": "2119915890",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Chelsea Finn",
    "id": "2387218305",
    "h_index": 3,
    "papers": 4
   }
  ],
  "comment": "10 pages, 12 figures. This arXiv version includes an appendix with qualitative simulation rollouts and additional ablations. Published at ICRA 2025",
  "topics": [
   "imitation-diffusion",
   "rl-control",
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.09138v2",
  "pdf_url": "https://arxiv.org/pdf/2608.09138v2",
  "html_url": "https://arxiv.org/html/2608.09138v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.2
 },
 {
  "id": "2608.09127",
  "slug": "high-fidelity-capture-reconstruction-and-transfer-of-human-demonstrati",
  "title": "High Fidelity Capture, Reconstruction, and Transfer of Human Demonstrations for Robot-Assisted Bathing",
  "abstract": "Despite the demand for robots in high-value clinical tasks like bathing, contemporary systems still lack the safety and reliability required for complex, sustained physical interaction with humans. A key challenge hindering the development of such systems is that collecting, understanding, and effectively transferring highly dynamic, contact-rich human bathing demonstrations is difficult, even with modern motion and tactile sensing equipment. We present a straightforward, but effective framework for doing so with high fidelity by utilizing contact regions as a key processing primitive. We use our framework to build a dataset of bathing demonstrations performed by trained clinicians on human subjects. We then use this dataset to design and control an arm-mounted dexterous soft hand to perform bathing tasks on a mannequin using open- and closed-loop strategies. Our dataset is the first to provide high quality synchronized motion, shape, contact, and force during sustained, contact-rich human-human interaction, and our transfer strategies demonstrate effective use of these data across multiple levels of the robotics stack. All relevant materials will be publicly released to enable further advancements in physical human-robot interaction (pHRI) research.",
  "published": "2026-08-10",
  "updated": "2026-08-10",
  "year": "2026",
  "authors": [
   "Arjun S. Lakshmipathy",
   "Jonathan P. King",
   "Ethan Zuo",
   "Rohit Satishkumar",
   "Hongyi Chen",
   "Jeffrey Ichnowski",
   "Dan Ding",
   "Zackory Erickson",
   "Nancy S. Pollard"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO",
   "cs.GR"
  ],
  "primary_category": "cs.RO",
  "venue": "RSS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This dataset is the first to provide high quality synchronized motion, shape, contact, and force during sustained, contact-rich human-human interaction, and the transfer strategies demonstrate effective use of these data across multiple levels of the robotics stack.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "A. Lakshmipathy",
    "id": "1405450779",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "J. King",
    "id": "50655498",
    "h_index": 12,
    "papers": 24
   },
   {
    "name": "Ethan Zuo",
    "id": "2419579056",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Rohit Satishkumar",
    "id": "2424079155",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Hongyi Chen",
    "id": "2309203483",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Jeffrey Ichnowski",
    "id": "2269146110",
    "h_index": 9,
    "papers": 29
   },
   {
    "name": "Dan Ding",
    "id": "2325008964",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Zackory Erickson",
    "id": "2360174150",
    "h_index": 4,
    "papers": 20
   },
   {
    "name": "Nancy S. Pollard",
    "id": "2277529387",
    "h_index": 3,
    "papers": 13
   }
  ],
  "comment": "Accepted to Robotics: Science and Systems (RSS) 2026. Official conference page: https://www.roboticsproceedings.org/rss22/p094.pdf",
  "topics": [
   "dexterous-manipulation",
   "egocentric-data",
   "tactile",
   "hri",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.09127v1",
  "pdf_url": "https://arxiv.org/pdf/2608.09127v1",
  "html_url": "https://arxiv.org/html/2608.09127v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.09125",
  "slug": "trajectory-divergence-horizon-decision-for-reliable-dual-arm-surgical",
  "title": "Trajectory Divergence Horizon Decision for Reliable Dual-Arm Surgical Subtask Manipulation",
  "abstract": "Surgical robotic systems are increasingly being adopted as clinical workload rises, motivating autonomous solutions for repetitive manipulation subtasks. Learning-based controllers improve generalization compared with rule-based and analytic approaches, but most are trained for individual tasks and remain difficult to reuse across procedures. Vision-Language-Action (VLA) models provide a unified framework that integrates visual perception, language grounding, and action generation, offering a promising path toward more composable surgical autonomy. However, existing VLA policies rely on fixed-length open-loop action sequences, where changing scene conditions can lead to accumulated errors and potential risks in surgical manipulation. To mitigate this issue, we formulate surgical VLA deployment as an adaptive execution-horizon decision problem and propose Trajectory Divergence Horizon Decision (TDHD), a test-time mechanism that estimates step-wise action reliability by measuring the divergence between two flow-matching-generated trajectories under small noise perturbations and truncates execution using a dual-threshold rule to trigger timely replanning. We further establish a real-world da Vinci-like dual-arm benchmark with synchronized multi-view perception and language instructions, and collect 600 teleoperated demonstrations across needle (reach, pick, regrasp) and tissue (reach, lift, resection) manipulation suites. On real hardware with 20 trials per task setting, TDHD consistently improves performance over the latest VLA baselines: success increases from 55\\% to 60\\% for needle manipulation and from 55\\% to 80\\% for tissue manipulation, with the largest gains observed in the final manipulation stages. These results highlight the importance of adaptive execution control for reliable deployment of VLA models in surgical robotic manipulation.",
  "published": "2026-08-10",
  "updated": "2026-08-10",
  "year": "2026",
  "authors": [
   "Mingwu Su",
   "Guankun Wang",
   "Jinsong Lin",
   "Rulin Zhou",
   "Ziyi Hao",
   "Zhiwei Fang",
   "Huxin Gao",
   "Jiewen Lai",
   "Jiazheng Wang",
   "Fan Zhang",
   "Hongliang Ren"
  ],
  "author_count": 11,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Trajectory Divergence Horizon Decision (TDHD), a test-time mechanism that estimates step-wise action reliability by measuring the divergence between two flow-matching-generated trajectories under small noise perturbations and truncates execution using a dual-threshold rule to trigger timely replanning, is proposed.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Mingwu Su",
    "id": "2367808998",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Guan-Feng Wang",
    "id": "2185172547",
    "h_index": 8,
    "papers": 29
   },
   {
    "name": "Jinsong Lin",
    "id": "2445481374",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Rulin Zhou",
    "id": "2353204753",
    "h_index": 2,
    "papers": 15
   },
   {
    "name": "Ziyi Hao",
    "id": "2088548476",
    "h_index": 5,
    "papers": 18
   },
   {
    "name": "Zhiwei Fang",
    "id": "2291304502",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Huxin Gao",
    "id": "1653073415",
    "h_index": 13,
    "papers": 49
   },
   {
    "name": "Jiewen Lai",
    "id": "2269206316",
    "h_index": 3,
    "papers": 23
   },
   {
    "name": "Jiazheng Wang",
    "id": "2308365041",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Fan Zhang",
    "id": "2341692584",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Hongliang Ren",
    "id": "2260612957",
    "h_index": 13,
    "papers": 53
   }
  ],
  "comment": "8 pages, 3 figures",
  "topics": [
   "vla",
   "dexterous-manipulation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.09125v1",
  "pdf_url": "https://arxiv.org/pdf/2608.09125v1",
  "html_url": "https://arxiv.org/html/2608.09125v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.09073",
  "slug": "latent-world-models-with-monotone-planning-costs-for-image-goal-naviga",
  "title": "Latent World Models with Monotone Planning Costs for Image-Goal Navigation",
  "abstract": "Image-goal navigation with latent world models requires not only accurate future prediction, but also a planning cost that reliably ranks candidate action sequences. We define the cost as the cosine distance between the predicted future embedding and the goal embedding, and show that poor cost ordering can mislead sampling-based planners such as Cross-Entropy Method (CEM). To address this, we propose a latent world model built on a frozen DINO-family encoder and train it with two complementary objectives. An autoregressive rollout loss reduces the gap between training and multi-step planning rollouts, while a Monotone Cost Ranking (MCR) loss directly encourages increasingly perturbed action sequences to receive higher planning costs. We also study InfoNCE-based action-contrastive training and find that temporal permutation negatives distort the latent geometry and degrade planning performance. On the GNM navigation dataset, our method outperforms Navigation World Models (NWM), DINO-WM, OmniVLA, and NoMaD, achieving state-of-the-art image-goal navigation performance while reducing orientation error by $2.7\\times$ over the same-encoder DINO WM baseline. We also deploy the model zero-shot on a physical robot, where it follows goal-directed paths in unseen indoor and outdoor environments.",
  "published": "2026-08-10",
  "updated": "2026-08-10",
  "year": "2026",
  "authors": [
   "Amirhosein Chahe",
   "Siwei Cai",
   "Lifeng Zhou"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes a latent world model built on a frozen DINO-family encoder and trains it with two complementary objectives, achieving state-of-the-art image-goal navigation performance while reducing orientation error by $2.7\\times over the same-encoder DINO WM baseline.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Amirhosein Chahe",
    "id": "2203816671",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Siwei Cai",
    "id": "2248167210",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Lifeng Zhou",
    "id": "2273914981",
    "h_index": 3,
    "papers": 11
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.09073v1",
  "pdf_url": "https://arxiv.org/pdf/2608.09073v1",
  "html_url": "https://arxiv.org/html/2608.09073v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.10025",
  "slug": "the-impact-of-operational-data-fidelity-when-assessing-safety-critical",
  "title": "The Impact of Operational-Data Fidelity when Assessing Safety-Critical Autonomous-Vehicle Software",
  "abstract": "For safety-critical software, data from the software's operational past (e.g. a sequence of success and failure events experienced by the software) can provide strong statistical support for reliability claims about the software. However, such data might not describe past software failure events in sufficient detail, and this might leave a reliability assessment (based on this data) unable to account for important features of past software failures. In this paper, by extending conservative Bayesian inference (CBI) techniques used in reliability assessment, we illustrate a principled statistical approach for checking the robustness of reliability claims derived from insufficiently detailed operational data. We demonstrate the extent to which insufficient detail in operational data can undermine software reliability claims in autonomous vehicle (AV) safety assessment scenarios. Reliability claims derived from insufficiently fine-grained data might be dangerously optimistic, despite a concerted effort by an assessor to use such data conservatively during the assessment. While these findings are consistent with previous work on the impact of statistical model fidelity in Bayesian software reliability assessments, our work clarifies why attempts to use low-fidelity data conservatively can be naive, and we give the first conservative estimates of the impact of data fidelity on assessments.",
  "published": "2026-08-09",
  "updated": "2026-08-09",
  "year": "2026",
  "authors": [
   "Kizito Salako",
   "Rabiu Tsoho Muhammad"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO",
   "cs.SE"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "By extending conservative Bayesian inference techniques used in reliability assessment, this work illustrates a principled statistical approach for checking the robustness of reliability claims derived from insufficiently detailed operational data and gives the first conservative estimates of the impact of data fidelity on assessments.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Kizito Salako",
    "id": "2863014",
    "h_index": 9,
    "papers": 32
   },
   {
    "name": "Rabiu Tsoho Muhammad",
    "id": "2391713555",
    "h_index": 0,
    "papers": 3
   }
  ],
  "comment": "16 pages, 12 figures",
  "topics": [
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.10025v1",
  "pdf_url": "https://arxiv.org/pdf/2608.10025v1",
  "html_url": "https://arxiv.org/html/2608.10025v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.10023",
  "slug": "protection-levels-for-vision-based-pose-estimation",
  "title": "Protection Levels for Vision-Based Pose Estimation",
  "abstract": "Vision-based navigation complements Global Navigation Satellite Systems, but certification demands integrity guarantees that account for faulty measurements. Previous work presented a probabilistic computer vision pipeline for runway-based pose estimation with fault detection inspired by Receiver Autonomous Integrity Monitoring. This work extends that framework by deriving protection levels, which provide probabilistic bounds on pose error that remain valid under undetected faults. We present an algorithm for computing protection levels for the nonlinear Perspective-$n$-Point problem applied to an aviation setting. The algorithm covers all six degrees of freedom of the aircraft pose (position and orientation) directly. We analyze the effect of measurement redundancy, pixel-level prediction uncertainty, and runway distance on the resulting protection levels. To make the results tangible, we demonstrate tradeoffs in the protection levels on an illustrative runway example.",
  "published": "2026-08-09",
  "updated": "2026-08-09",
  "year": "2026",
  "authors": [
   "Olivia Beyer Bruvik",
   "Romeo Valentin",
   "Marc R. Schlichting",
   "Don Walker",
   "Mykel J. Kochenderfer"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.CV",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work presents an algorithm for computing protection levels for the nonlinear Perspective-$n$-Point problem applied to an aviation setting and analyzes the effect of measurement redundancy, pixel-level prediction uncertainty, and runway distance on the resulting protection levels.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "O. Bruvik",
    "id": "2328245109",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Romeo Valentin",
    "id": "2312394645",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Marc R. Schlichting",
    "id": "2159279123",
    "h_index": 3,
    "papers": 14
   },
   {
    "name": "Don Walker",
    "id": "2312391598",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "M. Kochenderfer",
    "id": "79262652",
    "h_index": 24,
    "papers": 213
   }
  ],
  "comment": "11 pages, 5 figures. Accepted for publication at the 2026 AIAA DATC/IEEE 45th Digital Avionics Systems Conference (DASC). O. Beyer Bruvik and R. Valentin contributed equally",
  "topics": [
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.10023v1",
  "pdf_url": "https://arxiv.org/pdf/2608.10023v1",
  "html_url": "https://arxiv.org/html/2608.10023v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.08871",
  "slug": "hierarchical-topology-aware-planning-and-control-of-underwater-vehicle",
  "title": "Hierarchical Topology-Aware Planning and Control of Underwater Vehicle-Manipulator Systems in Confined Environments",
  "abstract": "This paper addresses autonomous intervention with an underwater vehicle--manipulator system (UVMS) in confined, cluttered, and partially known environments, where poor maneuverability, narrow passages, and uncertain execution may cause the robot to enter unrecoverable regions. We propose MANTA, a three-layer hierarchical planning-and-control framework that couples passage accessibility, manipulation feasibility, and closed-loop execution. The first layer performs global connectivity reasoning in a conservative reduced base space to extract traversable corridor candidates toward the task region. The second layer refines each candidate corridor by jointly optimizing the continuous base motion and arm trajectory, producing a collision-free base--arm trajectory. The third layer learns a reach-and-hold base policy using Gaussian-process model-based reinforcement learning (MBRL) through MC-PILCO, enabling trajectory tracking and station keeping at the planned manipulation state. During execution, the framework monitors map updates and can trigger recovery and route repair when the active passage becomes infeasible. MANTA is evaluated in confined UVMS planning and closed-loop tracking experiments. Across 120 matched planning queries, it achieves higher task success than full-state sampling-based baselines while producing larger clearance margins and lower arm motion. The learned MC-PILCO policy further reduces position and yaw tracking errors on both training and unseen tube-like references. These results show MANTA as a structured and data-efficient framework for safe autonomous underwater intervention in caves, tubes, and cluttered subsea structures.",
  "published": "2026-08-09",
  "updated": "2026-08-09",
  "year": "2026",
  "authors": [
   "Mohamed Abdelwahab",
   "Ruggero Carli",
   "Damiano Varagnolo",
   "Alberto Dalla Libera"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Mohamed Abdelwahab",
    "id": "2307472471",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Ruggero Carli",
    "id": "2201436597",
    "h_index": 4,
    "papers": 45
   },
   {
    "name": "Damiano Varagnolo",
    "id": "50412265",
    "h_index": 22,
    "papers": 160
   },
   {
    "name": "A. D. Libera",
    "id": "51450820",
    "h_index": 10,
    "papers": 45
   }
  ],
  "comment": "",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.08871v1",
  "pdf_url": "https://arxiv.org/pdf/2608.08871v1",
  "html_url": "https://arxiv.org/html/2608.08871v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.08839",
  "slug": "sg-wam-text-grounded-and-spatial-aware-semantic-guidance-for-world-act",
  "title": "SG-WAM: Text-Grounded and Spatial-aware Semantic Guidance for World-Action Models",
  "abstract": "World-Action Models (WAMs) have emerged as a promising paradigm for robotic manipulation. However, most existing WAMs generate future videos and actions by relying mainly on visual cues rather than language instructions, since off-the-shelf text encoders embed instructions independently of visual observations. As a result, the videos predicted by these WAMs are often semantically misaligned with their corresponding language instructions, which degrades the accuracy of the predicted actions. To overcome this limitation, we propose SG-WAM, a semantic guidance method for world-action models that leverages a vision-language model (VLM) as a semantic planner to enhance the instruction-grounding capacity of world-action models. Specifically, we train a VLM-based planner to predict text-grounded and spatial-aware semantic foresight. The text-grounded semantic foresight grounds the instruction by identifying the correct target objects, and the spatial-aware semantic foresight provides the scene geometry for precise manipulation. We then inject this foresight into the world-action model as high-level semantic guidance, ensuring that both future-video generation and action prediction faithfully follow the language instruction. Extensive experiments in simulation and the real world demonstrate the superiority of our semantic guidance method, showcasing precise manipulation and strong instruction-following capabilities.",
  "published": "2026-08-09",
  "updated": "2026-08-09",
  "year": "2026",
  "authors": [
   "Junjie He",
   "Junfeng Li",
   "Zhide Zhong",
   "Haodong Yan",
   "Ruixin Li",
   "Yangyang Zheng",
   "Jiaguan Zhu",
   "Tianran Zhang",
   "Yuqiao Du",
   "Wen Chen",
   "Shunbo Zhou",
   "Haoang Li"
  ],
  "author_count": 12,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "SG-WAM is proposed, a semantic guidance method for world-action models that leverages a vision-language model (VLM) as a semantic planner to enhance the instruction-grounding capacity of world-action models.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Junjie He",
    "id": "2316016558",
    "h_index": 3,
    "papers": 16
   },
   {
    "name": "Junfeng Li",
    "id": "2376547586",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Zhide Zhong",
    "id": "2349315841",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Haodong Yan",
    "id": "2321603038",
    "h_index": 5,
    "papers": 16
   },
   {
    "name": "Ruixin Li",
    "id": "2455424656",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Yangyang Zheng",
    "id": "2233047956",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Jiaguang Zhu",
    "id": "1575707808",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Tianran Zhang",
    "id": "2308051940",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Yuqiao Du",
    "id": "2455824586",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Wen Chen",
    "id": "2444700942",
    "h_index": 0,
    "papers": 5
   },
   {
    "name": "Shunbo Zhou",
    "id": "2349004355",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Haoang Li",
    "id": "2384363611",
    "h_index": 9,
    "papers": 38
   }
  ],
  "comment": "",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.08839v1",
  "pdf_url": "https://arxiv.org/pdf/2608.08839v1",
  "html_url": "https://arxiv.org/html/2608.08839v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.08749",
  "slug": "onevomemory-evolving-memory-through-online-robot-rollouts-for-pretrain",
  "title": "OnEvoMemory: Evolving Memory through Online Robot Rollouts for Pretrained Robot Policies",
  "abstract": "Long-horizon robot manipulation requires policies to track completed subtasks and critical interaction events. However, existing memory mechanisms heavily rely on external models or predefined update rules. To address this, we propose OnEvoMemory, a value-guided memory module for pretrained robot policies. It maintains recent context, high-value experiences, and salient transitions, while learning which experiences should be retained from trajectory outcomes. Offline demonstrations initialize the memory prior, whereas successful and unsuccessful online rollouts refine memory selection, helping the policy recognize task-stage transitions and avoid repeating completed subtasks. Experiments on long-horizon manipulation benchmarks show that OnEvoMemory improves the performance of the base VLA policy through both offline initialization and online memory evolution.",
  "published": "2026-08-09",
  "updated": "2026-08-09",
  "year": "2026",
  "authors": [
   "Zhongxi Chen",
   "Shenqi Zong"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "ECCV 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "OnEvoMemory is proposed, a value-guided memory module for pretrained robot policies that maintains recent context, high-value experiences, and salient transitions, while learning which experiences should be retained from trajectory outcomes.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhongxi Chen",
    "id": "2456882931",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Shenqi Zong",
    "id": "2456703296",
    "h_index": 0,
    "papers": 1
   }
  ],
  "comment": "6 pages, 1 figure. Accepted as a poster at the ECCV 2026 Workshop on Embodied Multimodal Reasoning in Physical Environments (EMR)",
  "topics": [
   "vla"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.08749v1",
  "pdf_url": "https://arxiv.org/pdf/2608.08749v1",
  "html_url": "https://arxiv.org/html/2608.08749v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.08725",
  "slug": "wa-specdec-world-aware-speculative-decoding-for-vision-language-action",
  "title": "WA-SpecDec: World-Aware Speculative Decoding for Vision-Language-Action Models",
  "abstract": "Vision-language-action (VLA) policies generate robot controls autoregressively, making closed-loop latency dominated by repeated target-model forward passes. Speculative decoding reduces this cost by verifying blocks of draft action tokens in parallel, and recent VLA methods further relax token-level acceptance because small differences in action-token space often map to similar continuous controls. However, this relaxation remains scene-agnostic. A fixed token-distance tolerance treats the same action-token deviation as equally safe across states, although deviations that are harmless in free space can cause collisions or grasp failures near contact. We propose WA-SpecDec, a world-aware speculative decoding framework that injects world-model-derived physical scene awareness during the VLA prefill stage, producing shared world-aware prefill states for draft proposal and target verification without changing the relaxed acceptance rule. Across three state-of-the-art relaxed acceptance schemes, WA-SpecDec preserves higher task success under looser relaxation and enables longer accepted prefixes. At comparable-success operating points, WA-SpecDec achieves a 1.5x matched-success speedup over VLA speculative decoding alone and reduces near-contact failure (NCF) by 18.6% on average relative to the corresponding speculative baselines.",
  "published": "2026-08-09",
  "updated": "2026-08-09",
  "year": "2026",
  "authors": [
   "Zikang Wen",
   "Yuning Zhang",
   "Dong Yuan"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "WA-SpecDec is proposed, a world-aware speculative decoding framework that injects world-model-derived physical scene awareness during the VLA prefill stage, producing shared world-aware prefill states for draft proposal and target verification without changing the relaxed acceptance rule.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zikang Wen",
    "id": "2370458458",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Yuning Zhang",
    "id": "2108208735",
    "h_index": 4,
    "papers": 18
   },
   {
    "name": "Dong Yuan",
    "id": "2271655539",
    "h_index": 2,
    "papers": 16
   }
  ],
  "comment": "Preprint",
  "topics": [
   "world-models",
   "vla",
   "dexterous-manipulation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.08725v1",
  "pdf_url": "https://arxiv.org/pdf/2608.08725v1",
  "html_url": "https://arxiv.org/html/2608.08725v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.08598",
  "slug": "knowledge-distilled-end-to-end-reinforcement-learning-for-smooth-6-dof",
  "title": "Knowledge-Distilled End-to-End Reinforcement Learning for Smooth 6-DOF Thrust Control and Rapid Adaptation to Ocean Currents in Remotely Operated Vehicles",
  "abstract": "With the continuous improvement of computational capabilities, end-to-end reinforcement learning has been rapidly developed for remotely operated vehicles control. Nevertheless, existing end-to-end reinforcement-learningbased methods still face challenges in achieving optimal control under oceancurrent disturbances. In particular, there remains a lack of a unified control framework that can simultaneously achieve low steady-state tracking error, rapid transient response, energy-efficient operation, and smooth controlforce outputs under disturbances. To address the issue, this paper proposes the thrust smoothness rapid current adaptation proximal policy optimization (TSRCA-PPO) method which learns a near-optimal strategy by a twostage distillation learning framework. The core innovations of this work lie in the reward-function design and the privileged multi-encoder architecture. Ablation studies validate the effectiveness of each module. Simulation results demonstrate that the proposed TSRCA-PPO method consistently outperforms the conventional cascaded P-PID controller across all evaluation metrics. Specifically, TSRCA-PPO reduces the steady-state position error, steady-state attitude error, settling time, energy index, and thrustsmoothness index to 42.7%, 76.5%, 10.6%, 93.5%, and 15.9% of the corresponding P-PID values, respectively.",
  "published": "2026-08-09",
  "updated": "2026-08-09",
  "year": "2026",
  "authors": [
   "Tiankuang Wen",
   "Huiping Li",
   "Gang Liu",
   "Yong Jiang"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The thrust smoothness rapid current adaptation proximal policy optimization (TSRCA-PPO) method which learns a near-optimal strategy by a twostage distillation learning framework is proposed which consistently outperforms the conventional cascaded P-PID controller across all evaluation metrics.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Tiankuang Wen",
    "id": "2433005712",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Huiping Li",
    "id": "2352193962",
    "h_index": 1,
    "papers": 10
   },
   {
    "name": "Gangfu Liu",
    "id": "2373670785",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Yongkang Jiang",
    "id": "2456260303",
    "h_index": 0,
    "papers": 2
   }
  ],
  "comment": "",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.08598v1",
  "pdf_url": "https://arxiv.org/pdf/2608.08598v1",
  "html_url": "https://arxiv.org/html/2608.08598v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.08558",
  "slug": "vid2wam-distilling-video-diffusion-priors-into-world-action-models",
  "title": "Vid2WAM: Distilling Video Diffusion Priors into World Action Models",
  "abstract": "World Action Models (WAMs) improve robot policy learning by jointly modeling future visual dynamics and actions. However, their scalability and generalization remain constrained by their reliance on costly expert demonstrations. We challenge this by asking whether future supervision for WAMs must originate from target-task expert trajectories. In this paper, we propose Vid2WAM, an offline distillation framework that transfers visual diffusion priors from a large video foundation model into a compact WAM student. Given an observation and language instruction, Vid2WAM distills supervision through two complementary channels: task-conditioned future rollouts directly supervise the student's future prediction branch, while an inverse dynamics model recovers embodiment-specific pseudo-actions for action learning. To robustly integrate synthetic and real supervision, we introduce source-aware residual action adaptation that learns source-specific corrections around a shared action backbone and mitigates interference from noisy pseudo-actions. During inference, both the video teacher and inverse dynamics model are discarded, leaving only the WAM student for efficient deployment. Simulation and real-world experiments demonstrate that Vid2WAM improves novel-task generalization and data efficiency under limited expert demonstrations while preserving low-latency inference.",
  "published": "2026-08-09",
  "updated": "2026-08-09",
  "year": "2026",
  "authors": [
   "Chenhao Qiu",
   "Ruixiang Wang",
   "Runyi Zhao",
   "Sixu Lin",
   "Songen Gu",
   "Shufeng Nan",
   "Guiliang Liu",
   "Kui Jia",
   "Yanwei Fu",
   "Simo Wu"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Vid2WAM is proposed, an offline distillation framework that transfers visual diffusion priors from a large video foundation model into a compact WAM student and introduces source-aware residual action adaptation that learns source-specific corrections around a shared action backbone and mitigates interference from noisy pseudo-actions.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Chenhao Qiu",
    "id": "2372602182",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Ruixiang Wang",
    "id": "2342353109",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Runyi Zhao",
    "id": "2359793561",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Sixu Lin",
    "id": "2348309829",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Songen Gu",
    "id": "2351410250",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "S. Nan",
    "id": "20966091",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Guiliang Liu",
    "id": "2319185234",
    "h_index": 3,
    "papers": 18
   },
   {
    "name": "Kui Jia",
    "id": "2370507",
    "h_index": 63,
    "papers": 249
   },
   {
    "name": "Yanwei Fu",
    "id": "2405950194",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Si Wu",
    "id": "2355121952",
    "h_index": 3,
    "papers": 21
   }
  ],
  "comment": "Project website: https://qch-fa.github.io/vid2wam-website/",
  "topics": [
   "foundation-pretraining",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.08558v1",
  "pdf_url": "https://arxiv.org/pdf/2608.08558v1",
  "html_url": "https://arxiv.org/html/2608.08558v1",
  "code_url": "https://qch-fa.github.io/vid2wam-website/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2608.08545",
  "slug": "curriculum-generation-under-structured-parametric-environments-for-rob",
  "title": "Curriculum Generation under Structured Parametric Environments for Robust Navigation Policies",
  "abstract": "Robust navigation policies for autonomous agents must generalize across continuously varying environmental conditions such as turn rates, obstacles, friction, pits, and slopes. Curriculum generation provides a principled mechanism for improving generalization by progressively adapting training environments, but designing such curricula in a sample-efficient and automated manner remains challenging. This paper proposes a reparameterized curriculum generation framework for structured continuous environment parameters using unidirectional gradient-based optimization. To improve robustness in multimodal observation spaces consisting of image-based and scalar inputs, a distribution-shift regularization objective is incorporated to encourage the learning of finer-grained latent representations. The proposed method is evaluated across two continuous-control OpenAI Gym environments: a 2D obstacle-based Car Racing variant and Bipedal Walker variant, where coupled environment parameters jointly influence policy performance. Across five random seeds, our method consistently outperforms vanilla policy training, random parameter sampling, manual curricula, frontier-based methods, Self-Paced Reinforcement Learning (SPRL), Absolute Learning Progress with Gaussian Mixture Models (ALP-GMM), and reverse curriculum learning baselines. Ablation studies further demonstrate the effectiveness of the reparameterized curriculum mechanism across both environments, while highlighting environment-dependent benefits of the auxiliary regularization objective.",
  "published": "2026-08-09",
  "updated": "2026-08-09",
  "year": "2026",
  "authors": [
   "Prishita Ray"
  ],
  "author_count": 1,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A reparameterized curriculum generation framework for structured continuous environment parameters using unidirectional gradient-based optimization to improve robustness in multimodal observation spaces consisting of image-based and scalar inputs and a distribution-shift regularization objective is incorporated to encourage the learning of finer-grained latent representations.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Prishita Ray",
    "id": "1491548229",
    "h_index": 3,
    "papers": 8
   }
  ],
  "comment": "22 pages, 5 figures",
  "topics": [
   "humanoids",
   "rl-control",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.08545v1",
  "pdf_url": "https://arxiv.org/pdf/2608.08545v1",
  "html_url": "https://arxiv.org/html/2608.08545v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.08356",
  "slug": "stochastic-physics-informed-neural-networks-on-lie-groups-for-learning",
  "title": "Stochastic Physics-Informed Neural Networks on Lie Groups for Learning Underwater Vehicle Dynamics",
  "abstract": "Accurate models of underwater vehicle motion are needed for autonomous execution of marine tasks like infrastructure inspection and scientific sampling. However, such motion is challenging to characterize using traditional physics-based methods. This paper presents a novel data-driven framework for learning stochastic underwater vehicle dynamics. Using Euler-Poincar\u00e9 dynamics and the geometry of Lie groups, we develop a stochastic physics-informed neural network architecture that respects the physical and geometric constraints of underwater vehicles. Our approach leverages structure-preserving stochastic integration and builds upon moment matching and finite dimensional matching to ensure geometrically-consistent training. We evaluate our approach in simulation and on an underwater vehicle navigating dock pylons in a harbor environment. The results demonstrate that our method learns accurate and robust dynamics models, enabling safe model-based control in challenging marine environments.",
  "published": "2026-08-08",
  "updated": "2026-08-08",
  "year": "2026",
  "authors": [
   "Evan F. Palmer",
   "Ross L. Hatton",
   "Geoffrey A. Hollinger"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper develops a stochastic physics-informed neural network architecture that respects the physical and geometric constraints of underwater vehicles and leverages structure-preserving stochastic integration and builds upon moment matching and finite dimensional matching to ensure geometrically-consistent training.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Evan Palmer",
    "id": "2293409029",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Ross L. Hatton",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Geoffrey A. Hollinger",
    "id": "2293436790",
    "h_index": 2,
    "papers": 7
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.08356v1",
  "pdf_url": "https://arxiv.org/pdf/2608.08356v1",
  "html_url": "https://arxiv.org/html/2608.08356v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.08323",
  "slug": "mppi-planning-with-gaussian-based-human-cost-function-for-social-navig",
  "title": "MPPI Planning with Gaussian-Based Human Cost Function for Social Navigation",
  "abstract": "Safe robot navigation in crowded spaces requires planning that accounts for where people will be, not only where they are now. Model Predictive Path Integral (MPPI) control is an effective sampling-based planner, but many implementations encode humans as static point obstacles at their current positions, underestimating risk in dynamic scenes. We propose Predictive Gaussian Interaction Fields (PGIF), a spatiotemporal cost formulation that propagates pedestrian predictions forward over the full planning horizon and encodes them as anisotropic Gaussian repulsive fields aligned with each pedestrian's direction of motion. The forward spread of each field grows with the pedestrian's speed, creating a motion cone danger zone that penalises robot trajectories entering the pedestrian's path of travel more strongly than those approaching from behind. The formulation is closed-form and fully parallelisable across rollouts, adding no measurable computational overhead. Evaluated over 300 randomised crowd scenarios at three density levels, PGIF-MPPI achieves a 0% collision rate at every density level, compared with up to 82% for vanilla MPPI, while maintaining real-time planning performance.",
  "published": "2026-08-08",
  "updated": "2026-08-11",
  "year": "2026",
  "authors": [
   "Chinmay Mundane"
  ],
  "author_count": 1,
  "categories": [
   "cs.RO",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Predictive Gaussian Interaction Fields (PGIF), a spatiotemporal cost formulation that propagates pedestrian predictions forward over the full planning horizon and encodes them as anisotropic Gaussian repulsive fields aligned with each pedestrian's direction of motion, is proposed.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Chinmay Mundane",
    "id": "2327876889",
    "h_index": 0,
    "papers": 3
   }
  ],
  "comment": "Will appear in Springer's Proceedings in Advanced Robotics series",
  "topics": [
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.08323v2",
  "pdf_url": "https://arxiv.org/pdf/2608.08323v2",
  "html_url": "https://arxiv.org/html/2608.08323v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.08285",
  "slug": "ego-oscar-egocentric-open-source-stereo-capture-system",
  "title": "Ego-OSCAR: Egocentric Open source Stereo CAptuRe System",
  "abstract": "We present Ego-OSCAR, an open-hardware, low-cost, head-mounted stereo-inertial capture device for egocentric data collection in the wild. EgoOSCAR pairs a hardware-synchronized global-shutter stereo camera with a 6- axis IMU, an embedded Linux SBC for on-device video encoding, and a realtime microcontroller for user feedback and watchdog functions. The complete bill of materials is under USD 200 per unit, using only commercially available components and 3D-printed parts. Alongside the device, we release a complete software stack (hardware-accelerated recording pipeline, IMU sampling daemon, time-synchronization tooling, and watchdog firmware) and roughly 550 hours of egocentric stereo video per camera with synchronized IMU, collected by a distributed contributor network across everyday indoor environments. The release is annotated rather than raw: free-form action captions cover essentially the entire recorded timeline with an open vocabulary, and per-frame 3D hand reconstructions ship alongside per-session stereo calibration. Ego-OSCAR does not aim to match the per-unit fidelity of research-grade systems such as Project Aria; it aims to be the cheapest defensible substrate for crowdsourced egocentric capture, and to lower the activation energy for any team that wants to collect egocentric data at scale. All hardware designs, software, and the dataset are open-sourced",
  "published": "2026-08-08",
  "updated": "2026-08-12",
  "year": "2026",
  "authors": [
   "Gunjan Paul",
   "Senthil Palanisamy",
   "Satpal Singh Rathore",
   "Pratyush Kumar Patnaik",
   "Shubhanshu Khatana",
   "Abhishek Anand"
  ],
  "author_count": 6,
  "categories": [
   "cs.CV",
   "cs.AR",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Ego-OSCAR is presented, an open-hardware, low-cost, head-mounted stereo-inertial capture device for egocentric data collection in the wild that aims to be the cheapest defensible substrate for crowdsourced egocentric capture, and to lower the activation energy for any team that wants to collect egocentric data at scale.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Gunjan Paul",
    "id": "2004767408",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Senthil Palanisamy",
    "id": "2399325212",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Satpal Singh Rathore",
    "id": "2439904654",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "P. Patnaik",
    "id": "35366109",
    "h_index": 12,
    "papers": 43
   },
   {
    "name": "Shubhanshu Khatana",
    "id": "2433249283",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Abhishek Anand",
    "id": "2064200957",
    "h_index": 1,
    "papers": 4
   }
  ],
  "comment": "",
  "topics": [
   "egocentric-data",
   "data-teleop",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.08285v2",
  "pdf_url": "https://arxiv.org/pdf/2608.08285v2",
  "html_url": "https://arxiv.org/html/2608.08285v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.08281",
  "slug": "exploring-llm-capabilities-for-situational-understanding-and-colreg-co",
  "title": "Exploring LLM Capabilities for Situational Understanding and COLREG compliance on real-world maritime navigation scenarios",
  "abstract": "Recently, Large Language Models (LLMs) have shown considerable capability for situational understanding, reasoning, and decision making in different domains, most notable in the automotive sector. Therefore, we explore current state-of-the-art LLMs as a tool for maritime navigation, which includes both codified rules in the Collision Regulations (COLREGs) and uncodified best practices summarized in the concept of ``Good Seamanship''. We construct a dataset consisting of 50 diverse, real-world navigation scenarios from AIS data, label scenarios with applicable COLREG rules, recommended actions, and the reasoning for the action. We explore a variety of different LLM architectures and sizes to determine their understanding of maritime navigation tasks as well as evaluate their reasoning capabilities in this domain. The results obtained indicate that the maritime navigation task remains difficult to solve without fine-tuning, even for larger online models.",
  "published": "2026-08-08",
  "updated": "2026-08-08",
  "year": "2026",
  "authors": [
   "Julius Wirbel",
   "P. Nicholas Hansen",
   "Line K. H. Clemmensen",
   "Roberto Galeazzi"
  ],
  "author_count": 4,
  "categories": [
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.AI",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Current state-of-the-art LLMs as a tool for maritime navigation, which includes both codified rules in the Collision Regulations and uncodified best practices summarized in the concept of ``Good Seamanship'' are explored.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Julius Wirbel",
    "id": "2383172401",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "P. N. Hansen",
    "id": "153628398",
    "h_index": 8,
    "papers": 20
   },
   {
    "name": "Line Clemmensen",
    "id": "2298272766",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Roberto Galeazzi",
    "id": "2308275731",
    "h_index": 2,
    "papers": 17
   }
  ],
  "comment": "Submitted and accepted to the IFAC WC 2026 as an invited session paper for track 7.2 Transportation and Vehicle Systems - Marine Systems",
  "topics": [
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.08281v1",
  "pdf_url": "https://arxiv.org/pdf/2608.08281v1",
  "html_url": "https://arxiv.org/html/2608.08281v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.08200",
  "slug": "spatiotemporal-context-dependent-personalized-movement-compensation-in",
  "title": "Spatiotemporal Context-dependent Personalized Movement Compensation in Delayed Telemanipulation",
  "abstract": "Communication delay remains a central challenge in telerobotics, where it disrupts visuomotor coordination and reduces task precision. Motion scaling is an effective countermeasure to delay-induced overshoot, yet typical deployments rely on uniform gains that neglect individual and contextual variability. We propose a human-centered method that fits personalized delay-, direction-, and distance-specific scaling parameters for each participant. We conducted experiments with twenty participants who performed delayed reaching tasks in a virtual simulator. Scaling gains were computed to minimize mean overshoot in simulation in each combination of experimental conditions. Evaluation was done in simulation and on a telesurgical robot to evaluate assistance benefits. Performance was assessed across multiple delays, distances, and movement directions using overshoot, endpoint error, trajectory smoothness, economy of motion, and a composite error-time metric. Motion scaling consistently improved performance relative to unassisted trials, yielding up to 20-25% performance gains in key metrics. Effects were most pronounced at longer delays. Personalization demonstrated additional accuracy benefits for inward reaching at a short distance under moderate delay. The results highlight the potential of personalized scaling as a foundation for more adaptive frameworks that integrate contextual information to improve the safety and precision of teleoperated procedures.",
  "published": "2026-08-08",
  "updated": "2026-08-08",
  "year": "2026",
  "authors": [
   "Sai Jiang",
   "Zonghe Chua"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO",
   "cs.HC",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Motion scaling consistently improved performance relative to unassisted trials, yielding up to 20-25% performance gains in key metrics, and personalization demonstrated additional accuracy benefits for inward reaching at a short distance under moderate delay.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Sai Jiang",
    "id": "2456976913",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Z. Chua",
    "id": "72270569",
    "h_index": 6,
    "papers": 16
   }
  ],
  "comment": "10 pages, 6 figures",
  "topics": [
   "sim2real",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.08200v1",
  "pdf_url": "https://arxiv.org/pdf/2608.08200v1",
  "html_url": "https://arxiv.org/html/2608.08200v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.08077",
  "slug": "explore-map-remember-decide-are-embodied-vlms-ready-for-safety-critica",
  "title": "Explore, Map, Remember, Decide: Are Embodied VLMs Ready for Safety-Critical Scenarios?",
  "abstract": "Theory of Space framework (ToS) assesses the spatial understanding of curiosity-driven Vision-Language Models (VLMs) under partial observability. As AI techniques are increasingly applied to safety-critical scenarios, it is crucial to understand whether VLMs possess robust spatial memory and make reliable decisions. In this paper, we assess whether VLMs' decisions are based on physical evidence or are corrupted by visual-language biases, if their memory processes align with human cognitive patterns, and how they respond to environmental hazards. We extend the ToS framework into a safety-critical, goal-driven pipeline, named Explore, Map, Remember, and Decide (EMRD). We then quantify Exploration Competence (Explore) through metrics of environmental coverage and temporal efficiency, assess Spatial Fidelity (Map), evaluate, with a suite of psychological metrics, Memory Persistence (Remember), and measure, using focal-point metrics, Cognitive Decision-Making (Decide). Our results show that in terms of decision-making capabilities, VLMs frequently select evacuation points based on pre-trained textual priors while lacking the spatial grounding to justify their choices. We also show that spatial reasoning degrades in low-light conditions, but it is not affected by texture and colour tampering. Our findings suggest that VLM memory fundamentally diverges from human cognition, creating unpredictable risks of misalignment.",
  "published": "2026-08-08",
  "updated": "2026-08-08",
  "year": "2026",
  "authors": [
   "Gabriele La Malfa",
   "Nitay Alon",
   "Emanuele La Malfa",
   "Reuth Mirsky",
   "Stefan Sarkadi"
  ],
  "author_count": 5,
  "categories": [
   "cs.AI",
   "cs.MA",
   "cs.RO"
  ],
  "primary_category": "cs.AI",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The findings suggest that VLM memory fundamentally diverges from human cognition, creating unpredictable risks of misalignment.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "G. Malfa",
    "id": "1582731195",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "Nitay Alon",
    "id": "2123022632",
    "h_index": 7,
    "papers": 16
   },
   {
    "name": "Emanuele La Malfa",
    "id": "1582740100",
    "h_index": 12,
    "papers": 29
   },
   {
    "name": "Reuth Mirsky",
    "id": "3402763",
    "h_index": 16,
    "papers": 84
   },
   {
    "name": "Stefan Sarkadi",
    "id": "2348743244",
    "h_index": 1,
    "papers": 4
   }
  ],
  "comment": "",
  "topics": [
   "spatial-3d",
   "navigation",
   "video-generation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.08077v1",
  "pdf_url": "https://arxiv.org/pdf/2608.08077v1",
  "html_url": "https://arxiv.org/html/2608.08077v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.08070",
  "slug": "surgwmbench-a-vision-based-benchmark-for-world-modeling-surgical-instr",
  "title": "SurgWMBench: A Vision-Based Benchmark for World-Modeling Surgical Instrument Motion Planning",
  "abstract": "Reliable surgical planning requires models that move beyond recognizing the current surgical step or imitating expert demonstrations, and instead anticipate how instrument motion reshapes subsequent operative states. Most surgical video understanding methods focus on recognizing phases, actions, or workflow states, while providing limited support for explicitly modeling instrument motion. Conversely, existing tool motion prediction methods can forecast instrument trajectories, but they generally do not capture the coupled evolution of future surgical video states. World models offer a natural framework for jointly modeling visual state transitions and instrument motion dynamics. However, existing surgical world model studies remain largely centered on visual generation quality, relying on generation-oriented metrics such as FVD and CD-FVD. These metrics are poorly aligned with instrument motion planning, as they do not directly measure whether predicted trajectories are geometrically accurate, temporally coherent, or actionable for downstream planning. This limitation is partly structural, since the field lacks public datasets and standardized evaluation protocols that provide the benchmarking infrastructure needed to assess motion-centric capabilities in surgical world models. In this paper, we introduce SurgWMBench, a vision-based benchmark for short-horizon surgical motion planning and dynamics prediction. Given intraoperative image sequences and historical instrument trajectory, SurgWMBench evaluates both near-future instrument motion prediction and stability under continuous rollout or input perturbations.",
  "published": "2026-08-08",
  "updated": "2026-08-08",
  "year": "2026",
  "authors": [
   "Huanrong Liu",
   "Weiliang Huang",
   "Bob Zhang",
   "Weichao Cai",
   "Chunlin Tian",
   "Qingbiao Li"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "SurgWMBench is introduced, a vision-based benchmark for short-horizon surgical motion planning and dynamics prediction, that evaluates both near-future instrument motion prediction and stability under continuous rollout or input perturbations.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Huanrong Liu",
    "id": "2269890777",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Weiliang Huang",
    "id": "2353661364",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Bobo Zhang",
    "id": "2456270150",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Weichao Cai",
    "id": "2295590609",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Chunlin Tian",
    "id": "2370481029",
    "h_index": 2,
    "papers": 13
   },
   {
    "name": "Qingbiao Li",
    "id": "2108053899",
    "h_index": 12,
    "papers": 33
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "sim2real"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.08070v1",
  "pdf_url": "https://arxiv.org/pdf/2608.08070v1",
  "html_url": "https://arxiv.org/html/2608.08070v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2608.08025",
  "slug": "da-nbv-a-direction-aware-next-best-view-planner-for-efficient-3d-recon",
  "title": "DA-NBV: A Direction-Aware Next-Best-View Planner for Efficient 3D Reconstruction of Ships at Sea",
  "abstract": "Accurate 3D reconstruction of ships at sea is important for maritime supervision, damage assessment, and autonomous maritime operations. Although 3D reconstruction has advanced considerably, high-quality data acquisition still largely relies on manually designed trajectories or skilled operators, resulting in high costs and limited scalability. Next-best-view (NBV) planning automates this process by selecting subsequent viewpoints based on the current state. However, existing NBV policies mainly model spatial occupancy while overlooking directional observation history. This limitation is particularly problematic for ships: their complex superstructures and severe self-occlusions require observations from multiple viewpoints, and insufficient directional coverage often yields incomplete reconstructions. These challenges are further amplified at sea, where wave-induced heave, roll, and pitch continuously alter the ship's pose and surface visibility. Meanwhile, wind disturbances and limited onboard power impose stricter requirements on scanning efficiency. To address these challenges, we propose DA-NBV, a direction-aware NBV policy that augments the conventional occupancy state with directional observation statistics. We introduce a learnable Position Advantage Field (PAF) that uses directional information to guide viewpoint selection. The policy further adopts a locally constrained action space and a nonlinear coverage-shaping reward to improve scanning efficiency. We also develop the ship-oriented SeaShip-3D dataset and a configurable sea-state simulation environment. Experiments under varying heave, roll, and pitch conditions show that DA-NBV improves reconstruction completeness by approximately 3 percentage points and reduces Chamfer distance by 43% while achieving higher path efficiency.",
  "published": "2026-08-08",
  "updated": "2026-08-08",
  "year": "2026",
  "authors": [
   "Jiaming Chen",
   "Juntao Yang",
   "Zhentao Zou",
   "Qi Ming",
   "Yi Yu",
   "Zhihang Zhong",
   "Xue Yang",
   "Xue Jiang",
   "Yue Zhou"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jiaming Chen",
    "id": "2456714714",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Juntao Yang",
    "id": "2456841976",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Zhentao Zou",
    "id": "2263819625",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Qi Ming",
    "id": "1401451771",
    "h_index": 11,
    "papers": 24
   },
   {
    "name": "Yi Yu",
    "id": "2456886719",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Zhihang Zhong",
    "id": "2266469531",
    "h_index": 7,
    "papers": 44
   },
   {
    "name": "Xue Yang",
    "id": "143989318",
    "h_index": 31,
    "papers": 52
   },
   {
    "name": "Xue Jiang",
    "id": "2331680587",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Yue Zhou",
    "id": "2306083274",
    "h_index": 5,
    "papers": 11
   }
  ],
  "comment": "16pages, 11 figures",
  "topics": [
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.08025v1",
  "pdf_url": "https://arxiv.org/pdf/2608.08025v1",
  "html_url": "https://arxiv.org/html/2608.08025v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.08023",
  "slug": "4d-wam-infusing-spatiotemporal-awareness-into-world-action-models-thro",
  "title": "4D-WAM: Infusing Spatiotemporal Awareness into World Action Models through Trajectory Fields",
  "abstract": "Building on recent advances in world models, World Action Models (WAMs) jointly model video prediction and action generation. However, they typically represent videos in 2D pixel space, creating a representation gap with 3D space in which robotic actions are executed. Recent 3D approaches introduce 3D information, but fail to fully exploit the dynamics of 3D structures. In this work, we propose 4D-WAM, a model-agnostic training strategy that injects spatiotemporal knowledge from 3D trajectory fields into WAMs through representation alignment. To this end, we introduce two complementary objectives: 1) motion alignment, which aligns temporal feature variations across adjacent frames and encourages the model to build local 4D awareness during training, and 2) destination alignment, which guides the model to infer the final destination from the source frame by minimizing the gap between their attention-like similarity distributions. Together, these objectives provide both local motion supervision and long-horizon goal guidance, enabling WAMs to learn trajectory-level spatiotemporal representations. Extensive in-distribution and out-of-distribution experiments across different base models demonstrate the model's improvements in spatial understanding, execution precision, robustness, generalization, and versatility.",
  "published": "2026-08-08",
  "updated": "2026-08-12",
  "year": "2026",
  "authors": [
   "Lishan Yang",
   "Wenxuan Song",
   "Xi Wang",
   "Pingyue Sheng",
   "Zheng Fang",
   "Ziyang Zhou",
   "Junjie He",
   "Haodong Yan",
   "Jiayi Chen",
   "Nan Sun",
   "Qiao Sun",
   "Pengwei Wang",
   "Lingqiao Liu",
   "Yan Wang",
   "Yuxiang Gao",
   "Feras Dayoub",
   "Haoang Li"
  ],
  "author_count": 17,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "4D-WAM is proposed, a model-agnostic training strategy that injects spatiotemporal knowledge from 3D trajectory fields into WAMs through representation alignment, enabling WAMs to learn trajectory-level spatiotemporal representations.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Lishan Yang",
    "id": "2456828225",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Wenxuan Song",
    "id": "2293142288",
    "h_index": 14,
    "papers": 46
   },
   {
    "name": "Xi Wang",
    "id": "2309115352",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Pingyue Sheng",
    "id": "2327340052",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Zheng Fang",
    "id": "2456649571",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Ziyang Zhou",
    "id": "2376390365",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Junjie He",
    "id": "2316016558",
    "h_index": 3,
    "papers": 16
   },
   {
    "name": "Haodong Yan",
    "id": "2321603038",
    "h_index": 5,
    "papers": 16
   },
   {
    "name": "Jiayi Chen",
    "id": "2348394168",
    "h_index": 8,
    "papers": 22
   },
   {
    "name": "Nan Sun",
    "id": "2322924729",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Qiao Sun",
    "id": "2107458720",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Lingqiao Liu",
    "id": "2264248423",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Yan Wang",
    "id": "2386705642",
    "h_index": 3,
    "papers": 17
   },
   {
    "name": "Yuxiang Gao",
    "id": "2449168885",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Feras Dayoub",
    "id": "1757942",
    "h_index": 30,
    "papers": 122
   },
   {
    "name": "Haoang Li",
    "id": "2384363611",
    "h_index": 9,
    "papers": 38
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.08023v2",
  "pdf_url": "https://arxiv.org/pdf/2608.08023v2",
  "html_url": "https://arxiv.org/html/2608.08023v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.07895",
  "slug": "auditing-instruction-trajectory-mismatches-in-multimodal-robot-demonst",
  "title": "Auditing Instruction-Trajectory Mismatches in Multimodal Robot Demonstrations",
  "abstract": "Robot demonstration datasets used to train vision-language-action policies can contain a subtle but harmful failure mode: trajectories that are behaviorally correct but paired with the wrong language instruction. We study post-hoc auditing of these Instruction-Trajectory Mismatches (ITMs). Unlike failed rollouts, ITMs often look plausible, and can corrupt the language-behavior mapping learned by the policy. We propose Multimodal Probabilistic Fusion (MMPF), a training-free auditing framework that treats each modality as an expert, estimates a task-label distribution from local neighborhood agreement and global prototype similarity, and then fuses modalities with predictive-entropy weighting in a product of experts. Across LIBERO benchmarks with injected instruction mismatches and noisy real-robot data, MMPF achieves the strongest overall ITM detection and label correction accuracy. We also show that auditing improves most downstream policy learning in settings where language is needed to disambiguate the task. We demonstrate in real robot experiments that our method can achieve improved policy performance and show the trade-off of filtering demonstrations compared to relabeling.",
  "published": "2026-08-08",
  "updated": "2026-08-08",
  "year": "2026",
  "authors": [
   "Simon Holk",
   "Ryosuke Takanami",
   "Tatsuya Matsushima",
   "Yusuke Iwasawa",
   "Yutaka Matsuo",
   "Yueh-Hua Wu",
   "Kei Ota"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "RA-L",
  "venue_source": "semantic-scholar",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Multimodal Probabilistic Fusion (MMPF), a training-free auditing framework that treats each modality as an expert, estimates a task-label distribution from local neighborhood agreement and global prototype similarity, and then fuses modalities with predictive-entropy weighting in a product of experts is proposed.",
  "doi": "10.1109/lra.2026.3723337",
  "oa_pdf": "https://arxiv.org/pdf/2608.07895",
  "s2_authors": [
   {
    "name": "Simon Holk",
    "id": "2140634515",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Ryosuke Takanami",
    "id": "2220216007",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "T. Matsushima",
    "id": "145930468",
    "h_index": 10,
    "papers": 35
   },
   {
    "name": "Yusuke Iwasawa",
    "id": "1715282",
    "h_index": 23,
    "papers": 171
   },
   {
    "name": "Yutaka Matsuo",
    "id": "2241471533",
    "h_index": 13,
    "papers": 81
   },
   {
    "name": "Yueh-Hua Wu",
    "id": "2253837067",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Keita Ota",
    "id": "104236766",
    "h_index": 10,
    "papers": 36
   }
  ],
  "comment": "Accepted for publication in IEEE Robotics and Automation Letters (RA-L). 8 pages, 3 figures, 7 tables",
  "topics": [
   "vla",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.07895v1",
  "pdf_url": "https://arxiv.org/pdf/2608.07895v1",
  "html_url": "https://arxiv.org/html/2608.07895v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.07870",
  "slug": "v-simba-unleashing-the-architectural-potential-of-rl-in-visual-continu",
  "title": "V-Simba: Unleashing the Architectural Potential of RL in Visual Continuous Control",
  "abstract": "Improving sample efficiency remains a core challenge in reinforcement learning (RL), especially in real-world settings like robotics, where data collection is costly. This challenge is pronounced in visual RL, where high-dimensional inputs often obscure learning signals. While prior work in visual RL has focused on algorithmic solutions, such as better dynamics models or exploration strategies, recent advances in state-based RL show that architectural design alone can lead to significant gains in sample efficiency. This raises an important question: Can these architectural principles transfer to visual RL? In response, we introduce V-Simba, a simple yet effective visual RL architecture inspired by the Simba architecture from state-based RL. Built on top of Soft Actor-Critic (SAC) with data augmentation, V-Simba modifies the architecture by adding normalization layers to stabilize training and using pointwise convolutions to reduce computation. Despite its simplicity, V-Simba matches or outperforms the state-of-the-art methods across the DMC, Adroit, and Meta-World benchmarks, while being more computationally efficient than DrQ-v2. We make our code publicly available at https://github.com/DAVIAN-Robotics/V-Simba.",
  "published": "2026-08-08",
  "updated": "2026-08-08",
  "year": "2026",
  "authors": [
   "Donghu Kim",
   "Youngdo Lee",
   "Hojoon Lee",
   "Johan Obando-Ceron",
   "Byungkun Lee",
   "Aaron Courville",
   "Pablo Samuel Castro",
   "Jaegul Choo",
   "Clare Lyle"
  ],
  "author_count": 9,
  "categories": [
   "cs.LG",
   "cs.RO"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work introduces V-Simba, a simple yet effective visual RL architecture inspired by the Simba architecture from state-based RL, which matches or outperforms the state-of-the-art methods across the DMC, Adroit, and Meta-World benchmarks, while being more computationally efficient than DrQ-v2.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Donghu Kim",
    "id": "2304514936",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "youngdon lee",
    "id": "2159466597",
    "h_index": 3,
    "papers": 20
   },
   {
    "name": "Hojoon Lee",
    "id": "2163406260",
    "h_index": 10,
    "papers": 26
   },
   {
    "name": "Johan S. Obando-Ceron",
    "id": "2284765306",
    "h_index": 4,
    "papers": 14
   },
   {
    "name": "Byungkun Lee",
    "id": "2263206527",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "A. Courville",
    "id": "2253653152",
    "h_index": 10,
    "papers": 18
   },
   {
    "name": "Pablo Samuel Castro",
    "id": "2256994343",
    "h_index": 10,
    "papers": 20
   },
   {
    "name": "Jaegul Choo",
    "id": "2260653165",
    "h_index": 11,
    "papers": 88
   },
   {
    "name": "Clare Lyle",
    "id": "2373276541",
    "h_index": 3,
    "papers": 5
   }
  ],
  "comment": "Accepted at RLC'26",
  "topics": [
   "rl-control",
   "navigation",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.07870v1",
  "pdf_url": "https://arxiv.org/pdf/2608.07870v1",
  "html_url": "https://arxiv.org/html/2608.07870v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.07797",
  "slug": "drone-assisted-uav-ugv-collaboration-for-autonomous-navigation-in-snow",
  "title": "Drone-Assisted UAV-UGV Collaboration for Autonomous Navigation in Snow-Covered Terrain",
  "abstract": "This paper presents a collaborative UAV-UGV navigation framework for high-altitude, snow-covered terrain, where reduced visibility and unstable ground render conventional methods ineffective. We introduce a custom efficient U-Net architecture that falls under the computational constraints for real-time road segmentation, utilizing a novel synthetic snow data augmentation technique to achieve 96.5% segmentation accuracy. For UAV localization, we implement an Extended Kalman Filter (EKF) fusing onboard GPS and IMU data, achieving a maximum observed positional error of +-0.5 meters. The UGV position is determined via a visual tracking pipeline using YOLOv5 and depth data from the UAV's RGB-D camera. A dynamic path planning algorithm utilizes this segmentation to adjust for snow drifts, enabling successful navigation in obscured test environment with minimal deviation.",
  "published": "2026-08-07",
  "updated": "2026-08-07",
  "year": "2026",
  "authors": [
   "Shreyam Gupta",
   "P. Agrawal",
   "Priyam Gupta",
   "R. Gautam"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A custom efficient U-Net architecture is introduced that falls under the computational constraints for real-time road segmentation, utilizing a novel synthetic snow data augmentation technique to achieve 96.5% segmentation accuracy.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Shreya Gupta",
    "id": "2284549906",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "P. Agrawal",
    "id": "2342503481",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Priya Gupta",
    "id": "2453324728",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "R. Group",
    "id": "2309135757",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Indian Institute of Technology",
    "id": "102280816",
    "h_index": 10,
    "papers": 36
   },
   {
    "name": "Varanasi",
    "id": "2104599942",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "India",
    "id": "2456630437",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "U. Colorado",
    "id": "2238707398",
    "h_index": 9,
    "papers": 25
   },
   {
    "name": "Boulder",
    "id": "83691542",
    "h_index": 42,
    "papers": 149
   },
   {
    "name": "Usa",
    "id": "2456995220",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Intelligent Field Robotic Systems",
    "id": "2426789768",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "University of Girona",
    "id": "2189542154",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Spain",
    "id": "2456962817",
    "h_index": 1,
    "papers": 3
   }
  ],
  "comment": "",
  "topics": [
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.07797v1",
  "pdf_url": "https://arxiv.org/pdf/2608.07797v1",
  "html_url": "https://arxiv.org/html/2608.07797v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.07751",
  "slug": "coconav-conformal-control-for-safe-robot-navigation-in-crowds",
  "title": "CoCoNav: Conformal Control for Safe Robot Navigation in Crowds",
  "abstract": "Safe and efficient robot navigation in crowds requires anticipating pedestrian motion despite uncertain and potentially shifting prediction errors. Existing reactive methods can produce oscillatory behavior, while predictive planners often treat forecasts as exact or rely on restrictive error models. Incorporating conservative uncertainty sets as hard constraints can also render model predictive control (MPC) infeasible. We propose \\textit{CoCoNav}, a crowd-navigation framework that combines online conformal calibration with runtime-certified planning. A horizon-specific conformal proportional--integral controller adapts trajectory-error bounds to regulate long-run empirical coverage, enabling the framework to respond to changing prediction errors. A \\textit{relax-then-verify} planner preserves solver feasibility by generating nominal trajectories with soft-constrained MPC and separately certifying them, together with contingency maneuvers, against the calibrated bounds before execution. Simulations and quadruped experiments show that CoCoNav achieves a favorable balance among collision avoidance, task success, and navigation efficiency relative to the evaluated baselines.",
  "published": "2026-08-07",
  "updated": "2026-08-07",
  "year": "2026",
  "authors": [
   "Cheng Guo",
   "Mingzhe Ni",
   "Zheng Liang",
   "Yihu Ling",
   "Yuan Hu",
   "Michele Caprio",
   "Daniele Pucci",
   "Wei Pan"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes CoCoNav, a crowd-navigation framework that combines online conformal calibration with runtime-certified planning and achieves a favorable balance among collision avoidance, task success, and navigation efficiency relative to the evaluated baselines.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Cheng Guo",
    "id": "2277651638",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Mingzhe Ni",
    "id": "2456629612",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Zheng Liang",
    "id": "2354607070",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Yihu Ling",
    "id": "2179105514",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Yuan Hu",
    "id": "2456527274",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Michele Caprio",
    "id": "2364055942",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Daniele Pucci",
    "id": "2287942224",
    "h_index": 6,
    "papers": 21
   },
   {
    "name": "Wei Pan",
    "id": "2377994256",
    "h_index": 1,
    "papers": 5
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "rl-control",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.07751v1",
  "pdf_url": "https://arxiv.org/pdf/2608.07751v1",
  "html_url": "https://arxiv.org/html/2608.07751v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.07746",
  "slug": "lucid-latent-skill-unified-control-via-imagined-dynamics-for-long-hori",
  "title": "LUCID: Latent-Skill Unified Control via Imagined Dynamics for Long-Horizon Humanoid Loco-Manipulation",
  "abstract": "Long-horizon humanoid loco-manipulation requires composing versatile whole-body skills and reliable high-level decision making. Existing methods often coordinate pretrained skills with scripted planners, finite-state machines or task-specific model-free policies, restricting their ability to handle complex task sequences. To address this limitation, we propose \\textbf{LUCID}, a hierarchical model-based reinforcement learning framework that plans over reusable skills through imagined rollouts of a learned dynamics model. LUCID first trains a structured latent-conditioned low-level policy via adversarial imitation and then freezes it while jointly learning a high-level policy and macro-dynamics world model. The world model predicts the temporally extended state transitions induced by latent decisions, enabling high-level policy optimization through imagined rollouts. We evaluate our framework across various simulated multi-object rearrangement scenarios. Experimental results show that LUCID improves the full-task success and partial-completion rates compared to prior baseline methods, demonstrating its effectiveness in complex sequential loco-manipulation tasks.",
  "published": "2026-08-07",
  "updated": "2026-08-07",
  "year": "2026",
  "authors": [
   "Cheng Guo",
   "Mingzhe Ni",
   "Angelo Cangelosi",
   "Arash Ajoudani"
  ],
  "author_count": 4,
  "categories": [
   "cs.LG",
   "cs.RO"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Experimental results show that LUCID improves the full-task success and partial-completion rates compared to prior baseline methods, demonstrating its effectiveness in complex sequential loco-manipulation tasks.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Cheng Guo",
    "id": "2277651638",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Mingzhe Ni",
    "id": "2456629612",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "A. Cangelosi",
    "id": "1692929",
    "h_index": 48,
    "papers": 413
   },
   {
    "name": "Arash Ajoudani",
    "id": "2349803262",
    "h_index": 7,
    "papers": 34
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "humanoids",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.07746v1",
  "pdf_url": "https://arxiv.org/pdf/2608.07746v1",
  "html_url": "https://arxiv.org/html/2608.07746v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.07621",
  "slug": "cmu-drive-and-v2v-vla-cooperative-multi-agent-unified-driving-with-rea",
  "title": "CMU-Drive and V2V-VLA: Cooperative Multi-agent Unified Driving with Reasoning Benchmark and Vehicle-to-Vehicle Vision-Language-Action Models",
  "abstract": "Vision-Language-Action (VLA) models have recently achieved impressive performance for end-to-end autonomous driving, yet existing approaches are primarily designed for an individual single autonomous driving agent with limited support for cooperative perception, reasoning, and planning. We present Cooperative Multi-agent Unified Driving with Reasoning (CMU-Drive), a closed-loop end-to-end benchmark for evaluating cooperative autonomous driving with multiple connected autonomous vehicles (CAVs) operating in safety-critical driving scenarios with background traffic participants. We further propose Vehicle-to-Vehicle Vision-Language-Action (V2V-VLA), a cooperative VLA model that integrates cooperative driving into a single forward pass by jointly generating driving actions, future waypoints, language reasoning, and communication policies. Experiments on CMU-Drive establish the first benchmark and baseline for cooperative VLA driving and provide a foundation for future research on multi-agent, closed-loop, end-to-end cooperative autonomous driving. Our code, benchmark, and model checkpoint will be publicly released to facilitate open-source research.",
  "published": "2026-08-07",
  "updated": "2026-08-07",
  "year": "2026",
  "authors": [
   "Hsu-kuang Chiu",
   "Stephen F. Smith"
  ],
  "author_count": 2,
  "categories": [
   "cs.AI",
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.AI",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Cooperative Multi-agent Unified Driving with Reasoning (CMU-Drive) is presented, a closed-loop end-to-end benchmark for evaluating cooperative autonomous driving with multiple connected autonomous vehicles (CAVs) operating in safety-critical driving scenarios with background traffic participants.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hsu-kuang Chiu",
    "id": "2345696515",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Stephen F. Smith",
    "id": "2345724283",
    "h_index": 2,
    "papers": 5
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "navigation",
   "safety-eval"
  ],
  "orgs": [
   "CMU"
  ],
  "abs_url": "https://arxiv.org/abs/2608.07621v1",
  "pdf_url": "https://arxiv.org/pdf/2608.07621v1",
  "html_url": "https://arxiv.org/html/2608.07621v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2608.07619",
  "slug": "gwm-vla-geometry-aware-latent-world-modeling-for-vision-language-actio",
  "title": "GWM-VLA: Geometry-Aware Latent World Modeling for Vision-Language-Action Learning",
  "abstract": "Vision-Language-Action (VLA) models achieve strong robotic manipulation performance but often degrade under visual and environmental shifts. Latent world modeling offers a promising approach to improving robustness, yet existing methods commonly encode camera views independently and predict holistic scene dynamics without explicitly modeling their geometric relationships. We propose GWM-VLA, a geometry-aware latent world modeling framework for VLA learning. GWM-VLA combines geometry-aware multi-view state encoding, global context-conditioned target-view prediction, and shared latent-action representations grounded by robot-action supervision. Specifically, VGGT-$\u03a9$ jointly aggregates multi-view observations at each timestep to construct geometry-aware multi-view states. The latent world model predicts the next-step patch tokens of a selected target view using patch and register tokens obtained after multi-view aggregation, thereby retaining multi-view geometric information without predicting the complete multi-view state. We use the wrist view as the target in our experiments, placing greater emphasis on end-effector motion and local gripper-object interactions. Finally, the shared latent-action representations condition both the latent world model and the flow-matching action head, allowing latent-prediction supervision and ground-truth robot-action supervision to jointly shape the same latent-action representations. Experiments across both simulation and real-world environments demonstrate the effectiveness and robustness of GWM-VLA.",
  "published": "2026-08-07",
  "updated": "2026-08-07",
  "year": "2026",
  "authors": [
   "Yanping Zhao",
   "Hang Yu",
   "Yiwei Wang",
   "Chen Ye",
   "Siyu Tian",
   "Di Zhang",
   "Qingjun Wang",
   "Qian Chen",
   "Junqiao Zhao",
   "Chen Ye",
   "Guang Chen"
  ],
  "author_count": 11,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "GWM-VLA is proposed, a geometry-aware latent world modeling framework for VLA learning that combines geometry-aware multi-view state encoding, global context-conditioned target-view prediction, and shared latent-action representations grounded by robot-action supervision.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yanping Zhao",
    "id": "2314528068",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Hang Yu",
    "id": "2245270655",
    "h_index": 6,
    "papers": 16
   },
   {
    "name": "Yiwei Wang",
    "id": "2446321685",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Chen Ye",
    "id": "2326115154",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Siyu Tian",
    "id": "2456646821",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Di Zhang",
    "id": "2220688800",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Qingjun Wang",
    "id": "2145778766",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Qian Chen",
    "id": "2455479296",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Junqiao Zhao",
    "id": "2282541691",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Guang Chen",
    "id": "2395872020",
    "h_index": 1,
    "papers": 7
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "vla",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.07619v1",
  "pdf_url": "https://arxiv.org/pdf/2608.07619v1",
  "html_url": "https://arxiv.org/html/2608.07619v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.07361",
  "slug": "depth-wise-probing-and-pruning-of-the-planning-token-in-a-driving-visi",
  "title": "Depth-Wise Probing and Pruning of the Planning Token in a Driving Vision-Language-Action Model",
  "abstract": "Vision-language-action (VLA) models route driving decisions through a deep language model, but it is unclear how much of that depth the action itself requires. We study a representative driving VLA whose entire plan is carried by a single planning token that a generative planner decodes into a trajectory. Borrowing the planner as a trajectory-space logit lens, we decode the planning token from every one of the 32 decoder layers and measure two signals: the linear decodability of the navigation command and trajectory compatibility with the frozen native planner. Our diagnostic shows that semantic intent is linearly decodable early: command-probe accuracy reaches 97.7\\% after the first decoder layer, compared with 16.7\\% chance. In contrast, compatibility with the frozen native planner improves gradually across depth, with open-loop Avg-L2 reaching its minimum of 2.11\\,m only at the final layer. Learned readouts from the first layer recover much of this gap, indicating that planning information is already present early but is not yet represented in the format expected by the deployed planner. Ranking decoder layers by the angular deviation they induce in the planning token permits removal of 8 of 32 layers within an approximately 5\\% relative open-loop error increase and yields a measured 1.33$\\times$ decoder speedup. At the evaluated sample size, no family-specific degradation is statistically resolved. These findings are limited to the evaluated ORION checkpoint and Bench2Drive setup.",
  "published": "2026-08-07",
  "updated": "2026-08-07",
  "year": "2026",
  "authors": [
   "Harisankar Babu",
   "Benjamin Coors",
   "Christopher Lang",
   "Hendrik Berkemeyer",
   "Tamim Asfour",
   "Simon Foell"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "ECCV 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A representative driving VLA whose entire plan is carried by a single planning token that a generative planner decodes into a trajectory is studied, showing that semantic intent is linearly decodable early and compatibility with the frozen native planner improves gradually across depth.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Harisankar Babu",
    "id": "2370929046",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Benjamin Coors",
    "id": "2326991006",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Christopher Lang",
    "id": "2146550450",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Hendrik Berkemeyer",
    "id": "2133409956",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Tamim Asfour",
    "id": "2248801384",
    "h_index": 11,
    "papers": 79
   },
   {
    "name": "Simon Foell",
    "id": "2456581796",
    "h_index": 0,
    "papers": 1
   }
  ],
  "comment": "Accepted at the 6th DriveX Workshop (Foundation Models for Autonomous Driving), ECCV 2026. 14 pages, 8 figures, 4 tables",
  "topics": [
   "vla",
   "navigation",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.07361v1",
  "pdf_url": "https://arxiv.org/pdf/2608.07361v1",
  "html_url": "https://arxiv.org/html/2608.07361v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.07328",
  "slug": "learning-fault-tolerant-locomotion-with-adaptive-gait-timing",
  "title": "Learning Fault-Tolerant Locomotion with Adaptive Gait Timing",
  "abstract": "Hardware failures require legged robots to rapidly reorganize coordination and gait timing to maintain stability and mobility. This is particularly challenging for larger quadrupeds, where increased mass and tighter actuation limits reduce the feasibility of aggressive, high-frequency compensation strategies often observed on smaller platforms. In this work, we propose a deep reinforcement learning approach for fault-tolerant locomotion under actuator power loss. The method employs an asymmetric actor-critic architecture in which the critic has access to privileged information during training, while the actor learns to reconstruct a corresponding latent representation from proprioceptive observations. We introduce a latent-alignment loss that encourages consistency between actor and critic representations. Additionally, we augment the action space with a learnable gait frequency parameter, enabling adaptive gait timing in response to terrain variations and actuator degradation without predefined faulty-leg strategies. The approach is validated in high-fidelity simulation on uneven terrain and real-world experiments on flat ground using a 68 kg quadruped robot.",
  "published": "2026-08-07",
  "updated": "2026-08-07",
  "year": "2026",
  "authors": [
   "Giovanbattista Gravina",
   "Luca Rossini",
   "Carlo Rizzardo",
   "Arturo Laurenzi",
   "Nikos Tsagarakis"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A deep reinforcement learning approach for fault-tolerant locomotion under actuator power loss that employs an asymmetric actor-critic architecture in which the critic has access to privileged information during training, while the actor learns to reconstruct a corresponding latent representation from proprioceptive observations.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Giovanbattista Gravina",
    "id": "2349430677",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Luca Rossini",
    "id": "2104941966",
    "h_index": 5,
    "papers": 15
   },
   {
    "name": "Carlo Rizzardo",
    "id": "1419510462",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Arturo Laurenzi",
    "id": "20815291",
    "h_index": 18,
    "papers": 69
   },
   {
    "name": "Nikos G. Tsagarakis",
    "id": "2307917489",
    "h_index": 2,
    "papers": 11
   }
  ],
  "comment": "Accepted at the 2026 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS)",
  "topics": [
   "humanoids",
   "rl-control",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.07328v1",
  "pdf_url": "https://arxiv.org/pdf/2608.07328v1",
  "html_url": "https://arxiv.org/html/2608.07328v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.07314",
  "slug": "tempo-semantic-action-decoupled-rl-post-training-for-vision-language-a",
  "title": "TEMPO: Semantic-Action Decoupled RL Post-Training for Vision-Language-Action Models",
  "abstract": "Vision-language-action (VLA) models are commonly adapted to downstream manipulation tasks via supervised fine-tuning (SFT) or online reinforcement learning (RL) post-training. SFT is prone to distribution mismatch, and existing RL approaches typically apply a single, uniform update strategy to all model components, ignoring their distinct functional roles. We propose TEMPO, a semantic-action decoupled, two-timescale RL post-training framework for VLA models. TEMPO freezes the pretrained vision-language backbone to preserve general semantic representations, and restricts adaptation to two components with dedicated RL optimization loops: the semantic projection layer and the low-level action expert. We update them at different rates--the semantic projection layer infrequently, to keep the latent action stable, and the action expert frequently, to rapidly incorporate control feedback from online interaction. This decoupling RL fine-tuning strategy prevents fast policy updates from destabilizing high-level semantic representations while still allowing the action expert to learn efficiently from online feedback. Experiments on the CALVIN benchmark and real-world manipulation tasks demonstrate that TEMPO consistently outperforms both pretrained state-of-the-art VLA models and the RL post-training baseline, while reaching and maintaining higher evaluation rewards on two real-world tasks.",
  "published": "2026-08-07",
  "updated": "2026-08-07",
  "year": "2026",
  "authors": [
   "Ziheng Liu",
   "Quantao Yang"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "TEMPO is proposed, a semantic-action decoupled, two-timescale RL post-training framework for VLA models that consistently outperforms both pretrained state-of-the-art VLA models and the RL post-training baseline, while reaching and maintaining higher evaluation rewards on two real-world tasks.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ziheng Liu",
    "id": "2456619271",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Quantao Yang",
    "id": "2321668723",
    "h_index": 4,
    "papers": 9
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.07314v1",
  "pdf_url": "https://arxiv.org/pdf/2608.07314v1",
  "html_url": "https://arxiv.org/html/2608.07314v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.07267",
  "slug": "wnm-3d-a-world-navigation-model-with-3d-scene-conditioning-for-closed",
  "title": "WNM-3D: A World Navigation Model with 3D Scene Conditioning for Closed-Loop VLN",
  "abstract": "Recent vision-language navigation (VLN) systems increasingly adapt pretrained vision-language models (VLMs) into vision-language-action (VLA) policies that map egocentric observations and language instructions directly to navigation actions. Although semantically capable, such action-centric training does not explicitly model how the agent's visual observations should evolve under its predicted motion. Generative world-action models (WAMs) jointly predict future observations and actions, yet existing WAMs for continuous VLN do not condition joint future-view and action generation on geometry-aware representations inferred from the observed history. We present WNM-3D, a generative World Navigation Model with 3D scene conditioning for continuous VLN. To consolidate past observations into persistent scene context, a frozen feed-forward geometry encoder extracts geometry-aware representations from the monocular egocentric RGB history, and a trainable 3D Scene-to-Token Adapter converts them into a fixed-length prefix in the token space of the world-action Diffusion Transformer. Through block-causal attention, this prefix conditions every future video-action block, providing a shared geometric context for both future-view and action generation. We train WNM-3D through supervised world-action fine-tuning on A*-generated demonstrations, DAgger-style adaptation on policy-visited states, and Counterfactual DanceGRPO refinement for closed-loop execution. Experiments on GN-Bench show that WNM-3D outperforms strong VLM-based navigation policies and its 2D-conditioned counterpart in closed-loop navigation. Stage-wise ablations further show that DAgger-SFT provides the larger success-rate gain, while Counterfactual DanceGRPO subsequently improves both navigation success and path efficiency.",
  "published": "2026-08-07",
  "updated": "2026-08-20",
  "year": "2026",
  "authors": [
   "Yuehao Huang",
   "Yunzi Wu",
   "Xiaotao Zhang",
   "Xinhai Li",
   "Jiankun Dong",
   "Jiajun Lv",
   "Chi Zhang",
   "Chenjia Bai",
   "Yong Liu",
   "Xuelong Li"
  ],
  "author_count": 10,
  "categories": [
   "cs.AI",
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.AI",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "WNM-3D, a generative World Navigation Model with 3D scene conditioning for continuous VLN, is presented, showing that WNM-3D outperforms strong VLM-based navigation policies and its 2D-conditioned counterpart in closed-loop navigation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yuehao Huang",
    "id": "2282599261",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Yunzi Wu",
    "id": "2326255864",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Xiaotao Zhang",
    "id": "2277983539",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Xinhai Li",
    "id": "2267385564",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Jian Dong",
    "id": "2335320190",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Jiajun Lv",
    "id": "2054668538",
    "h_index": 12,
    "papers": 32
   },
   {
    "name": "Chi Zhang",
    "id": "2376189893",
    "h_index": 5,
    "papers": 21
   },
   {
    "name": "Chenjia Bai",
    "id": "2303257958",
    "h_index": 7,
    "papers": 18
   },
   {
    "name": "Yong Liu",
    "id": "2317960641",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Xuelong Li",
    "id": "2295686463",
    "h_index": 10,
    "papers": 30
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "egocentric-data",
   "navigation",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.07267v2",
  "pdf_url": "https://arxiv.org/pdf/2608.07267v2",
  "html_url": "https://arxiv.org/html/2608.07267v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.07154",
  "slug": "representation-handoffs-for-openarm-based-laboratory-mobile-manipulati",
  "title": "Representation Handoffs for OpenArm-Based Laboratory Mobile Manipulation",
  "abstract": "Open-source robotics and foundation models have lowered the barrier to embodied AI, yet language-guided laboratory automation still requires reliable alignment from instructions and observations to safe actions. This field report presents an OpenArm-based mobile manipulation prototype for laboratory-style tasks, built by integrating dual OpenArm manipulators with a mobile base, vertical slide, RGB-D sensing, lidar-based mapping, ROS2/MoveIt execution, and profile-defined skill interfaces. The system is organized around representation handoffs: natural language requests are constrained into registered skill calls, sensor observations are grounded into maps and object poses, object priors provide role and skill constraints, and runtime bindings compile validated skills into executable motion goals. We use dry-run traces and startup checks to evaluate this integration path, showing how the prototype exposes missing calibration, incomplete object assets, and unfinished real-scene visual grounding as explicit deployment blockers. These intermediate representations serve as practical debugging interfaces for integrating language, perception, planning, and robot safety in embodied systems.",
  "published": "2026-08-07",
  "updated": "2026-08-07",
  "year": "2026",
  "authors": [
   "Yang Shen",
   "Chonghao Cheng",
   "Ziyi Zhao",
   "Jialuo Zhu",
   "Zhenyi Yi",
   "Qi Zhao",
   "Jian Yang",
   "Yuhui Shi",
   "Chin-Teng Lin"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "RSS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This field report presents an OpenArm-based mobile manipulation prototype for laboratory-style tasks, built by integrating dual OpenArm manipulators with a mobile base, vertical slide, RGB-D sensing, lidar-based mapping, ROS2/MoveIt execution, and profile-defined skill interfaces.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yang Shen",
    "id": "2346673966",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Chong Cheng",
    "id": "2329321364",
    "h_index": 8,
    "papers": 19
   },
   {
    "name": "Ziyi Zhao",
    "id": "2315855828",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Jialu Zhu",
    "id": "2428509249",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Zhenyi Yi",
    "id": "2426853746",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Qi Zhao",
    "id": "2256983621",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Jian Yang",
    "id": "2118801121",
    "h_index": 7,
    "papers": 21
   },
   {
    "name": "Yuhui Shi",
    "id": "2408229234",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Chin-teng Lin",
    "id": "2324070926",
    "h_index": 3,
    "papers": 7
   }
  ],
  "comment": "Robotics: Science and Systems (RSS) Workshop 2026",
  "topics": [
   "navigation",
   "foundation-pretraining",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.07154v1",
  "pdf_url": "https://arxiv.org/pdf/2608.07154v1",
  "html_url": "https://arxiv.org/html/2608.07154v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.07079",
  "slug": "lifelongcrossnav-persistent-3d-semantic-memory-for-cross-floor-multi-o",
  "title": "LifelongCrossNav: Persistent 3D Semantic Memory for Cross-Floor Multi-Object Navigation",
  "abstract": "Object-goal navigation has made substantial progress in semantic perception and exploration, yet persistent memory for multi-object navigation and cross-floor navigation are still commonly addressed separately. We present LifelongCrossNav, a framework for sequential multi-object ObjectNav in unknown multi-floor indoor environments. Within each episode, the agent receives an ordered sequence of object-goal queries while continuously maintaining a shared sparse 3D semantic voxel memory. This memory incrementally accumulates geometric structure, traversability states, and vision-language features, allowing subsequent object-goal queries to retrieve previously acquired scene information without rebuilding the map. To support persistent search across floors, LifelongCrossNav combines support-aware 3D traversability mapping, stair-specific perception, and direction-aware stair traversal. A unified navigation policy coordinates same-floor frontier exploration, live and historical point-of-interest retrieval, stair navigation, and target-object search and approach. We further introduce HM3D-MFMON, a benchmark for sequential Multi-Floor Multi-Object Navigation built on HM3D scenes, including a dedicated subset in which completing the full sequence of object-goal subtasks requires at least one floor transition. Experimental results show that LifelongCrossNav consistently outperforms a representative planar persistent semantic-map baseline on HM3D-MFMON, demonstrating that persistent 3D semantic memory and cross-floor traversability modeling effectively support sequential multi-object navigation in multi-floor environments. Project page: https://flageval-baai.github.io/LifelongCrossNavPage.",
  "published": "2026-08-07",
  "updated": "2026-08-07",
  "year": "2026",
  "authors": [
   "Zehui Li",
   "Zihao Sun",
   "Jiawei Xu",
   "Zheqi He",
   "Xiaoqiang Zhang",
   "Jing-Shu Zheng",
   "Lu Liu",
   "Dahui Gao",
   "Xiuwan Chen"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Experimental results show that LifelongCrossNav consistently outperforms a representative planar persistent semantic-map baseline on HM3D-MFMON, demonstrating that persistent 3D semantic memory and cross-floor traversability modeling effectively support sequential multi-object navigation in multi-floor environments.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zehui Li",
    "id": "2337975192",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Zihao Sun",
    "id": "2456517276",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Jiawei Xu",
    "id": "2456612310",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Zheqi He",
    "id": "2281226931",
    "h_index": 6,
    "papers": 20
   },
   {
    "name": "Xiaoqian Zhang",
    "id": "2455488113",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Jingshu Zheng",
    "id": "2366533739",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Lu Liu",
    "id": "2448609730",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Dahui Gao",
    "id": "2456581844",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Xiuwan Chen",
    "id": "2456610559",
    "h_index": 0,
    "papers": 1
   }
  ],
  "comment": "",
  "topics": [
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.07079v1",
  "pdf_url": "https://arxiv.org/pdf/2608.07079v1",
  "html_url": "https://arxiv.org/html/2608.07079v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.07075",
  "slug": "detection-and-ranging-of-transient-extrinsic-contacts-based-on-6d-dyna",
  "title": "Detection and Ranging of Transient Extrinsic Contacts Based on 6D Dynamic Tactile Sensing",
  "abstract": "Delicate manipulation often involves transient and subtle collisions between a grasped object and the environment. While the human hand localizes these contacts effortlessly thanks to superior tactile sensitivity, robotic systems often lack the requisite resolution to acquire the information necessary for motion planning, resulting in clumsy manipulation or even task failure. Here, we propose transient extrinsic contact detection and ranging (TECDAR), a simple yet fast and efficient method for detecting and ranging extrinsic contact of grasped objects. Our design of gripper tips employs dynamic tactile sensing leveraging a single 2.5$\\times$3 mm 6D inertial measurement unit. The sensor captures sub-millisecond tip deformations at a 7 kHz sampling rate, but operating on a data stream of only 84 KB/s. High bandwidth and compact data size enable the system to rapidly detect and localize contact between grasped objects and their surroundings. Specifically, fusing tactile data with robot pose via an extended Kalman filter enables fast and precise localization of extrinsic contact, reaching millimeter-level accuracy within 180 ms. Experimental results demonstrate that the system achieves an average localization accuracy of approximately 7\\,mm in both line-contact and point-contact localization tasks. Furthermore, this near-instantaneous localization enables the robot to rectify its trajectory on a millisecond scale, facilitating precise tool manipulation and enhanced perception of complex environments purely through tactile exploration and mapping. We envision such techniques advancing the future of robotics across domains requiring delicate manipulation, including precision assembly, surgical assistance, and autonomous exploration in touch-dominant environments. Project page: humitlab.github.io/TECDAR/",
  "published": "2026-08-07",
  "updated": "2026-08-07",
  "year": "2026",
  "authors": [
   "Haowen Zheng",
   "Yinghao Wu",
   "Fuyuan Liu",
   "Yichen Li",
   "Yitian Shao"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Haowen Zheng",
    "id": "2277811715",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Yinghao Wu",
    "id": "2239418767",
    "h_index": 4,
    "papers": 30
   },
   {
    "name": "Fuyuan Liu",
    "id": "1785369750",
    "h_index": 6,
    "papers": 19
   },
   {
    "name": "Yichen Li",
    "id": "2293023373",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Yitian Shao",
    "id": "36768094",
    "h_index": 11,
    "papers": 40
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "navigation",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.07075v1",
  "pdf_url": "https://arxiv.org/pdf/2608.07075v1",
  "html_url": "https://arxiv.org/html/2608.07075v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.07074",
  "slug": "m2-smap-memory-efficient-semantic-mapping-with-hierarchical-multi-mode",
  "title": "M2-SMap: Memory-Efficient Semantic Mapping with Hierarchical Multi-Model Representation",
  "abstract": "Dense point cloud maps, as a typically used mapping representation, are difficult to deploy on resource-constrained robots because their memory consumption grows rapidly with scene scale. Although compact single-model representations reduce memory cost, their fixed geometric expressiveness is insufficient for structurally diverse environments. Existing multi-model methods improve representational flexibility, yet their feature extraction and model selection are often dominated by local geometry, which can cause overfitting and adhesion between objects. To address these issues, this paper presents M2-SMap, a memory-efficient semantic mapping framework based on hierarchical multi-model representation. First, a hierarchical geometric decomposition partitions RGB-D point clouds into compact Gaussian components. Then, a projection-guided semantic annotation mechanism assigns instance identities to each component. Subsequently, these annotations are incorporated into an object-aware Gaussian fusion strategy. Furthermore, a multi-scale feature extraction strategy separates large planar regions, semantic objects, and complex residual structures, which are respectively represented by bounded planes, object-level superquadrics, and GMM primitives. Experiments on three RGB-D sequences show that M2-SMap runs in real time at no less than 29.37 Hz while achieving the lowest primitive count, with an average reduction of 18.7% over the best baseline. It also reduces the mean per-frame number of measured inter-object adhesion cases from 2.808 to 0, demonstrating efficient and semantically consistent scene representation.",
  "published": "2026-08-07",
  "updated": "2026-08-07",
  "year": "2026",
  "authors": [
   "QiYing Deng",
   "ZhongLai Wang",
   "Yuan Gao",
   "Wei Dong"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "M2-SMap is presented, a memory-efficient semantic mapping framework based on hierarchical multi-model representation that reduces the mean per-frame number of measured inter-object adhesion cases from 2.808 to 0, demonstrating efficient and semantically consistent scene representation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Qi Deng",
    "id": "2448221392",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Zhong Wang",
    "id": "2452627928",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yuan Gao",
    "id": "2290018823",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Wei Dong",
    "id": "2242975135",
    "h_index": 3,
    "papers": 12
   }
  ],
  "comment": "8 pages, 10 figures",
  "topics": [
   "spatial-3d",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.07074v1",
  "pdf_url": "https://arxiv.org/pdf/2608.07074v1",
  "html_url": "https://arxiv.org/html/2608.07074v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.07045",
  "slug": "c2dex-contact-consistent-reconstruction-and-retargeting-for-dexterous",
  "title": "C2Dex: Contact-Consistent Reconstruction and Retargeting for Dexterous Manipulation from Monocular Video",
  "abstract": "High-quality demonstrations for dexterous robot manipulation are costly and difficult to collect, whereas monocular human videos provide a scalable source of diverse manipulation behaviors. However, transferring such demonstrations to dexterous robots remains challenging: monocular hand-object interaction (HOI) reconstruction often produces temporally unstable contacts and physically implausible interactions, while conventional retargeting methods struggle to preserve task-relevant contacts and local interaction geometry across different hand embodiments. We present C2Dex, a video-to-dexterous-manipulation framework built around a shared interaction representation: stable object-side contacts recovered by aggregating noisy frame-wise observations in the canonical object space. These stable contacts serve a dual role: as trajectory-level constraints that guide reconstruction toward temporally coherent and physically plausible human HOI trajectories, and as explicit transfer targets for the dexterous hand, where Laplacian interaction optimization preserves the local hand-object geometry across embodiments and residual reinforcement learning refines the trajectory in simulation. Experiments on DexYCB and TACO show that C2Dex achieves end-to-end trajectory success rates of 57.78% and 26.67%, respectively, substantially outperforming the strongest baselines (17.78% and 10.00%) under identical evaluation criteria. Real-robot replay experiments further demonstrate physical feasibility across diverse contact-rich manipulation tasks. Project page: https://k-jie.github.io/C2Dex/",
  "published": "2026-08-07",
  "updated": "2026-08-07",
  "year": "2026",
  "authors": [
   "Jie Ren",
   "Zhehao Jiang",
   "Yinhong Yang",
   "Haorui Jia",
   "Han Jiang",
   "Ben Li",
   "Yao Yao",
   "Cheng Lin",
   "Qiu Shen",
   "Zhenshan Bing",
   "Xiao-Xiao Long",
   "Xun Cao"
  ],
  "author_count": 12,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "C2Dex is presented, a video-to-dexterous-manipulation framework built around a shared interaction representation: stable object-side contacts recovered by aggregating noisy frame-wise observations in the canonical object space that serve a dual role: as trajectory-level constraints that guide reconstruction toward temporally coherent and physically plausible human HOI trajectories.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jie Ren",
    "id": "2456619398",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Zhehao Jiang",
    "id": "2456613073",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yinhong Yang",
    "id": "2456626521",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Haorui Jia",
    "id": "2456567104",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Han Jiang",
    "id": "2387249927",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Benshun Li",
    "id": "2362165575",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Yao Yao",
    "id": "2380551629",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Cheng Lin",
    "id": "2268380305",
    "h_index": 12,
    "papers": 26
   },
   {
    "name": "Qiu Shen",
    "id": "2261820351",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Zhenshan Bing",
    "id": "13974169",
    "h_index": 25,
    "papers": 164
   },
   {
    "name": "Xiaoxiao Long",
    "id": "2325298178",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Xun Cao",
    "id": "2371143530",
    "h_index": 3,
    "papers": 5
   }
  ],
  "comment": "9 pages, 5 figures. Submitted to IEEE Robotics and Automation Letters (RA-L). Project page: https://k-jie.github.io/C2Dex/",
  "topics": [
   "dexterous-manipulation",
   "egocentric-data",
   "tactile",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.07045v1",
  "pdf_url": "https://arxiv.org/pdf/2608.07045v1",
  "html_url": "https://arxiv.org/html/2608.07045v1",
  "code_url": "https://k-jie.github.io/C2Dex/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2608.07005",
  "slug": "real-time-whole-body-motion-planning-for-mobile-manipulators-carrying",
  "title": "Real-time Whole-Body Motion Planning for Mobile Manipulators Carrying Arbitrarily Shaped Payloads via Kinematically-Coupled SVSDF",
  "abstract": "Mobile manipulators are increasingly tasked with transporting large, non-convex payloads through cluttered environments, yet existing planners either oversimplify the payload geometry or fail to handle the kinematic coupling between manipulator links, leading to lost feasible space or stalled optimization. This letter presents a real-time whole-body motion planning framework for mobile manipulators carrying arbitrarily shaped payloads. The front-end employs a chain-decomposed kernel-based collision check that preserves the true geometry of the robot and payload, with compact storage and fast bit-level queries. A mid-end preprocessing stage converts the front-end path into a continuous trajectory enforcing smoothness and feasibility, and executes it directly when collision-free to bypass the costly back-end. When refinement is required, the back-end performs trajectory optimization built on a Kinematically-Coupled SVSDF (KC-SVSDF), which propagates collision-avoidance gradients along the kinematic chain to produce coherent whole-body escape directions. Ablation studies, comparative benchmarks against state-of-the-art baselines, and real-world experiments on a differential-drive mobile manipulator demonstrate that the proposed framework reliably transports large, non-convex payloads through tight passages and cluttered environments.",
  "published": "2026-08-07",
  "updated": "2026-08-07",
  "year": "2026",
  "authors": [
   "Yisheng Li",
   "Longji Yin",
   "Tingrui Zhang",
   "Ruize Xue",
   "Haoda Zhu",
   "Nan Chen",
   "Siqi Liang",
   "Yuxi Liu",
   "Fu Zhang"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Alation studies, comparative benchmarks against state-of-the-art baselines, and real-world experiments on a differential-drive mobile manipulator demonstrate that the proposed framework reliably transports large, non-convex payloads through tight passages and cluttered environments.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yisheng Li",
    "id": "2111151733",
    "h_index": 6,
    "papers": 31
   },
   {
    "name": "Longji Yin",
    "id": "2274190845",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Ting Zhang",
    "id": "2146319616",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "Ruize Xue",
    "id": "2322946547",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Haoda Zhu",
    "id": "2222856723",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Nan Chen",
    "id": "2182887311",
    "h_index": 10,
    "papers": 18
   },
   {
    "name": "Siqi Liang",
    "id": "2164497086",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Yuxi Liu",
    "id": "2456988542",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Fu Zhang",
    "id": "2157800072",
    "h_index": 18,
    "papers": 40
   }
  ],
  "comment": "",
  "topics": [
   "humanoids"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.07005v1",
  "pdf_url": "https://arxiv.org/pdf/2608.07005v1",
  "html_url": "https://arxiv.org/html/2608.07005v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.07002",
  "slug": "a-haptic-robot-finger-designed-for-guqin-instrument-playing",
  "title": "A Haptic Robot Finger Designed for Guqin Instrument Playing",
  "abstract": "With the rapid advancement of humanoid robotics and embodied intelligence technologies, numerous musical instrument-playing robots have emerged in recent years, such as pianos, chime bells, and taiko drums. These robots primarily employ open-loop positional control, rendering them incapable of operating instruments requiring dexterous hands and precise tactile perception, such as a violin, guitar, and guqin. This paper describes the design and validation of a high-precision tactile-sensing finger. By mimicking the shape of the fingertip and fingernail found on a human finger, we develop a biomimetic multimodal haptic fingertip and validate it on selected guqin string-contact tasks, including open-string and stopped-note comparisons, harmonic-tuning, and tactile-triggered bimanual coordination, using the guqin, a traditional Chinese musical instrument, as a challenging validation scenario rather than as a fully demonstrated robotic performance system. This research integrates tactile sensing with robotics technology, thereby contributing to applications in world heritage conservation and cultural dissemination.",
  "published": "2026-08-07",
  "updated": "2026-08-07",
  "year": "2026",
  "authors": [
   "Tianwei Zhang",
   "Hanming Yan",
   "Yang Yang. Ziya Wang"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A biomimetic multimodal haptic fingertip is developed and validated on selected guqin string-contact tasks, including open-string and stopped-note comparisons, harmonic-tuning, and tactile-triggered bimanual coordination, using the guqin, a traditional Chinese musical instrument.",
  "doi": "10.1109/TOH.2026.3720822",
  "oa_pdf": "https://arxiv.org/pdf/2608.07002",
  "s2_authors": [
   {
    "name": "Tianwei Zhang",
    "id": "2265215291",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Hanming Yan",
    "id": "2392145699",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Yang Yang",
    "id": "2358489936",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Ziya Wang",
    "id": "2321046964",
    "h_index": 2,
    "papers": 6
   }
  ],
  "comment": "Accepted by IEEE Transactions on Haptics",
  "topics": [
   "dexterous-manipulation",
   "humanoids",
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.07002v1",
  "pdf_url": "https://arxiv.org/pdf/2608.07002v1",
  "html_url": "https://arxiv.org/html/2608.07002v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.06965",
  "slug": "cross-view-action-consistency-for-camera-robust-vision-language-action",
  "title": "Cross-View Action Consistency for Camera-Robust Vision-Language-Action Policies",
  "abstract": "Vision-language-action (VLA) policies fine-tuned from a fixed scene camera can fail when the camera is moved, even when the task, objects, language, and robot state are unchanged. We study scene-camera viewpoint robustness using only a scene RGB image, language, and proprioception, without camera labels, extrinsics, depth, or point-cloud inputs. The wrist stream is masked throughout to prevent an unperturbed visual shortcut from confounding attribution to scene-camera variation. For flow-based VLAs, we propose to regularize the action-flow velocity field, the quantity directly integrated to generate continuous action chunks. We construct action-equivalent view pairs by resetting original LIBERO demonstrations to the same MuJoCo state and rendering nominal and perturbed scene-camera views. Both views are supervised by flow matching, while a cross-view loss encourages their predicted action-flow velocities to agree at the same sampled flow coordinates. On the LIBERO-Plus camera-perturbation track, our method reaches 87.2$\\pm$0.4% (4,797 rollouts per seed across 3 training seeds), +7.4pp over flow-matching-only training on the same paired data (79.8$\\pm$0.8%, also 3 seeds) and +12.5pp over naive mixed-camera SFT, while maintaining nominal-camera ID performance (95.0$\\pm$0.8%; same-data FM-only: 95.0$\\pm$4.3%). A shuffled-pair control collapses to 25.8%, showing that the gain depends on action-equivalent pairing. On a real robot, we evaluate three tabletop tasks with 10 rollouts per task and camera placement; held-out-camera success improves from 53.3% to 74.4% under the same single-scene-RGB inference interface.",
  "published": "2026-08-07",
  "updated": "2026-08-07",
  "year": "2026",
  "authors": [
   "Bingqi Huang",
   "Bingchuan Wei",
   "Xuan Wang",
   "Yingkai Cai",
   "Zhaokui Wang"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work constructs action-equivalent view pairs by resetting original LIBERO demonstrations to the same MuJoCo state and rendering nominal and perturbed scene-camera views, supervised by flow matching, while a cross-view loss encourages their predicted action-flow velocities to agree at the same sampled flow coordinates.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Bingqi Huang",
    "id": "2446907310",
    "h_index": 0,
    "papers": 5
   },
   {
    "name": "Bingchuan Wei",
    "id": "6147636",
    "h_index": 14,
    "papers": 31
   },
   {
    "name": "Xuan Wang",
    "id": "2456624787",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yingkai Cai",
    "id": "1471705598",
    "h_index": 4,
    "papers": 18
   },
   {
    "name": "Zhaokui Wang",
    "id": "2284829171",
    "h_index": 3,
    "papers": 15
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.06965v1",
  "pdf_url": "https://arxiv.org/pdf/2608.06965v1",
  "html_url": "https://arxiv.org/html/2608.06965v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.06919",
  "slug": "vernata-self-supervised-learning-of-lidar-point-representations",
  "title": "Vernata: Self-Supervised Learning of LiDAR Point Representations",
  "abstract": "LiDAR serves as a primary sensing modality for robots operating in outdoor environments. However, the performance of deep learning models in this domain is severely limited by the scarcity of labeled data, a direct result of the high cost of 3D annotation. Self-supervised learning addresses this scarcity by learning general-purpose features from unlabeled data. In this work, we present a multi-modal, multi-teacher distillation framework for self-supervised learning on outdoor LiDAR point clouds. Building upon the Sonata architecture, we introduce Vernata, consisting of three extensions: sparse view augmentation to improve robustness against varying point densities, a memory bank mechanism to stabilize resource-constrained training, and cross-modal distillation utilizing dense, high-resolution 2D image features to enable fine-grained semantic guidance. We evaluate our method on the GrandTour, TartanGround, and Waymo datasets, as well as data collected from our own robotic platforms. Our experiments demonstrate a significant performance improvement over Sonata baselines, yielding mIoU scores of 54.7 on TartanGround (+5.9 points, +12.1%) and 57.1 on Waymo (+7.3 points, +14.7%). Finally, we show that the self-supervised approach maintains strong performance even in reduced-modality settings (lacking color or normals), achieving competitive mIoU scores of 49.4 and 50.2 on the respective datasets.",
  "published": "2026-08-07",
  "updated": "2026-08-07",
  "year": "2026",
  "authors": [
   "Oliver Lemke",
   "Alexander Liniger",
   "Abel Gawel",
   "Marco Hutter"
  ],
  "author_count": 4,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Vernata is introduced, consisting of three extensions: sparse view augmentation to improve robustness against varying point densities, a memory bank mechanism to stabilize resource-constrained training, and cross-modal distillation utilizing dense, high-resolution 2D image features to enable fine-grained semantic guidance.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Oliver Lemke",
    "id": "2004521305",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Alexander Liniger",
    "id": "2373969",
    "h_index": 27,
    "papers": 56
   },
   {
    "name": "Abel Gawel",
    "id": "2261282959",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Marco Hutter",
    "id": "2273650854",
    "h_index": 4,
    "papers": 13
   }
  ],
  "comment": "IROS 2026. Implementation: https://github.com/rai-opensource/vernata",
  "topics": [
   "spatial-3d",
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.06919v1",
  "pdf_url": "https://arxiv.org/pdf/2608.06919v1",
  "html_url": "https://arxiv.org/html/2608.06919v1",
  "code_url": "https://github.com/rai-opensource/vernata",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.5
 },
 {
  "id": "2608.06907",
  "slug": "spatiotemporal-agility-time-constrained-reinforcement-learning-for-vis",
  "title": "Spatiotemporal Agility: Time-Constrained Reinforcement Learning for Vision-Guided Dynamic Quadrupedal Interception",
  "abstract": "Legged robots require robust agility to perceive and interact with complex and dynamic environments within a constrained time. However, most existing quadruped locomotion works rely on velocity-tracking policy, which struggle to reach precise targets within strict temporal constraints. Moreover, integrating real-time perception with agile locomotion for highly dynamic targets remains challenging due to sensor latency and processing delays. To concretely study and benchmark such agility in dynamic settings, we introduce a challenging ball-catching task for legged robots. This paper proposes an integrated framework that combines a vision module for landing point and time prediction with a direct position and time conditioned RL locomotion policy, instead of intermediate velocity commands. Beyond the method design, this work presents a system-level contribution that completes real-time robotic interception system that integrates multi-camera perception, online trajectory prediction, low-latency target communication, and sim-to-real locomotion control into a closed-loop deployment pipeline. By explicitly predicting the future spatial-temporal target, our approach mitigates perception latency during dynamic interception. We conducted extensive ball-catching experiments for the legged robot. Through comparative experiments against a velocity-tracking baseline, our direct target-conditioned approach achieves a higher success rate in catching balls with predicted landing spots within 2 meters and flight times between 0.8 and 1.2 seconds. This shows that the robot has successfully completed the dynamic ball-catching task under our tested setup. Furthermore, our policy exhibits a smaller performance gap after deployment, suggesting improved sim-to-real behavior in these trials.",
  "published": "2026-08-07",
  "updated": "2026-08-07",
  "year": "2026",
  "authors": [
   "Yidong Zhu",
   "Zibo Dai",
   "Tongning Zhang",
   "Leixin Chang",
   "Hua Chen"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "An integrated framework that combines a vision module for landing point and time prediction with a direct position and time conditioned RL locomotion policy, instead of intermediate velocity commands is proposed, which mitigates perception latency during dynamic interception.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yidong Zhu",
    "id": "2456617949",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Zibo Dai",
    "id": "2284737288",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Tongning Zhang",
    "id": "2456611337",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Leixin Chang",
    "id": "2291142443",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Hua Chen",
    "id": "2331031298",
    "h_index": 4,
    "papers": 11
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "sim2real",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.06907v1",
  "pdf_url": "https://arxiv.org/pdf/2608.06907v1",
  "html_url": "https://arxiv.org/html/2608.06907v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.06898",
  "slug": "how-should-i-pick-a-foundation-model-for-my-robot-in-favor-of-a-commun",
  "title": "How Should I Pick a Foundation Model for My Robot? In Favor of a Community Evaluation Framework for Social Robots",
  "abstract": "Researchers who seek to build social robot applications on foundation models are faced with a difficult question: how should we pick a model? Public leaderboards offer little guidance: the demands of real-time, embodied social interaction lie largely outside their focus. And direct evaluation is impractical at scale: each embodied study requires scarce participant, robot, and experimenter time. In this paper, we identify five evaluation dimensions for foundation models in social robots: (i) conversational competence, (ii) user safety, (iii) embodied character, (iv) target scene effectiveness, and (v) audience appropriateness. To make model selection cheaper and better informed, we propose a three-tiered evaluation funnel paradigm that first filters with general metrics, then extends to simulated interactions, and terminates in more expensive, robot-specific evaluation. We map all five dimensions across all three tiers, chart where applicable evaluation methods exist and are missing, and close with a call to action: let's build the evaluation framework together as a community.",
  "published": "2026-08-07",
  "updated": "2026-08-07",
  "year": "2026",
  "authors": [
   "Eric Nichols",
   "Alva Markelius",
   "Hatice Gunes"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.CL",
   "cs.HC"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper proposes a three-tiered evaluation funnel paradigm that first filters with general metrics, then extends to simulated interactions, and terminates in more expensive, robot-specific evaluation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Eric Nichols",
    "id": "2272865341",
    "h_index": 5,
    "papers": 26
   },
   {
    "name": "A. Markelius",
    "id": "2208973765",
    "h_index": 5,
    "papers": 19
   },
   {
    "name": "Hatice Gunes",
    "id": "2351608112",
    "h_index": 2,
    "papers": 14
   }
  ],
  "comment": "5 pages, 1 figure, 1 table. Accepted at the FoRMA workshop (Foundation Models in the RO-MAN Age: Responsible Development for Social Robotics) at IEEE RO-MAN 2026, Kitakyushu, Japan. Workshop homepage: https://sites.google.com/cam.ac.uk/forma/",
  "topics": [
   "foundation-pretraining",
   "hri",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.06898v1",
  "pdf_url": "https://arxiv.org/pdf/2608.06898v1",
  "html_url": "https://arxiv.org/html/2608.06898v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.06833",
  "slug": "unordered-landmark-visual-navigation",
  "title": "Unordered Landmark Visual Navigation",
  "abstract": "Image-goal navigation is a fundamental capability for embodied AI, yet its practical deployment is strained by strong prior assumptions. Existing methods predominantly rely on temporally ordered video streams or auxiliary sensors (e.g., depth, LiDAR) to maintain spatial consistency. These sequential and multimodal dependencies severely restrict scalability, especially when deploying robots using crowd-sourced or pre-recorded unordered image collections. When temporal priors are removed, current methods struggle with severe perceptual aliasing, noisy associations, and catastrophic mapping failures. To address this underexplored challenge, we propose Unordered Landmark Visual Navigation (ULVN), a unified RGB-only framework free from temporal and odometric priors. ULVN systematically mitigates error accumulation by integrating mapping, localization, and planning. Specifically, it constructs a robust 2D topological map directly from unstructured images via calibrated geometric verification and maximum spanning forest refinement. For closed-loop execution, ULVN abandons sequential heuristics, utilizing a graph-based belief propagation filter with entropy-adaptive fusion for global localization and dynamic subgoal planning. Extensive experiments in simulation and real-world deployments demonstrate that ULVN significantly outperforms state-of-the-art methods.",
  "published": "2026-08-07",
  "updated": "2026-08-10",
  "year": "2026",
  "authors": [
   "Hao Ren",
   "Junzhe Zhu",
   "Yihan Li",
   "Zetong Bi",
   "Le Zheng",
   "Zhi Li",
   "Yiqing Yuan",
   "Zhaoliang Wan",
   "Dizhe Zhang",
   "Lu Qi",
   "Hui Cheng"
  ],
  "author_count": 11,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Unordered Landmark Visual Navigation (ULVN) is proposed, a unified RGB-only framework free from temporal and odometric priors that constructs a robust 2D topological map directly from unstructured images via calibrated geometric verification and maximum spanning forest refinement.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hao Ren",
    "id": "2356375867",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "Junzhe Zhu",
    "id": "2454691404",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yihan Li",
    "id": "2373275880",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Zetong Bi",
    "id": "2355350918",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Le Zheng",
    "id": "2378909227",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Zhi Li",
    "id": "2155344114",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Yiqing Yuan",
    "id": "2387361542",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Zhaoliang Wan",
    "id": "2337114741",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Dizhe Zhang",
    "id": "2367731713",
    "h_index": 6,
    "papers": 16
   },
   {
    "name": "Lu Qi",
    "id": "2307083138",
    "h_index": 8,
    "papers": 21
   },
   {
    "name": "Hui Cheng",
    "id": "2269708435",
    "h_index": 4,
    "papers": 16
   }
  ],
  "comment": "ECCV2026 Oral & Spotlight",
  "topics": [
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.06833v2",
  "pdf_url": "https://arxiv.org/pdf/2608.06833v2",
  "html_url": "https://arxiv.org/html/2608.06833v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.06827",
  "slug": "r2s-ego-dual-proxy-refinement-for-sparse-capture-real-to-sim",
  "title": "R2S-EGO: Dual-Proxy Refinement for Sparse-Capture Real-to-Sim",
  "abstract": "Real-to-sim (R2S) depends on scene representations that render observations along robot ego trajectories, yet dense multi-view capture limits per-environment real-image capture-count efficiency, and sparse human capture can leave behavior-scoped robot views under-supported. Camera-controlled synthesis can fill missing views, but its use in R2S requires behavior-admissible queries and capture-anchored structural conditioning. We present R2S-EGO, which couples a simulator-derived robot proxy that represents the behavior-scoped executable query domain with a capture-anchored geometry proxy that supplies scene-specific structural conditions. Within this domain, fixed- budget selection targets current support deficits for which geometry support is available. The generated observations are assimilated as pseudo-observations to refine the visual asset, while real captures remain anchors. The fused geometry proxy also supplies the scene collision surface, which is refreshed between rounds. Together, these updates refine the existing simulation scene while its robot dynamics and control stack stay fixed. Across 48 frozen Unitree G1 ego views in three Replica scenes, six-view R2S-EGO reaches 19.062 dB PSNR, compared with 14.226 dB for the strongest reported R2S baseline. Across five paired policy-training seeds, R2S-EGO achieves 82.5% +/- 6.8% real-G1 sitting success, compared with 10.0% +/- 10.5% for GaussGym.",
  "published": "2026-08-07",
  "updated": "2026-08-07",
  "year": "2026",
  "authors": [
   "Shuai Fang",
   "Xin Deng",
   "Yuchen Kang",
   "Zhenjiang Li",
   "Jie Chen"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.CV",
   "cs.GR"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "This work presents R2S-EGO, which couples a simulator-derived robot proxy that represents the behavior-scoped executable query domain with a capture-anchored geometry proxy that supplies scene-specific structural conditions.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Shuai Fang",
    "id": "2456280553",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Xin Deng",
    "id": "2453289990",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yuchen Kang",
    "id": "2445709480",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Zhenjiang Li",
    "id": "2453951630",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Jie Chen",
    "id": "2445573778",
    "h_index": 0,
    "papers": 4
   }
  ],
  "comment": "11 pages, 6 figures, 4 tables, and 1 algorithm",
  "topics": [
   "sim2real",
   "spatial-3d"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2608.06827v1",
  "pdf_url": "https://arxiv.org/pdf/2608.06827v1",
  "html_url": "https://arxiv.org/html/2608.06827v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.8
 },
 {
  "id": "2608.06799",
  "slug": "is-forward-prediction-enough-physical-state-grounding-for-jepa-world-m",
  "title": "Is Forward Prediction Enough? Physical State Grounding for JEPA World Models",
  "abstract": "Learning structured and control-relevant latent representations remains a key challenge for world models. Recent JEPA-based world models learn action-conditioned predictive latent dynamics from observation sequences. However, their forward-prediction objectives do not explicitly enforce reliable identifiability of robot-centric physical state from individual latents or state changes from latent pairs, which can limit downstream planning and policy performance. We propose PSG-JEPA, a physically grounded JEPA world model that shapes its latent space with two complementary grounding objectives beyond forward prediction: grounding individual latents in robot proprioceptive state, and grounding latent pairs in multi-horizon joint-angle changes. Both objectives are applied only during training, leaving the inference architecture and computational cost unchanged. To comprehensively evaluate PSG-JEPA, we conduct experiments at three levels: (1) latent identifiability via probing, (2) goal-conditioned planning on frozen latents, and (3) policy learning in simulation and on a real robot. Experiments demonstrate that our PSG-JEPA consistently outperforms state-of-the-art latent world-model baselines at all three levels.",
  "published": "2026-08-07",
  "updated": "2026-08-07",
  "year": "2026",
  "authors": [
   "Haodong Yan",
   "Jiaguan Zhu",
   "Mingyuan Jia",
   "Ruiqing Yin",
   "Junjie He",
   "Zhide Zhong",
   "Junfeng Li",
   "Jinxuan Lu",
   "Hengtao Li",
   "Tianran Zhang",
   "Jiayi Chen",
   "Wenxuan Song",
   "Wen Chen",
   "Yuxiang Gao",
   "Haoang Li"
  ],
  "author_count": 15,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 2,
  "influential_citations": 0,
  "tldr": "PSG-JEPA is proposed, a physically grounded JEPA world model that shapes its latent space with two complementary grounding objectives beyond forward prediction: grounding individual latents in robot proprioceptive state, and grounding latent pairs in multi-horizon joint-angle changes.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Haodong Yan",
    "id": "2321603038",
    "h_index": 5,
    "papers": 16
   },
   {
    "name": "Jiaguang Zhu",
    "id": "1575707808",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Ming-Ming Jia",
    "id": "2322839115",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Ruiqing Yin",
    "id": "2401540544",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Junjie He",
    "id": "2316016558",
    "h_index": 3,
    "papers": 16
   },
   {
    "name": "Zhide Zhong",
    "id": "2349315841",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Junfeng Li",
    "id": "2376547586",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Jinxuan Lu",
    "id": "2456616441",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Hengtao Li",
    "id": "2218230392",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "Tianran Zhang",
    "id": "2308051940",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Jiayi Chen",
    "id": "2348394168",
    "h_index": 8,
    "papers": 22
   },
   {
    "name": "Wenxuan Song",
    "id": "2293142288",
    "h_index": 14,
    "papers": 46
   },
   {
    "name": "Wen Chen",
    "id": "2444700942",
    "h_index": 0,
    "papers": 5
   },
   {
    "name": "Yuxiang Gao",
    "id": "2449168885",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Haoang Li",
    "id": "2384363611",
    "h_index": 9,
    "papers": 38
   }
  ],
  "comment": "",
  "topics": [
   "world-models"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.06799v1",
  "pdf_url": "https://arxiv.org/pdf/2608.06799v1",
  "html_url": "https://arxiv.org/html/2608.06799v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.48
 },
 {
  "id": "2608.06729",
  "slug": "atlasvla-persistent-world-ego-state-modeling-for-vision-language-actio",
  "title": "AtlasVLA: Persistent World-Ego State Modeling for Vision-Language-Action Models",
  "abstract": "While Vision-Language-Action (VLA) models have advanced embodied AI, their fundamentally reactive paradigm severely limits performance in partially observable and long-horizon tasks. When restricted to a single wrist-mounted camera, they inevitably suffer from perception forgetting as objects exit the field of view, and temporal task-progress forgetting} during multi-step execution. To overcome these bottlenecks, we propose AtlasVLA, a novel framework that transitions from direct reactive manipulation to proactive reasoning through a persistent world-ego state. AtlasVLA features a dual-memory architecture: a 4D Persistent World State Memory that lifts transient 2D observations into a globally updated, voxel-hashed spatial state to resolve visual blind spots, and an Ego-Working State Memory that tracks historical ego state and task progress. By conditioning a diffusion transformer (DiT) on this joint World-Ego state, AtlasVLA enables robust spatial reasoning. Extensive evaluations across LIBERO, RLBench, and real-world benchmarks demonstrate that AtlasVLA achieves state-of-the-art performance using solely a wrist camera. Remarkably, it decisively outperforms multi-view baselines, yielding absolute success rate improvements of 9.4% on LIBERO-Long and 17.5% in real-world long-horizon tasks.",
  "published": "2026-08-07",
  "updated": "2026-08-07",
  "year": "2026",
  "authors": [
   "Guiyu Zhao",
   "Longteng Guo",
   "Yanghong Mei",
   "Zilin Zhu",
   "Yu Zhang",
   "Bin Cao",
   "Mingming Yu",
   "Xingjian He",
   "Jie Jiang",
   "Jing Liu"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "AtlasVLA is a novel framework that transitions from direct reactive manipulation to proactive reasoning through a persistent world-ego state and decisively outperforms multi-view baselines, yielding absolute success rate improvements of 9.4% on LIBERO-Long and 17.5% in real-world long-horizon tasks.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Guiyu Zhao",
    "id": "2260857503",
    "h_index": 5,
    "papers": 16
   },
   {
    "name": "Longteng Guo",
    "id": "26982950",
    "h_index": 18,
    "papers": 73
   },
   {
    "name": "Yanghong Mei",
    "id": "2397618849",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Zilin Zhu",
    "id": "2373433533",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Yu Zhang",
    "id": "2355247122",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Bin Cao",
    "id": "2307455673",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Mingming Yu",
    "id": "2387333550",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Xingjian He",
    "id": "153003010",
    "h_index": 11,
    "papers": 41
   },
   {
    "name": "Jie Jiang",
    "id": "2297824922",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Jing Liu",
    "id": "2287961447",
    "h_index": 5,
    "papers": 13
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.06729v1",
  "pdf_url": "https://arxiv.org/pdf/2608.06729v1",
  "html_url": "https://arxiv.org/html/2608.06729v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.06688",
  "slug": "crosstracer-cross-embodiment-navigation-via-vla-model-reasoning-and-tr",
  "title": "CrossTracer: Cross-Embodiment Navigation via VLA Model Reasoning and Trace Residuals Adapting",
  "abstract": "Vision-language-action (VLA) models provide strong semantic priors for robot navigation, but they often ignore embodiment-specific mobility constraints. A path that is semantically plausible for one robot may be physically infeasible for another. We propose CrossTracer, a hierarchical framework for cross-embodiment navigation through adaptive trace residuals. CrossTracer represents navigation plans as normalized image-plane waypoints, forming a unified pixel-space interface between semantic reasoning and physical grounding. First, Vision-Language Trace Proposer (VL-Tracer) adapts a pretrained VLA model to predict an initial navigation trace from egocentric observations and flexible goal specifications. Second, CE-Adapter refines this trace by predicting embodiment-conditioned residual corrections from visual traversability cues, robot identity, and the initial trace. To train the refinement module without costly manual annotation, Cross-Embodiment RRT* (CE-RRT*) converts panoptic segmentation into robot-conditioned traversability cost maps and generates cost-minimizing pixel-space traces. We evaluate CrossTracer on the NaviTrace benchmark, which tests whether a model can generate embodiment-consistent navigation traces from egocentric observations, language instructions, and robot embodiment types. CrossTracer achieves a total score of 45.68, outperforming the strongest evaluated general-purpose baseline, Gemini-2.5-Pro, by 10.01 points, corresponding to a 28.1% relative improvement. Real-world deployment on wheeled and legged robots further shows improved navigation success and execution efficiency.",
  "published": "2026-08-07",
  "updated": "2026-08-07",
  "year": "2026",
  "authors": [
   "Yao Wang",
   "Siyuan Wang",
   "Zhirui Sun",
   "Wenzheng Chi",
   "Liang Lin",
   "Jiankun Wang",
   "Wenjun Xu"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes CrossTracer, a hierarchical framework for cross-embodiment navigation through adaptive trace residuals, and evaluates it on the NaviTrace benchmark, which tests whether a model can generate embodiment-consistent navigation traces from egocentric observations, language instructions, and robot embodiment types.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yao Wang",
    "id": "2367480047",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Siyuan Wang",
    "id": "2456617651",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Zhi-Qiang Sun",
    "id": "2133864216",
    "h_index": 6,
    "papers": 21
   },
   {
    "name": "Wenzheng Chi",
    "id": "1913866",
    "h_index": 17,
    "papers": 79
   },
   {
    "name": "Liang Lin",
    "id": "2456613967",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Jiankun Wang",
    "id": "51068901",
    "h_index": 25,
    "papers": 126
   },
   {
    "name": "Wenjun Xu",
    "id": "2379807474",
    "h_index": 1,
    "papers": 11
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "humanoids",
   "egocentric-data",
   "navigation",
   "foundation-pretraining",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.06688v1",
  "pdf_url": "https://arxiv.org/pdf/2608.06688v1",
  "html_url": "https://arxiv.org/html/2608.06688v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.07606",
  "slug": "enhanced-real-time-6-dof-extended-reality-catheter-tracking-for-evalua",
  "title": "Enhanced Real-Time 6-DOF Extended Reality Catheter Tracking for Evaluating Potential Improvement in Efficiency, Precision, and Depth Perception for Cardiac Interventions",
  "abstract": "Despite advances in 3D ultrasound, most percutaneous cardiac interventions still rely on 2D visualization, limiting depth perception and spatial understanding. To address this challenge, we developed an Extended Reality (XR)-based platform that enables real-time six-degree-of-freedom (6-DOF) catheter tracking and visualization within a patient-specific 3D heart model. The system combines a custom machine-vision algorithm for 5-DOF catheter tracking with a 3D-printed electromechanical encoder that measures catheter roll, providing complete 6-DOF motion reconstruction. In a proof-of-concept study, 20 novice medical students navigated an intracardiac echocardiography (ICE) catheter to six anatomical targets using either immersive 3D visualization or a conventional 2D cathlab-style view. Participants in the 3D condition completed the task in 54.6 seconds and traveled 1,939 mm on average, compared with 267.5 seconds and 7,854 mm in the 2D condition. Therefore, the XR-based 3D system was more than 5x faster and required ~5x less catheter travel. The 3D mode also improved targeting precision and reduced performance variability. Participants consistently rated immersive visualization higher for accuracy, speed, usability, and clinical value. Kinematic analysis showed smoother depth-axis navigation in 3D, whereas 2D users relied on repeated corrective movements. These findings demonstrate that XR-based visualization can substantially improve procedural training efficiency, precision, and motor control.",
  "published": "2026-08-06",
  "updated": "2026-08-06",
  "year": "2026",
  "authors": [
   "Mohsen Annabestani",
   "Sandhya Sriram",
   "Andrew Kuzemczak",
   "S. Chiu Wong",
   "Alexandros Sigaras",
   "Bobak Mosadegh"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.HC",
   "eess.IV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "An Extended Reality (XR)-based platform that enables real-time six-degree-of-freedom catheter tracking and visualization within a patient-specific 3D heart model demonstrates that XR-based visualization can substantially improve procedural training efficiency, precision, and motor control.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Mohsen Annabestani",
    "id": "2284288405",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Sandhya Sriram",
    "id": "2329186485",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Andrew Kuzemczak",
    "id": "2254136779",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "S. Wong",
    "id": "2307422839",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Alexandros Sigaras",
    "id": "2835550",
    "h_index": 16,
    "papers": 38
   },
   {
    "name": "B. Mosadegh",
    "id": "2283126",
    "h_index": 38,
    "papers": 126
   }
  ],
  "comment": "",
  "topics": [
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.07606v1",
  "pdf_url": "https://arxiv.org/pdf/2608.07606v1",
  "html_url": "https://arxiv.org/html/2608.07606v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.07600",
  "slug": "adadexgrasp-adaptive-dexterous-grasping-via-3d-visuo-tactile-represent",
  "title": "AdaDexGrasp: Adaptive Dexterous Grasping via 3D Visuo-Tactile Representation Fusion",
  "abstract": "Humans achieve stable and adaptive grasps by seamlessly integrating visual perception and tactile feedback, a capability that remains challenging to replicate in robotic systems. Existing robotic grasping approaches predominantly rely on visual inputs and lack mechanisms for tactile-guided adaptation after contact, limiting robustness and generalization. To address this challenge, we propose a unified visuo-tactile-fusion grasping framework that integrates grasp generation, feasibility prediction, and adaptive refinement. At its core, our method introduces an efficient visuo-tactile representation that tightly fuses object geometry with tactile feedback by associating tactile signals with finger identities. This unified representation supports contact-aware grasp pose generation during planning and tactile-guided refinement after contact, enabling the system to reason about fine-grained finger-object interactions and adjust grasps dynamically. Comprehensive experiments in both simulation and real-world environments demonstrate that our approach significantly enhances grasp success rates and generalization across diverse objects.",
  "published": "2026-08-06",
  "updated": "2026-08-06",
  "year": "2026",
  "authors": [
   "Xirui Liang",
   "Jiaqi Liang",
   "Jingkai Xu",
   "Yuran Wang",
   "Ruochong Li",
   "Yuanpei Chen",
   "Masayoshi Tomizuka",
   "Wei Zhan",
   "Ruihai Wu"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO",
   "eess.IV"
  ],
  "primary_category": "cs.RO",
  "venue": "ECCV 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes a unified visuo-tactile-fusion grasping framework that integrates grasp generation, feasibility prediction, and adaptive refinement and introduces an efficient visuo-tactile representation that tightly fuses object geometry with tactile feedback by associating tactile signals with finger identities.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xirui Liang",
    "id": "2455423695",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Jiaqi Liang",
    "id": "2362279070",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Jingkai Xu",
    "id": "2456657821",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yuran Wang",
    "id": "2349738743",
    "h_index": 5,
    "papers": 19
   },
   {
    "name": "Ruochong Li",
    "id": "2348185103",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Yuanpei Chen",
    "id": "2261728034",
    "h_index": 13,
    "papers": 33
   },
   {
    "name": "Masayoshi Tomizuka",
    "id": "2261974717",
    "h_index": 7,
    "papers": 27
   },
   {
    "name": "Wei Zhan",
    "id": "144267500",
    "h_index": 42,
    "papers": 164
   },
   {
    "name": "Ruihai Wu",
    "id": "2382450653",
    "h_index": 3,
    "papers": 13
   }
  ],
  "comment": "Accepted at ECCV 2026",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.07600v1",
  "pdf_url": "https://arxiv.org/pdf/2608.07600v1",
  "html_url": "https://arxiv.org/html/2608.07600v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.07596",
  "slug": "lira-local-cross-layer-information-routing-for-vision-language-action",
  "title": "LIRA: Local Cross-Layer Information Routing for Vision-Language-Action Decoding",
  "abstract": "Vision-Language-Action (VLA) models transform representations from pretrained vision-language models (VLMs) into robot actions, yet the interface that routes intermediate VLM features into action decoders remains underexplored. Existing designs either expose only a narrow part of the representation hierarchy or rigidly match each decoder block to one VLM layer, restricting access to complementary task evidence across depths. We introduce LIRA, a local cross-layer action-conditioning mechanism that formulates VLM-to-action conditioning as depth-aware information routing. LIRA operates on task-token features and LIRA Query features derived from intermediate VLM states, then assigns each Parallel Fusion Block a depth-aligned local window centered on its corresponding VLM layer. Parallel Fusion Blocks aggregate neighboring LIRA Query features and integrate them with task-token features and proprioceptive inputs before action prediction. This routing interface leaves the backbone architecture, action decoder, and supervised training recipe unchanged. Across LIBERO, LIBERO-Plus, CALVIN ABC$\\rightarrow$D, and real-world manipulation, LIRA improves the principal aggregate metrics over the VLA-Adapter baseline under the same 0.5B-parameter configuration. In zero-shot transfer to LIBERO-Plus, LIRA increases average success from 59.1% to 78.0%, an 18.9-point gain indicating improved robustness under controlled distribution shifts.",
  "published": "2026-08-06",
  "updated": "2026-08-06",
  "year": "2026",
  "authors": [
   "Zhewei Zhang",
   "Puyue Wang",
   "Guanren Qiao",
   "Yijie Weng",
   "Jiawei Hu",
   "Guo Li",
   "Lujia Wang",
   "Junyan Wang",
   "Tao Gu",
   "Hongliang Lu",
   "Guiliang Liu",
   "Hong Jia",
   "Xinhu Zheng"
  ],
  "author_count": 13,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "LIRA is introduced, a local cross-layer action-conditioning mechanism that formulates VLM-to-action conditioning as depth-aware information routing and improves the principal aggregate metrics over the VLA-Adapter baseline under the same 0.5B-parameter configuration.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhewei Zhang",
    "id": "2456839231",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Pu Wang",
    "id": "2338038877",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Guanren Qiao",
    "id": "2273970471",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Yijie Weng",
    "id": "2307780905",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Jiawei Hu",
    "id": "2329300012",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Guo Li",
    "id": "2184570319",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Lujia Wang",
    "id": "2456632597",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Junyan Wang",
    "id": "2408723623",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Tao Gu",
    "id": "2363596361",
    "h_index": 1,
    "papers": 9
   },
   {
    "name": "Hongliang Lu",
    "id": "2299307996",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Guiliang Liu",
    "id": "2319185234",
    "h_index": 3,
    "papers": 18
   },
   {
    "name": "Hong Jia",
    "id": "2316274006",
    "h_index": 5,
    "papers": 30
   },
   {
    "name": "Xinhu Zheng",
    "id": "2309671026",
    "h_index": 5,
    "papers": 14
   }
  ],
  "comment": "9 pages, 4 figures. Code and model checkpoints will be released upon acceptance of the paper",
  "topics": [
   "vla",
   "video-generation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.07596v1",
  "pdf_url": "https://arxiv.org/pdf/2608.07596v1",
  "html_url": "https://arxiv.org/html/2608.07596v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.06650",
  "slug": "soromox-fast-differentiable-and-parallelizable-soft-robot-models",
  "title": "SoRoMoX: Fast, Differentiable, and Parallelizable Soft Robot Models",
  "abstract": "Reduced-order models based on Cosserat-rod theory are now well established, and modeling theory is no longer the primary bottleneck in soft-robot control. Their implementations, however, do not support the differentiable, GPU-parallel, and control-oriented workflows that underpin advanced rigid-robotics applications. Here, we fill this gap with SoRoMoX (Soft Robot Models in JAX), a fully numerical, JIT-compilable Python/JAX framework. SoRoMoX implements articulated, Piecewise Constant Strain, and Variable Strain models through a unified, control-ready interface that provides inertia matrices, gravitational and elastic forces, Jacobians, and their derivatives. To our knowledge, it is the first rod/strain-based soft-robot modeling framework that runs directly on GPUs and is end-to-end differentiable with respect to states, inputs, and parameters. Sequential CPU rollouts are up to 18.1x faster than state-of-the-art alternatives, while GPU-parallel rollouts increase throughput by up to 234.6x. This performance enables workflows that were previously impractical or impossible: static-equilibrium system identification with 66% lower marker RMSE; residual-force learning with a further 64% reduction; computed-torque tracking with RMSE reduced by a factor of approximately 500 relative to model-free PD; control-gain optimization with up to 62% lower loss than untuned gains; safety-constrained control using high-order control barrier functions to keep the peak contact force within a prescribed 5 N bound, compared with 33.5 N without the safety constraint; and reinforcement-learning policy training up to 7x faster than a CPU PyElastica discrete-rod baseline through massively parallel rollouts.",
  "published": "2026-08-06",
  "updated": "2026-08-06",
  "year": "2026",
  "authors": [
   "Maximilian St\u00f6lzle",
   "Solange Gribonval",
   "Daniel Feliu-Talegon",
   "Vito Daniele Perfetta",
   "Michele Martini",
   "Chuhan Zhang",
   "Kiwan Wong",
   "Mohammed Tarnini",
   "Anup Teejo Mathew",
   "Federico Renda",
   "Daniela Rus",
   "Cosimo Della Santina"
  ],
  "author_count": 12,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "SoRoMoX (Soft Robot Models in JAX), a fully numerical, JIT-compilable Python/JAX framework, is presented, the first rod/strain-based soft-robot modeling framework that runs directly on GPUs and is end-to-end differentiable with respect to states, inputs, and parameters.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Maximilian St\u00f6lzle",
    "id": "2127775868",
    "h_index": 8,
    "papers": 18
   },
   {
    "name": "Solange Gribonval",
    "id": "2456565085",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "D. Feli\u00fa-Talegon",
    "id": "1413560935",
    "h_index": 13,
    "papers": 43
   },
   {
    "name": "Vito Daniele Perfetta",
    "id": "2400589514",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Michele Martini",
    "id": "2454243818",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Chuhan Zhang",
    "id": "2273553654",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Kiwan Wong",
    "id": "2275135074",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Mohammed Tarnini",
    "id": "2333362422",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "A. Mathew",
    "id": "51251266",
    "h_index": 12,
    "papers": 42
   },
   {
    "name": "Federico Renda",
    "id": "2363344946",
    "h_index": 3,
    "papers": 15
   },
   {
    "name": "Daniela Rus",
    "id": "2261287511",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "C. D. Santina",
    "id": "35178897",
    "h_index": 25,
    "papers": 177
   }
  ],
  "comment": "",
  "topics": [
   "rl-control",
   "hardware-codesign",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.06650v1",
  "pdf_url": "https://arxiv.org/pdf/2608.06650v1",
  "html_url": "https://arxiv.org/html/2608.06650v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.06481",
  "slug": "lyevo-lyapunov-guided-evolutionary-optimization-for-safe-and-robust-si",
  "title": "LyEvO: Lyapunov-Guided Evolutionary Optimization for Safe and Robust Sim-to-Real Policy Learning",
  "abstract": "Training controllers that are safe and robust in simulation, and systematically assessing their readiness for real-world deployment, remain key challenges in sim-to-real transfer. To address this, we propose LyEvO, a physics-grounded framework that combines constrained Evolutionary Optimization and Statistical Model Checking (SMC)-based verification with Lyapunov-based stability analysis. Leveraging prior knowledge of the system dynamics, LyEvO uses Lyapunov analysis to compute an initial candidate stability region. An iterative loop then uses operational scenarios drawn from this region to jointly optimize and statistically verify a policy, and subsequently expands the region's boundaries based on the verification outcome. This integrated procedure provides a practical criterion for assessing deployment readiness. We evaluate LyEvO on Cartpole and 3D Quadrotor benchmarks through extensive simulations and targeted real-world experiments, demonstrating safe and robust sim-to-real transfer.",
  "published": "2026-08-06",
  "updated": "2026-08-06",
  "year": "2026",
  "authors": [
   "Riccardo Curcio",
   "Hongpeng Cao",
   "Marco Caccamo"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG",
   "cs.NE"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "LyEvO, a physics-grounded framework that combines constrained Evolutionary Optimization and Statistical Model Checking (SMC)-based verification with Lyapunov-based stability analysis, is proposed, demonstrating safe and robust sim-to-real transfer.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Riccardo Curcio",
    "id": "2449494171",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Hongpeng Cao",
    "id": "2280065247",
    "h_index": 4,
    "papers": 14
   },
   {
    "name": "Marco Caccamo",
    "id": "2237987292",
    "h_index": 4,
    "papers": 13
   }
  ],
  "comment": "",
  "topics": [
   "sim2real",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.06481v1",
  "pdf_url": "https://arxiv.org/pdf/2608.06481v1",
  "html_url": "https://arxiv.org/html/2608.06481v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.06434",
  "slug": "fast-and-accurate-an-adaptive-vla-inference-framework-through-environm",
  "title": "Fast and Accurate: An Adaptive VLA Inference Framework through Environment-aware Model Selection",
  "abstract": "Embodied intelligence demands both long-horizon reasoning and real-time closed-loop responsiveness. Recent dual-system Vision-Language-Action (VLA) architectures combine fast reactive control with slow deliberative reasoning to balance inference speed and task success rate. However, existing dual-process VLAs tightly couple the fast module to intermediate representations of the slow module, necessitating end-to-end joint training and limiting modularity, extensibility and flexible system switching. In this paper, we propose Environment-aware Model Selection (EMS), an adaptive VLA inference framework that switches between two fully decoupled systems of different scales through environment-aware model selection. The large-scale deliberative system provides globally consistent trajectory planning to ensure task success, while a lightweight reactive system enables high-frequency closed-loop control. A reinforcement-learning-based switching policy dynamically selects which system to invoke based on real-time feedback, enabling sparse use of the slow system and thereby balancing pretrained knowledge utilisation with runtime efficiency. Our design offers three key advantages over prior hierarchical VLA frameworks: (1) a fully decoupled and modular dual-system architecture that supports plug-and-play model replacement; (2) an adaptive, environment-aware switching strategy; (3) high-frequency inference for responsive closed-loop control. We extensively evaluate EMS in both simulation and real-world environments. On the LIBERO benchmark, EMS achieves success rates comparable to the large-scale baseline while increasing the effective action frequency to 93.4 Hz. The framework further demonstrates strong extensibility in real-world dual-arm manipulation tasks, where it accelerates task completion while maintaining robust performance.",
  "published": "2026-08-06",
  "updated": "2026-08-06",
  "year": "2026",
  "authors": [
   "Yuewei Sun",
   "Lang Qin",
   "Zechuan Tian",
   "Jingwen Li",
   "Guiqin Wang",
   "Shengzeng Huo",
   "Wenxin Ren",
   "Tao Fang",
   "Xiaochen Zhang",
   "Guanqing Deng",
   "Xiang Wang",
   "Xiaowen Dong",
   "Qinghai Guo",
   "Yuxin Ma"
  ],
  "author_count": 14,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Environment-aware Model Selection (EMS), an adaptive VLA inference framework that switches between two fully decoupled systems of different scales through environment-aware model selection, and offers three key advantages over prior hierarchical VLA frameworks: a fully decoupled and modular dual-system architecture that supports plug-and-play model replacement; an adaptive, environment-aware switching strategy; and high-frequency inference for responsive closed-loop control.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yuewei Sun",
    "id": "2456611252",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Lang Qin",
    "id": "2292205839",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Zechuan Tian",
    "id": "2456580561",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Jingwen Li",
    "id": "2456983476",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Guiqin Wang",
    "id": "2164973680",
    "h_index": 4,
    "papers": 16
   },
   {
    "name": "Shengzeng Huo",
    "id": "1905639488",
    "h_index": 9,
    "papers": 24
   },
   {
    "name": "Wenxin Ren",
    "id": "2456566415",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Tao Fang",
    "id": "2446399088",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Xiaochen Zhang",
    "id": "2456365013",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Guanqing Deng",
    "id": "2456566161",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Xiang Wang",
    "id": "2404036586",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Xiaowen Dong",
    "id": "2456659608",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Qinghai Guo",
    "id": "2289677552",
    "h_index": 2,
    "papers": 13
   },
   {
    "name": "Yuxin Ma",
    "id": "2359780774",
    "h_index": 0,
    "papers": 4
   }
  ],
  "comment": "",
  "topics": [
   "vla"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.06434v1",
  "pdf_url": "https://arxiv.org/pdf/2608.06434v1",
  "html_url": "https://arxiv.org/html/2608.06434v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.06375",
  "slug": "0-a-latent-predictive-world-action-model-for-concurrent-humanoid-loco",
  "title": "$\u03c9$-0: A Latent Predictive World Action Model for Concurrent Humanoid Loco-Manipulation",
  "abstract": "Humanoid household tasks often require concurrent loco-manipulation, where the robot must move, adjust posture, maintain balance, and manipulate objects as a single coordinated behavior. Yet existing humanoid policies typically decompose locomotion and manipulation, while recent world-action models remain either arm-centric or video-centered. We present $\u03c9$-0, a latent predictive whole-body world-action model for real-world humanoid concurrent loco-manipulation. Given a language instruction, current visual observation, and robot proprioceptive state, $\u03c9$-0 directly predicts controller-compatible whole-body action latents for real-robot execution. Rather than reconstructing future videos, $\u03c9$-0 learns compact future observation embeddings as a lightweight predictive objective, coupling latent visual foresight with diffusion-based whole-body action generation. The model supports egocentric RGB, exocentric RGB, and exocentric depth inputs, and leverages controller-based simulation replay to ground human/public visual-motion priors into robot-executable action latents. We further collect $\u03c9$-HOME, a 40+ hour real-world household humanoid dataset with synchronized multi-view observations, whole-body SMPL motions, robot states, and action latents. Real-world experiments on 11 household tasks demonstrate that a single $\u03c9$-0 model can produce smooth manipulate-while-moving behaviors and consistently outperform representative imitation learning, VLA, humanoid, and WAM baselines.",
  "published": "2026-08-06",
  "updated": "2026-08-09",
  "year": "2026",
  "authors": [
   "Zhe Li",
   "Zhenzhe Zhang",
   "Yangyang Wei",
   "Wenjie Zhang",
   "Xichen Yuan",
   "Peiyuan Zhi",
   "Gen Li",
   "Xinying Guo",
   "Fengjie Gao",
   "Jianfei Yang",
   "Shanghang Zhang"
  ],
  "author_count": 11,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Real-world experiments demonstrate that a single $\\omega$-0 model can produce smooth manipulate-while-moving behaviors and consistently outperform representative imitation learning, VLA, humanoid, and WAM baselines.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhe Li",
    "id": "2385507969",
    "h_index": 7,
    "papers": 15
   },
   {
    "name": "Zhenzhen Zhang",
    "id": "2397804431",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Yangyang Wei",
    "id": "2386810501",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Wenjie Zhang",
    "id": "2455806423",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Xichen Yuan",
    "id": "2239070261",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Peiyuan Zhi",
    "id": "2296711192",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Gen Li",
    "id": "2409917984",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Xinying Guo",
    "id": "2183512577",
    "h_index": 7,
    "papers": 14
   },
   {
    "name": "Feng Gao",
    "id": "2449161703",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Jianfei Yang",
    "id": "2404007795",
    "h_index": 2,
    "papers": 26
   },
   {
    "name": "Shanghang Zhang",
    "id": "2346116279",
    "h_index": 16,
    "papers": 51
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "egocentric-data",
   "imitation-diffusion"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.06375v2",
  "pdf_url": "https://arxiv.org/pdf/2608.06375v2",
  "html_url": "https://arxiv.org/html/2608.06375v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.06374",
  "slug": "dypes-vla-learning-shared-dynamics-priors-and-embodiment-specific-cont",
  "title": "DyPES-VLA: Learning Shared Dynamics Priors and Embodiment-Specific Control for Cross-Embodiment Manipulation",
  "abstract": "Vision-Language-Action (VLA) models have become a powerful paradigm for robot manipulation, but training a single generalist policy for heterogeneous robot embodiments remains an open problem. Existing methods have two main limitations. First, they underuse dynamics priors shared across diverse visual and interaction data, limiting cross-embodiment transfer. Second, they require extensive manual preprocessing to convert embodiment-specific actions into a common format. To overcome these limitations, we propose DyPES-VLA, a cross-embodiment VLA that learns shared Dynamics Priors and Embodiment-Specific control. First, we learn shared dynamics priors by training the vision-language model (VLM) with a future-prediction objective on cross-embodiment data, driving the shared query representation to capture object motion, contact, and interaction-induced scene changes. Second, an embodiment-specific Mixture-of-Experts (MoE) action head translates these shared dynamics priors into executable controls directly in each embodiment's native action space, without manually pre-aligning heterogeneous actions into a common format. This head shares attention layers to capture common temporal action structures, while its embodiment-specific feed-forward experts resolve the unique kinematic constraints and control semantics of distinct embodiments. As a generalist policy, our \\ourmethod achieves state-of-the-art performance across simulation and real-world evaluations, reaching 98.0% success on LIBERO, 59.25% on RoboCasa-GR1, and 89.02% on RoboTwin~2.0.",
  "published": "2026-08-06",
  "updated": "2026-08-06",
  "year": "2026",
  "authors": [
   "Junfeng Li",
   "Junjie He",
   "Zhide Zhong",
   "Yangyang Zheng",
   "Pingyue Sheng",
   "Jiayu Dong",
   "Ruixin Li",
   "Haodong Yan",
   "Jiaguan Zhu",
   "Tianran Zhang",
   "Runze Yu",
   "Wen Chen",
   "Liuqing Yang",
   "Yuxiang Gao",
   "Haoang Li"
  ],
  "author_count": 15,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "DyPES-VLA is proposed, a cross-embodiment VLA that learns shared Dynamics Priors and Embodiment-Specific control, and achieves state-of-the-art performance across simulation and real-world evaluations.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Junfeng Li",
    "id": "2376547586",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Junjie He",
    "id": "2316016558",
    "h_index": 3,
    "papers": 16
   },
   {
    "name": "Zhide Zhong",
    "id": "2349315841",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Yangyang Zheng",
    "id": "2454984594",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Pingyue Sheng",
    "id": "2327340052",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Jiayue Dong",
    "id": "2449163267",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Ruixin Li",
    "id": "2455424656",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Haodong Yan",
    "id": "2321603038",
    "h_index": 5,
    "papers": 16
   },
   {
    "name": "Jiaguang Zhu",
    "id": "1575707808",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Tianran Zhang",
    "id": "2308051940",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Runze Yu",
    "id": "2455712668",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Wen Chen",
    "id": "2444700942",
    "h_index": 0,
    "papers": 5
   },
   {
    "name": "Liuqing Yang",
    "id": "2311864071",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Yuxiang Gao",
    "id": "2449168885",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Haoang Li",
    "id": "2384363611",
    "h_index": 9,
    "papers": 38
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "foundation-pretraining",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.06374v1",
  "pdf_url": "https://arxiv.org/pdf/2608.06374v1",
  "html_url": "https://arxiv.org/html/2608.06374v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.06332",
  "slug": "geniworld-a-generalizable-interactive-world-model-for-robotic-manipula",
  "title": "GeniWorld: A Generalizable Interactive World Model for Robotic Manipulation via Visual Actions",
  "abstract": "Generalist robot policies exhibit strong capabilities, but their robustness in complex and unseen environments remains limited. Scaling robot learning and evaluation in diverse real-world environments remains costly and challenging. Action-conditioned world models offer a promising alternative, but they often suffer from limited action controllability and poor generalization to out-of-distribution (OOD) scenarios. To this end, we present GeniWorld, an interactive world model for robots that generalizes robustly across unseen scenarios. Building on pretrained video generative models, we use URDF-based rendering to transform numerical actions into visual action representations, enabling spatially grounded action control. By explicitly decoupling embodiment kinematics from environmental dynamics, our model mitigates scene overfitting and facilitates modeling of robot-environment interactions. To achieve closed-loop control, we construct an autoregressive video prediction model integrated with high-frequency robot kinematic control, enabling interaction with both robot policies and human teleoperators. In our experiments, even when trained solely on limited fixed-scene data, our model achieves superior in-domain performance and robust zero-shot generalization to highly randomized, unseen environments. For downstream applications, GeniWorld serves as a scalable policy evaluator that remains reliable under environmental perturbations. Furthermore, even with limited real-world demonstrations, GeniWorld generates diverse manipulation trajectories within the world model, improving downstream policy performance and robustness in complex environments.",
  "published": "2026-08-06",
  "updated": "2026-08-06",
  "year": "2026",
  "authors": [
   "Chenghao Gu",
   "Hanyang Yu",
   "Jingbo Zhang",
   "Haitao Lin",
   "Wenyao Zhang",
   "Jinghe Wang",
   "Hanglei Jin",
   "Shuzhao Xie",
   "Jingyan Jiang",
   "Zhi Wang"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "GeniWorld is presented, an interactive world model for robots that generalizes robustly across unseen scenarios by explicitly decoupling embodiment kinematics from environmental dynamics, and generates diverse manipulation trajectories within the world model, improving downstream policy performance and robustness in complex environments.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Chenghao Gu",
    "id": "2305618507",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Hanyang Yu",
    "id": "2324225754",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Jingbo Zhang",
    "id": "2453816376",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Haitao Lin",
    "id": "2363848505",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Wenyao Zhang",
    "id": "2282545418",
    "h_index": 10,
    "papers": 26
   },
   {
    "name": "Jinghe Wang",
    "id": "2332300881",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Hanglei Jin",
    "id": "2455793376",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "S. Xie",
    "id": "148165054",
    "h_index": 7,
    "papers": 30
   },
   {
    "name": "Jingyan Jiang",
    "id": "2296747178",
    "h_index": 6,
    "papers": 26
   },
   {
    "name": "Zhi Wang",
    "id": "2305646996",
    "h_index": 5,
    "papers": 16
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "video-generation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.06332v1",
  "pdf_url": "https://arxiv.org/pdf/2608.06332v1",
  "html_url": "https://arxiv.org/html/2608.06332v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.06221",
  "slug": "robot-learning-from-human-demonstrations-handwritten-alphabet-trajecto",
  "title": "Robot Learning from Human Demonstrations: Handwritten Alphabet Trajectories and Human-Likeness Evaluation",
  "abstract": "Learning from demonstration (LfD) provides a developmental framework through which robots can develop motor skills by observing and imitating human dynamics, reducing reliance on explicit programming to teach a skill to a robot. The resulting human-like robot motion is recognised as a key factor in building trust and enabling natural collaboration in human-robot interaction. This paper presents a framework for learning human-like robot motion from demonstration, including data collection, probabilistic trajectory learning, and perceptual user evaluation. A dataset of 3,142 handwriting demonstrations was collected from 22 participants across all 52 Latin alphabet character-case combinations via a touchscreen teleoperation interface, capturing planar position, contact force, and timing. Building on the widely used Gaussian Mixture Model and Gaussian Mixture Regression approach for learning from demonstration, the framework is extended in this work by incorporating force and normalised time dimensions to enable richer representation of human dynamics, and adapting it to handle non-continuous, multi-segment trajectories, enabling generalisation across demonstrations. A user study with 21 participants evaluated the perceived human-likeness of the generated trajectories using a continuous scale anchored between robotic and human-like motion, normalised to 0-100 where 50 represents the neutral midpoint. The generated trajectories achieved an overall human-likeness score of 71.50 (SD=22.56), indicating that the majority of trajectories were perceived as more human-like. Participants identified geometric positioning and trajectory sequence as the most influential perceptual factors, and reported positive attitudes toward human-like robot behaviour. The datasets are released as open-source, providing a reproducible benchmark for developing and evaluating human-like robot motion methods.",
  "published": "2026-08-06",
  "updated": "2026-08-06",
  "year": "2026",
  "authors": [
   "Alperen Kenan",
   "Paul Bremner",
   "Manuel Giuliani"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.HC",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "A framework for learning human-like robot motion from demonstration, including data collection, probabilistic trajectory learning, and perceptual user evaluation is presented, extending the widely used Gaussian Mixture Model and Gaussian Mixture Regression approach for learning from demonstration.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Alperen Kenan",
    "id": "2380387326",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Paul A. Bremner",
    "id": "2375498894",
    "h_index": 1,
    "papers": 10
   },
   {
    "name": "Manuel Giuliani",
    "id": "2268226527",
    "h_index": 3,
    "papers": 22
   }
  ],
  "comment": "9 pages, 7 figures, 4 tables, accepted for presentation at the IEEE International Conference on Development and Learning (ICDL) 2026, Kyoto, Japan, 15-18 September 2026",
  "topics": [
   "egocentric-data",
   "data-teleop",
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.06221v1",
  "pdf_url": "https://arxiv.org/pdf/2608.06221v1",
  "html_url": "https://arxiv.org/html/2608.06221v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2608.06219",
  "slug": "design-and-evaluation-of-a-touchscreen-based-teleoperation-interface-f",
  "title": "Design and Evaluation of a Touchscreen-Based Teleoperation Interface for Robotic Manipulators",
  "abstract": "Intuitive teleoperation interfaces are crucial for the safe and effective operation of robotic manipulators in challenging environments. In the nuclear industry, surface contact tasks such as swab sampling require precise path and force tracking, obstacle avoidance, and sustained operator attention, which conventional joystick interfaces struggle to support effectively. This study designs and evaluates a novel touchscreen teleoperation interface that maps continuous finger movements directly to robotic manipulator motions, provides finer velocity control, and integrates control with visualization, enabling more natural, precise, and intuitive surface interaction than conventional controllers. A comparative user study with 20 participants evaluated task performance and workload using the proposed touchscreen, a conventional joystick, and a single-click autonomous mode. Tasks simulated realistic surface manipulation using a Franka Emika Panda arm, remotely controlled from another country. Kinematic, physiological, and behavioral data were recorded to comprehensively assess task performance, cognitive load, and operator trust across each control condition. Participants completed teleoperation tasks more efficiently and accurately with the touchscreen interface, achieving a 53.5% reduction in completion time (median: 2.50 vs. 5.38 min), higher in-area coverage on the sinusoidal path (90.7% vs. 84.1%), and lower overshoot on both path geometries compared with the joystick. Cognitive load, quantified via NASA-TLX (0-100), decreased from joystick to touchscreen (mean TLX 52 to 43; -9 points, -17.3%) and was lowest under the autonomous one-click mode (31; -21 points vs. joystick, -40.4%; -12 vs. touchscreen, -27.9%). This research presents an easy-to-implement touchscreen interface that improves performance in teleoperated surface tasks while reducing cognitive load.",
  "published": "2026-08-06",
  "updated": "2026-08-06",
  "year": "2026",
  "authors": [
   "Juan Jos\u00e9 Garc\u00eda C\u00e1rdenas",
   "Alperen Kenan",
   "Hamidreza Raei",
   "Paul Bremner",
   "Manuel Giuliani",
   "Arash Ajoudani",
   "Adriana Tapus"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "cs.HC"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "An easy-to-implement touchscreen interface is presented that improves performance in teleoperated surface tasks while reducing cognitive load and provides finer velocity control, and integrates control with visualization.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Juan Jose Garcia Cardenas",
    "id": "2329741220",
    "h_index": 3,
    "papers": 13
   },
   {
    "name": "Alperen Kenan",
    "id": "2380387326",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Hamidreza Raei",
    "id": "2148874169",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Paul A. Bremner",
    "id": "2375498894",
    "h_index": 1,
    "papers": 10
   },
   {
    "name": "Manuel Giuliani",
    "id": "2268226527",
    "h_index": 3,
    "papers": 22
   },
   {
    "name": "Arash Ajoudani",
    "id": "2372791",
    "h_index": 40,
    "papers": 244
   },
   {
    "name": "Adriana Tapus",
    "id": "2266495390",
    "h_index": 4,
    "papers": 40
   }
  ],
  "comment": "9 pages, 7 figures, accepted for presentation at the IEEE International Conference on Robot and Human Interactive Communication (RO-MAN 2026), Kitakyushu, Japan, 24-28 August 2026",
  "topics": [
   "navigation",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.06219v1",
  "pdf_url": "https://arxiv.org/pdf/2608.06219v1",
  "html_url": "https://arxiv.org/html/2608.06219v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2608.06210",
  "slug": "vidp-variable-impedance-diffusion-policy-for-compliant-robot-manipulat",
  "title": "VIDP: Variable Impedance Diffusion Policy for Compliant Robot Manipulation from Diverse Demonstrations",
  "abstract": "Contact-rich manipulation requires precise tracking and mechanical compliance, where variable impedance control can improve robustness in task success, whereas static compliance cannot adapt to varying contact constraints. Variable impedance skills can be learned from demonstrations, avoiding complex modeling, but compliance is a hidden variable in force-agnostic kinematic data. While existing methods infer compliance from trajectory variations, these variations may reflect geometric adaptation and not intentional compliance when subject to changing spatial layouts. Therefore, this letter introduces Variable Impedance Diffusion Policy (VIDP), an imitation learning-based variable impedance control framework leveraging a Task-Parameterized Directionality-Aware Mixture Model (TP-DAMM) to extract physically consistent trajectory distributions from diverse demonstrations. By mapping distributions to stiffness profiles, VIDP jointly predicts pose actions and task compliance without force sensors. Real-world experiments show that VIDP significantly outperforms fixed-impedance baselines in task success rate while reducing interaction forces with respect to high stiffness controllers and tracking errors with respect to low stiffness baselines.",
  "published": "2026-08-06",
  "updated": "2026-08-06",
  "year": "2026",
  "authors": [
   "Hisham Khalil",
   "Neil Fernandes",
   "Thomas M. Kwok",
   "Hsiu-Chin Lin",
   "Yue Hu"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Variable Impedance Diffusion Policy (VIDP), an imitation learning-based variable impedance control framework leveraging a Task-Parameterized Directionality-Aware Mixture Model (TP-DAMM) to extract physically consistent trajectory distributions from diverse demonstrations is introduced.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hisham Khalil",
    "id": "2124488063",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Neil Fernandes",
    "id": "2312453925",
    "h_index": 1,
    "papers": 10
   },
   {
    "name": "Thomas M. Kwok",
    "id": "2351054889",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Hsiu-Chin Lin",
    "id": "38637273",
    "h_index": 19,
    "papers": 98
   },
   {
    "name": "Yue Hu",
    "id": "2269466571",
    "h_index": 2,
    "papers": 11
   }
  ],
  "comment": "8 pages, 5 figures",
  "topics": [
   "tactile",
   "imitation-diffusion",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.06210v1",
  "pdf_url": "https://arxiv.org/pdf/2608.06210v1",
  "html_url": "https://arxiv.org/html/2608.06210v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.06208",
  "slug": "ergosurf-ergodic-control-for-the-coverage-of-unknown-surfaces",
  "title": "ErgoSurf: Ergodic Control for the Coverage of Unknown Surfaces",
  "abstract": "Contact-centric tasks on surfaces, ranging from inspection and cleaning to sanding and polishing, require robots to systematically cover the surface while maintaining stable contact. Ergodic control generates trajectories that spend time at a location proportional to a desired, task-specific spatial distribution, enabling efficient information gathering and coverage. However, traditional ergodic control methods rely on prior knowledge of surface geometry or require a vision sensory input to scan the geometry beforehand, limiting their applicability in real-world scenarios with unknown or dynamic environments. This paper introduces a novel online ergodic control framework that achieves systematic surface coverage while simultaneously reconstructing unknown surface geometry. We employ a Gaussian Process Implicit Surface (GPIS) model that learns global surface geometry from intrinsic tactile sensing during execution. For efficient online planning, we approximate the surface locally using point clouds sampled from tangent planes at observed contact points and iteratively fit them to the Gaussian Process. This approximation simultaneously serves as the sampling domain for both the target and the coverage distributions. We employ a heat-diffusion analogy to compute potential fields that guide ergodic exploration, translating spatial coverage objectives into smooth robot trajectories. We demonstrate our framework through simulation and real-robot experiments, validating simultaneous ergodic coverage and online surface geometry learning with reconstruction errors approaching the ground truth.",
  "published": "2026-08-06",
  "updated": "2026-08-06",
  "year": "2026",
  "authors": [
   "Stefan Schneyer",
   "Timo Bachmann",
   "Maged Iskandar",
   "Korbinian Nottensteiner",
   "Alin Albu-Sch\u00e4ffer",
   "Freek Stulp",
   "Jo\u00e3o Silv\u00e9rio"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A novel online ergodic control framework that achieves systematic surface coverage while simultaneously reconstructing unknown surface geometry from intrinsic tactile sensing during execution is introduced, employing a Gaussian Process Implicit Surface (GPIS) model that learns global surface geometry from intrinsic tactile sensing during execution.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Stefan Schneyer",
    "id": "115680632",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Timo Bachmann",
    "id": "2133413205",
    "h_index": 2,
    "papers": 11
   },
   {
    "name": "Maged Iskandar",
    "id": "51300406",
    "h_index": 13,
    "papers": 30
   },
   {
    "name": "Korbinian Nottensteiner",
    "id": "3414652",
    "h_index": 12,
    "papers": 26
   },
   {
    "name": "Alin Albu-Sch\u00e4ffer",
    "id": "2398204373",
    "h_index": 2,
    "papers": 17
   },
   {
    "name": "F. Stulp",
    "id": "50707365",
    "h_index": 36,
    "papers": 179
   },
   {
    "name": "Jo\u00e3o Silv\u00e9rio",
    "id": "2430988257",
    "h_index": 1,
    "papers": 3
   }
  ],
  "comment": "",
  "topics": [
   "tactile",
   "spatial-3d",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.06208v1",
  "pdf_url": "https://arxiv.org/pdf/2608.06208v1",
  "html_url": "https://arxiv.org/html/2608.06208v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.06088",
  "slug": "icfuzz-fuzzing-isaac-sim-with-semantic-stage-guidance-and-multi-level",
  "title": "IcFuzz: Fuzzing Isaac Sim with Semantic Stage Guidance and Multi-level Mutation",
  "abstract": "Robotics simulators serve as a foundational infrastructure for embodied AI, facilitating safe and scalable robotic system development. NVIDIA Isaac Sim has emerged as one of the most popular simulators, distinguished by its GPU-accelerated physics engine and photorealistic rendering, which enable high-fidelity modeling of complex environments. However, its inherent complexity inevitably introduces software bugs that can compromise simulation reliability. Existing fuzzing approaches struggle to test Isaac Sim effectively due to challenges of context-aware object semantics, hierarchical simulation control, and a vast simulation state space. In this paper, we propose IcFuzz, the first fuzzing approach for Isaac Sim. IcFuzz first performs an LLM-based semantic stage segmentation, decomposing simulation programs into structured stages that capture context-aware object semantics. Guided by this information, IcFuzz designs multi-level mutation operators to systematically exercise the simulator across hierarchical granularities. To efficiently navigate the vast simulation state space, IcFuzz employs a multi-armed bandit algorithm to adaptively schedule mutation operators. Experimental results show that IcFuzz outperforms the baselines in terms of both code coverage and bug detection. Specifically, IcFuzz achieves approximately 190\\%--205\\% of the code coverage of the baselines and detects an average of 3.7 unique crashes over three rounds of 12-hour tests, while no crashes are detected by the baselines. Moreover, IcFuzz has uncovered 11 bugs over approximately four months, 9 of which have been confirmed or fixed by the developers.",
  "published": "2026-08-06",
  "updated": "2026-08-06",
  "year": "2026",
  "authors": [
   "Zhixiang Chen",
   "Zhuangbin Chen",
   "Ruoxi Jia",
   "Zeqin Liao",
   "Wei Li",
   "Jinyang Liu",
   "Zibin Zheng"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "cs.SE"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "IcFuzz is proposed, the first fuzzing approach for Isaac Sim, which performs an LLM-based semantic stage segmentation, decomposing simulation programs into structured stages that capture context-aware object semantics and efficiently navigate the vast simulation state space.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhixiang Chen",
    "id": "2284267712",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Zhuangbin Chen",
    "id": "2344859146",
    "h_index": 4,
    "papers": 14
   },
   {
    "name": "Ruoxi Jia",
    "id": "2291965197",
    "h_index": 9,
    "papers": 14
   },
   {
    "name": "Zeqin Liao",
    "id": "2176395754",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Wei Li",
    "id": "2372188008",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Jinyang Liu",
    "id": "2108510525",
    "h_index": 18,
    "papers": 35
   },
   {
    "name": "Zibin Zheng",
    "id": "2372157735",
    "h_index": 2,
    "papers": 12
   }
  ],
  "comment": "Accepted at the 41st IEEE/ACM International Conference on Automated Software Engineering (ASE 2026)",
  "topics": [
   "sim2real"
  ],
  "orgs": [
   "NVIDIA"
  ],
  "abs_url": "https://arxiv.org/abs/2608.06088v1",
  "pdf_url": "https://arxiv.org/pdf/2608.06088v1",
  "html_url": "https://arxiv.org/html/2608.06088v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2608.06008",
  "slug": "adaptive-wam-quality-guided-early-exit-planning-from-intermediate-vide",
  "title": "Adaptive-WAM: Quality-Guided Early-Exit Planning from Intermediate Video-Diffusion Features",
  "abstract": "Large video diffusion models provide rich spatiotemporal priors for autonomous driving, but existing world-action models often inherit the cost of iterative future-video generation even though deployment only requires an ego trajectory. We ask a more basic question: how much of a video diffusion model must be executed to make a reliable driving decision? Through a controlled study of video denoising timesteps and Diffusion Transformer (DiT) depth, we find that planning performance is largely insensitive to the tested video-noise levels, whereas strong trajectories can already be decoded from intermediate layers. Based on this observation, we introduce Adaptive-WAM, a quality-aware multi-exit planner built on a Wan2.2-5B backbone. Trajectory diffusion heads are attached to selected DiT blocks, and a lightweight trajectory-quality scorer terminates inference once the best trajectory decoded so far satisfies a quality threshold; otherwise, computation continues from the cached hidden state to a deeper exit. The deployed planner therefore avoids the iterative classifier-free denoising loop and VAE decoding required for future-video synthesis, while dynamically allocating backbone depth according to trajectory quality. On NAVSIM, the adaptive single-trajectory planner achieves 90.8 PDMS; a separate fixed-exit variant reaches 92.6 PDMS with 64 proposals. It further obtains 89.9 EPDMS on NAVSIM v2, yielding the best reported results among the compared front-view video world-model planners. Without target-domain fine-tuning, Adaptive-WAM transfers to nuScenes with 0.88 m average L2 error and a 0.08\\% collision rate. On an A100, adaptive routing improves PDMS from 90.62 to 90.79 while averaging 170 ms end-to-end planning latency, approximately 10\\% below the 190 ms fixed block-15 planner and 47\\% below the 320 ms fixed full-depth planner. Code will be released.",
  "published": "2026-08-06",
  "updated": "2026-08-06",
  "year": "2026",
  "authors": [
   "Sining Ang",
   "Yuguang Yang",
   "Yan Wang"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "Adaptive-WAM is introduced, a quality-aware multi-exit planner built on a Wan2.2-5B backbone that avoids the iterative classifier-free denoising loop and VAE decoding required for future-video synthesis, while dynamically allocating backbone depth according to trajectory quality.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Sining Ang",
    "id": "2410030760",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Yuguang Yang",
    "id": "2239394661",
    "h_index": 11,
    "papers": 60
   },
   {
    "name": "Yan Wang",
    "id": "2386705642",
    "h_index": 3,
    "papers": 17
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "navigation",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.06008v1",
  "pdf_url": "https://arxiv.org/pdf/2608.06008v1",
  "html_url": "https://arxiv.org/html/2608.06008v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2608.05999",
  "slug": "beyond-flat-policies-hierarchical-post-training-for-embodied-agents-in",
  "title": "Beyond Flat Policies: Hierarchical Post-Training for Embodied Agents in Robotic Manipulation",
  "abstract": "Vision-language-action (VLA) models have demonstrated remarkable capabilities in robotic manipulation by leveraging pretrained vision-language models. However, existing post-training methods predominantly optimize VLA models as flat policies, making it difficult to explicitly model task progression and perform robust long-horizon manipulation. Although hierarchical approaches introduce task decomposition, they mainly rely on supervised learning from offline demonstrations and cannot effectively improve execution through online interaction. To address this limitation, we propose Hierarchical Robotic Control (HiRoC), a hierarchical post-training framework that decouples high-level task planning from low-level action execution. The planner decomposes complex tasks into executable subgoals to provide explicit semantic guidance, while the executor continuously improves subgoal-conditioned action generation through reinforcement learning. To enable effective collaboration between the two modules, we further align the executor with planner-generated subgoals before reinforcement learning, mitigating the distribution misalignment between planning and execution. Extensive experiments across diverse robotic manipulation benchmarks demonstrate that HiRoC consistently outperforms strong baselines. Comprehensive analyses further validate the effectiveness of hierarchical post-training and the contribution of each key component.",
  "published": "2026-08-06",
  "updated": "2026-08-06",
  "year": "2026",
  "authors": [
   "He Kong",
   "Zengjue Chen",
   "Qi Wang",
   "Qianli Xing",
   "Runliang Niu",
   "Peidong Liu",
   "Jiawei Li",
   "Shiqi Wang",
   "Yi Chang"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Hierarchical Robotic Control (HiRoC) is proposed, a hierarchical post-training framework that decouples high-level task planning from low-level action execution and aligns the executor with planner-generated subgoals before reinforcement learning, mitigating the distribution misalignment between planning and execution.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "He Kong",
    "id": "2283935174",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Ze Chen",
    "id": "2381132754",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Qi Wang",
    "id": "2358236936",
    "h_index": 4,
    "papers": 15
   },
   {
    "name": "Qianli Xing",
    "id": "9123083",
    "h_index": 5,
    "papers": 17
   },
   {
    "name": "Runliang Niu",
    "id": "2174434599",
    "h_index": 4,
    "papers": 14
   },
   {
    "name": "Peidong Liu",
    "id": "2268467993",
    "h_index": 3,
    "papers": 15
   },
   {
    "name": "Jiawei Li",
    "id": "2456515595",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Shiqi Wang",
    "id": "2283983135",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Yi Chang",
    "id": "2243466364",
    "h_index": 3,
    "papers": 11
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "rl-control",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.05999v1",
  "pdf_url": "https://arxiv.org/pdf/2608.05999v1",
  "html_url": "https://arxiv.org/html/2608.05999v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.05989",
  "slug": "observation-grounded-self-predictive-reinforcement-learning-for-visual",
  "title": "Observation-Grounded Self-Predictive Reinforcement Learning for Visual Continuous Control",
  "abstract": "Sample-efficient policy learning from pixels is a long-standing challenge in reinforcement learning (RL). Recent dynamics-based representation learning methods have significantly improved the sample efficiency of model-free visual RL by learning dynamics-aware representations through auxiliary prediction performed either in latent space (self-prediction) or observation space (observation prediction). However, state-of-the-art methods from both categories still struggle on challenging visual control tasks when training data is limited. We posit that relying on either predictive objective alone may be insufficient. In contrast, observation prediction grounds learned representations in observation-level dynamics, but does not directly regularize the temporal predictability of latent representations over extended horizons. In this paper, we propose Observation-Grounded Self-Predictive Representations (OG-SPR), a model-free visual RL algorithm for continuous control that learns representations that are both temporally predictive in latent space and grounded in observation-level dynamics. OG-SPR incorporates two core auxiliary objectives: multi-step latent self-prediction and next-observation prediction. We empirically show that directly imposing latent self-prediction on the shared representation may over-constrain it and does not necessarily improve performance. To address this issue, OG-SPR introduces two lightweight adapters for latent self-prediction, allowing the shared representation to benefit from temporally predictive signals without being forced to directly satisfy the self-prediction objective. Experiments on 28 visual control tasks from the DeepMind Control Suite show that OG-SPR improves aggregate performance over state-of-the-art self-predictive and observation-predictive RL methods, with particularly pronounced gains in challenging domains such as dog and humanoid.",
  "published": "2026-08-06",
  "updated": "2026-08-06",
  "year": "2026",
  "authors": [
   "Xinwei Liu",
   "Junyuan Liang",
   "Jianting Zhang",
   "Wuhui Chen"
  ],
  "author_count": 4,
  "categories": [
   "cs.LG",
   "cs.RO"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Observation-Grounded Self-Predictive Representations (OG-SPR), a model-free visual RL algorithm for continuous control that learns representations that are both temporally predictive in latent space and grounded in observation-level dynamics, is proposed.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xinwei Liu",
    "id": "2445899418",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Junyuan Liang",
    "id": "2271719725",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Jianting Zhang",
    "id": "2267780732",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Wuhui Chen",
    "id": "2257370299",
    "h_index": 9,
    "papers": 35
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "rl-control"
  ],
  "orgs": [
   "Google DeepMind"
  ],
  "abs_url": "https://arxiv.org/abs/2608.05989v1",
  "pdf_url": "https://arxiv.org/pdf/2608.05989v1",
  "html_url": "https://arxiv.org/html/2608.05989v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2608.05975",
  "slug": "trace-learned-proprioceptive-odometry-for-legged-robots-under-unreliab",
  "title": "TRACE: Learned Proprioceptive Odometry for Legged Robots under Unreliable Contact Conditions",
  "abstract": "In this paper, we present TRACE (Tokenized Robust Attention for Contact-Aware Estimation), an end-to-end learned proprioceptive odometry estimator for legged robots under unreliable contact conditions. The proposed estimator directly predicts relative displacement, relative rotation, and body-frame velocity from a recent history of onboard inertial and joint measurements. To improve robustness under unreliable contact conditions, we introduce a foot-aware cross-attention module that adaptively weights IMU and leg-wise kinematic tokens without relying on manually defined contact or slip thresholds. The estimator is trained with direct supervision and two physics-inspired auxiliary losses that promote kinematic consistency and reliable use of leg information. To reduce policy-specific overfitting and consequently improve sim-to-real transfer, simulation training incorporates policy randomization, followed by partial real-world fine-tuning of the temporal encoder and prediction head. Experiments across diverse indoor and outdoor terrains demonstrate consistent reductions in position drift compared with classical filtering-based, hybrid, and purely learning-based baselines. Ablation studies further validate the contributions of the proposed training objectives, policy randomization, and real-world fine-tuning, particularly under unreliable contacts and sim-to-real mismatch.",
  "published": "2026-08-06",
  "updated": "2026-08-06",
  "year": "2026",
  "authors": [
   "Taehyeon Kong",
   "Woojin Kim",
   "Jemin Hwangbo"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "An end-to-end learned proprioceptive odometry estimator for legged robots under unreliable contact conditions that directly predicts relative displacement, relative rotation, and body-frame velocity from a recent history of onboard inertial and joint measurements.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Taehyeon Kong",
    "id": "2455710362",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "W. Kim",
    "id": "2453582798",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Jemin Hwangbo",
    "id": "1707297",
    "h_index": 25,
    "papers": 48
   }
  ],
  "comment": "8 pages, 7 figures. Submitted to IEEE Robotics and Automation Letters (RA-L)",
  "topics": [
   "humanoids",
   "sim2real",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.05975v1",
  "pdf_url": "https://arxiv.org/pdf/2608.05975v1",
  "html_url": "https://arxiv.org/html/2608.05975v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.05970",
  "slug": "skillmemo-expert-guided-skill-memory-framework-for-compositional-embod",
  "title": "SkillMemo: Expert-guided Skill Memory Framework for Compositional Embodied Manipulation",
  "abstract": "Embodied visuomotor models, including Diffusion Policy (DP) and Vision-Language-Action (VLA) models, have demonstrated promising performance on robotic manipulation benchmarks. However, their potential remains fundamentally constrained by the scarcity of large-scale embodied trajectory datasets, leading to insufficient compositional generalization in out-of-distribution (OOD) scenarios with limited capability to capture reusable skill structures. To address this limitation, we propose Skill-Based Memory (SkillMemo) framework that implicitly decomposes long-horizon demonstrations into latent atomic skills and integrates skill-level features into a dynamic episodic memory bank for solving compositional tasks. Specifically, we first introduce an expert-guided trajectory segmentation module built upon a Mixture-of-Experts (MoE) architecture, which implicitly partitions trajectories into distinct skill primitives represented by learned gating coefficients. We further design a skill-level episodic memory architecture that stores compact skill representations as retrievable key-value pairs. During inference, the memory bank retrieves the most relevant skill primitives which are subsequently fused with the model's current gating distribution, providing a robust contextual prior to refine action predictions. Extensive experiments on the simulation benchmark and real-world manipulation tasks demonstrate that SkillMemo consistently enhances both DP and VLA backbones, achieving state-of-the-art performance and outperforming $\u03c0_{0.5}$, while exhibiting strong compositional generalization to unseen task configurations.",
  "published": "2026-08-06",
  "updated": "2026-08-06",
  "year": "2026",
  "authors": [
   "Changyuan Wang",
   "Chubin Zhang",
   "Zhenyu Wu",
   "Runhao Li",
   "Angyuan Ma",
   "Ke Chao",
   "Yinan Liang",
   "Xiuwei Xu",
   "Ziwei Wang",
   "Yansong Tang",
   "Jiwen Lu"
  ],
  "author_count": 11,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Skill-Based Memory (SkillMemo) framework is proposed that implicitly decomposes long-horizon demonstrations into latent atomic skills and integrates skill-level features into a dynamic episodic memory bank for solving compositional tasks and consistently enhances both DP and VLA backbones.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Changyuan Wang",
    "id": "2218903807",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Chubin Zhang",
    "id": "2274193361",
    "h_index": 8,
    "papers": 10
   },
   {
    "name": "Zhenyu Wu",
    "id": "2257436035",
    "h_index": 7,
    "papers": 22
   },
   {
    "name": "Runhao Li",
    "id": "2200076428",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Angyuan Ma",
    "id": "2378923451",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Ke Chao",
    "id": "2449488617",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yinan Liang",
    "id": "2261898108",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Xiuwei Xu",
    "id": "2158440998",
    "h_index": 14,
    "papers": 42
   },
   {
    "name": "Ziwei Wang",
    "id": "2284636662",
    "h_index": 8,
    "papers": 17
   },
   {
    "name": "Yansong Tang",
    "id": "2335553792",
    "h_index": 6,
    "papers": 14
   },
   {
    "name": "Jiwen Lu",
    "id": "2257098405",
    "h_index": 12,
    "papers": 39
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "sim2real",
   "imitation-diffusion",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.05970v1",
  "pdf_url": "https://arxiv.org/pdf/2608.05970v1",
  "html_url": "https://arxiv.org/html/2608.05970v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.05948",
  "slug": "gauge-a-measurement-grounded-benchmark-for-physical-fidelity-in-simula",
  "title": "GAUGE: A Measurement-Grounded Benchmark for Physical Fidelity in Simulation Engines and Video World Models",
  "abstract": "Physics engines facilitate large-scale training and evaluation for embodied intelligence, while generative video world models are emerging as implicit simulators of future states and interactions. However, existing evaluations of physical fidelity are often conducted in isolation and rely heavily on perceptual similarity or human judgments, providing limited insight into which physical principles or parameters are violated. We introduce GAUGE, a real-world-grounded diagnostic benchmark for jointly evaluating how numerical simulators and generative video world models reproduce or deviate from real-world physics. It comprises 22 controlled task families covering rigid bodies, flexible cables, textiles, and volumetric deformable objects. Grounded in real-world trajectories and paired with calibrated physical metadata, uncertainty annotations, and task-specific observables, these tasks cover fundamental physical processes including collision, friction, momentum transfer, oscillation, self-contact, and deformation across diverse materials and conditions. We benchmark Isaac Sim, Genesis, and Newton on 14 task families using generalized trajectory errors, and evaluate 6 image-to-video models on 5 rigid-body tasks by testing physical-law consistency and the temporal stability of inferred parameters. Our results reveal no uniformly faithful physics engine, with the largest discrepancies arising in impulsive contact, rapid textile motion, and volumetric deformation. We further find that video world models can produce trajectories with the expected equation form while recovering incorrect accelerations, momentum transfer, and oscillation timing. GAUGE lays the groundwork for developing more physically faithful simulators and world models for embodied intelligence.",
  "published": "2026-08-06",
  "updated": "2026-08-06",
  "year": "2026",
  "authors": [
   "Shuai Wang",
   "Yaxin Feng",
   "Xuekun Jiang",
   "Shihan Tian",
   "Ningyu Yan",
   "Xing Shen",
   "Chaoyang Lyu",
   "Hui Wang",
   "Yunsong Zhou",
   "Hanqing Wang",
   "Jiangmiao Pang",
   "Yang Xiang",
   "Xing Gao",
   "Chunhua Shen",
   "Weinan Zhang"
  ],
  "author_count": 15,
  "categories": [
   "cs.AI",
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.AI",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "GAUGE, a real-world-grounded diagnostic benchmark for jointly evaluating how numerical simulators and generative video world models reproduce or deviate from real-world physics, is introduced and no uniformly faithful physics engine is revealed.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Shuai Wang",
    "id": "2445399831",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Yaxin Feng",
    "id": "2253854053",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Xuekun Jiang",
    "id": "80180784",
    "h_index": 9,
    "papers": 16
   },
   {
    "name": "Shihan Tian",
    "id": "2274002608",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Ning Yan",
    "id": "2347357165",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Xing Shen",
    "id": "2398705609",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Chaoyang Lyu",
    "id": "146393915",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Hui Wang",
    "id": "2281486680",
    "h_index": 7,
    "papers": 15
   },
   {
    "name": "Yunsong Zhou",
    "id": "2118117099",
    "h_index": 9,
    "papers": 18
   },
   {
    "name": "Hanqing Wang",
    "id": "2311307527",
    "h_index": 11,
    "papers": 27
   },
   {
    "name": "Jiangmiao Pang",
    "id": "2405891293",
    "h_index": 2,
    "papers": 17
   },
   {
    "name": "Yang Xiang",
    "id": "2337672446",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Xing Gao",
    "id": "2445502394",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Chunhua Shen",
    "id": "2257475525",
    "h_index": 7,
    "papers": 15
   },
   {
    "name": "Weinan Zhang",
    "id": "2344034124",
    "h_index": 5,
    "papers": 16
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "sim2real",
   "data-teleop",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.05948v1",
  "pdf_url": "https://arxiv.org/pdf/2608.05948v1",
  "html_url": "https://arxiv.org/html/2608.05948v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2608.05903",
  "slug": "robust-wam-bridging-generative-pretraining-and-semantic-foresight-in-w",
  "title": "Robust-WAM: Bridging Generative Pretraining and Semantic Foresight in World-Action Models",
  "abstract": "Mainstream World-Action Models (WAMs) adapt pretrained video generation models (VGMs) for robot control, transferring their learned dynamics prior for action prediction. These VGMs are typically trained in a variational autoencoder (VAE) latent space. However, the VAE latent space is optimized for pixel reconstruction, which rewards fine appearance detail and leaves the action prediction fragile under visual shifts. Recent works build WAMs in semantic latent space, which are more robust to appearance shifts. However, these models cannot leverage the large-scale VGM pretraining that exists only in VAE space. To overcome this dilemma, we propose Robust-WAM, a general post-training method for video-generation-based WAMs that preserves the VAE-based generative path and adds a lightweight semantic foresight alignment objective on the action stream. This retains the large-scale VGM pretraining while grounding actions in appearance-invariant dynamics that stay reliable under illumination shifts and other visual out-of-distribution conditions. Specifically, we employ learnable query tokens to bring future-scene semantics into the action stream by aligning their output hidden states with the semantic foresight of future ground-truth frames. To establish the temporal correspondence between each query and the future step it describes, we give it the positional encoding of the matching action tokens. Experiments on out-of-distribution generalization simulation benchmarks and a real-robot setup show that our Robust-WAM consistently improves the success rates of multiple WAM baselines without sacrificing in-distribution performance.",
  "published": "2026-08-06",
  "updated": "2026-08-07",
  "year": "2026",
  "authors": [
   "Haodong Yan",
   "Junfeng Li",
   "Junjie He",
   "Zhide Zhong",
   "MingMing Yu",
   "Wenxuan Song",
   "Jiaguan Zhu",
   "Yangyang Zheng",
   "Yuqiao Du",
   "Jiadi You",
   "Yingjie Cai",
   "Xu Yan",
   "Guanyi Zhao",
   "Bingbing Liu",
   "Haoang Li"
  ],
  "author_count": 15,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The Robust-WAM is a general post-training method for video-generation-based WAMs that preserves the VAE-based generative path and adds a lightweight semantic foresight alignment objective on the action stream to retain the large-scale VGM pretraining while grounding actions in appearance-invariant dynamics.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Haodong Yan",
    "id": "2321603038",
    "h_index": 5,
    "papers": 16
   },
   {
    "name": "Junfeng Li",
    "id": "2376547586",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Junjie He",
    "id": "2316016558",
    "h_index": 3,
    "papers": 16
   },
   {
    "name": "Zhide Zhong",
    "id": "2349315841",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Mingming Yu",
    "id": "2387333550",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Wenxuan Song",
    "id": "2293142288",
    "h_index": 14,
    "papers": 46
   },
   {
    "name": "Jiaguang Zhu",
    "id": "1575707808",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Yangyang Zheng",
    "id": "2233047956",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Yuqiao Du",
    "id": "2455824586",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Jiadi You",
    "id": "2364354912",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Yingjie Cai",
    "id": "2323851085",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Xu Yan",
    "id": "2279659190",
    "h_index": 8,
    "papers": 17
   },
   {
    "name": "Guanyi Zhao",
    "id": "2276829544",
    "h_index": 4,
    "papers": 16
   },
   {
    "name": "Bingbing Liu",
    "id": "2323042847",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Haoang Li",
    "id": "2384363611",
    "h_index": 9,
    "papers": 38
   }
  ],
  "comment": "",
  "topics": [
   "sim2real",
   "foundation-pretraining",
   "data-teleop",
   "video-generation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.05903v2",
  "pdf_url": "https://arxiv.org/pdf/2608.05903v2",
  "html_url": "https://arxiv.org/html/2608.05903v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.05799",
  "slug": "xeworld-can-action-conditioned-world-models-generalize-to-unseen-robot",
  "title": "XEWorld: Can Action-Conditioned World Models Generalize to Unseen Robot Embodiments?",
  "abstract": "Action-conditioned world models are promising learned simulators for robotic manipulation, yet evaluating them exclusively on training robots fails to reveal whether they capture physical dynamics or merely memorize visual patterns. To answer whether a model can faithfully render a robot it has never seen, we introduce XEWorld, a controlled cross-embodiment testbed for world models that isolates embodiments by evaluating held-out robots within physically identical scenes. Our systematic analysis uncovers a shared architectural bottleneck: current models act primarily as 2D visual pattern matchers whose generalization is governed by visual similarity rather than physical kinematic similarity. Driven by this limitation, they struggle to translate abstract numeric joint actions into coherent visual trajectories, and fail to predict dynamic visual changes from static initial observations. Consequently, successfully rendering an unseen embodiment zero-shot strictly requires heavily grounded cues, specifically pixel-space actions and explicit spatial-temporal alignment. Even when bypassing this zero-shot barrier via few-shot adaptation, the forced appearance recovery triggers catastrophic forgetting of seen embodiments. Together, these failures expose a critical inability to apply learned physical dynamics to novel visual appearances, highlighting that achieving true cross-embodiment generalization requires architectural innovations that decouple visual appearance from underlying physical dynamics.",
  "published": "2026-08-06",
  "updated": "2026-08-06",
  "year": "2026",
  "authors": [
   "Yixiang Chen",
   "Jiabing Yang",
   "Yuan Xu",
   "Qisen Ma",
   "Keji He",
   "Peiyan Li",
   "Kai Wang",
   "Ziheng He",
   "Xiangnan Wu",
   "Jing Liu",
   "Nianfeng Liu",
   "Yan Huang",
   "Liang Wang"
  ],
  "author_count": 13,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "XEWorld is introduced, a controlled cross-embodiment testbed for world models that isolates embodiments by evaluating held-out robots within physically identical scenes, highlighting that achieving true cross-embodiment generalization requires architectural innovations that decouple visual appearance from underlying physical dynamics.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yixiang Chen",
    "id": "2366155958",
    "h_index": 4,
    "papers": 20
   },
   {
    "name": "Jiabing Yang",
    "id": "2376278703",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Yuan Xu",
    "id": "2313357459",
    "h_index": 4,
    "papers": 21
   },
   {
    "name": "Qisen Ma",
    "id": "2215860505",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Keji He",
    "id": "51054943",
    "h_index": 7,
    "papers": 16
   },
   {
    "name": "Peiyan Li",
    "id": "2305635155",
    "h_index": 7,
    "papers": 26
   },
   {
    "name": "Kai Wang",
    "id": "2382828533",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Ziheng He",
    "id": "2338038073",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Xiang Wu",
    "id": "2240816230",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Jing Liu",
    "id": "2309490169",
    "h_index": 3,
    "papers": 14
   },
   {
    "name": "Nianfeng Liu",
    "id": "2391015095",
    "h_index": 2,
    "papers": 11
   },
   {
    "name": "Yan Huang",
    "id": "2375031786",
    "h_index": 4,
    "papers": 22
   },
   {
    "name": "Liang Wang",
    "id": "2317093163",
    "h_index": 4,
    "papers": 24
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "sim2real",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.05799v1",
  "pdf_url": "https://arxiv.org/pdf/2608.05799v1",
  "html_url": "https://arxiv.org/html/2608.05799v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.05746",
  "slug": "acoustic-driven-millimetric-helical-robot-ultrasonic-synergistic-manip",
  "title": "Acoustic-driven millimetric helical robot: ultrasonic synergistic manipulation in confined fluidic environment",
  "abstract": "Acoustic field-driven manipulation provides a non-contact and non-invasive strategy for controlling microscale and nanoscale objects, yet its extension to millimeter-scale robots was limited by insufficient propulsion efficiency in confined biological environments. Here, a coordinated multi-acoustic-field approach is introduced, which harnesses the synergistic action of acoustic radiation forces and acoustic streaming flows to enable controlled locomotion of millimeter-scale helical robots and enhance propulsion. Multiphysics simulations captured the dynamics of millimeter-scale helical robots under combined acoustic fields, and experimental validation demonstrated their locomotion capabilities, including planar navigation, inclined climbing, and vertical motion. Semi-autonomous navigation experiments further confirmed that ultrasonic synergy substantially improved maneuverability. In vitro tests in porcine venous vessels demonstrated that coordinated acoustic fields supported both unidirectional and reciprocating motion under biologically relevant confinement. These findings provide mechanistic insight into scaling acoustic micromanipulation to the millimetre regime and support biomedical applications requiring versatile and controllable robotic mobility.",
  "published": "2026-08-06",
  "updated": "2026-08-06",
  "year": "2026",
  "authors": [
   "Hanlin Wang",
   "Xin Wang",
   "Xinwei Wei",
   "Jiaxu Liu",
   "Le Wang",
   "Shengze Cai",
   "Chao Xu"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "Ultrasonics",
  "venue_source": "semantic-scholar",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "10.1016/j.ultras.2026.108233",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hanlin Wang",
    "id": "2390987578",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Xin Wang",
    "id": "2244170309",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Xinwei Wei",
    "id": "2211895071",
    "h_index": 11,
    "papers": 16
   },
   {
    "name": "Jiaxu Liu",
    "id": "2144131112",
    "h_index": 5,
    "papers": 21
   },
   {
    "name": "Le Wang",
    "id": "2144668171",
    "h_index": 7,
    "papers": 28
   },
   {
    "name": "Shengze Cai",
    "id": "40756038",
    "h_index": 20,
    "papers": 75
   },
   {
    "name": "Chao Xu",
    "id": "2389798144",
    "h_index": 1,
    "papers": 6
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.05746v1",
  "pdf_url": "https://arxiv.org/pdf/2608.05746v1",
  "html_url": "https://arxiv.org/html/2608.05746v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.05738",
  "slug": "in-context-vla-endowing-vision-language-action-models-with-language-vi",
  "title": "In-Context VLA: Endowing Vision-Language-Action Models with Language via In-Context Post-Training and Agentic Tool Use",
  "abstract": "Vision-Language-Action (VLA) models have become the dominant recipe for generalist manipulation, yet they are almost universally trained by behavior cloning: a policy imitates expert action chunks conditioned on a static image and a fixed instruction. A natural remedy is to inject explicit reasoning through textual chain-of-thought (CoT). We show, both empirically and analytically, that free-form textual CoT degrades low-level control: the reasoning it produces is ungrounded, its latency breaks closed-loop timing, and, crucially, the reasoning and action tokens are optimized against conflicting objectives so that the policy learns to narrate rather than to act. We argue that what a VLA needs is not the ability to generate language, but the ability to consume grounded language. To this end we introduce \\textbf{\\ourmethod{}}, a framework that endows a VLA with language competence through (i) in-context post-training, in which perceptual evidence is injected as structured context and the model is supervised only on actions, and (ii) an agentic tool-use interface, in which the policy queries open-vocabulary detectors, monocular depth, and a vision--language model to actively acquire task-relevant information. Rather than emitting a single templated caption, our data engine produces diverse, paraphrased, and evidence-conditioned spatial descriptions, so that the policy learns to interpret language it has never seen verbatim. Across the RoboCasa-GR1, SimplerEnv, and LIBERO simulation benchmarks, together with 8 real-world robot manipulation tasks, our method consistently achieves SOTA results in both performance and efficiency when compared with CoT-based approaches under matched configurations.",
  "published": "2026-08-06",
  "updated": "2026-08-06",
  "year": "2026",
  "authors": [
   "Jiarui Yang",
   "Wen Huang",
   "Jiale Zhang",
   "Maowei Hu",
   "Hang Guo"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper argues that what a VLA needs is not the ability to generate language, but the ability to consume grounded language, and introduces a framework that endows a VLA with language competence through in-context post-training and an agentic tool-use interface.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jiaru Yang",
    "id": "2446874920",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Wen Huang",
    "id": "2375903953",
    "h_index": 1,
    "papers": 9
   },
   {
    "name": "Jiale Zhang",
    "id": "2353134126",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "M. Hu",
    "id": "2455761281",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Hang Guo",
    "id": "2274202835",
    "h_index": 12,
    "papers": 32
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "sim2real",
   "imitation-diffusion"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.05738v1",
  "pdf_url": "https://arxiv.org/pdf/2608.05738v1",
  "html_url": "https://arxiv.org/html/2608.05738v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.05725",
  "slug": "near-sensor-computing-for-rapid-visuotactile-perception",
  "title": "Near-sensor Computing for Rapid Visuotactile Perception",
  "abstract": "Visuotactile sensors reconstruct dense contact geometry from measured surface gradients, but host-based processing increases power consumption and introduces data-transfer delays and variable scheduling latency, limiting the sensing and response speed of robotic systems. To address these limitations, we implement a near-sensor computing framework that includes a spectral Poisson solver as a fully streaming hardware pipeline. The computational core logic has an estimated power consumption of 347 mW and achieves high throughput without data-dependent branching or iterative convergence, thereby providing deterministic latency. Operating at 166 MHz, the pipeline produces the first depth value of each 128x128 frame 35,107 cycles after receiving the first input pixel, corresponding to a fixed latency of 0.211 ms. Across 15 contact geometries, the reconstructed depths differ from a double-precision reference by 0.17 % of the peak contact depth. On-chip decisions based on these reconstructions close a robot protective reflex loop in 28.3 +/- 4.9 ms, compared with 169.9 +/- 27.8 ms for an equivalent host-based loop using the same actuator. These results demonstrate that near-sensor reconstruction can provide accurate, energy-efficient, and deterministic tactile geometry on timescales suitable for rapid robotic contact responses.",
  "published": "2026-08-06",
  "updated": "2026-08-06",
  "year": "2026",
  "authors": [
   "Zhengying Zhu",
   "Ruilin Zhang",
   "Runze Hu",
   "Chenxi Xiao"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A near-sensor computing framework that includes a spectral Poisson solver as a fully streaming hardware pipeline that achieves high throughput without data-dependent branching or iterative convergence, thereby providing deterministic latency and demonstrating that near-sensor reconstruction can provide accurate, energy-efficient, and deterministic tactile geometry on timescales suitable for rapid robotic contact responses.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhengying Zhu",
    "id": "2325892040",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Ruilin Zhang",
    "id": "2455790672",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Runze Hu",
    "id": "2420098239",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Chenxi Xiao",
    "id": "2356920855",
    "h_index": 3,
    "papers": 14
   }
  ],
  "comment": "14 pages, 4 figures",
  "topics": [
   "tactile",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.05725v1",
  "pdf_url": "https://arxiv.org/pdf/2608.05725v1",
  "html_url": "https://arxiv.org/html/2608.05725v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.05723",
  "slug": "atp-anatomical-torque-with-passivity-based-control-framework-for-safe",
  "title": "ATP: Anatomical Torque with Passivity-based Control Framework for Safe Upper-Limb Exoskeleton Assistance",
  "abstract": "Providing assistance across diverse movements is a central objective of exoskeletons, and anatomical knowledge can enable responsive support that generalizes across tasks. However, anatomical assistance has mainly been studied for lower-limb exoskeletons, where periodic, weight-bearing motions impose lower demands on torque precision. Extending such assistance to complex, nonperiodic upper-limb movements remains challenging. This paper proposes Anatomical Torque with Passivity-Based Control (ATP) for safe upper-limb exoskeleton assistance. First, a scalable musculoskeletal simulation framework trains a unified reinforcement-learning muscle controller that generalizes across upper-limb movements and generates anatomical reference torques without complex biomechanical computations. Second, an online torque-refinement scheme adapts the reference to diverse movements, suppresses tendon-induced spikes, and incorporates a learned anomaly score for safe and comfortable assistance. Third, an interaction torque controller delivers assistance through a cable-driven compliant exoskeleton without constraining motion to predefined trajectories, while an energy tank preserves passivity with theoretical guarantees on torque tracking and system passivity. Simulations and real-world experiments show accurate tracking of long-duration motion sequences and generalization to real-time human movements. The controller achieves accurate torque tracking while preserving passivity and resumes tracking after energy-tank replenishment. An EMG study with five participants further shows reduced target-muscle activity during static and dynamic tasks compared with gravity compensation and open-loop assistance, with reductions of up to 48% relative to movement without the exoskeleton in a dynamic multi-joint task.",
  "published": "2026-08-06",
  "updated": "2026-08-06",
  "year": "2026",
  "authors": [
   "Yu Chen",
   "Gong Chen",
   "Xiang Li"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper proposes Anatomical Torque with Passivity-Based Control (ATP) for safe upper-limb exoskeleton assistance and simulations and real-world experiments show accurate tracking of long-duration motion sequences and generalization to real-time human movements.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yu Chen",
    "id": "2144838669",
    "h_index": 5,
    "papers": 15
   },
   {
    "name": "Gong Chen",
    "id": "2155017538",
    "h_index": 9,
    "papers": 23
   },
   {
    "name": "Xiang Li",
    "id": "2144440112",
    "h_index": 5,
    "papers": 13
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.05723v1",
  "pdf_url": "https://arxiv.org/pdf/2608.05723v1",
  "html_url": "https://arxiv.org/html/2608.05723v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.05674",
  "slug": "joyai-ra-0-5-scaling-robot-manipulation-learning-via-dual-action-align",
  "title": "JoyAI-RA 0.5: Scaling Robot Manipulation Learning via Dual Action Alignment",
  "abstract": "Robot data is scarce, so generalist policies need to learn from heterogeneous sources, including human egocentric video, simulation, and real robots, which differ in supervision and embodiment, with action labels missing or mutually incompatible. Human egocentric data scale best but sit farthest from robot data, and naive pooling causes negative transfer rather than knowledge sharing. We propose JoyAI-RA 0.5, a generalist Vision-Language-World-Action (VLWA) framework that couples physical world-dynamics priors with visual semantics and scales manipulation learning across such data via dual action alignment. Implicit action alignment infers latent actions from visual transitions, enabling action-free human, simulation, and robot data to guide a latent-action-conditioned world model in learning physical dynamics. Explicit alignment grounds reliable human and robot trajectories in a unified physical action space through a canonical action representation and camera-frame chunk-relative end-effector actions. An inner-outer-loop reinforcement stage then pairs efficient task adaptation with foundation-policy improvement. On a real-world AgiBot benchmark, JoyAI-RA performs strongly on both seen tasks and unseen variations. The task score improves consistently as the volume of human egocentric pretraining data increases and shows no sign of plateauing at our largest scale. This suggests that abundant but weakly labeled human experience can be converted into a transferable training signal, making human video not merely a weak auxiliary source but a primary axis along which manipulation capability can be scaled. Project page can be found at https://joyai-ra-05.github.io/.",
  "published": "2026-08-06",
  "updated": "2026-08-06",
  "year": "2026",
  "authors": [
   "JoyAI-RA Team"
  ],
  "author_count": 1,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "JoyAI-RA 0.5 is proposed, a generalist Vision-Language-World-Action framework that couples physical world-dynamics priors with visual semantics and scales manipulation learning across such data via dual action alignment, suggesting that abundant but weakly labeled human experience can be converted into a transferable training signal.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "JoyAI-RA Team",
    "id": "2455673731",
    "h_index": 0,
    "papers": 1
   }
  ],
  "comment": "Project Page: https://joyai-ra-05.github.io/",
  "topics": [
   "world-models",
   "egocentric-data",
   "foundation-pretraining"
  ],
  "orgs": [
   "AgiBot"
  ],
  "abs_url": "https://arxiv.org/abs/2608.05674v1",
  "pdf_url": "https://arxiv.org/pdf/2608.05674v1",
  "html_url": "https://arxiv.org/html/2608.05674v1",
  "code_url": "https://joyai-ra-05.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.05647",
  "slug": "kilvo-kinematic-inertial-lidar-visual-odometry-with-robust-multimodal",
  "title": "KILVO: Kinematic-Inertial-LiDAR-Visual Odometry with Robust Multimodal Adaptation for Humanoid Robots",
  "abstract": "This article presents a kinematic-inertial-LiDAR-visual odometry for humanoid robots, called KILVO. Tailored to the platform features, requirements, and real-world complexity, it fully utilizes the sensors commonly equipped on humanoid robots, including joint encoders, IMU, LiDAR, and camera, within an asynchronous-sequential hybrid error-state iterated Kalman filter (ESIKF). Specifically, inertial data are used for prediction, leg kinematics are processed asynchronously at a high rate and provide proprioceptive constraints, while exteroception is updated sequentially, first by registering LiDAR points for geometric priors and then by updating the visual component via photometric errors. Moreover, the framework is elaborately designed with multimodal adaptation for resilience to sensor failures. A compact contact estimation module is also developed, sharing information with state estimation without additional sensors. Extensive experiments on public datasets and in the real world across multiple humanoid robots, gait patterns, and scenarios demonstrate that KILVO achieves highly competitive accuracy, efficiency, and output rates, with strong robustness against sensor degradation and failures, making it more suitable for humanoid robots than state-of-the-art fusion methods. Our code and datasets are released on GitHub.",
  "published": "2026-08-06",
  "updated": "2026-08-06",
  "year": "2026",
  "authors": [
   "Jixin Gao",
   "Fucheng Liu",
   "Teng Zhang",
   "Fusheng Zha"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Extensive experiments demonstrate that KILVO achieves highly competitive accuracy, efficiency, and output rates, with strong robustness against sensor degradation and failures, making it more suitable for humanoid robots than state-of-the-art fusion methods.",
  "doi": "10.1109/TMECH.2026.3721778",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jixin Gao",
    "id": "2233436874",
    "h_index": 3,
    "papers": 13
   },
   {
    "name": "Fucheng Liu",
    "id": "2360848552",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Teng Zhang",
    "id": "2243380896",
    "h_index": 2,
    "papers": 11
   },
   {
    "name": "Fusheng Zha",
    "id": "2344258085",
    "h_index": 1,
    "papers": 5
   }
  ],
  "comment": "This article has been accepted for publication in IEEE/ASME Transactions on Mechatronics. Personal use is permitted. All other uses require IEEE permission",
  "topics": [
   "humanoids",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.05647v1",
  "pdf_url": "https://arxiv.org/pdf/2608.05647v1",
  "html_url": "https://arxiv.org/html/2608.05647v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.05594",
  "slug": "jta-joint-testability-architecture-for-scenario-based-validation-of-sa",
  "title": "JTA: Joint Testability Architecture for Scenario-Based Validation of Safety-Critical Software",
  "abstract": "Validation adequacy in safety-critical software depends on more than the system under test. Critical scenarios must be constructed under controlled conditions, execution evidence must be aligned into verdict-ready form, and abnormal outcomes must be attributable to actionable causes. Existing testability research remains largely artifact-centric and offers little architectural support for reasoning about the combined capability of the scenario, the test system, and the system under test. Joint Testability Architecture (JTA) addresses this gap by treating those three elements as a single object of analysis and design. It characterizes validation capability along three dimensions--controllability, observability, and isolability--and organizes them through three domains, three bridges, and an analysis-design-evaluation-refinement loop. JTA also introduces scenario contracts, joint capability assessment, validation blind-spot identification, and bridge-oriented design actions that map capability gaps to concrete improvements in control points, evidence organization, and attribution boundaries. An illustrative analysis of ArduPilot failsafe validation shows that link-loss scenarios are comparatively mature, whereas state-estimation anomaly scenarios remain harder to validate because evidence alignment and attribution semantics are weaker. JTA is not a replacement for existing testing or safety-analysis techniques; it provides an architectural basis for modeling, designing, and assessing scenario-based validation in safety-critical software.",
  "published": "2026-08-06",
  "updated": "2026-08-06",
  "year": "2026",
  "authors": [
   "Wenyao Xue",
   "Jiandi Wang",
   "Yichen Wang"
  ],
  "author_count": 3,
  "categories": [
   "cs.SE",
   "cs.RO"
  ],
  "primary_category": "cs.SE",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "JTA provides an architectural basis for modeling, designing, and assessing scenario-based validation in safety-critical software and shows that link-loss scenarios are comparatively mature, whereas state-estimation anomaly scenarios remain harder to validate because evidence alignment and attribution semantics are weaker.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Wenyao Xue",
    "id": "2328335515",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Jiandi Wang",
    "id": "2456610796",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yichen Wang",
    "id": "2246683426",
    "h_index": 1,
    "papers": 4
   }
  ],
  "comment": "Accepted by QRS 2026",
  "topics": [
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.05594v1",
  "pdf_url": "https://arxiv.org/pdf/2608.05594v1",
  "html_url": "https://arxiv.org/html/2608.05594v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.05588",
  "slug": "search-aided-joint-agent-environment-reinforcement-learning-for-robust",
  "title": "Search-Aided Joint Agent-Environment Reinforcement Learning for Robust Lifelong Multi-Agent Path Finding with Rotations",
  "abstract": "Lifelong Multi-Agent Path Finding (LMAPF) requires repeatedly planning collision-free paths for agents that continuously receive new goals upon reaching their current ones. While many learning-based planners have been proposed for LMAPF, most rely on oversimplified kinematic assumptions that may overlook motion constraints critical to real-world performance. In this work, we study a more realistic LMAPF model derived from many real-world automated warehouse systems, termed LMAPF-R2, which incorporates robust safety constraints and in-place rotation constraints. These constraints substantially increase coordination difficulty, particularly in highly constrained spaces. To address these challenges, we propose Search-Aided Joint Reinforcement Learning (SJRL). We first augment neural policies with Causal PIBT, a single-step search-based planner that resolves agents' collisions and propagates their intentions. We then introduce a unified RL formulation that jointly optimizes agent and environment policies, where the environment policy learns graph edge costs to provide global movement guidance via backward Dijkstra search. Experiments demonstrate that SJRL achieves significant improvements over the strong search-based planner, Causal-PIBT, across multiple high-density maps. We further validate SJRL in a challenging mixed-reality warehouse environment with 8 physical robots and 248 virtual robots.",
  "published": "2026-08-06",
  "updated": "2026-08-06",
  "year": "2026",
  "authors": [
   "He Jiang",
   "Jingtian Yan",
   "Yulun Zhang",
   "Yimin Tang",
   "Tanishq Duhan",
   "Rishi Veerapaneni",
   "Guillaume Sartoretti",
   "Jiaoyang Li"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.MA"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work introduces a unified RL formulation that jointly optimizes agent and environment policies, where the environment policy learns graph edge costs to provide global movement guidance via backward Dijkstra search and achieves significant improvements over the strong search-based planner, Causal-PIBT, across multiple high-density maps.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "He Jiang",
    "id": "2282501831",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Jingtian Yan",
    "id": "2185745209",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Yulun Zhang",
    "id": "2108094463",
    "h_index": 9,
    "papers": 25
   },
   {
    "name": "Yimin Tang",
    "id": "2292781160",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "T. Duhan",
    "id": "2257346752",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Rishi Veerapaneni",
    "id": "9567136",
    "h_index": 8,
    "papers": 32
   },
   {
    "name": "G. Sartoretti",
    "id": "2292917033",
    "h_index": 11,
    "papers": 67
   },
   {
    "name": "Jiaoyang Li",
    "id": "2294313888",
    "h_index": 7,
    "papers": 27
   }
  ],
  "comment": "",
  "topics": [
   "rl-control",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.05588v1",
  "pdf_url": "https://arxiv.org/pdf/2608.05588v1",
  "html_url": "https://arxiv.org/html/2608.05588v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.05586",
  "slug": "pathcover-a-fast-convex-decomposition-along-a-path-via-randomized-iter",
  "title": "PathCover: A Fast Convex Decomposition along a Path via Randomized Iterative Space Partitioning (RISP) on Point Clouds",
  "abstract": "Autonomous robot navigation requires the rapid generation of obstacle-free regions for trajectory planning. However, existing corridor generators struggle to meet real-time, sensor-rate computational constraints. To resolve this bottleneck, we introduce PathCover, a framework driven by RISP; a novel randomized algorithm that constructs convex polytopes directly from raw point cloud data in expected linear time under a mild probabilistic elimination condition. PathCover generates sequences of overlapping, obstacle-free polytopes that safely constrain downstream MPC and trajectory optimization. We mathematically guarantee that the algorithm terminates in finite steps while ensuring continuous progress along any obstacle-free reference path. Extensive benchmarks on synthetic and real-world LiDAR datasets demonstrate an order-of-magnitude speedup over state-of-the-art methods while maintaining comparable corridor volumes. The complete pipeline is validated via high-fidelity quadrotor simulations and physical deployment on a quadrupedal robot navigating constrained environments using live LiDAR perception.",
  "published": "2026-08-06",
  "updated": "2026-08-06",
  "year": "2026",
  "authors": [
   "Kunal S. Narkhede",
   "Abhijeet M. Kulkarni",
   "Guoquan Huang",
   "Ioannis Poulakakis"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "PathCover is introduced, a framework driven by RISP; a novel randomized algorithm that constructs convex polytopes directly from raw point cloud data in expected linear time under a mild probabilistic elimination condition, which guarantees that the algorithm terminates in finite steps while ensuring continuous progress along any obstacle-free reference path.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "K. S. Narkhede",
    "id": "51937588",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Abhijeet M. Kulkarni",
    "id": "2164038075",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Guoquan Huang",
    "id": "2243256784",
    "h_index": 8,
    "papers": 18
   },
   {
    "name": "Ioannis Poulakakis",
    "id": "2378864343",
    "h_index": 2,
    "papers": 10
   }
  ],
  "comment": "13 pages, 5 figures",
  "topics": [
   "humanoids",
   "spatial-3d",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.05586v1",
  "pdf_url": "https://arxiv.org/pdf/2608.05586v1",
  "html_url": "https://arxiv.org/html/2608.05586v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.05369",
  "slug": "world-to-wrist-task-conditioned-future-wrist-modeling-for-fine-grained",
  "title": "World-to-Wrist: Task-Conditioned Future Wrist Modeling for Fine-Grained Robot Manipulation",
  "abstract": "Vision-language-action (VLA) models often treat main-view and wrist-view observations as parallel visual inputs, overlooking their distinct roles in robot manipulation. Fine-grained manipulation, however, benefits from anticipating how wrist-local interactions may evolve under the global task context. To address this limitation, we present World-to-Wrist VLA (W2-VLA), a VLA model for fine-grained robot manipulation with task-conditioned future wrist modeling. Given current multi-view observations and a task instruction, W2-VLA contextualizes a set of latent modeling tokens as a compact interface between the vision-language model and the wrist predictor. Conditioned on this interface and the observed wrist history, the predictor forecasts future wrist latents, which are transformed into future-aware context for action prediction. In addition, we introduce W2-CoT, a synthesis pipeline that produces structured annotations describing manipulation progress, physical transition cues, and wrist-local evidence. These annotations provide auxiliary supervision that shapes the task-conditioned latent interface. Experiments on LIBERO, RoboTwin 2.0, and real-world manipulation tasks demonstrate improved fine-grained and contact-sensitive manipulation across both single-arm and bimanual settings, while maintaining action-generation rates above 80 Hz.",
  "published": "2026-08-05",
  "updated": "2026-08-05",
  "year": "2026",
  "authors": [
   "Yuhao Pan",
   "Haosong Peng",
   "Zhengshen Zhang",
   "Zhengyang Yan",
   "Yalun Dai",
   "Fushuo Huo",
   "Chujie Wang",
   "Tianyu Qi",
   "Xiucheng Wang",
   "Nan Cheng",
   "Wenchao Xu"
  ],
  "author_count": 11,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "World-to-Wrist VLA (W2-VLA), a VLA model for fine-grained robot manipulation with task-conditioned future wrist modeling, and W2-CoT, a synthesis pipeline that produces structured annotations describing manipulation progress, physical transition cues, and wrist-local evidence.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yuhao Pan",
    "id": "2307593824",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Haosong Peng",
    "id": "2188574915",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "Zhengsheng Zhang",
    "id": "2221777902",
    "h_index": 6,
    "papers": 17
   },
   {
    "name": "Zhengyang Yan",
    "id": "2398982194",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Yalun Dai",
    "id": "2176843801",
    "h_index": 9,
    "papers": 23
   },
   {
    "name": "Fushuo Huo",
    "id": "1904860264",
    "h_index": 14,
    "papers": 31
   },
   {
    "name": "Chujie Wang",
    "id": "2394241774",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Tianyu Qi",
    "id": "2187597620",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Xiucheng Wang",
    "id": "2118419158",
    "h_index": 15,
    "papers": 76
   },
   {
    "name": "Nan Cheng",
    "id": "2351809523",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Wenchao Xu",
    "id": "2319271216",
    "h_index": 5,
    "papers": 8
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "data-teleop",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.05369v1",
  "pdf_url": "https://arxiv.org/pdf/2608.05369v1",
  "html_url": "https://arxiv.org/html/2608.05369v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.05365",
  "slug": "unified-planning-learning-framework-for-robust-uuv-navigation-under-pa",
  "title": "Unified Planning-Learning Framework for Robust UUV Navigation Under Partial Observability",
  "abstract": "This paper presents an observation-only autonomy framework for Unmanned Underwater Vehicles (UUVs) navigation in dynamic underwater environments that integrates persistent occupancy mapping, global clearance-aware planning, and risk-aware local control. The proposed pipeline constructs occupancy maps solely from onboard sonar and depth image observations, adapts a clearance-constrained global planner (GP) to provide long-horizon structure, and integrates a reinforcement learning (RL) policy to handle short-range tracking and reactive avoidance. To further support decision-making under partial observability, the system learns a compact latent state representation from onboard sensor data, encoding environmental structure, obstacle dynamics, and uncertainty. Behavior tree (BT) distillation with staged supervision is introduced to improve safety and training stability, while an uncertainty-calibrated distillation mechanism reweights teacher guidance using online latent-model uncertainty, emphasizing uncertain regimes during learning, with time-to-collision (TTC) and clearance cues remaining explicit in planning and local policy features. To demonstrate the efficacy of the framework, a reproducible multi-seed evaluation protocol is established in high-fidelity GPU-accelerated simulation using NVIDIA Isaac Sim, and performance is benchmarked against BT-only and standard RL baselines. The results obtained demonstrate improved robustness and safety under dynamic conditions, thus providing a general pipeline with a unified hybrid planning learning architecture and a reproducible methodology for robust UUV autonomy under partial observability.",
  "published": "2026-08-05",
  "updated": "2026-08-05",
  "year": "2026",
  "authors": [
   "Md Ether Deowan",
   "Eleni Kelasidi"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "M. E. Deowan",
    "id": "2158512697",
    "h_index": 3,
    "papers": 13
   },
   {
    "name": "Eleni Kelasidi",
    "id": "2322501765",
    "h_index": 2,
    "papers": 5
   }
  ],
  "comment": "8 pages, 6 figures. Accepted for presentation at the 2026 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS 2026), Philadelphia, PA, USA",
  "topics": [
   "sim2real",
   "rl-control",
   "spatial-3d",
   "navigation",
   "safety-eval"
  ],
  "orgs": [
   "NVIDIA"
  ],
  "abs_url": "https://arxiv.org/abs/2608.05365v1",
  "pdf_url": "https://arxiv.org/pdf/2608.05365v1",
  "html_url": "https://arxiv.org/html/2608.05365v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.0
 },
 {
  "id": "2608.05215",
  "slug": "vlaff-vision-language-affordance-model-for-unified-actionable-affordan",
  "title": "VLAff: Vision-Language-Affordance Model for Unified Actionable Affordances",
  "abstract": "Learning manipulation skills from human videos is promising for scalable robot learning. However, the embodiment mismatch between humans and robots makes this challenging. One promising solution is to learn object-centric actionable affordances that are embodiment-agnostic. In this work, we propose a framework that leverages egocentric human videos with state-of-the-art 3D Structure-from-Motion and hand mesh reconstruction to extract actionable affordances such as visual, grasp, and trajectory affordances that explicitly encode where to interact, how to grasp, and how to move. We construct EgoAffordance, a large-scale dataset comprising 204K episodes with 5.6M visual affordances and 11.6M grasp and trajectory affordances. Building on this, we introduce VLAff, a large vision-language model-based unified foundation model that learns cross-modal correlations across all actionable affordances. Given a visual observation and instruction, VLAff generates visual affordance heatmaps, grasp poses, and trajectories, which are then converted into directly executable actions by utilizing 3D scene information. Through extensive experiments, we demonstrate that VLAff not only achieves state-of-the-art performance on visual affordance prediction, but can also be effectively applied to real robot applications such as zero-shot manipulation and affordance-guided robot learning.",
  "published": "2026-08-05",
  "updated": "2026-08-05",
  "year": "2026",
  "authors": [
   "Jihoon Oh",
   "Kento Kawaharazuka",
   "Kei Okada"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes a framework that leverages egocentric human videos with state-of-the-art 3D Structure-from-Motion and hand mesh reconstruction to extract actionable affordances such as visual, grasp, and trajectory affordances that explicitly encode where to interact, how to grasp, and how to move.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jihoon Oh",
    "id": "2256484322",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Kento Kawaharazuka",
    "id": "8308607",
    "h_index": 17,
    "papers": 220
   },
   {
    "name": "Kei Okada",
    "id": "2248244895",
    "h_index": 5,
    "papers": 68
   }
  ],
  "comment": "8 pages, 5 figures. Accepted to IEEE/RSJ IROS 2026. Project page: https://ojh6404.github.io/vlaff/",
  "topics": [
   "dexterous-manipulation",
   "egocentric-data",
   "foundation-pretraining",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.05215v1",
  "pdf_url": "https://arxiv.org/pdf/2608.05215v1",
  "html_url": "https://arxiv.org/html/2608.05215v1",
  "code_url": "https://ojh6404.github.io/vlaff/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.5
 },
 {
  "id": "2608.05084",
  "slug": "learning-when-to-stop-prefix-optimal-dynamic-diffusion-policies-for-co",
  "title": "Learning When to Stop: Prefix-Optimal Dynamic Diffusion Policies for Continuous Control",
  "abstract": "Diffusion policies are a powerful policy class for continuous control, but their iterative denoising process creates a substantial computational bottleneck. Reducing this cost requires adapting the number of denoising steps to the difficulty of each action while preserving task performance. We introduce Prefix-Optimal Generative Policies (POGP), a framework that learns a prefix value function at every intermediate denoising step through a Bellman-style recursion over the denoising chain. The prefix value function serves two purposes: it provides an auxiliary training objective that encourages intermediate outputs to become high-quality actions, and it enables a test-time stopping rule that terminates denoising when additional steps are unlikely to produce meaningful improvement. Across four MuJoCo environments and comparisons with 12 baselines, POGP reduces the required number of denoising iterations by approximately 2.7-fold while retaining near-full task performance. Compared with state-of-the-art dynamic diffusion baselines, prefix training also improves final task performance by approximately 3.5%. These results indicate that supervising intermediate denoising steps is useful not only for adaptive early stopping, but also as an auxiliary objective that improves the learned policy.",
  "published": "2026-08-05",
  "updated": "2026-08-05",
  "year": "2026",
  "authors": [
   "Rohit Kumar Salla",
   "Manoj Saravanan",
   "Simon Stepputtis"
  ],
  "author_count": 3,
  "categories": [
   "cs.LG",
   "cs.RO"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "POGP is introduced, a framework that learns a prefix value function at every intermediate denoising step through a Bellman-style recursion over the denoising chain, and indicates that supervising intermediate denoising steps is useful not only for adaptive early stopping, but also as an auxiliary objective that improves the learned policy.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Rohit Kumar Salla",
    "id": "2401828927",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "M. Saravanan",
    "id": "2254291865",
    "h_index": 1,
    "papers": 10
   },
   {
    "name": "Simon Stepputtis",
    "id": "8083127",
    "h_index": 16,
    "papers": 45
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.05084v1",
  "pdf_url": "https://arxiv.org/pdf/2608.05084v1",
  "html_url": "https://arxiv.org/html/2608.05084v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.05078",
  "slug": "spikingnav-robust-embodied-navigation-with-spiking-neural-policies",
  "title": "SpikingNav: Robust Embodied Navigation with Spiking Neural Policies",
  "abstract": "Embodied navigation requires an agent to make sequential decisions from egocentric observations in a physical environment. Existing Artificial Neural Network (ANN)-based navigation models have achieved strong performance, yet they often rely on dense computation and may degrade under visual corruptions. Spiking neural networks (SNNs) provide event-driven computation and intrinsic temporal dynamics, which are promising for compact and robust navigation on resource-constrained platforms. However, whether spike-based sensing and policy dynamics can improve robustness in visually rich embodied navigation remains an open problem. This paper proposes SpikingNav, a spiking framework for robust indoor embodied navigation. It contains a Spiking Sensing Encoder (SSE) and a Spiking Policy Network (SPN). The SSE extracts task-conditioned visual features with a spike-based backbone. The SPN maintains a recurrent policy state through membrane integration, thresholding, and spike-triggered reset. In this way, SpikingNav exploits the dynamic properties and spike activations of SNNs to improve navigation performance and robustness. We evaluate SpikingNav on PointNav and ObjectNav under clean observations and visual corruptions. SpikingNav achieves competitive clean performance and stronger robustness with fewer parameters and lower per-step computation than a matched ANN baseline. For instance, SpikingNav improves ObjectNav success from 31.05% to 34.12%, and raises the average success under visual corruptions from 8.45% to 13.71%, demonstrating the benefits of spike-based sensing and policy dynamics. We further validate the deployability of our spike-based sensing method on the Thruster-V2 neuromorphic chip. This physical hardware validation shows that SpikingNav can be instantiated on a real neuromorphic substrate for cyber-physical systems.",
  "published": "2026-08-05",
  "updated": "2026-08-05",
  "year": "2026",
  "authors": [
   "Jiahong Zhang",
   "Sijun Shen",
   "Dehua Wu",
   "Yifan Lin",
   "Xuechen Xia",
   "Xu Chu",
   "Youhui Zhang",
   " GuoqiLi"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "SpikingNav achieves competitive clean performance and stronger robustness with fewer parameters and lower per-step computation than a matched ANN baseline, and physical hardware validation shows that SpikingNav can be instantiated on a real neuromorphic substrate for cyber-physical systems.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jiahong Zhang",
    "id": "2314781146",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Sijun Shen",
    "id": "2398661223",
    "h_index": 0,
    "papers": 7
   },
   {
    "name": "Dehua Wu",
    "id": "2398801779",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Yifan Lin",
    "id": "2451917912",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Xuecheng Xia",
    "id": "2403535751",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Xu Chu",
    "id": "2455597047",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Youhui Zhang",
    "id": "2455657398",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "GuoqiLi",
    "id": "2455596876",
    "h_index": 0,
    "papers": 1
   }
  ],
  "comment": "",
  "topics": [
   "egocentric-data",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.05078v1",
  "pdf_url": "https://arxiv.org/pdf/2608.05078v1",
  "html_url": "https://arxiv.org/html/2608.05078v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.05047",
  "slug": "exact-model-free-policy-iteration-for-co-safe-ltl-planning",
  "title": "Exact Model-Free Policy Iteration for Co-safe LTL Planning",
  "abstract": "This work studies model-free reinforcement learning for co-safe linear temporal logic (sc-LTL) objectives in finite Markov decision processes, which can be reduced to maximal reachability objectives via the standard product construction. For this problem, direct sample-based bootstrap methods (e.g., TD or Q-learning) may fail to converge to optimal policies due to the noncontractive nature and nonuniqueness of solutions to the Bellman equation. We develop a new two-step model-free reinforcement learning method that first uses a discounted surrogate to identify a clamp set that resolves this nonuniqueness, and then applies undiscounted policy evaluation and greedy policy improvement with guarantees of finding an optimal solution. We prove almost-sure convergence of the policy evaluation step and finite termination of the policy iteration algorithm at an optimal policy. These theoretical results are validated through numerical experiments on a stochastic grid world.",
  "published": "2026-08-05",
  "updated": "2026-08-05",
  "year": "2026",
  "authors": [
   "Zetong Xuan",
   "Yu Wang"
  ],
  "author_count": 2,
  "categories": [
   "eess.SY",
   "cs.FL",
   "cs.RO"
  ],
  "primary_category": "eess.SY",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A new two-step model-free reinforcement learning method that first uses a discounted surrogate to identify a clamp set that resolves this nonuniqueness, and then applies undiscounted policy evaluation and greedy policy improvement with guarantees of finding an optimal solution.",
  "doi": "10.1109/LCSYS.2026.3702192",
  "oa_pdf": "https://arxiv.org/pdf/2608.05047",
  "s2_authors": [
   {
    "name": "Zetong Xuan",
    "id": "2295670811",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Yu Wang",
    "id": "2296049944",
    "h_index": 2,
    "papers": 6
   }
  ],
  "comment": "7 pages, 1 figure. Accepted for publication in IEEE Control Systems Letters. To be presented at the 2026 IEEE Conference on Decision and Control",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.05047v1",
  "pdf_url": "https://arxiv.org/pdf/2608.05047v1",
  "html_url": "https://arxiv.org/html/2608.05047v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.05042",
  "slug": "bridgevla-a-data-efficient-generalizable-and-memory-augmented-vision-l",
  "title": "BridgeVLA++: A Data-Efficient, Generalizable, and Memory-Augmented Vision-Language-Action Framework for 3D Manipulation",
  "abstract": "Leveraging pre-trained vision-language models (VLMs) to construct vision-language-action (VLA) models has emerged as a promising paradigm for 3D robot manipulation. However, existing 3D VLA methods remain data-hungry, exhibit limited generalization under distribution shifts, and lack explicit memory of past observations. These limitations hinder their application to data-scarce, open-world, and memory-dependent manipulation scenarios. Our previous work, BridgeVLA, improves data efficiency and generalization by preserving the input--output alignment of a pre-trained VLM during 3D action learning: raw point clouds are projected into multi-view images, and intermediate heatmaps are predicted before generating robot actions. In this work, we develop BridgeVLA++ by equipping BridgeVLA with a unified spatio-temporal memory architecture that models persistent spatial context and temporal interaction history. The resulting memory-augmented framework can reason over observation histories while preserving BridgeVLA's data efficiency and generalization capabilities. Extensive experiments show that our framework achieves strong performance on spatial manipulation tasks while exhibiting robust generalization. BridgeVLA++ further achieves state-of-the-art performance on two challenging memory-dependent manipulation benchmarks without sacrificing the data efficiency and generalization of the original BridgeVLA. In addition, BridgeVLA++ performs effectively in bimanual manipulation settings and is validated on an additional real-world robotic platform, demonstrating its scalability across tasks, environments, and robotic platforms. These results establish BridgeVLA++ as a unified 3D vision-language-action framework that simultaneously supports data-efficient learning, robust generalization, and effective memory-aware robot manipulation. Project website: https://bridgevla-plus.github.io/.",
  "published": "2026-08-05",
  "updated": "2026-08-05",
  "year": "2026",
  "authors": [
   "Peiyan Li",
   "Yuze Zhu",
   "Yixiang Chen",
   "Qisen Ma",
   "Yuan Xu",
   "Jiabing Yang",
   "He Guan",
   "Yan Huang",
   "Hongtao Wu",
   "Xiao Ma",
   "Tao Kong",
   "Liang Wang",
   "Tieniu Tan"
  ],
  "author_count": 13,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "BridgeVLA++ is developed by equipping BridgeVLA with a unified spatio-temporal memory architecture that models persistent spatial context and temporal interaction history that can reason over observation histories while preserving BridgeVLA's data efficiency and generalization capabilities.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Peiyan Li",
    "id": "2305635155",
    "h_index": 7,
    "papers": 26
   },
   {
    "name": "Yuze Zhu",
    "id": "2455656339",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yixiang Chen",
    "id": "2366155958",
    "h_index": 4,
    "papers": 20
   },
   {
    "name": "Qisen Ma",
    "id": "2215860505",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Yuan Xu",
    "id": "2313357459",
    "h_index": 4,
    "papers": 21
   },
   {
    "name": "Jiabing Yang",
    "id": "2376278703",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "He Guan",
    "id": "2055337654",
    "h_index": 7,
    "papers": 15
   },
   {
    "name": "Yan Huang",
    "id": "2375031786",
    "h_index": 4,
    "papers": 22
   },
   {
    "name": "Hongtao Wu",
    "id": "2264188661",
    "h_index": 9,
    "papers": 11
   },
   {
    "name": "Xiao Ma",
    "id": "2125110703",
    "h_index": 13,
    "papers": 45
   },
   {
    "name": "Tao Kong",
    "id": "2366394394",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Liang Wang",
    "id": "2317093163",
    "h_index": 4,
    "papers": 24
   },
   {
    "name": "Tieniu Tan",
    "id": "2346480480",
    "h_index": 7,
    "papers": 15
   }
  ],
  "comment": "This work has been submitted to the IEEE TPAMI for possible publication. Copyright may be transferred without notice, after which this version may no longer be accessible",
  "topics": [
   "vla",
   "spatial-3d",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.05042v1",
  "pdf_url": "https://arxiv.org/pdf/2608.05042v1",
  "html_url": "https://arxiv.org/html/2608.05042v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.04996",
  "slug": "dreamwam-beyond-rgb-future-prediction-for-world-action-models",
  "title": "DreamWAM: Beyond RGB Future Prediction for World Action Models",
  "abstract": "World Action Models (WAMs) learn action-relevant representations by predicting how the observed world will evolve. Most existing WAMs define this future in RGB space, where task-relevant state transitions are entangled with nuisance variations in texture, illumination, background, and viewpoint. We argue that WAMs should explicitly predict action-relevant future state rather than relying on RGB prediction alone. We introduce DreamWAM, which reformulates future prediction as structured world modeling beyond RGB, representing future states through complementary views of appearance, motion, geometry, and semantics. During training, DreamWAM combines joint latent denoising of RGB and motion with lightweight gated residual branches for geometry and semantics. Shared attention between VideoDiT and ActionDiT allows the action branch to learn from these future-state predictions, while all beyond-RGB supervision branches are disabled at inference and deployment remains RGB-only. Across both no-rollout and joint video-action inference, DreamWAM consistently improves the matched RGB-only baselines on LIBERO, from 97.30\\% to 98.40\\% and from 98.00\\% to 98.90\\%, respectively. The gains become larger under unseen LIBERO-Plus perturbations, from 51.36\\% to 63.44\\% and from 69.16\\% to 75.47\\%. The same robustness extends to real-world manipulation, where DreamWAM attains an average success rate of 74.4\\% across unseen changes in lighting, background, and object layout, compared with 55.6\\% for Fast-WAM-Joint. These results show that robust world-action learning depends not only on predicting the future, but on representing it in a form that matters for action. The code and models are publicly released at https://github.com/hustvl/DreamWAM.",
  "published": "2026-08-05",
  "updated": "2026-08-05",
  "year": "2026",
  "authors": [
   "Shanglin Yuan",
   "Weiheng Zhao",
   "Xin Shi",
   "Haoyi Jiang",
   "Xianda Guo",
   "Liu Liu",
   "Wenyu Liu",
   "Wei Sui",
   "Xinggang Wang"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "DreamWAM is introduced, which reformulates future prediction as structured world modeling beyond RGB, representing future states through complementary views of appearance, motion, geometry, and semantics, showing that robust world-action learning depends not only on predicting the future, but on representing it in a form that matters for action.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Shanglin Yuan",
    "id": "2395600009",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Weiheng Zhao",
    "id": "2441143790",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Xin Shi",
    "id": "2325918080",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Haoyi Jiang",
    "id": "2317648374",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Xianda Guo",
    "id": "2284674027",
    "h_index": 8,
    "papers": 23
   },
   {
    "name": "Liu Liu",
    "id": "2332597209",
    "h_index": 8,
    "papers": 31
   },
   {
    "name": "Wenyu Liu",
    "id": "2257432695",
    "h_index": 15,
    "papers": 33
   },
   {
    "name": "Wei Sui",
    "id": "2350672341",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Xinggang Wang",
    "id": "2320181046",
    "h_index": 8,
    "papers": 25
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.04996v1",
  "pdf_url": "https://arxiv.org/pdf/2608.04996v1",
  "html_url": "https://arxiv.org/html/2608.04996v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.04905",
  "slug": "primal3-pathfinding-via-reinforcement-and-imitation-multi-agent-learni",
  "title": "PRIMAL3: Pathfinding via Reinforcement and Imitation Multi-Agent Learning - Leveraging LaCAM3",
  "abstract": "We present PRIMAL3, an ultra-large-scale learning-based framework for multi-agent pathfinding (MAPF) that integrates reinforcement learning, topology-aware communication, LaCAM3-guided training, and PIBT-based action refinement. PRIMAL3 targets failures at topologically critical states, where agents must coordinate decisively around bottlenecks, dead ends, and persistent conflicts. Each agent is represented using features derived from cut vertices, dead-end regions, shortest-path distances, and blocking estimates. Two complementary graphs capture agent interactions: a same-direction following graph propagates multihop context along compatible paths, while a different-direction conflict graph differentiates agents competing for shared space through masked attention and relative features. During training, we propose to let policy entropy identify uncertain agents, for which LaCAM3 provides confidence-triggered action interventions and label-smoothed imitation targets. During execution, a priority-aware PIBT module refines the proposed joint actions using persistent, learned, and distance-aware priorities together with policy-aware fallback preferences while maintaining collision-free execution. The resulting framework combines learned exploration with structured expert guidance without requiring LaCAM3 at inference. Experiments demonstrate that PRIMAL3 substantially outperforms state-of-the-art learning-based baselines and scales to ultra-large instances with up to city-level 100,000 agents. Real-world experiments further demonstrate the feasibility of deploying PRIMAL3 on physical robotic systems and ablation studies validate the individual contributions the components we proposed. Project page: https://marmotlab.github.io/PRIMAL3/",
  "published": "2026-08-05",
  "updated": "2026-08-05",
  "year": "2026",
  "authors": [
   "Chengyang He",
   "Tanishq Duhan",
   "Gadiel Sznaier Camps",
   "Fangyuan Wang",
   "Yuhong Cao",
   "Jiankai Sun",
   "Ge Sun",
   "Mac Schwager",
   "Guillaume Sartoretti"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The resulting framework combines learned exploration with structured expert guidance without requiring LaCAM3 at inference and substantially outperforms state-of-the-art learning-based baselines and scales to ultra-large instances with up to city-level 100,000 agents.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Chengyang He",
    "id": "2257636330",
    "h_index": 5,
    "papers": 17
   },
   {
    "name": "T. Duhan",
    "id": "2257346752",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Gadiel Sznaier Camps",
    "id": "1504361763",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Fangyuan Wang",
    "id": "2276506399",
    "h_index": 3,
    "papers": 15
   },
   {
    "name": "Yuhong Cao",
    "id": "2146175818",
    "h_index": 12,
    "papers": 39
   },
   {
    "name": "Jiankai Sun",
    "id": "2282025",
    "h_index": 22,
    "papers": 61
   },
   {
    "name": "Ge Sun",
    "id": "2257130026",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Mac Schwager",
    "id": "2360172520",
    "h_index": 6,
    "papers": 14
   },
   {
    "name": "G. Sartoretti",
    "id": "2292917033",
    "h_index": 11,
    "papers": 67
   }
  ],
  "comment": "Under Review",
  "topics": [
   "rl-control",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.04905v1",
  "pdf_url": "https://arxiv.org/pdf/2608.04905v1",
  "html_url": "https://arxiv.org/html/2608.04905v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.04842",
  "slug": "rora-realistic-object-reconstruction-with-articulation",
  "title": "RORA: Realistic Object Reconstruction with Articulation",
  "abstract": "Replicating real-world environments into simulation by realistic visual representation like NeRF and 3D Gaussian Splatting (3DGS) has emerged as an effective strategy to reduce the sim-to-real gap in robot learning. However, implementing object articulation during the real-to-sim process is still a challenging task. Existing motion tracking or learning based articulation methods shows low success rates on complex kinematic structures having multiple joints. Furthermore, those methods require scan of dynamic motion of objects, which makes reconstruction process much complicated. In this work, we propose the first end-to-end pipeline that reconstructs simulation-ready assets with accurate articulation from a single static object video input through suggestion based human-in-the-loop process. Our approach exports a hybrid representation combining 3DGS for photorealistic rendering and mesh-based geometry for physical interaction. In the reconstruction process, our pipeline performs convex decomposition followed by user grouping for intuitive part segmentation, subsequently binding 3D Gaussians to the corresponding mesh parts. An Automatic Joint Suggestion Algorithm then calculates candidate joint axes from local boundary geometries and presents them to users for efficient articulated asset reconstruction. We have shown that our method achieves precise articulation results on partnet-mobility-v0 dataset and real objects. Additionally we presented a potential usage of our framework on robot learning, deploying the reconstructed assets in Unreal Engine and NVIDIA Isaac Sim, demonstrating real-time dexterous hand manipulation tasks.",
  "published": "2026-08-05",
  "updated": "2026-08-05",
  "year": "2026",
  "authors": [
   "Hyesung Lee",
   "Youngseon Lee",
   "Kyutae Lee",
   "Dongjun Lee",
   "Yongseok Lee"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.GR"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes the first end-to-end pipeline that reconstructs simulation-ready assets with accurate articulation from a single static object video input through suggestion based human-in-the-loop process, and achieves precise articulation results on partnet-mobility-v0 dataset and real objects.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hyesung Lee",
    "id": "2274062278",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Youngseon Lee",
    "id": "2303802511",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Kyutae Lee",
    "id": "2455635410",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Dongjun Lee",
    "id": "2248090105",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Yongseok Lee",
    "id": "2110034056",
    "h_index": 9,
    "papers": 19
   }
  ],
  "comment": "This work has been submitted to the IEEE for possible publication. Copyright may be transferred without notice, after which this version may no longer be accessible",
  "topics": [
   "dexterous-manipulation",
   "sim2real",
   "spatial-3d",
   "hri"
  ],
  "orgs": [
   "NVIDIA"
  ],
  "abs_url": "https://arxiv.org/abs/2608.04842v1",
  "pdf_url": "https://arxiv.org/pdf/2608.04842v1",
  "html_url": "https://arxiv.org/html/2608.04842v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2608.04825",
  "slug": "deliberate-before-you-fly-vision-guided-spatial-deliberation-for-uav-s",
  "title": "Deliberate Before You Fly: Vision-Guided Spatial Deliberation for UAV See-and-Reach Navigation",
  "abstract": "UAV see-and-reach navigation requires an aerial agent to approach a language-specified target visible in its initial view and stop reliably near it. Existing methods typically map vision-language representations directly to action outputs without explicitly modeling intermediate fine-grained spatial decisions. This direct mapping causes semantic-control misalignment, leading to inconsistent maneuvers and unreliable termination. To address this issue, we propose DBFly, a vision-language waypoint prediction framework that introduces explicit vision-guided spatial deliberation before waypoint generation. Specifically, DBFly introduces a spatial maneuver decision chain that progressively performs target-direction anchoring, spatial diagnosis, and maneuver decision, enabling high-level maneuver intent to explicitly guide continuous waypoint generation. DBFly further constructs an implicit flight corridor by transforming the initial target-direction prior into a persistent geometric reference and deriving an online corridor state from the UAV's current position, thereby providing soft geometric guidance for spatial diagnosis and maneuver correction. In addition, DBFly develops a terminal-convergence-aware stopping strategy that characterizes terminal states through both target proximity and short-horizon motion convergence, enabling more reliable stopping near the target. Extensive experiments across seen, unseen-object, and unseen-scene test sets demonstrate that DBFly improves the success rate over the SOTA baseline by an average of 25.07 percentage points. The project homepage is available at https://xuefanfu.github.io/DBFly-Page.",
  "published": "2026-08-05",
  "updated": "2026-08-05",
  "year": "2026",
  "authors": [
   "Fanfu Xue",
   "En Yu",
   "Bohang Liu",
   "Hongjun Wang",
   "Yang Yang",
   "Xindi Wang",
   "Jiande Sun"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "DBFly is proposed, a vision-language waypoint prediction framework that introduces explicit vision-guided spatial deliberation before waypoint generation and develops a terminal-convergence-aware stopping strategy that characterizes terminal states through both target proximity and short-horizon motion convergence, enabling more reliable stopping near the target.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Fanfu Xue",
    "id": "2175365324",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "En Yu",
    "id": "144118668",
    "h_index": 6,
    "papers": 17
   },
   {
    "name": "Bohang Liu",
    "id": "2269021573",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Hongjun Wang",
    "id": "2271255474",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Yang Yang",
    "id": "2363933376",
    "h_index": 4,
    "papers": 16
   },
   {
    "name": "Xindi Wang",
    "id": "2322827331",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Jiande Sun",
    "id": "2239940291",
    "h_index": 1,
    "papers": 8
   }
  ],
  "comment": "13 pages, 9 figures",
  "topics": [
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.04825v1",
  "pdf_url": "https://arxiv.org/pdf/2608.04825v1",
  "html_url": "https://arxiv.org/html/2608.04825v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.04765",
  "slug": "explicit-language-memory-for-long-horizon-planning-in-vision-language",
  "title": "Explicit Language Memory for Long-Horizon Planning in Vision-Language-Action Models",
  "abstract": "Vision-language-action (VLA) models provide a unified paradigm for connecting visual perception, language understanding, and robotic control. However, existing VLA models still face major challenges in long-horizon tasks: sparse expert demonstrations constrain cross-task compositional generalization; the non-Markovian nature of long-horizon tasks makes it difficult for policies conditioned only on current observations to maintain temporal consistency; limited closed-loop error correction allows execution errors to accumulate; and end-to-end action fine-tuning may weaken the high-level semantic representations of vision-language model (VLM) backbones. To address these issues, we propose a hierarchical long-horizon VLA architecture with an explicit language-memory module. The central idea is to convert discrete temporal observations into a coherent textual memory sequence with temporal logic. The system is decoupled into a high-level VLM and a low-level VLA: the high-level VLM performs semantic reasoning through a visual question answering training paradigm, while the low-level VLA executes precise continuous control conditioned on subtask instructions and visual observations. The high-level VLM recursively updates both language memory and subtask instructions using the previous memory as a contextual anchor, enabling persistent temporal tracking and dynamic correction during long-horizon execution. We evaluate the proposed method in multiple simulation environments and conduct sim-to-real experiments on a real robotic platform. The results demonstrate that explicit language memory improves the success rate and robustness of VLA models on complex long-horizon tasks while providing an interpretable semantic account of the decision process.",
  "published": "2026-08-05",
  "updated": "2026-08-05",
  "year": "2026",
  "authors": [
   "Houze Xu",
   "Jizhong Li",
   "Ziyi Ye"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The proposed hierarchical long-horizon VLA architecture with an explicit language-memory module improves the success rate and robustness of VLA models on complex long-horizon tasks while providing an interpretable semantic account of the decision process.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Houze Xu",
    "id": "153194451",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Jizhong Li",
    "id": "2454490378",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Ziyi Ye",
    "id": "2249050664",
    "h_index": 8,
    "papers": 44
   }
  ],
  "comment": "11 pages, 4 figures",
  "topics": [
   "vla",
   "sim2real",
   "video-generation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.04765v1",
  "pdf_url": "https://arxiv.org/pdf/2608.04765v1",
  "html_url": "https://arxiv.org/html/2608.04765v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.04732",
  "slug": "toward-integrating-adaptive-experience-replay-and-online-uncertainty-e",
  "title": "Toward Integrating Adaptive Experience Replay and Online Uncertainty Estimation in Safe Actor-Critic Optimal Control",
  "abstract": "Safe actor-critic control often treats barrier filtering, uncertainty estimation, and experience replay as separate modules, even though each changes the data used for learning and control. We develop an integrated architecture in which the uncertainty estimate updates the obstacle geometry used by a control barrier function, filter interventions and estimation residuals determine replay priority, and the critic learns from the executed rather than nominal action. We instantiate the architecture on a two-dimensional robot-navigation task with corrupted obstacle measurements and compare six component-matched configurations under common training budgets, random seeds, sensor streams, exploration, and disturbances. Evaluation includes a moderate post-training test, an eleven-level perception-noise sweep, and an exploratory extreme-stress test at multiplier $6.0$. In the extreme test, the integrated configuration recorded no contacts and reached the goal in all five evaluation seeds. Its mean cost was $7.63\\pm0.44$ and its obstacle-belief root-mean-square error was $3.52\\pm0.55$ cm. The uncertainty-estimation ablation also recorded no contacts but reached the goal in four of five seeds, with mean cost $8.96\\pm2.08$ and belief error $11.08\\pm1.23$ cm. A finite-training bound clarifies replay exposure, and a robust barrier condition states the required estimation-error and feasibility assumptions. The results support coupling estimation, safety filtering, and replay on this benchmark; broader safety and convergence claims require further study.",
  "published": "2026-08-05",
  "updated": "2026-08-05",
  "year": "2026",
  "authors": [
   "Mahshad Rastegarmoghaddam",
   "Davoud Nikkhouy",
   "Shima Samadzadeh"
  ],
  "author_count": 3,
  "categories": [
   "eess.SY",
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "eess.SY",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "An integrated architecture in which the uncertainty estimate updates the obstacle geometry used by a control barrier function, filter interventions and estimation residuals determine replay priority, and the critic learns from the executed rather than nominal action is developed.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Mahshad Rastegarmoghaddam",
    "id": "2376342944",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Davoud Nikkhouy",
    "id": "2376343364",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Shima Samadzadeh",
    "id": "2370313056",
    "h_index": 1,
    "papers": 4
   }
  ],
  "comment": "Code, deterministic seeds, data, figures, and protocol files are archived at https://doi.org/10.5281/zenodo.21515850 and https://github.com/SDNT8810/safe-actor-critic-aer-ue-reproducibility",
  "topics": [
   "rl-control",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.04732v1",
  "pdf_url": "https://arxiv.org/pdf/2608.04732v1",
  "html_url": "https://arxiv.org/html/2608.04732v1",
  "code_url": "https://github.com/SDNT8810/safe-actor-critic-aer-ue-reproducibility",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2608.04723",
  "slug": "a-vision-based-control-framework-for-real-time-autonomous-uuv-operatio",
  "title": "A Vision-based Control Framework for Real-time Autonomous UUV Operations",
  "abstract": "This paper presents a fully integrated vision-based framework for real-time and robust localization, autonomous navigation, and mapping for unmanned underwater vehicles (UUVs) in dynamic, visually challenging environments. The proposed pipeline enables both net-relative and global localization while generating continuous 3D maps of the surroundings in real-time. The framework was validated on synthetic datasets with ground truth and tested onboard an UUV during autonomous net-relative navigation experiments. Results demonstrate real-time performance and enhanced robustness, supporting vision-driven autonomous navigation and enabling the field deployment of marine robots for critical inspection and mapping tasks in complex underwater environments.",
  "published": "2026-08-05",
  "updated": "2026-08-05",
  "year": "2026",
  "authors": [
   "Erik Tj\u00e6rand Fr\u00f8land",
   "Marco Job",
   "Md Ether Deowan",
   "Eleni Kelasidi"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper presents a fully integrated vision-based framework for real-time and robust localization, autonomous navigation, and mapping for unmanned underwater vehicles (UUVs) in dynamic, visually challenging environments and demonstrates real-time performance and enhanced robustness.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Erik Tj\u00e6rand Fr\u00f8land",
    "id": "2455577883",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Marco Job",
    "id": "2323735394",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "M. E. Deowan",
    "id": "2158512697",
    "h_index": 3,
    "papers": 13
   },
   {
    "name": "Eleni Kelasidi",
    "id": "2322501765",
    "h_index": 2,
    "papers": 5
   }
  ],
  "comment": "Accepted to IFAC WC 2026",
  "topics": [
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.04723v1",
  "pdf_url": "https://arxiv.org/pdf/2608.04723v1",
  "html_url": "https://arxiv.org/html/2608.04723v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.04692",
  "slug": "suppression-sticks-locality-is-fragile-a-closed-loop-target-and-contro",
  "title": "Suppression Sticks, Locality Is Fragile: A Closed-Loop Target-and-Control Audit of Task-Vector Negation in VLA Policies",
  "abstract": "Task-vector arithmetic offers a closed-form way to modify a model, yet its behavioral locality remains unclear in closed-loop robot control. We present a target-and-control audit of per-skill task-vector subtraction from multitask vision-language-action (VLA) policies. Across all ten LIBERO-Goal skills, subtraction produces three qualitatively different regimes: target-control separation for five skills, resistance for three, and global collapse for two. On held-out initial states, the five suppressible targets remain at 0% success; however, mean baseline-normalized control retention is only 52%, and each target-suppressing edit materially harms at least one nominally unrelated control. Additional Goal panels show separation across tested policies with continuous-regression, discrete-token, and flow-matching action heads, whereas we observe no clean separation on Spatial and control collapse on the tested Object and Long-horizon panels. Mean task-vector cosine does not account for this variation. A matched-norm control identifies a local sign asymmetry around one Goal anchor, while multi-vector outcomes vary with anchor and scale. Retain-aware gradient baselines provide data-dependent comparators but require removal-time data and optimization; subtraction is data- and gradient-free only at edit time, assuming precomputed expert deltas. Finally, a single-skill relearning probe is consistent with behavioral masking, not certified unlearning. These results characterize task-vector subtraction as a fast but brittle intervention and underscore the need for closed-loop target-and-control evaluation when assessing locality in embodied model editing.",
  "published": "2026-08-05",
  "updated": "2026-08-05",
  "year": "2026",
  "authors": [
   "Shaoguang Wang",
   "Weiyu Guo",
   "Rushi Dai",
   "Yiren Zhao",
   "Yandong Guo",
   "Hui Xiong"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A target-and-control audit of per-skill task-vector subtraction from multitask vision-language-action policies underscores the need for closed-loop target-and-control evaluation when assessing locality in embodied model editing.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Shaoguang Wang",
    "id": "2347552474",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Weiyu Guo",
    "id": "2257320765",
    "h_index": 7,
    "papers": 27
   },
   {
    "name": "Rushi Dai",
    "id": "2455447767",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yiren Zhao",
    "id": "2109919449",
    "h_index": 21,
    "papers": 88
   },
   {
    "name": "Yandong Guo",
    "id": "2284207445",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Hui Xiong",
    "id": "2346984163",
    "h_index": 3,
    "papers": 13
   }
  ],
  "comment": "28 pages, 14 figures, 40 tables. Preprint",
  "topics": [
   "vla",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.04692v1",
  "pdf_url": "https://arxiv.org/pdf/2608.04692v1",
  "html_url": "https://arxiv.org/html/2608.04692v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.04653",
  "slug": "overcoming-statistical-bias-in-action-controllable-world-models",
  "title": "Overcoming Statistical Bias in Action-Controllable World Models",
  "abstract": "Action-conditioned world models aim to predict how visual environments evolve under an agent's actions. Yet future frames are often highly predictable from visual inertia and recurring motion patterns alone. This creates a shortcut: models can fit the data by exploiting statistical biases without making their visible dynamics meaningfully depend on the action. As a result, different actions may produce similar futures, while motion may persist even under zero action. The key question is how to reduce reliance on statistical shortcuts from dominating action-conditioned prediction. We argue that action control requires more than injecting action features; it requires enforcing consistency under counterfactual changes to actions and observations. Based on this insight, we introduce CoCo, a Counterfactual Consistency framework to enhance action controllability through two complementary constraints. Multi-step counterfactual consistency constrains reference, inverse-action, and zero-action rollouts, while action-spatial counterfactual consistency enforces consistent predictions under mirrored scenes and transformed actions. Together, they reduce reliance on statistical shortcuts from substituting for action-dependent dynamics. We further introduce Action Response Consistency (ARC) and Drift Energy (DE) to assess action controllability, together with Mini-SSMB for same-state, multi-action counterfactual evaluation. On Mini-SSMB, our full model achieved ARC_inv of 0.412 and ARC_ref of 0.483, while reducing DE by 17.07% relative to the baseline. On VP2 visual planning, it achieves the highest average success rate among SOTA models, at 73.1%. Experiments on BAIR and RoboNet further show that these gains preserve video prediction quality and transfer across model settings.",
  "published": "2026-08-05",
  "updated": "2026-08-05",
  "year": "2026",
  "authors": [
   "Yuhong Shi",
   "Zhenhao Chu",
   "Jie Wei",
   "Jun Hao",
   "Jianyi Liu",
   "Jingwen Fu"
  ],
  "author_count": 6,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Co is introduced, a Counterfactual Consistency framework to enhance action controllability through two complementary constraints: Multi-step counterfactual consistency constrains reference, inverse-action, and zero-action rollouts, while action-spatial counterfactual consistency enforces consistent predictions under mirrored scenes and transformed actions.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yuhong Shi",
    "id": "2308038002",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Zhenhao Chu",
    "id": "2455574161",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Jie Wei",
    "id": "2447561686",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Jun Hao",
    "id": "2455595650",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Jianyi Liu",
    "id": "2284130439",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Jingwen Fu",
    "id": "2312017817",
    "h_index": 5,
    "papers": 24
   }
  ],
  "comment": "",
  "topics": [
   "world-models"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.04653v1",
  "pdf_url": "https://arxiv.org/pdf/2608.04653v1",
  "html_url": "https://arxiv.org/html/2608.04653v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.04633",
  "slug": "mind-vla-instruction-aware-spatial-representation-alignment-for-vision",
  "title": "Mind-VLA: Instruction-Aware Spatial Representation Alignment for Vision-Language-Action Models",
  "abstract": "Recent Vision-Language-Action (VLA) methods improve generalization by aligning their representations with 3D scene geometry. However, these methods are fundamentally instruction-agnostic: the representations align the entire scene uniformly, neglecting the 3D geometry of the specific target object designated by the language instruction. This causes failures on fine-grained manipulation and target occlusion tasks, where success depends on accurate 3D understanding of the target object rather than the entire scene. To address this, we present Mind-VLA, an instruction-aware spatial representation alignment method for VLA models. Specifically, Mind-VLA first obtains the target object specified by the language instruction, then prepares its target-object tri-view and extracts the corresponding VAE and VGGT features. Finally, the latent representation of the VLA model is aligned with these features to enable instruction-aware 3D understanding. Mind-VLA reaches 93.9% on LIBERO and 4.47 on CALVIN with a compact 345M-parameter backbone. On real-robot tasks with target occlusion, Mind-VLA reaches 54% average success, outperforming the best-performing instruction-agnostic method in real-robot comparison by 32 percentage points. Code will be publicly available.",
  "published": "2026-08-05",
  "updated": "2026-08-05",
  "year": "2026",
  "authors": [
   "Xingyu Ding",
   "Yuzhong Zhao",
   "Yang Wu",
   "Chaoyang Zhao",
   "Chunhai Zhao",
   "Yifan Zhang",
   "Jian Cheng"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Mind-VLA is presented, an instruction-aware spatial representation alignment method for VLA models that obtains the target object specified by the language instruction, then prepares its target-object tri-view and extracts the corresponding VAE and VGGT features to enable instruction-aware 3D understanding.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "X. Ding",
    "id": "2455597432",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yuzhong Zhao",
    "id": "2279864665",
    "h_index": 7,
    "papers": 15
   },
   {
    "name": "Yang Wu",
    "id": "2453814073",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Chaoyang Zhao",
    "id": "2349548443",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Chunhai Zhao",
    "id": "2455656321",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yifan Zhang",
    "id": "2454379449",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Jian Cheng",
    "id": "2163382256",
    "h_index": 5,
    "papers": 14
   }
  ],
  "comment": "",
  "topics": [
   "vla"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.04633v1",
  "pdf_url": "https://arxiv.org/pdf/2608.04633v1",
  "html_url": "https://arxiv.org/html/2608.04633v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.04612",
  "slug": "gasp-gpu-accelerated-safe-planner-for-real-time-collision-aware-motion",
  "title": "GASP: GPU-Accelerated Safe Planner for Real-Time Collision-Aware Motion Generation with Latent Trajectory Sampling",
  "abstract": "We present GASP, a GPU-Accelerated Safe Planner for real-time, collision-aware joint-space motion generation in known environments. GASP combines a clamped B-spline trajectory parameterization with a convolutional residual neural network that predicts the free interior control points, while analytically inserted boundary control points enforce initial and final derivative constraints for collision-aware planning under non-stationary conditions. A conditional variational autoencoder samples multiple trajectory candidates, which are decoded and validated in parallel on the GPU, yielding a batched planner for collision-aware coupled joint-space motion with near-millisecond inference. We validate GASP as an online motion-generation module, where it achieves analytical-level success rates with high collision-aware feasibility and substantially reduces inference time relative to GPU-based trajectory optimization. We further deploy GASP as a reinforcement-learning reset planner in competitive robotic table tennis, matching the baseline return rate while roughly halving training-time collisions.",
  "published": "2026-08-05",
  "updated": "2026-08-05",
  "year": "2026",
  "authors": [
   "Colin Merk",
   "Stefanos Charalambous",
   "Peter D\u00fcrr",
   "Farshad Khadivar"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "GASP combines a clamped B-spline trajectory parameterization with a convolutional residual neural network that predicts the free interior control points, while analytically inserted boundary control points enforce initial and final derivative constraints for collision-aware planning under non-stationary conditions.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Colin Merk",
    "id": "2377313211",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "S. Charalambous",
    "id": "91880520",
    "h_index": 19,
    "papers": 160
   },
   {
    "name": "Peter D\u00fcrr",
    "id": "2066182179",
    "h_index": 11,
    "papers": 27
   },
   {
    "name": "Farshad Khadivar",
    "id": "1505653722",
    "h_index": 9,
    "papers": 18
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.04612v1",
  "pdf_url": "https://arxiv.org/pdf/2608.04612v1",
  "html_url": "https://arxiv.org/html/2608.04612v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.04527",
  "slug": "retrieve-in-time-correct-in-frequency",
  "title": "Retrieve in Time, Correct in Frequency",
  "abstract": "Frozen vision-language-action (VLA) policies generate temporally extended action chunks, but long-horizon manipulation remains vulnerable to accumulated execution error and visual aliasing across task stages. Successful rollouts provide useful corrective evidence, yet current frame retrieval can return progress-misaligned actions,while direct replay or time-domain fusion can overwrite the reactive structure of the policy proposal. We introduce Retrieve in Time, Correct in Frequency (RTCF), a training-free test-time correction framework that improves frozen VLA performance with low model-side overhead.RTCF separates which experience to retrieve from which part of its action to transfer. Progressive Memory Alignment (PMA) causally aligns the growing visual execution history with complete successful trajectories through incrementally updated monotonic frontiers, jointly identifying a relevant memory and the current aligned memory position without stage labels. From the aligned action chunk,RTCF transfers a coefficient-wise-clipped low-frequency residual on motion channels. Higher-frequency components and gripper decisions remain inherited from the frozen policy. Across four LIBERO suites and 2,000 episodes per condition, RTCF raises aggregate success from 86.4% to 88.4% and improves LIBERO-Long from 61.6% to 68.6%.These gains require no parameter updates, repeated VLA inference, or additional GPU resources: correction can be performed on the client CPU after a single policy invocation, and the median latencies sum to only 10.99 ms per action chunk",
  "published": "2026-08-05",
  "updated": "2026-08-07",
  "year": "2026",
  "authors": [
   "Yuze Fan",
   "Yue Cao",
   "Pengjie Gao",
   "Haojia Gao",
   "Guangqiu Guo",
   "Ziyue Zhang",
   "Junbo Tan",
   "Bokui Chen",
   "Zhuo Zou",
   "Xueqian Wang"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Retrieve in Time, Correct in Frequency (RTCF), a training-free test-time correction framework that improves frozen VLA performance with low model-side overhead and requires no parameter updates, repeated VLA inference, or additional GPU resources.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yuze Fan",
    "id": "2298235010",
    "h_index": 8,
    "papers": 21
   },
   {
    "name": "Yue Cao",
    "id": "2455664620",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Pengjie Gao",
    "id": "2455615387",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Haojia Gao",
    "id": "2331263068",
    "h_index": 2,
    "papers": 15
   },
   {
    "name": "Guangqiu Guo",
    "id": "2455578376",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Ziyue Zhang",
    "id": "2454699654",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Junbo Tan",
    "id": "5653211",
    "h_index": 15,
    "papers": 83
   },
   {
    "name": "Bokui Chen",
    "id": "2366470843",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Zhuowu Zou",
    "id": "2419646039",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Xueqian Wang",
    "id": "2337851514",
    "h_index": 1,
    "papers": 18
   }
  ],
  "comment": "",
  "topics": [
   "vla"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.04527v2",
  "pdf_url": "https://arxiv.org/pdf/2608.04527v2",
  "html_url": "https://arxiv.org/html/2608.04527v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.04510",
  "slug": "guard-grounding-uncertainty-and-ablation-based-risk-detection-for-diff",
  "title": "GUARD: Grounding Uncertainty and Ablation-Based Risk Detection for Diffusion-Based VLAs",
  "abstract": "Diffusion-based vision-language-action (VLA) policies can generate plausible actions even when their predictions are weakly grounded in the visual and language evidence defining the task. We introduce GUARD, a test-time failure detection method that measures this grounding without modifying the pretrained policy. GUARD estimates the influence of token-indexed entries in the final vision-language model key-value (KV) cache, constructs counterfactual caches by ablating salient KV entries, and compares their denoising responses with the original conditioning. Based on the comparison, we derive GUARD diagnostic stream including sensitivity, attention entropy, modality bias, and grounding efficiency, which are calibrated online and processed by a lightweight temporal classifier. We evaluate GUARD under task-held-out splits across five policy-benchmark settings, using Pi0, SmolVLA, and Alpamayo-1.5 on LIBERO, SimplerEnv, MetaWorld, and PhysicalAI-AV. GUARD achieves the best ROC-AUC on four of five unseen-task settings and ranks second on the remaining setting, improving the average unseen-task ROC-AUC by 5.73 percentage points over the strongest competing runtime monitor while remaining within 0.19 points of the best seen-task average. These results show that directly probing action-head dependence on multimodal evidence provides a transferable failure signal across policies, tasks, embodiments, and domains.",
  "published": "2026-08-05",
  "updated": "2026-08-05",
  "year": "2026",
  "authors": [
   "Suhas Hegde",
   "Jitendra Yasaswi Bharadwaj Katta"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The results show that directly probing action-head dependence on multimodal evidence provides a transferable failure signal across policies, tasks, embodiments, and domains.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Suhas G. Hegde",
    "id": "2352278242",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Jitendra Y. Katta",
    "id": "2189753785",
    "h_index": 1,
    "papers": 2
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "video-generation",
   "safety-eval"
  ],
  "orgs": [
   "Physical Intelligence"
  ],
  "abs_url": "https://arxiv.org/abs/2608.04510v1",
  "pdf_url": "https://arxiv.org/pdf/2608.04510v1",
  "html_url": "https://arxiv.org/html/2608.04510v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2608.04420",
  "slug": "scope-field-of-view-aware-path-planning-in-unknown-3d-environments-via",
  "title": "SCOPE: Field-of-View-Aware Path Planning in Unknown 3D Environments via Safety-Volume Certification",
  "abstract": "Safe navigation with a body-mounted limited-field-of-view sensor requires the complete robot-inflated volume of an intended motion to be observed and verified free before execution. We formulate this requirement as online safety-volume certification in an unknown voxel map and construct a certified graph whose vertices correspond exactly to positions with fully known-free safety volumes. Based on this representation, we propose SCOPE (Safety Certification through Observation Planning and Execution), a planning framework that decouples optimistic goal-directed guidance from certified execution. SCOPE converts the first uncertified point along an optimistic route into an explicit observation obligation, resolves it through target-centric viewpoint search, and recursively clears intermediate obligations when useful viewpoints are not yet certified-reachable. A certified preview mechanism and an observation-aware trajectory optimization backend enable smooth execution. We prove conditional complete planning: under ideal monotone sensing and exhaustive finite-domain graph search, SCOPE reaches the goal whenever a finite feasible sequence of certified sensing actions exists within its planning primitives. Across 60 randomized tasks in three unknown 3D environments, SCOPE reaches every goal while maintaining near-zero entry into non-certified inflated space. Preview reduces mean mission time by 27%, and real-robot demonstrations in two representative scenarios validate the complete system.",
  "published": "2026-08-05",
  "updated": "2026-08-05",
  "year": "2026",
  "authors": [
   "Junbin Yuan",
   "Muqing Cao",
   "Yunwoo Lee",
   "Brady Moon",
   "Sebastian Scherer"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "It is proved conditional complete planning: under ideal monotone sensing and exhaustive finite-domain graph search, SCOPE reaches the goal whenever a finite feasible sequence of certified sensing actions exists within its planning primitives.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Junbin Yuan",
    "id": "2347000063",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Muqing Cao",
    "id": "2350759312",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Yunwoo Lee",
    "id": "31130931",
    "h_index": 8,
    "papers": 28
   },
   {
    "name": "Brady G. Moon",
    "id": "2322505161",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Sebastian A. Scherer",
    "id": "2280069515",
    "h_index": 3,
    "papers": 10
   }
  ],
  "comment": "Project website: https://yuanjunbin.github.io/scope-planner/",
  "topics": [
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.04420v1",
  "pdf_url": "https://arxiv.org/pdf/2608.04420v1",
  "html_url": "https://arxiv.org/html/2608.04420v1",
  "code_url": "https://yuanjunbin.github.io/scope-planner/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2608.04309",
  "slug": "structured-llm-reasoning-for-zero-shot-human-robot-coordination-under",
  "title": "Structured LLM Reasoning for Zero-Shot Human--Robot Coordination Under Hidden Goals",
  "abstract": "We present a structured large-language-model (LLM) architecture for zero-shot human--robot coordination in a cooperative construction task with private goal views. Guided by a Dec-POMDP formulation, the architecture decomposes decision-making into (i) action-conditioned Theory-of-Mind (ToM) inference, (ii) hierarchical planning, (iii) conversation interpretation, (iv) action verification, and (v) feedback-based replanning. We compare the proposed method with an ablation without ToM inference and a multi-agent reinforcement-learning policy trained offline over many goal pairs. In human-participant experiments, the proposed method required fewer interaction steps and yielded higher post-interaction trust ratings than both baselines. These results suggest that systematically decomposing the team decision problem, using LLMs as tractable surrogates for otherwise intractable inference and planning computations, and retaining conventional verification for physical feasibility can improve both task coordination and the human experience.",
  "published": "2026-08-05",
  "updated": "2026-08-05",
  "year": "2026",
  "authors": [
   "Dong Hae Mangalindan",
   "Anand Gokhale",
   "Francesco Bullo",
   "Vaibhav Srivastava"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The results suggest that systematically decomposing the team decision problem, using LLMs as tractable surrogates for otherwise intractable inference and planning computations, and retaining conventional verification for physical feasibility can improve both task coordination and the human experience.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Dong Hae Mangalindan",
    "id": "2221120366",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Anand Gokhale",
    "id": "2246832796",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Francesco Bullo",
    "id": "2265493221",
    "h_index": 6,
    "papers": 23
   },
   {
    "name": "Vaibhav Srivastava",
    "id": "2303847756",
    "h_index": 2,
    "papers": 8
   }
  ],
  "comment": "",
  "topics": [
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.04309v1",
  "pdf_url": "https://arxiv.org/pdf/2608.04309v1",
  "html_url": "https://arxiv.org/html/2608.04309v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.07573",
  "slug": "projection-retraction-mppi-exact-constraint-manifold-control-for-manip",
  "title": "Projection-Retraction MPPI: Exact Constraint-Manifold Control for Manipulators",
  "abstract": "Model Predictive Path Integral (MPPI) control is widely used in manipulation for its gradient-free, parallel handling of non-convex costs. Manipulation tasks, however, often impose constraints that hold throughout the motion: a closed kinematic chain that two grasping arms keep exactly, or joint limits and obstacle clearances that are never crossed. MPPI handles such constraints only through the cost, as soft penalties that hold approximately and fail under a strong task cost. To address this, we propose Projection-Retraction MPPI (PR-MPPI), which enforces the constraints inside the sampled dynamics. At every rollout step, the sampled velocity is projected to satisfy both constraint types: the equality restricts it to a subspace, and each inequality to a half-space within that subspace, so inequality handling never breaks the equality. This projection, however, satisfies the constraints only to first order, and a finite step leaves a small drift off the equality. Therefore, we retract the returned command back onto the constraint to numerical tolerance and independent of task weighting. We validate PR-MPPI on 14-DoF dual-arm systems. In simulation, the returned commands satisfy the closed-chain equality to numerical tolerance through a joint-limit stress test and randomized obstacle avoidance. On real hardware, the arms of a Unitree H1-2 humanoid reactively avoid a moving obstacle. Code and experiment videos are available at https://rcilab.github.io/prmppi.",
  "published": "2026-08-04",
  "updated": "2026-08-04",
  "year": "2026",
  "authors": [
   "Seulchan Lee",
   "Leesai Park",
   "Minhyeong Kang",
   "Sanghyun Kim"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes Projection-Retraction MPPI (PR-MPPI), which enforces the constraints inside the sampled dynamics inside the sampled dynamics, and validate PR-MPPI on 14-DoF dual-arm systems.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Seulchan Lee",
    "id": "2314335325",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Leesai Park",
    "id": "2314263170",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Minhyeong Kang",
    "id": "2456632781",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Sanghyun Kim",
    "id": "2314327043",
    "h_index": 2,
    "papers": 7
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "humanoids",
   "navigation"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2608.07573v1",
  "pdf_url": "https://arxiv.org/pdf/2608.07573v1",
  "html_url": "https://arxiv.org/html/2608.07573v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2608.04246",
  "slug": "safecast-robust-failure-detection-for-vla-policies-with-contrast-set-t",
  "title": "SAFECAST: Robust Failure Detection for VLA Policies with Contrast-Set Training and Calibration",
  "abstract": "Vision-language-action policies often fail under deployment-time distribution shifts such as clutter, distractor objects, lighting changes, novel objects, altered initial states, and reworded instructions. Hidden-state-based risk probes combined with functional conformal prediction can detect rollout failures, but their reliability depends on calibration data matching deployment conditions. We introduce SAFECAST, which leverages contrast set perturbations to improve hidden-state probe training and calibration for deployment time shift. SAFECAST statistically significantly improves failure detection ROC-AUC scores over a state of the art baseline in both real-world DROID and LIBERO simulation experiments across multiple VLM backbones. We further find that SAFECAST benefits most when both visual and language contrast set perturbations are used to augment data, and that with contrast set perturbations, sim-to-real calibration leads to better probes than using real rollout data only.",
  "published": "2026-08-04",
  "updated": "2026-08-04",
  "year": "2026",
  "authors": [
   "Harshitha Rajaprakash",
   "Aditeya Prajapati",
   "Rong Xue",
   "Abrar Anwar",
   "Jesse Thomason"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work introduces SAFECAST, which leverages contrast set perturbations to improve hidden-state probe training and calibration for deployment time shift and finds that SAFECAST benefits most when both visual and language contrast set perturbations are used to augment data.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Harshitha Rajaprakash",
    "id": "2363578636",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Aditeya Prajapati",
    "id": "2442354641",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Rong Xue",
    "id": "2382928865",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Abrar Anwar",
    "id": "2266391775",
    "h_index": 9,
    "papers": 21
   },
   {
    "name": "Jesse Thomason",
    "id": "2266392436",
    "h_index": 4,
    "papers": 13
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "sim2real",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.04246v1",
  "pdf_url": "https://arxiv.org/pdf/2608.04246v1",
  "html_url": "https://arxiv.org/html/2608.04246v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.04242",
  "slug": "feasibility-of-embedded-photoplethysmography-sensing-in-short-duration",
  "title": "Feasibility of Embedded Photoplethysmography Sensing in Short-Duration Tactile Interactions With Pocket-Sized Robots Using IMU- and Confidence-Based Filtering",
  "abstract": "Ubiquitous companion robots offer a promising avenue for immediate anxiety relief in children, yet their effectiveness relies on the ability to monitor physiological states continuously and unobtrusively. Current solutions often depend on external wearables, which impose usability barriers and limit the robot's autonomy. This paper investigates the integration of an embedded photoplethysmography (PPG) sensor directly into a pocket-sized companion robot, AffectaPocket, to enable self-contained heart rate monitoring during tactile interaction. We address the significant challenge of motion artifacts inherent in handheld usage by implementing a two-stage filtering pipeline that utilizes an onboard Inertial Measurement Unit (IMU) to reject high-variance segments and a confidence-based smoothing algorithm for recovery periods. We evaluated the system against a commonly used wrist worn sensor in a Within-Subjects Study with 26 participants. Our results demonstrate that the filtering strategy significantly reduced the Mean Absolute Percentage Error and achieved statistical equivalence to the ground truth measurements (p<0.05). Analysis of short-duration interactions shows that the sensor requires stability over longer periods to converge.",
  "published": "2026-08-04",
  "updated": "2026-08-04",
  "year": "2026",
  "authors": [
   "Turjja Datta",
   "Morten Roed Frederiksen"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper investigates the integration of an embedded photoplethysmography sensor directly into a pocket-sized companion robot to enable self-contained heart rate monitoring during tactile interaction and addresses the significant challenge of motion artifacts inherent in handheld usage by implementing a two-stage filtering pipeline.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Turjja Datta",
    "id": "2008244621",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Morten Roed Frederiksen",
    "id": "89219216",
    "h_index": 4,
    "papers": 23
   }
  ],
  "comment": "8 pages, 3 figures, 4 tables",
  "topics": [
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.04242v1",
  "pdf_url": "https://arxiv.org/pdf/2608.04242v1",
  "html_url": "https://arxiv.org/html/2608.04242v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.04196",
  "slug": "simdex-mining-similar-egocentric-videos-for-cross-embodiment-dexterous",
  "title": "SiMDex: Mining Similar Egocentric Videos for Cross-Embodiment Dexterous Manipulation",
  "abstract": "Recent years have witnessed an explosive trend of scaling ego-centric human videos for robot manipulation, yet it remains unclear which data actually benefits dexterous manipulation. We present SiMDex, a similarity-based data mining framework that casts human data selection for VLA post-training in dexterous manipulation as a recommendation problem. For each robot demonstration, SiMDex employs a three-layer recall-ranking-re-ranking pipeline to extract task-relevant subsets from a pool of ~32M egocentric human samples, operating in a morphology-agnostic action space that requires no changes to VLA architecture or training. Against a strong baseline trained with an equal amount of randomly sampled human data, SiMDex uses only ~1.49M mined samples (<5% of the pool) yet improves the overall success rate from 47.7% to 61.1%, showing that selective curation outperforms indiscriminate data mixing.",
  "published": "2026-08-04",
  "updated": "2026-08-04",
  "year": "2026",
  "authors": [
   "Nie Lin",
   "Takehiko Ohkawa",
   "Sijin Chen",
   "Ruoshi Wen",
   "Zhuohang Li",
   "Liqun Huang",
   "Zhengming Zhu",
   "Yiming Bao",
   "Yunfei Li",
   "Minjie Cai",
   "Xiao Ma",
   "Wei Xu",
   "Yoichi Sato"
  ],
  "author_count": 13,
  "categories": [
   "cs.RO",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "SiMDex is presented, a similarity-based data mining framework that casts human data selection for VLA post-training in dexterous manipulation as a recommendation problem, and shows that selective curation outperforms indiscriminate data mixing.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Nie Lin",
    "id": "2293324964",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Takehiko Ohkawa",
    "id": "1515548135",
    "h_index": 8,
    "papers": 21
   },
   {
    "name": "Sijin Chen",
    "id": "2374373837",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Ruoshi Wen",
    "id": "30932282",
    "h_index": 8,
    "papers": 19
   },
   {
    "name": "Zhuohang Li",
    "id": "2290740132",
    "h_index": 4,
    "papers": 17
   },
   {
    "name": "Liqun Huang",
    "id": "2373416739",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Zhengming Zhu",
    "id": "2374086349",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Yiming Bao",
    "id": "2455592655",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yunfei Li",
    "id": "2401852871",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Minjie Cai",
    "id": "3172280",
    "h_index": 14,
    "papers": 27
   },
   {
    "name": "Xiao Ma",
    "id": "2125110703",
    "h_index": 13,
    "papers": 45
   },
   {
    "name": "Wei Xu",
    "id": "2373713682",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Yoichi Sato",
    "id": "2268723767",
    "h_index": 4,
    "papers": 14
   }
  ],
  "comment": "12 pages, 4 figures. Project page: https://lin-nie.github.io/SiMDex/",
  "topics": [
   "vla",
   "dexterous-manipulation",
   "egocentric-data",
   "foundation-pretraining",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.04196v1",
  "pdf_url": "https://arxiv.org/pdf/2608.04196v1",
  "html_url": "https://arxiv.org/html/2608.04196v1",
  "code_url": "https://lin-nie.github.io/SiMDex/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2608.04043",
  "slug": "tactus-open-vocabulary-object-recognition-from-low-cost-pressure-array",
  "title": "Tactus: Open-Vocabulary Object Recognition from Low-Cost Pressure Arrays",
  "abstract": "Resistive pressure arrays are the cheapest and most widely shipped tactile sensors, yet tactile representation learning has concentrated on optical sensors that image a deforming gel. We present Tactus, an open model that answers text queries from pressure data alone: on the STAG benchmark (27 objects, held-out recordings), it reaches 0.771 +/- 0.062 top-1 over four runs (top-3 0.935), matching, and at best exceeding, the dataset's supervised closed-set CNN at 0.76, with no trained classifier head. The recipe is small-data: 187 training recordings, masked-autoencoder pretraining on 144k unlabeled same-sensor frames, and the sensor's own calibration affine, which recovered more accuracy than every architecture change combined. The released model's errors concentrate in a few contact-ambiguous classes, are uncorrelated with text-target geometry (Spearman rho <= 0.05 over 702 class pairs), and survive paraphrased and even bare-name queries within one point; two diverse frames recover 89% of eight-frame accuracy. Failures are reported with equal precision: cross-sensor pretraining pooling gave no gain, vision co-training degraded touch, and a mis-normalized input pipeline silently discarded 97% of the sensor's dynamic range while producing plausible intermediate results. Weights, code, and the memory layer the model plugs into are released openly.",
  "published": "2026-08-04",
  "updated": "2026-08-04",
  "year": "2026",
  "authors": [
   "Abdul Basit Tonmoy"
  ],
  "author_count": 1,
  "categories": [
   "cs.LG",
   "cs.RO"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Tactus is an open model that answers text queries from pressure data alone: on the STAG benchmark, it reaches 0.771 +/- 0.062 top-1 over four runs, matching, and at best exceeding, the dataset's supervised closed-set CNN at 0.76, with no trained classifier head.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Abdul Basit Tonmoy",
    "id": "2342702745",
    "h_index": 1,
    "papers": 4
   }
  ],
  "comment": "6 pages, 4 figures. Weights: https://huggingface.co/EximiusLabs/fusion-embedding-2-tactus",
  "topics": [
   "tactile",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.04043v1",
  "pdf_url": "https://arxiv.org/pdf/2608.04043v1",
  "html_url": "https://arxiv.org/html/2608.04043v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.04042",
  "slug": "kitchen-robotic-manipulation-utilizing-foundation-models",
  "title": "Kitchen Robotic Manipulation utilizing Foundation Models",
  "abstract": "Deploying robots in everyday human environments requires perception systems that are both robust and adaptable to diverse, dynamic conditions. In this work, we present a modular perception pipeline for household manipulation tasks, with a focus on dishware handling in kitchen environments. The pipeline integrates open-vocabulary object detection, multi-view segmentation, instance-aware 3D reconstruction, and a 2D-3D feature fusion strategy for 6D pose estimation and grasp planning. Its modular design enables systematic substitution of multiple visual and geometric foundation models, allowing us to identify the best-performing configuration through extensive evaluation on a custom kitchen dataset. The best-performing configuration (LLMDet + SAMv2 + DINOv2 + GeoTransformer) achieves an ADI of 89.12\\% on the 20-scene kitchen benchmark with cluttered and occluded conditions. Furthermore, real-world demonstrations confirm that the best configuration can be deployed on physical robots without environment-specific retraining, successfully executing tasks such as sink-to-dishwasher transfer and cup stacking. It validates the adaptability and scalability of the pipeline and highlights its potential as a practical framework for household robotic systems. Our code and supplementary materials are available at https://raivlab.github.io/FM_kitchen .",
  "published": "2026-08-04",
  "updated": "2026-08-04",
  "year": "2026",
  "authors": [
   "Myung-Hwan Jeon",
   "Sankalp Yamsani",
   "Joohyung Kim"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A modular perception pipeline for household manipulation tasks, with a focus on dishware handling in kitchen environments, that integrates open-vocabulary object detection, multi-view segmentation, instance-aware 3D reconstruction, and a 2D-3D feature fusion strategy for 6D pose estimation and grasp planning is presented.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Myung-Hwan Jeon",
    "id": "151118755",
    "h_index": 9,
    "papers": 21
   },
   {
    "name": "Sankalp Yamsani",
    "id": "2229898158",
    "h_index": 5,
    "papers": 16
   },
   {
    "name": "Joohyung Kim",
    "id": "2266820244",
    "h_index": 5,
    "papers": 26
   }
  ],
  "comment": "Intelligent Service Robotics (ISR)",
  "topics": [
   "dexterous-manipulation",
   "spatial-3d",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.04042v1",
  "pdf_url": "https://arxiv.org/pdf/2608.04042v1",
  "html_url": "https://arxiv.org/html/2608.04042v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.03978",
  "slug": "stochastic-multiple-shooting-trajectory-optimization-via-sequential-lo",
  "title": "Stochastic Multiple Shooting Trajectory Optimization via Sequential Local Policy Evaluation",
  "abstract": "Stochastic single shooting trajectory optimization methods such as Model Predictive Path Integral control (MPPI) have been widely adopted in robotics due to their ability to reason about probabilistic dynamics and provide solutions where model gradients are noisy, costly to evaluate, or unavailable. However, satisfaction of terminal constraints when shooting over long action sequences is often sample inefficient, requiring a large number of iterations for convergence. In this paper, we present a stochastic multiple shooting method that optimizes short control action sequences connected via local feedback policies to improve sample efficiency and convergence to a terminal set. Additionally, we show that we are able to synthesize approximate system Jacobians purely from rollouts, making the method suitable for model-based reinforcement learning with black-box dynamics. We demonstrate the algorithm has improved sample efficiency and terminal set convergence for three nonlinear, underactuated optimization problems: a classic cartpole swingup task with analytical dynamics, a cartpole swingup task with learned neural network dynamics, and a VTOL quadplane performing a high angle-of-attack, precision post-stall landing maneuver.",
  "published": "2026-08-04",
  "updated": "2026-08-04",
  "year": "2026",
  "authors": [
   "Ashwin Gupta",
   "Joseph Moore"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper presents a stochastic multiple shooting method that optimizes short control action sequences connected via local feedback policies to improve sample efficiency and convergence to a terminal set and demonstrates the algorithm has improved sample efficiency and terminal set convergence for three nonlinear, underactuated optimization problems.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ashwin Gupta",
    "id": "2381311801",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Joseph L. Moore",
    "id": "2246509877",
    "h_index": 3,
    "papers": 11
   }
  ],
  "comment": "",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.03978v1",
  "pdf_url": "https://arxiv.org/pdf/2608.03978v1",
  "html_url": "https://arxiv.org/html/2608.03978v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.03938",
  "slug": "bimanual-manipulation-within-an-8-gb-budget-zero-copy-sensing-and-quan",
  "title": "Bimanual Manipulation Within an 8 GB Budget: Zero-Copy Sensing and Quantized ACT on an Entry-Level Jetson",
  "abstract": "Bimanual manipulation policies trained with imitation learning are typically evaluated on workstation or datacenter-class GPUs, leaving the cost of deploying them on embedded hardware largely uncharacterized. We present a bimanual SO-101 system running entirely on an NVIDIA Jetson Orin Nano Super (8 GB), the entry-level tier of NVIDIA's embedded line, using a desktop GPU (RTX 3070) only for offline training, evaluated on pick-and-place of a deformable beanbag. First, we build a GStreamer capture pipeline backed by NVMM buffers that removes redundant host-device copies from three-camera sensing. Contrary to expectation, the conventional path fit the memory budget and dropped no frames; what zero-copy sensing recovers is CPU headroom (peak single-core utilization 98.0% to 77.0%) and worst-case latency (117.31 ms to 101.52 ms). Second, we train ACT and Diffusion Policy on identical demonstrations, each at its own reference budget (100k gradient steps for ACT, 200k for Diffusion Policy). ACT converges to a task-competent policy (19/20 trials) while Diffusion Policy does not converge to a usable one (0/10) even at twice the step count, which we attribute to differing convergence costs rather than an accuracy ceiling. Third, we convert ACT to TensorRT. FP16 reduces mean inference latency from 114.02 ms to 17.93 ms (6.4x) and INT8 to 12.65 ms (9.0x), with task success preserved at all three precisions (19/20, 18/20, 19/20). We report two findings not previously documented for ACT: TensorRT's general-purpose INT8 calibration quantizes the ResNet18 backbone but accepts zero of 145 transformer layers, explaining INT8's negligible size reduction over FP16 (0.9%) despite a further 28% latency gain; and the need for quantization is conditional on ACT's action-chunking configuration, feasible in full precision at n_action_steps = 100 but not at the per-step re-prediction temporal ensembling requires.",
  "published": "2026-08-04",
  "updated": "2026-08-04",
  "year": "2026",
  "authors": [
   "Ekansh Singh",
   "Eva Samuel",
   "Alessandra Reneau",
   "Ryan Schmeelk",
   "Yashvi Gandhi"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A bimanual SO-101 system running entirely on an NVIDIA Jetson Orin Nano Super (8 GB), the entry-level tier of NVIDIA's embedded line, using a desktop GPU only for offline training, evaluated on pick-and-place of a deformable beanbag.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ekansh Singh",
    "id": "2378587978",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Eva Samuel",
    "id": "2377077789",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Alessandra Reneau",
    "id": "2455491942",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Ryan Schmeelk",
    "id": "2455491863",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Y. Gandhi",
    "id": "2428825160",
    "h_index": 0,
    "papers": 2
   }
  ],
  "comment": "9 pages, 8 tables. Work conducted at the Georgia Tech Research Institute (GTRI), Aerospace, Transportation and Advanced Systems Laboratory (ATAS)",
  "topics": [
   "imitation-diffusion"
  ],
  "orgs": [
   "NVIDIA",
   "Georgia Tech"
  ],
  "abs_url": "https://arxiv.org/abs/2608.03938v1",
  "pdf_url": "https://arxiv.org/pdf/2608.03938v1",
  "html_url": "https://arxiv.org/html/2608.03938v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2608.03872",
  "slug": "evohil-self-evolving-reward-and-flow-matched-policy-optimization-for-r",
  "title": "EvoHIL: Self-Evolving Reward and Flow-Matched Policy Optimization for Robust Human-in-the-Loop Reinforcement Learning",
  "abstract": "Human-in-the-loop reinforcement learning (HIL-RL) enables robots to learn contact-rich manipulation from limited real-world interaction, but deployment exposes three coupled limitations: static visual reward models fail under scene changes; independently sampled actions cause temporally inconsistent motion; and vision-based policies remain sensitive to appearance shifts. We present EvoHIL, a unified framework that adapts the reward model, action generator, and visual do main within a staged human-in-the-loop learning process. First, self-evolving reward (SER) adapts the success classifier from human-confirmed positives and provisional weak negatives. Second, Action Flow Stabilization (AFS) generates temporally coherent action chunks through flow matching, grounding policy updates in executed action prefixes and demonstrated behavior. Third, retention-aware offline fine-tuning replays relit interaction data while anchoring the AFS actor-critic to prior behavior, adapting the visual domain without additional robot interaction. Across six manipulation tasks on Franka FR3 and SO-101 arms under a controlled lighting shift, EvoHIL improves task success, agreement with human-confirmation labels, motion smoothness, and completion time relative to human-in-the-loop and imitation baselines.Project page: https://anonymous4366.github.io/EvoHIL/",
  "published": "2026-08-04",
  "updated": "2026-08-04",
  "year": "2026",
  "authors": [
   "Shuoqin Zhang",
   "Tongtong Cheng",
   "Xiru Gao",
   "Jinzhuo Peng",
   "Bin Zheng",
   "Jiahao Tu",
   "Ke Wang",
   "Jia Pan",
   "Zhe Hu",
   "Kai Liu"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "EvoHIL is presented, a unified framework that adapts the reward model, action generator, and visual do main within a staged human-in-the-loop learning process to improve task success, agreement with human-confirmation labels, motion smoothness, and completion time relative to human-in-the-loop and imitation baselines.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Shuoqing Zhang",
    "id": "2317730586",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Tongtong Cheng",
    "id": "2283824157",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Xiru Gao",
    "id": "2438701429",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Jinzhuo Peng",
    "id": "2455571030",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Binjie Zheng",
    "id": "2447718425",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Jiahao Tu",
    "id": "2455492647",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Ke Wang",
    "id": "2446824909",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Jia Pan",
    "id": "2390556259",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Zhe Hu",
    "id": "2111374729",
    "h_index": 11,
    "papers": 23
   },
   {
    "name": "Kai Liu",
    "id": "2377763386",
    "h_index": 4,
    "papers": 9
   }
  ],
  "comment": "",
  "topics": [
   "tactile",
   "rl-control",
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.03872v1",
  "pdf_url": "https://arxiv.org/pdf/2608.03872v1",
  "html_url": "https://arxiv.org/html/2608.03872v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.03753",
  "slug": "gordon-graph-based-object-centric-rewards-for-decomposition-of-long-ho",
  "title": "GORDON: Graph-based Object-centric Rewards for Decomposition of Long-Horizon Manipulation",
  "abstract": "Learning long-horizon manipulation skills with reinforcement learning remains challenging due to the complexity of reward design, the limited guidance of sparse rewards, and the high cost of manual subtask annotation. Visual demonstrations can provide supervision for reward learning, but rewards learned from raw pixels can be brittle and sensitive to visual variation, background appearance, and robot motion. In this work, we propose GORDON, a graph-based object-centric reward learning framework that learns dense rewards from action-free video demonstrations. Each visual scene is represented as a graph of detected objects and spatial relations, and a graph neural network is trained in a self-supervised manner to embed these graphs into a task-aligned latent space. To align the representation with semantic task progress, we introduce an activity-aware weighted pooling mechanism that emphasizes task-relevant objects while masking robot-dominated motion. The dense reward is then computed as distances in the learned latent space of the current state to demonstrated goal configurations, providing a measure of task progress. In long-horizon tasks, the temporal profile of this reward reveals stage-wise object-state transitions, enabling automatic subtask discovery without manual segmentation. The discovered segments are then used to train subtask-specific rewards and specialized policies that are composed sequentially. Experiments on seven manipulation tasks on MAGICAL and ManiSkill3 benchmarks show that our object-centric reward improves reinforcement learning in short-horizon settings and enables successful policy learning in complex long-horizon tasks through automatic decomposition, achieving an average success rate of 74.4% across the long-horizon tasks (on average approximately +35 p.p. vs. best learned baseline and approximately +25 p.p. vs. oracle).",
  "published": "2026-08-04",
  "updated": "2026-08-04",
  "year": "2026",
  "authors": [
   "Andrea Protopapa",
   "Davide Buoso",
   "Francesca Pistilli",
   "Georgia Chalvatzaki",
   "Giuseppe Averta"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes GORDON, a graph-based object-centric reward learning framework that learns dense rewards from action-free video demonstrations that improves reinforcement learning in short-horizon settings and enables successful policy learning in complex long-horizon tasks through automatic decomposition.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Andrea Protopapa",
    "id": "2210857922",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Davide Buoso",
    "id": "2436212907",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Francesca Pistilli",
    "id": "1796267555",
    "h_index": 7,
    "papers": 22
   },
   {
    "name": "Georgia Chalvatzaki",
    "id": "2333945187",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Giuseppe Averta",
    "id": "2256667696",
    "h_index": 9,
    "papers": 31
   }
  ],
  "comment": "9 pages, 6 figures, preprint. Project page: https://andreaprotopapa.github.io/graph-reward-learning/",
  "topics": [
   "rl-control",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.03753v1",
  "pdf_url": "https://arxiv.org/pdf/2608.03753v1",
  "html_url": "https://arxiv.org/html/2608.03753v1",
  "code_url": "https://andreaprotopapa.github.io/graph-reward-learning/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2608.03727",
  "slug": "track4action-distilling-world-centric-3d-tracker-into-vision-language",
  "title": "Track4Action: Distilling World-Centric 3D Tracker into Vision-Language-Action Policies",
  "abstract": "Action labels tell a vision-language-action (VLA) policy which robot commands to imitate, but not how those commands change the 3D world. The aligned demonstration clip contains this missing supervision because its $K$ frame transitions record the geometry, motion, visibility, and camera change produced during the corresponding $K$ actions. We introduce Track4Action, a framework that distills this realized transition from a frozen world-centric 3D tracker into a current-observation VLA policy. During training, Track4World encodes the clip $V_{t:t+K}$ into a pooled tracker feature. Learnable track queries infer this feature from current VLA hidden states, match it in a shared space, and condition a flow-matching action head through a feature-wise gate. The tracker feature only defines the alignment target, so neither the clip nor the tracker is used at deployment. Track4Action reaches 82.3% on zero-shot LIBERO-Plus, improving the alignment-free variant by 7.6 points and LaMP by 3.0 points. It obtains 80.44% and 81.48% on the clean and randomized RoboTwin 2.0 splits, and 67.5% average success across four physical bimanual tasks, 25.0 points above the alignment-free variant. The gains across simulation and physical tasks support action-aligned 3D tracker features as privileged supervision for tracker-free VLA deployment. Our project page is available at https://wing0night.github.io/track4action-project-page.",
  "published": "2026-08-04",
  "updated": "2026-08-04",
  "year": "2026",
  "authors": [
   "Chenyi Wang",
   "Xinkai Wang",
   "Bokai Lin",
   "Jialin Tian",
   "Fucheng Zhang",
   "Cewu Lu",
   "Lixin Yang"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Track4Action is introduced, a framework that distills this realized transition from a frozen world-centric 3D tracker into a current-observation VLA policy, and gains across simulation and physical tasks support action-aligned 3D tracker features as privileged supervision for tracker-free VLA deployment.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Chenyi Wang",
    "id": "2350835958",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Xinkai Wang",
    "id": "2330429786",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Bokai Lin",
    "id": "2260343640",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Jialin Tian",
    "id": "2087131827",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Fugang Zhang",
    "id": "2399175133",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Cewu Lu",
    "id": "2336035401",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Lixin Yang",
    "id": "2111809756",
    "h_index": 16,
    "papers": 41
   }
  ],
  "comment": "",
  "topics": [
   "vla"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.03727v1",
  "pdf_url": "https://arxiv.org/pdf/2608.03727v1",
  "html_url": "https://arxiv.org/html/2608.03727v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.03682",
  "slug": "phyai-real-time-physical-ai-at-the-edge-scalable-rollouts-in-the-cloud",
  "title": "PhyAI: Real-Time Physical AI at the Edge, Scalable Rollouts in the Cloud",
  "abstract": "Physical AI policies require inference throughout their lifecycle, including model evaluation, cloud reinforcement learning rollout, edge GPU serving, and onboard deployment. Although these settings share the same checkpoint and action semantics, they often rely on separate inference programs. To unify them, we build PhyAI, a Physical AI inference engine with a single runtime that keeps architecture-specific conditioning, solver, cache, and output logic in model adapters while sharing graph execution, kernels, memory management, and parallel services. The same codebase runs vision-language-action (VLA) models and world-action models (WAMs) on single or multiple GPUs across onboard, edge, and cloud deployments. We used the adapter interface to add MiniCPM-Robot on the day of its release. PhyAI achieves 1.40x-4.65x speedups over the official implementations of pi0, pi0.5, GR00T N1.7, and MiniCPM-Robot. On Cosmos3-Nano-Policy-DROID it reduces latency from 2.46 to 1.18 s on eight H20 GPUs (CFG=2, TP=4), a 2.08x speedup. Specialized runtimes remain faster in several configurations, so our goal is one runtime with competitive latency rather than the fastest result in every case. Detailed profiles reveal why different models need different execution policies: on a Hopper-series GPU at batch size one, the pi0.5 action expert accounts for 8.8% of FLOPs but 57.2% of latency; at batch size 32 its share drops to 13.5% and throughput reaches about 100 samples/s. Cosmos3 remains generation-dominated and gains only 14.3% throughput as batch size increases from 1 to 16. We further introduce the control-time Roofline, which distinguishes inference-bound from environment-bound control; the measured pi0.5 points on four LIBERO suites are environment-bound while Cosmos3 stays inference-bound. Code and benchmarks: https://github.com/mingti-org/phyai.",
  "published": "2026-08-04",
  "updated": "2026-08-14",
  "year": "2026",
  "authors": [
   "Chenghua Wang",
   "Daliang Xu",
   "Dongqi Cai",
   "Duojin Sun",
   "Hao Zhang",
   "Haoze Qian",
   "Huaiyuan Zhang",
   "Jinshuo Cui",
   "Junbo Cui",
   "Kezhao Zhao",
   "Longxi Gao",
   "Mengwei Xu",
   "Rongjie Yi",
   "Ruixin Liu",
   "Shangguang Wang",
   "Tam Sikyuen",
   "Tianyue Zhang",
   "Weikai Xie",
   "Xuanzhe Liu",
   "Yingying Qin",
   "Yiwen Lu",
   "Yuan Yao",
   "Yuezhi Zu",
   "Yunhan Guo",
   "Yuxin Zheng",
   "Ziqi Guo"
  ],
  "author_count": 26,
  "categories": [
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.AI",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "PhyAI, a Physical AI inference engine with a single runtime that keeps architecture-specific conditioning, solver, cache, and output logic in model adapters while sharing graph execution, kernels, memory management, and parallel services, is built.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Chenghua Wang",
    "id": "2382356880",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Daliang Xu",
    "id": "2238890503",
    "h_index": 6,
    "papers": 22
   },
   {
    "name": "Dongqi Cai",
    "id": "2324790861",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Duojin Sun",
    "id": "2455559614",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Hao Zhang",
    "id": "2363858048",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Hao Qian",
    "id": "2454537404",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Huaiyuan Zhang",
    "id": "2458018357",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Jinshuo Cui",
    "id": "2457987119",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Junbo Cui",
    "id": "2292213035",
    "h_index": 6,
    "papers": 19
   },
   {
    "name": "Kezhao Zhao",
    "id": "2455500123",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Longxi Gao",
    "id": "2298336923",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Mengwei Xu",
    "id": "2239403838",
    "h_index": 14,
    "papers": 47
   },
   {
    "name": "Rongjie Yi",
    "id": "2170360770",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "Ruixin Liu",
    "id": "2456846199",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Shangguang Wang",
    "id": "2166005800",
    "h_index": 21,
    "papers": 54
   },
   {
    "name": "Tam Sikyuen",
    "id": "2455577476",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Tianyue Zhang",
    "id": "2455591894",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Weikai Xie",
    "id": "2331009162",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Xuanzhe Liu",
    "id": "2237080638",
    "h_index": 13,
    "papers": 53
   },
   {
    "name": "Yingying Qin",
    "id": "2455558550",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yiwen Lu",
    "id": "2455561283",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yuan Yao",
    "id": "2446383625",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Y. Zu",
    "id": "2446241885",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yunhan Guo",
    "id": "2408735942",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Yuxin Zheng",
    "id": "2453809109",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Ziqi Guo",
    "id": "2454255423",
    "h_index": 0,
    "papers": 2
   }
  ],
  "comment": "25 pages, 9 figures",
  "topics": [
   "vla",
   "rl-control"
  ],
  "orgs": [
   "NVIDIA",
   "Physical Intelligence"
  ],
  "abs_url": "https://arxiv.org/abs/2608.03682v3",
  "pdf_url": "https://arxiv.org/pdf/2608.03682v3",
  "html_url": "https://arxiv.org/html/2608.03682v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2608.03563",
  "slug": "unified-visuomotor-targets-supervising-vlas-beyond-physical-actions",
  "title": "Unified Visuomotor Targets: Supervising VLAs Beyond Physical Actions",
  "abstract": "VLA models are trained to predict robot actions from visual and language observations. This is a natural choice, but it creates a mismatch: VLMs encode rich, high-level representations of scenes and goals, while robot actions are low-level signals with limited task structure. We ask whether changing what the policy is trained to predict, rather than how it is architecturally designed, can yield better and more efficiently trained policies. We propose UVT (Unified Visuomotor Target), a unified latent prediction target that jointly encodes motor control and visual scene transition information, requiring no architectural changes and no additional data. Applied to two representative VLA systems across simulation benchmarks and real bimanual manipulation tasks, UVT improves training efficiency, final task performance, and policy robustness, with particularly strong gains under limited training budgets and challenging environmental conditions. Rollout videos and additional qualitative results are available at our project webpage: https://unified-visuomotor-targets.github.io/",
  "published": "2026-08-04",
  "updated": "2026-08-04",
  "year": "2026",
  "authors": [
   "Zhenyang Feng",
   "Unnat Jain"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Applied to two representative VLA systems across simulation benchmarks and real bimanual manipulation tasks, UVT improves training efficiency, final task performance, and policy robustness, with particularly strong gains under limited training budgets and challenging environmental conditions.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhenyang Feng",
    "id": "2324708185",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Unnat Jain",
    "id": "2387697720",
    "h_index": 2,
    "papers": 9
   }
  ],
  "comment": "Accepted at IROS 2026. Project page: https://unified-visuomotor-targets.github.io/",
  "topics": [
   "vla",
   "sim2real",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.03563v1",
  "pdf_url": "https://arxiv.org/pdf/2608.03563v1",
  "html_url": "https://arxiv.org/html/2608.03563v1",
  "code_url": "https://unified-visuomotor-targets.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.5
 },
 {
  "id": "2608.03556",
  "slug": "human-centric-embodied-intelligence-for-soft-wearable-robotics",
  "title": "Human Centric Embodied Intelligence for Soft Wearable Robotics",
  "abstract": "Soft wearable robots have evolved rapidly from proof-of-concept devices into promising platforms for rehabilitation, occupational assistance, and human augmentation. As the field matures, its central challenge extends beyond the development of softer materials and more capable actuators to the integration of sensing, intelligence, and human adaptation into systems that users can wear comfortably, trust, and benefit from over extended periods. This transition motivates the concept of Human-Centric Embodied Intelligence (HCEI), in which intelligence emerges from the coupled human-robot system through the interaction of morphology, multimodal sensing, adaptive cognition, compliant actuation, and the wearer's own physiological and behavioral adaptation. To organize this perspective, this review introduces the Perception-Cognition-Actuation-Augmentation (PCAA) framework, which positions perception and cognition as the primary drivers of design, shifting development beyond the conventional actuator-first paradigm. Using this framework, the review synthesizes advances in soft materials, wearable sensing, artificial intelligence, actuation, human-robot interaction, digital twins, clinical translation, manufacturing, regulation, and ethics, highlighting how these interdependent components collectively shape long-term personalization and real-world deployment. By providing a unified conceptual framework and design perspective, this review aims to guide future research, foster interdisciplinary collaboration, and accelerate the translation of next-generation soft wearable robots toward personalized, predictive, and human-centric wearable intelligence.",
  "published": "2026-08-04",
  "updated": "2026-08-04",
  "year": "2026",
  "authors": [
   "Rainier Natividad",
   "Raye Chen-Hua Yeow"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This review introduces the Perception-Cognition-Actuation-Augmentation (PCAA) framework, which positions perception and cognition as the primary drivers of design, shifting development beyond the conventional actuator-first paradigm.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Rainier F. Natividad",
    "id": "21078291",
    "h_index": 9,
    "papers": 14
   },
   {
    "name": "R. C. Yeow",
    "id": "3108956",
    "h_index": 19,
    "papers": 67
   }
  ],
  "comment": "Review article; 48 pages, 6 figures, 4 tables, and 2 supplementary tables",
  "topics": [
   "hardware-codesign",
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.03556v1",
  "pdf_url": "https://arxiv.org/pdf/2608.03556v1",
  "html_url": "https://arxiv.org/html/2608.03556v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.03490",
  "slug": "lightweight-3d-object-detection-via-mamba-based-knowledge-distillation",
  "title": "Lightweight 3D Object Detection via Mamba-Based Knowledge Distillation",
  "abstract": "3D object detection using light detection and ranging (LiDAR) sensors requires a balance between accuracy and computational efficiency for onboard perception in autonomous driving and robotic navigation. Many existing LiDAR-based detection methods employ complex architectures to extract features, integrating large amounts of contextual information to enhance accuracy. This often results in significant computational costs, leading to suboptimal performance on resource-constrained embedded devices. In this study, we propose a knowledge distillation framework that transfers object-level voxel representations from a strong teacher model to lightweight student models through selective voxel-space feature alignment. Taking advantage of the linear-time sequence model with selective state spaces (Mamba), we design a multi-branch Mamba teacher backbone and a box-aware feature transfer mechanism that aligns spatially corresponding voxel features between teacher and student networks through a Mamba-based projection module. Experimental results on both a public dataset and real-world data show that our approach significantly reduces computational load while maintaining competitive accuracy compared with state-of-the-art methods.",
  "published": "2026-08-04",
  "updated": "2026-08-04",
  "year": "2026",
  "authors": [
   "Quoc Cuong Ninh",
   "Huy Xuan Pham",
   "Anh Tung Nguyen",
   "Dinh Hoan Trinh"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "RA-L",
  "venue_source": "semantic-scholar",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A knowledge distillation framework that transfers object-level voxel representations from a strong teacher model to lightweight student models through selective voxel-space feature alignment and significantly reduces computational load while maintaining competitive accuracy compared with state-of-the-art methods is proposed.",
  "doi": "10.1109/LRA.2026.3719203",
  "oa_pdf": "https://arxiv.org/pdf/2608.03490",
  "s2_authors": [
   {
    "name": "Quoc Cuong Ninh",
    "id": "2455343801",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "H. Pham",
    "id": "40897168",
    "h_index": 14,
    "papers": 25
   },
   {
    "name": "A. Nguyen",
    "id": "2454912069",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "D. Trinh",
    "id": "2915627",
    "h_index": 9,
    "papers": 25
   }
  ],
  "comment": "Accepted for publication in IEEE Robotics and Automation Letters (RA-L), 2026",
  "topics": [
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.03490v1",
  "pdf_url": "https://arxiv.org/pdf/2608.03490v1",
  "html_url": "https://arxiv.org/html/2608.03490v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.03483",
  "slug": "continue-or-replan-bernoulli-continuation-policy-learning-for-adaptive",
  "title": "Continue or Replan? Bernoulli-Continuation Policy Learning for Adaptive Horizon Execution",
  "abstract": "Existing chunk-based Vision-Language-Action (VLA) models execute a fixed number of actions (i.e., execution horizon) before replanning, turning replanning into a task-agnostic periodic schedule that is independent of task progress. As a result, when no replanning boundary falls before a critical manipulation stage, it is executed from a stale chunk rather than a freshly replanned one. To address this limitation, we propose Bernoulli-Continuation Policy (BCP), a lightweight, plug-and-play framework for adaptive horizon execution that keeps the base VLA frozen. Given a fixed-length action chunk, its continuation head decomposes execution-horizon selection into a sequence of continue-or-replan decisions, which imposes an ordinal, prefix-sharing inductive bias over candidate horizons rather than treating them as independent classes. Since the optimal horizon for each chunk is not observable, we train this head with reinforcement learning from trajectory-level outcomes and introduce a Replanning-Efficiency Reward that jointly rewards task success and efficient VLA usage, discouraging the policy from collapsing to unnecessarily short horizons. On RoboTwin 2.0 with LingBot-VLA as the base policy, BCP improves the average success rate by +11.08% on 13 low-success tasks and from 89.88% to 93.94% (+4.06%) across all 50 tasks. Although trained only under the Clean setting, BCP generalizes to the Randomized setting, raising the average success rate by +4.06%. It also transfers to a different base policy $\u03c0_{0.5}$, achieving a better result on LIBERO (+1.7%) and, notably, on the harder LIBERO-PRO (+6.8%). On a real robot, BCP lifts success from 74% to 92% and from 44% to 84% on two manipulation tasks. Meanwhile, its negligible overhead, combined with higher success, makes BCP's overall runtime even lower than the fixed-horizon baselines.",
  "published": "2026-08-04",
  "updated": "2026-08-04",
  "year": "2026",
  "authors": [
   "Weichen Xu",
   "Zhenhua Liu",
   "Lin Luo",
   "Yaobo Liang",
   "Chengtang Yao",
   "Qingyu Mei",
   "Jian Cao",
   "Xixin Cao",
   "Xing Zhang",
   "Jiaolong Yang",
   "Baining Guo"
  ],
  "author_count": 11,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Bernoulli-Continuation Policy is proposed, a lightweight, plug-and-play framework for adaptive horizon execution that keeps the base VLA frozen and introduces a Replanning-Efficiency Reward that jointly rewards task success and efficient VLA usage, discouraging the policy from collapsing to unnecessarily short horizons.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Weichen Xu",
    "id": "2282176240",
    "h_index": 3,
    "papers": 18
   },
   {
    "name": "Zhenhua Liu",
    "id": "2283438760",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Lin Luo",
    "id": "2333975449",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Yaobo Liang",
    "id": "2333468608",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Chengtang Yao",
    "id": "1739110837",
    "h_index": 7,
    "papers": 15
   },
   {
    "name": "Qingyu Mei",
    "id": "2455492580",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Jian Cao",
    "id": "2282543370",
    "h_index": 3,
    "papers": 17
   },
   {
    "name": "Xixin Cao",
    "id": "2292664784",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Xingyi Zhang",
    "id": "2455666159",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Jiaolong Yang",
    "id": "2237946707",
    "h_index": 10,
    "papers": 22
   },
   {
    "name": "Baining Guo",
    "id": "2400505652",
    "h_index": 3,
    "papers": 6
   }
  ],
  "comment": "Project page: https://fleetfootwork.github.io/BCP/",
  "topics": [
   "vla",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.03483v1",
  "pdf_url": "https://arxiv.org/pdf/2608.03483v1",
  "html_url": "https://arxiv.org/html/2608.03483v1",
  "code_url": "https://fleetfootwork.github.io/BCP/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2608.03387",
  "slug": "roboreact-agentic-skill-distillation-from-generated-egocentric-videos",
  "title": "RoboReact: Agentic Skill Distillation from Generated Egocentric Videos for Generalizable Whole-Body Manipulation",
  "abstract": "Humanoid robots have the potential to perform dexterous manipulation in human environments, yet acquiring diverse and generalizable skills remains costly due to expensive hardware data collection and labor-intensive annotation. Recent advances in video generative models provide a promising opportunity to synthesize rich manipulation experiences from visual observations, but transferring such imagined behaviors into executable whole-body humanoid skills remains largely unexplored. In this work, we present RoboReact, a framework that automatically synthesizes whole-body humanoid manipulation skills from a single egocentric RGB-D observation. RoboReact generates human manipulation videos, extracts geometry-preserving interaction keyframes through depth-aware 3D reconstruction, and retargets them to high-DoF humanoid platforms while preserving hand-object interaction geometry. To bridge the gap between imagined plans and physical execution, RoboReact performs online object-centric re-grounding and leverages a vision-language model-guided refinement loop to adapt skills under geometric mismatch and execution deviations. The refined skills are executed through a whole-body controller, enabling coordinated whole-body manipulation and dexterous interaction. Experiments on real humanoid robots demonstrate that RoboReact generalizes across diverse object configurations and robustly recovers from execution disturbances without requiring teleoperation or human demonstrations. These results highlight the potential of combining generative models, vision-language reasoning, and closed-loop control for scalable humanoid skill acquisition.",
  "published": "2026-08-04",
  "updated": "2026-08-05",
  "year": "2026",
  "authors": [
   "Shuliang He",
   "Shuai Wang",
   "Bo Yue",
   "Junchi Teng",
   "Changyu Wang",
   "Guiliang Liu"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work presents RoboReact, a framework that automatically synthesizes whole-body humanoid manipulation skills from a single egocentric RGB-D observation, and highlights the potential of combining generative models, vision-language reasoning, and closed-loop control for scalable humanoid skill acquisition.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Shuliang He",
    "id": "2455480230",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Shuai Wang",
    "id": "2451296810",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Bo Yue",
    "id": "2322506329",
    "h_index": 2,
    "papers": 12
   },
   {
    "name": "Junchi Teng",
    "id": "2455448705",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Changyu Wang",
    "id": "2455450758",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Guiliang Liu",
    "id": "2319185234",
    "h_index": 3,
    "papers": 18
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "humanoids",
   "egocentric-data",
   "spatial-3d",
   "data-teleop",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.03387v2",
  "pdf_url": "https://arxiv.org/pdf/2608.03387v2",
  "html_url": "https://arxiv.org/html/2608.03387v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.03234",
  "slug": "learning-context-aware-motion-priors-for-humanoid-control",
  "title": "Learning Context-Aware Motion Priors for Humanoid Control",
  "abstract": "Motion priors provide powerful guidance for learning naturalistic humanoid behaviors. However, existing methods typically learn a general, task-agnostic prior from the entire reference dataset and apply it uniformly throughout policy training. As a result, the prior cannot distinguish which reference motions are relevant to the current task context, potentially providing irrelevant or conflicting guidance. We present Context-Aware Motion Priors (CMP), a framework that adapts a general motion prior to the current task context without manual skill labels, dataset partitioning, or a separate skill discovery stage. Specifically, CMP learns context-motion compatibility using high-advantage policy rollouts, while a demonstration-based objective keeps the learned relevance grounded in the reference distribution. The resulting relevance scores reweight reference supervision for training a lightweight context-conditioned adapter. To evaluate the effectiveness and generality of CMP, we instantiate it with both Adversarial Motion Priors and Score-Matching Motion Priors. Across five humanoid control tasks, CMP consistently improves task performance and sample efficiency, learns meaningful context-motion alignment, and remains robust to imbalanced reference distributions. These results show that adapting motion priors to task contexts provides more relevant guidance for humanoid policy learning.",
  "published": "2026-08-04",
  "updated": "2026-08-04",
  "year": "2026",
  "authors": [
   "Yunyang Mo",
   "Yi Gu",
   "Yangchen Zhou",
   "Hanyang Cao",
   "Renjing Xu"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Context-Aware Motion Priors is presented, a framework that adapts a general motion prior to the current task context without manual skill labels, dataset partitioning, or a separate skill discovery stage, and remains robust to imbalanced reference distributions.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yunyang Mo",
    "id": "2436192080",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Yi Gu",
    "id": "2331691332",
    "h_index": 2,
    "papers": 15
   },
   {
    "name": "Yang Zhou",
    "id": "2315929219",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Hanyang Cao",
    "id": "2276192673",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Renjing Xu",
    "id": "2409963644",
    "h_index": 2,
    "papers": 8
   }
  ],
  "comment": "16 pages, including appendices. Code will be released publicly",
  "topics": [
   "humanoids"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.03234v1",
  "pdf_url": "https://arxiv.org/pdf/2608.03234v1",
  "html_url": "https://arxiv.org/html/2608.03234v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.03231",
  "slug": "structure-aware-robust-fine-tuning-defending-vision-language-action-ro",
  "title": "Structure-Aware Robust Fine-Tuning: Defending Vision-Language-Action Robots Against Physical Attention Hijacking",
  "abstract": "Vision-Language-Action (VLA) policies promise general robotic manipulation, but their robustness against physical-world attacks remains fragile. In particular, we show that physically realizable adversarial patches can reliably induce failures by triggering a mechanism we call policy-critical action-to-vision attention hijacking, where action-conditioned attention is diverted from task-relevant regions to a localized patch. To demonstrate the threat, we propose Attention-Guided Semantic Disruption (AGSD), an Expectation-over-Transformation (EOT) optimized printable patch that jointly (i) concentrates action-to-vision attention on the patch and (ii) disrupts vision-language semantic alignment, yielding strong cross-task and cross-architecture transfer. To mitigate such attacks, we introduce Structure-Aware Robust Fine-Tuning (SARF), a zero-inference-overhead defense that fine-tunes only the visual encoder using feature anchoring, policy-critical attention correction, and language-guided geometric consistency restricted to semantically relevant regions. On LIBERO, SARF reduces OpenVLA's failure rate under AGSD from 100% to 14.2%-56.8% (28.6% average) across suites while preserving clean performance, and on a real PiPER manipulator it improves average success under AGSD from 23.0% to 65.0%. These results highlight mechanism-level robustness as a practical path to securing VLA robots against physical attention hijacking.",
  "published": "2026-08-04",
  "updated": "2026-08-04",
  "year": "2026",
  "authors": [
   "Jinquan Zhang",
   "Dongfu Yin",
   "Run Yang",
   "Yufeng Yan",
   "Zhen Tian",
   "F. Richard Yu"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Attention-Guided Semantic Disruption (AGSD), an Expectation-over-Transformation optimized printable patch that jointly concentrates action-to-vision attention on the patch and disrupts vision-language semantic alignment, yielding strong cross-task and cross-architecture transfer is proposed.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jinquan Zhang",
    "id": "2455556894",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Dongfu Yin",
    "id": "2447600344",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Run Yang",
    "id": "2420084719",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Yu Yan",
    "id": "2453855872",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Zhen Tian",
    "id": "2395630036",
    "h_index": 1,
    "papers": 12
   },
   {
    "name": "F. Yu",
    "id": "2371197090",
    "h_index": 1,
    "papers": 10
   }
  ],
  "comment": "Accepted to the 2026 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS 2026)",
  "topics": [
   "vla",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.03231v1",
  "pdf_url": "https://arxiv.org/pdf/2608.03231v1",
  "html_url": "https://arxiv.org/html/2608.03231v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.03227",
  "slug": "pfm-hr-pose-flow-matching-for-humanoid-robots",
  "title": "PFM-HR: Pose Flow Matching for Humanoid Robots",
  "abstract": "Motion priors improve reinforcement learning for physics-based humanoid tracking, but temporal priors require ordered motion clips, while pose priors provide limited guidance for policy-induced pose transitions. We present Pose Flow Matching for Humanoid Robots (PFM-HR), a reusable flow matching prior trained directly on large scale unordered pose data. PFM-HR introduces the Pose Geometry Score (PGS), which quantifies how joint coordinate changes during rollouts align with the local geometry of pose variation captured by the prior. Using PGS to modulate the tracking reward guides policy exploration toward structured pose changes while keeping the prior frozen across tracking tasks. Experiments demonstrate that PFM-HR improves both single motion and general motion tracking, especially for highly dynamic motions.",
  "published": "2026-08-04",
  "updated": "2026-08-14",
  "year": "2026",
  "authors": [
   "Yukang Gao",
   "Yi Gu",
   "Yangchen Zhou",
   "Xingyu Chen",
   "Zhaorui Wang",
   "Fanghai Zhang",
   "Hanyang Cao",
   "Zhengyang Shen",
   "Ji Ma",
   "Runhan Zhang",
   "Lei Han",
   "Renjing Xu"
  ],
  "author_count": 12,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "PFM-HR introduces the Pose Geometry Score (PGS), which quantifies how joint coordinate changes during rollouts align with the local geometry of pose variation captured by the prior, and using PGS to modulate the tracking reward guides policy exploration toward structured pose changes while keeping the prior frozen across tracking tasks.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yukang Gao",
    "id": "120625060",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Yi Gu",
    "id": "2331691332",
    "h_index": 2,
    "papers": 15
   },
   {
    "name": "Yang Zhou",
    "id": "2315929219",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Xingyu Chen",
    "id": "2381665219",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Zhaorui Wang",
    "id": "2331699558",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Fan Zhang",
    "id": "2323601562",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Hanyang Cao",
    "id": "2276192673",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Zhengyang Shen",
    "id": "2383880622",
    "h_index": 1,
    "papers": 9
   },
   {
    "name": "Ji Ma",
    "id": "2336920494",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "Runhan Zhang",
    "id": "2445708812",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Lei Han",
    "id": "2362318590",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Renjing Xu",
    "id": "2409963644",
    "h_index": 2,
    "papers": 8
   }
  ],
  "comment": "7 pages",
  "topics": [
   "humanoids",
   "rl-control",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.03227v2",
  "pdf_url": "https://arxiv.org/pdf/2608.03227v2",
  "html_url": "https://arxiv.org/html/2608.03227v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.03211",
  "slug": "crossscope-a-role-asymmetric-world-model-for-joint-dual-scope-surgical",
  "title": "CrossScope: A Role-Asymmetric World Model for Joint Dual-Scope Surgical Video Prediction",
  "abstract": "Visual world models typically learn future dynamics from a single observation stream, limiting their ability to model cooperative systems with multiple independently moving observers. We investigate this challenge in Mother--Child endoscopic retrograde cholangiopancreatography (ERCP), where two flexible scopes provide complementary yet role-dependent views without a calibrated stereo relationship. Unlike conventional multi-view fusion that assumes symmetric information exchange, we formulate \\textbf{role-asymmetric dual-scope future prediction}, where cross-view evidence is selectively transferred according to the prediction target and its underlying spatial requirements. We propose \\textbf{CrossScope}, a dual-stream surgical world model that preserves view-specific experts while enabling target-specific evidence routing through geometry-guided residual interactions. CrossScope learns two complementary communication directions: geometric motion cues from the Mother view guide Child-view future dynamics, while pose-aligned Child appearance supports Mother-view prediction only when valid spatial correspondence is established. This design allows each scope to contribute task-relevant evidence without compromising its view-specific representation. To evaluate this problem, we establish a paired dual-scope benchmark comprising synchronized phantom and real-world ERCP episodes, with evaluations assessing visual fidelity, structural preservation, target localization, and motion consistency. Experiments demonstrate that CrossScope consistently outperforms strong surgical video generation baselines, validating the importance of role-aware evidence routing for multi-observer visual world modeling.",
  "published": "2026-08-04",
  "updated": "2026-08-04",
  "year": "2026",
  "authors": [
   "Wanhao Liu",
   "Jinsong Lin",
   "Rulin Zhou",
   "Chi Kit Ng",
   "Wenbin Pan",
   "Zhiqing Tang",
   "Dongyue Li",
   "Liwei Luo",
   "Yanshen Wu",
   "Panshuo Li",
   "Zhiyong Xiong",
   "Huxin Gao",
   "Tamas Haidegger",
   "Hongliang Ren"
  ],
  "author_count": 14,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Experiments demonstrate that CrossScope consistently outperforms strong surgical video generation baselines, validating the importance of role-aware evidence routing for multi-observer visual world modeling.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Wanhao Liu",
    "id": "2455435040",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Jinsong Lin",
    "id": "2445481374",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Rulin Zhou",
    "id": "2353204753",
    "h_index": 2,
    "papers": 15
   },
   {
    "name": "Chikit Ng",
    "id": "2307447792",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Wenbin Pan",
    "id": "2455487287",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Zhiqing Tang",
    "id": "2455447929",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Dongyue Li",
    "id": "2455450276",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Liwei Luo",
    "id": "2448554553",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Yanshen Wu",
    "id": "2434778835",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Panshuo Li",
    "id": "2243343954",
    "h_index": 4,
    "papers": 21
   },
   {
    "name": "Zhiyong Xiong",
    "id": "47845657",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Huxin Gao",
    "id": "1653073415",
    "h_index": 13,
    "papers": 49
   },
   {
    "name": "T. Haidegger",
    "id": "1683535",
    "h_index": 31,
    "papers": 226
   },
   {
    "name": "Hongliang Ren",
    "id": "2260612957",
    "h_index": 13,
    "papers": 53
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.03211v1",
  "pdf_url": "https://arxiv.org/pdf/2608.03211v1",
  "html_url": "https://arxiv.org/html/2608.03211v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.03155",
  "slug": "pomdps-for-autonomous-science-exploration",
  "title": "POMDPs for Autonomous Science Exploration",
  "abstract": "Autonomous exploration missions require decision-making under sensor uncertainty and computational constraints, yet integrating scientific representations into POMDP planning has remained intractable due to high-dimensional observation spaces. Information-theoretic planners overcome this by assuming deterministic observations, sacrificing the principled uncertainty quantification that POMDPs provide. We introduce the Science Hypothesis Map POMDP (SHM-POMDP), which makes science-driven belief-space planning more tractable by branching on inferred physical properties rather than raw sensor data. This preserves full sensor information through learned observation models while enabling the planner to reason jointly about navigation and scientific properties under uncertainty. On an extended RockSample domain with 50-dimensional observations, SHM-POMDP achieves 18.6\\% higher rewards and 32.9\\% reduced computation time per step than continuous-observation baselines. On realistic geologic exploration using Cuprite hyperspectral data, SHM-POMDP achieves 2.5$\\times$ higher information gain than the best information-theoretic baseline by maintaining beliefs and replanning adaptively---reaching 80\\% of oracle performance using only uniform priors. These results demonstrate that integrating hierarchical probabilistic models into belief-space planning enables tractable, principled autonomous science that outperforms both traditional POMDP methods and science-aware information-theoretic approaches.",
  "published": "2026-08-04",
  "updated": "2026-08-04",
  "year": "2026",
  "authors": [
   "Daniel Guirguis",
   "Nathan Wallace",
   "Hanna Kurniawati",
   "Salah Sukkarieh"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The Science Hypothesis Map POMDP is introduced, which makes science-driven belief-space planning more tractable by branching on inferred physical properties rather than raw sensor data, and enables tractable, principled autonomous science that outperforms both traditional POMDP methods and science-aware information-theoretic approaches.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Daniel Guirguis",
    "id": "2455447478",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Nathan D. Wallace",
    "id": "48266589",
    "h_index": 7,
    "papers": 22
   },
   {
    "name": "Hanna Kurniawati",
    "id": "2287938519",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Salah Sukkarieh",
    "id": "2405232817",
    "h_index": 1,
    "papers": 10
   }
  ],
  "comment": "8 pages, 3 figures, 2 tables. Accepted to the 2026 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS 2026)",
  "topics": [
   "navigation",
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.03155v1",
  "pdf_url": "https://arxiv.org/pdf/2608.03155v1",
  "html_url": "https://arxiv.org/html/2608.03155v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.03143",
  "slug": "from-routes-to-steps-separating-semantic-progress-from-local-execution",
  "title": "From Routes to Steps: Separating Semantic Progress from Local Execution in Vision-and-Language Navigation",
  "abstract": "Vision-and-Language Navigation (VLN) requires an agent to follow a route-level instruction by executing its constituent steps from egocentric visual observations. Existing VLM-based navigators typically supervise both capabilities through next-action prediction alone, making progress-tracking errors difficult to distinguish from execution errors. When an agent deviates from the route, a corrective action label may recover the next movement but does not indicate whether the agent selected the wrong sub-instruction or failed to execute the correct one. Consequently, the agent may continue making decisions from an erroneous progress state. To resolve this ambiguity, we propose \\textbf{Route2Step}, a framework that decouples semantic progress tracking from action generation through an explicit step-level interface. The Instruction Analysis Module ($\\mathcal{M}_{\\mathrm{IA}}$) predicts this state from the global instruction and visual history. Conditioned on the predicted state and recent observations, the Action Generation Module ($\\mathcal{M}_{\\mathrm{AG}}$) generates local action chunks. To supervise the progress state without manual temporal labels, E-SPA, a step-alignment procedure, associates sub-instructions with their corresponding portions of route-level demonstrations. These alignments enable state supervision for incorrect progress estimates, while direct action supervision is reserved for rollout groups that repeatedly fail under the correct active sub-instruction. On R2R-CE, Route2Step improves SR from 48.1\\% to 55.3\\% and SPL from 43.3\\% to 48.2\\%, using 190K state-level corrective samples while requiring only 11.5K directly action-supervised states. Experiments in real-world indoor and outdoor environments further demonstrate the practical applicability of Route2Step. The project page is: https://sisyphus-hxy.github.io/Route2Step/.",
  "published": "2026-08-04",
  "updated": "2026-08-04",
  "year": "2026",
  "authors": [
   "Xiangyun Huang",
   "Xiangchen Wang",
   "Runfeng Lin",
   "Yihao Xu",
   "Kangyu Huang",
   "Jiang Hengchen",
   "Xiwang Dong",
   "Lin Jiarong"
  ],
  "author_count": 8,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes Route2Step, a framework that decouples semantic progress tracking from action generation through an explicit step-level interface, and demonstrates the practical applicability of Route2Step in real-world indoor and outdoor environments.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xiangyu Huang",
    "id": "2383971798",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Xiangchen Wang",
    "id": "2375899094",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Runfeng Lin",
    "id": "2455489551",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yihao Xu",
    "id": "2451015392",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Kang Huang",
    "id": "2445393609",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Hengchen Jiang",
    "id": "2455547836",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Xiwang Dong",
    "id": "2116116549",
    "h_index": 1,
    "papers": 22
   },
   {
    "name": "Jiarong Lin",
    "id": "11663106",
    "h_index": 19,
    "papers": 30
   }
  ],
  "comment": "16 pages, 9 figures",
  "topics": [
   "egocentric-data",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.03143v1",
  "pdf_url": "https://arxiv.org/pdf/2608.03143v1",
  "html_url": "https://arxiv.org/html/2608.03143v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.03116",
  "slug": "shooting-for-contact-contact-implicit-multiple-shooting-for-dynamic-mo",
  "title": "Shooting for Contact: Contact-Implicit Multiple Shooting for Dynamic Motion Retargeting",
  "abstract": "Motion retargeting approaches often prioritize kinematic similarity over whole-body dynamics, contact consistency, and actuation limits, yielding references that are difficult for reinforcement learning (RL) policies to reproduce, particularly for contact-rich behaviors. We present a contact-implicit, direct simulation-based multiple shooting (DSMS) framework that transforms kinematically feasible references into dynamically feasible whole-body trajectories. By embedding a differentiable simulator within a nonlinear program, DSMS resolves contact, friction, impacts, self-collision, and joint limits internally while enforcing tracking, actuation, and task constraints without prescribing a contact schedule or introducing explicit contact constraints. Compared with existing retargeting methods, DSMS accelerates motion-imitation RL training and yields policies with high success rates and low tracking error. We further demonstrate zero-shot sim-to-real transfer on the Unitree G1 through command-conditioned contact-rich crawling and a highly dynamic 180-degree jump-turn.",
  "published": "2026-08-04",
  "updated": "2026-08-04",
  "year": "2026",
  "authors": [
   "Sergio A. Esteban",
   "Jason H. K. Siu",
   "Derrick Mach",
   "Junheng Li",
   "Vince Kurtz",
   "Joel W. Burdick",
   "Aaron D. Ames"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A contact-implicit, direct simulation-based multiple shooting framework that transforms kinematically feasible references into dynamically feasible whole-body trajectories, and demonstrates zero-shot sim-to-real transfer on the Unitree G1 through command-conditioned contact-rich crawling and a highly dynamic 180-degree jump-turn.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Sergio A. Esteban",
    "id": "2346836979",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Jason H. K. Siu",
    "id": "2455448602",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "D. Mach",
    "id": "2320275187",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Junheng Li",
    "id": "2108988997",
    "h_index": 7,
    "papers": 24
   },
   {
    "name": "Vince Kurtz",
    "id": "31574614",
    "h_index": 14,
    "papers": 32
   },
   {
    "name": "J. W. Burdick",
    "id": "2303258085",
    "h_index": 5,
    "papers": 23
   },
   {
    "name": "Aaron D. Ames",
    "id": "2338277217",
    "h_index": 4,
    "papers": 23
   }
  ],
  "comment": "Project website with additional material: https://shooting-for-contact.github.io/",
  "topics": [
   "humanoids",
   "tactile",
   "sim2real",
   "rl-control"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2608.03116v1",
  "pdf_url": "https://arxiv.org/pdf/2608.03116v1",
  "html_url": "https://arxiv.org/html/2608.03116v1",
  "code_url": "https://shooting-for-contact.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.03109",
  "slug": "multimodal-plant-root-phenotyping-with-integration-of-3d-skeleton-extr",
  "title": "Multimodal Plant Root Phenotyping with Integration of 3D Skeleton Extraction and Language Analysis",
  "abstract": "Plant root phenotyping is fundamental to understanding below-ground structures, optimizing crop management, and improving agricultural sustainability. This paper presents a multimodal robotic AI framework that integrates 3D skeleton extraction with language-guided reasoning for interpretable and data-efficient root analysis. We develop an unsupervised skeleton extraction network based on Weighted Laplacian Contraction (W-LBC) to generate high-fidelity structural representations from dense point clouds captured by robotic 3D sensing platforms. Quantitative morphological descriptors, including root count, length, branching angle, and density, are computed from the reconstructed skeleton graph to capture geometric and topological characteristics. Building on these features, we introduce an Evidence-First language modeling framework that fine-tunes GPT as an interactive analytical chatbot using automatically generated instruction--response pairs. Each training sample provides measurable evidence before natural-language reasoning, enabling the model to ground interpretation in quantitative morphology. Through supervised fine-tuning, GPT associates numerical structure with semantic meaning, producing biologically consistent explanations of growth patterns and adaptive traits. Experiments show that the structure-guided framework achieves robust, interpretable reasoning across 12 plant species with diverse root architectures. By integrating unsupervised 3D geometric perception with large-scale language understanding, our approach bridges quantitative analysis and semantic interpretation, establishing a unified paradigm for explainable robotic plant root phenotyping.",
  "published": "2026-08-04",
  "updated": "2026-08-04",
  "year": "2026",
  "authors": [
   "Jiakai Lin",
   "Zijun Li",
   "Guoyu Lu"
  ],
  "author_count": 3,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A multimodal robotic AI framework that integrates 3D skeleton extraction with language-guided reasoning for interpretable and data-efficient root analysis and an Evidence-First language modeling framework that fine-tunes GPT as an interactive analytical chatbot using automatically generated instruction--response pairs are introduced.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jiakai Lin",
    "id": "2344652256",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Zijun Li",
    "id": "2349637281",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Guoyu Lu",
    "id": "2315782372",
    "h_index": 4,
    "papers": 15
   }
  ],
  "comment": "",
  "topics": [
   "spatial-3d",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.03109v1",
  "pdf_url": "https://arxiv.org/pdf/2608.03109v1",
  "html_url": "https://arxiv.org/html/2608.03109v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.03103",
  "slug": "a-hierarchical-approach-to-imitation-learning-for-manipulation-tasks-r",
  "title": "A Hierarchical Approach to Imitation Learning for Manipulation Tasks Requiring Time Varying Forces",
  "abstract": "Diffusion policies have shown strong performance in learning complex, multi-modal behaviors for robotic manipulation. However, their application to contact-rich disassembly tasks remains limited by a key trade-off: the iterative denoising process introduces inference latencies that makes high frequency control difficult, which is essential for realizing dynamic interactions such as chiseling and prying. Recent action-chunking techniques mitigate latency but use an open-loop execution window, rendering the system blind to rapid force transients caused by fracture events. To bridge this gap, we introduce the Diffusion Policy Augmented by Fast Trajectory Generation (DPA-FTG). Compared to recent visual-tactile approaches that focus on positional correction, DPA-FTG decouples low-frequency planning from high-frequency force regulation. At the high level ($5$ Hz), a conditional diffusion model predicts a sequence of latent parameters for selecting a strategy from a learned vocabulary of task primitives. At the low level ($60$ Hz), a lightweight, force-conditioned policy acts as a neural impedance controller, modulating execution in real-time to maintain contact stability. We validate our approach on a bimanual battery disassembly task involving the separation of a compliant sheet. Experimental evaluation demonstrates that DPA-FTG outperforms state-of-the-art baselines, including Reactive Diffusion Policy (RDP).",
  "published": "2026-08-04",
  "updated": "2026-08-04",
  "year": "2026",
  "authors": [
   "Rishabh Shukla",
   "Adithya Santhosh",
   "Shaili Gandhi",
   "Samrudh Moode",
   "Satyandra K. Gupta"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Compared to recent visual-tactile approaches that focus on positional correction, DPA-FTG decouples low-frequency planning from high-frequency force regulation, and outperforms state-of-the-art baselines, including Reactive Diffusion Policy (RDP).",
  "doi": "10.1016/j.rcim.2026.103309",
  "oa_pdf": "https://arxiv.org/pdf/2608.03103",
  "s2_authors": [
   {
    "name": "Rishabh Shukla",
    "id": "2266537745",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Adithya Santhosh",
    "id": "2415489026",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Shaili Gandhi",
    "id": "2351946786",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Samrudh Moode",
    "id": "2337120977",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Satyandra K. Gupta",
    "id": "2385276261",
    "h_index": 1,
    "papers": 6
   }
  ],
  "comment": "",
  "topics": [
   "tactile",
   "imitation-diffusion",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.03103v1",
  "pdf_url": "https://arxiv.org/pdf/2608.03103v1",
  "html_url": "https://arxiv.org/html/2608.03103v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.03060",
  "slug": "passively-safe-convex-guidance-for-cislunar-rendezvous-and-proximity-o",
  "title": "Passively Safe Convex Guidance for Cislunar Rendezvous and Proximity Operations",
  "abstract": "This paper presents purely convex programs for passively safe impulsive rendezvous and proximity operations in cislunar orbits. Approach, arrival, and abort maneuvers are all designed and validated in the context of maneuver execution error and navigation uncertainty, and formulated for efficient onboard execution in the autonomous scenario. The outlined methods form the baseline onboard guidance routines for NASA's CAPSTONE 02 mission planned to demonstrate autonomous rendezvous and proximity operations capabilities in the southern 9:2 synodic near rectilinear halo orbit. High fidelity closed loop Monte Carlo simulations using the planned relative navigation sensor suite and measurement cadence verify the intended maneuver design performance.",
  "published": "2026-08-04",
  "updated": "2026-08-04",
  "year": "2026",
  "authors": [
   "Ian M. Down",
   "Connor Plaks",
   "Matthew Bolliger",
   "Michael Caudill"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "math.OC",
   "nlin.CD"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ian M. Down",
    "id": "2321404080",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Connor Plaks",
    "id": "2455446981",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Matt Bolliger",
    "id": "2359647487",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Michael Caudill",
    "id": "2323041593",
    "h_index": 2,
    "papers": 13
   }
  ],
  "comment": "Presented at the 2026 AAS/AIAA Astrodynamics Specialist Conference, Whistler, BC",
  "topics": [
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.03060v1",
  "pdf_url": "https://arxiv.org/pdf/2608.03060v1",
  "html_url": "https://arxiv.org/html/2608.03060v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.03052",
  "slug": "how-should-vision-language-action-models-use-proprioceptive-state",
  "title": "How Should Vision-Language-Action Models Use Proprioceptive State?",
  "abstract": "Recent Vision-Language-Action (VLA) models almost universally take robot proprioceptive state as input, yet wire it in incompatible ways -- serialized into text prompts, projected into the vision-language prefix, or fed directly to the action expert -- and almost always as a single current frame. Three questions remain open: (1) whether, and on which tasks, current state actually improves closed-loop control; (2) how much state history helps, and whether its benefit reflects genuine temporal variation rather than added conditioning capacity; and (3) where state should enter the model -- the vision-language backbone or the action-generation module. We answer these questions through controlled experiments on a flow-matching VLA, fixing the backbone, training data, action representation, and evaluation protocol throughout. We implement five representative interfaces -- discrete state prompt, VLM prefix, action prefix, state expert, and feature modulation -- under matched implementation details, and evaluate them on 45 atomic tasks spanning three task families plus 20 composite tasks; we then sweep the state-history length from 1 to 96 frames to examine how historical state information affects model performance. The experiments yield systematic answers to all three questions, distilled into testable design principles for state-aware VLAs.",
  "published": "2026-08-04",
  "updated": "2026-08-04",
  "year": "2026",
  "authors": [
   "Yiren Zhao",
   "Ziyang Chen",
   "Ziyang Rao",
   "Pengteng Li",
   "He Zhang",
   "Weiyu Guo",
   "Yandong Guo",
   "Rushi Dai"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Five representative interfaces are implemented -- discrete state prompt, VLM prefix, action prefix, state expert, and feature modulation -- under matched implementation details, and evaluated on 45 atomic tasks spanning three task families plus 20 composite tasks.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yiren Zhao",
    "id": "2109919449",
    "h_index": 21,
    "papers": 88
   },
   {
    "name": "Ziyang Chen",
    "id": "2347655949",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Ziyang Rao",
    "id": "2343504880",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Pengteng Li",
    "id": "2342359936",
    "h_index": 3,
    "papers": 14
   },
   {
    "name": "He Zhang",
    "id": "2257389052",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Weiyu Guo",
    "id": "2257320765",
    "h_index": 7,
    "papers": 27
   },
   {
    "name": "Yandong Guo",
    "id": "2284207445",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Rushi Dai",
    "id": "2455447767",
    "h_index": 0,
    "papers": 2
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "sim2real"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.03052v1",
  "pdf_url": "https://arxiv.org/pdf/2608.03052v1",
  "html_url": "https://arxiv.org/html/2608.03052v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.03051",
  "slug": "cuda-mpc-a-gpu-native-solver-for-model-predictive-control",
  "title": "CUDA MPC: A GPU-Native Solver for Model Predictive Control",
  "abstract": "Model Predictive Control (MPC) delivers constraint-aware control, but its reliance on online optimization limits its use on systems with fast dynamics, high-dimensional models, or long horizons. Existing GPU implementations typically treat the device as a linear-algebra accelerator, leaving the optimization loop dependent on repeated kernel launches and high-latency memory transfers. This paper introduces CUDA MPC, a GPU-native MPC framework that co-designs the optimization algorithm, execution model, and memory architecture for CUDA hardware. CUDA MPC pairs a parallel-in-horizon alternating direction method of multipliers (ADMM) splitting with a fused CUDA kernel that runs the entire iterative solve on the device. Intermediate optimization variables stay in low-latency, on-chip shared memory, and a localized atomic-flag protocol synchronizes only adjacent horizon blocks, minimizing host intervention, kernel-dispatch overhead, and global-memory traffic. Across six nonlinear robotics benchmarks spanning increasing state dimension and constraint density, CUDA MPC sustains real-time rates at horizons one to two orders of magnitude longer than CPU solvers: it solves an optimization-based collision-avoidance parking problem with 100 s of lookahead within a 0.1 s sampling interval, and is the only solver evaluated that achieves both real-time execution and collision-free coordination for a centralized 10-agent swarm, where acados and CasADi return no feasible solution and require 3.5 s and 4.5 s per solve. Against tensor-framework implementations of the same ADMM splitting, the fused kernel is up to $965\\times$ faster.",
  "published": "2026-08-04",
  "updated": "2026-08-04",
  "year": "2026",
  "authors": [
   "Babak Akbari",
   "Melissa Greeff"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO",
   "cs.DC",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "CUDA MPC is introduced, a GPU-native MPC framework that co-designs the optimization algorithm, execution model, and memory architecture for CUDA hardware and sustains real-time rates at horizons one to two orders of magnitude longer than CPU solvers.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Babak Akbari",
    "id": "2284593648",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Melissa Greeff",
    "id": "47297149",
    "h_index": 8,
    "papers": 25
   }
  ],
  "comment": "",
  "topics": [
   "rl-control",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.03051v1",
  "pdf_url": "https://arxiv.org/pdf/2608.03051v1",
  "html_url": "https://arxiv.org/html/2608.03051v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.03002",
  "slug": "a-wearable-stiffness-rendering-haptic-device-with-a-honeycomb-jamming",
  "title": "A Wearable Stiffness-Rendering Haptic Device with a Honeycomb Jamming Mechanism for Bilateral Teleoperation",
  "abstract": "This paper addresses the challenge of providing kinesthetic feedback in bilateral teleoperation by designing a wearable, lightweight (20 g), and compact haptic device, the HJ-Haptic, utilizing a honeycomb jamming mechanism for object stiffness rendering. The HJ-Haptic device can vary its stiffness, from 1.15 N/mm to 2.64 N/mm, using a 30 kPa vacuum pressure. We demonstrate its implementation in a teleoperation framework, enabling operators to adjust grip force based on a reliable haptic feedback on object stiffness. A three-point flexural test on the honeycomb jamming mechanism and teleoperated object-grasping tasks were conducted to evaluate the device's functionality. Our experiments demonstrated a small RMSE and strong correlations in teleoperated motion, stiffness rendering, and interaction force feedback. The HJ-Haptic effectively adjusts its stiffness in response to real-time gripper feedback, mimicking the sensation of direct object grasping with hands. The device's use of vacuum pressure ensures operator safety by preventing dangerous outcomes in case of gas leakage or material failure. Incorporating the HJ-Haptic into the teleoperation framework provided the reliable perception of object stiffness and stable teleoperation. This study highlights the potential of the honeycomb jamming mechanism for enhancing haptic feedback in various applications, including teleoperation scenarios, as well as interactions with extended-reality environments.",
  "published": "2026-08-04",
  "updated": "2026-08-04",
  "year": "2026",
  "authors": [
   "Thomas M. Kwok",
   "Bohan Zhang",
   "Wai Tuck Chow"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "Machines",
  "venue_source": "semantic-scholar",
  "citations": 6,
  "influential_citations": 0,
  "tldr": "The HJ-Haptic effectively adjusts its stiffness in response to real-time gripper feedback, mimicking the sensation of direct object grasping with hands, enabling operators to adjust grip force based on a reliable haptic feedback on object stiffness.",
  "doi": "10.3390/machines13010027",
  "oa_pdf": "https://www.mdpi.com/2075-1702/13/1/27/pdf?version=1736162452",
  "s2_authors": [
   {
    "name": "Thomas M. Kwok",
    "id": "46882707",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Bohan Zhang",
    "id": "2318890174",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Wai Tuck Chow",
    "id": "2242940469",
    "h_index": 5,
    "papers": 17
   }
  ],
  "comment": "18 pages, 10 figures. Published in Machines 2025, 13(1), 27",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.03002v1",
  "pdf_url": "https://arxiv.org/pdf/2608.03002v1",
  "html_url": "https://arxiv.org/html/2608.03002v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.35
 },
 {
  "id": "2608.02993",
  "slug": "neurosymbolic-reasoning-with-incremental-knowledge-for-sample-efficien",
  "title": "Neurosymbolic Reasoning with Incremental Knowledge for Sample Efficient Hierarchical Reinforcement Learning",
  "abstract": "(Flat) Reinforcement Learning (RL) agents face significant challenges in environments with sparse rewards that require long-horizon reasoning. A compelling approach to improve sample efficiency is to incorporate knowledge into learning and decision-making. In standard Hierarchical RL (HRL), knowledge is encoded in a fixed, non-updatable form, such as architectural choices, and remains unchanged throughout learning. With fixed HRL, reasoning with incremental knowledge learned during exploration is impractical before sufficient environmental knowledge is acquired, leading to poor sample efficiency. In this work, we propose neurosymbolic HRL with {\\em Incremental Knowledge (InK)}: symbolic high-level components perform {\\em symbolic planning} (e.g. using $D^*$) on an updatable representation of current InK, while low-level goal-conditioned neural modules learn motion primitives through experience using reward shaping. Experiments on navigation tasks demonstrate that incorporating InK substantially improves sample efficiency. Additionally, to perform {\\em optimal} symbolic planning given {\\em prior} knowledge about the world, we develop Belief World Tree Search. The code is available at https://github.com/CPS-research-group/ink_bwts.",
  "published": "2026-08-04",
  "updated": "2026-08-04",
  "year": "2026",
  "authors": [
   "Subrat Prasad Panda",
   "Blaise Genest",
   "Arvind Easwaran"
  ],
  "author_count": 3,
  "categories": [
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.AI",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes neurosymbolic HRL with {\\em Incremental Knowledge (InK), where symbolic high-level components perform symbolic planning on an updatable representation of current InK, while low-level goal-conditioned neural modules learn motion primitives through experience using reward shaping.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Subrat Prasad Panda",
    "id": "32770022",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "B. Genest",
    "id": "1725550",
    "h_index": 17,
    "papers": 96
   },
   {
    "name": "A. Easwaran",
    "id": "144266912",
    "h_index": 28,
    "papers": 195
   }
  ],
  "comment": "Published in ECML-PKDD 2026",
  "topics": [
   "rl-control",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.02993v1",
  "pdf_url": "https://arxiv.org/pdf/2608.02993v1",
  "html_url": "https://arxiv.org/html/2608.02993v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.02990",
  "slug": "embodiedvae-disentangled-video-vae-for-efficient-and-controllable-embo",
  "title": "EmbodiedVAE: Disentangled Video VAE for Efficient and Controllable Embodied Manipulation",
  "abstract": "Latent diffusion models (LDMs) have recently significantly advanced embodied learning in constructing powerful embodied manipulation world models. However, despite the remarkable performance, existing LDMs predominantly rely on Variational Autoencoders (VAEs) optimized for natural scenes while failing to account for the unique characteristics of embodied manipulation scenarios, yielding latent representations that are neither compact nor controllable, thereby hindering efficient training of LDMs and precise robotic control. To solve this problem, we present EmbodiedVAE, a novel video VAE that provides compact yet controllable latent representations tailored for the robotic manipulation world models. Specifically, EmbodiedVAE adopts a dual-encoder, single-decoder architecture with an asymmetric spatio-temporal compression module, which automatically disentangles the robot arm's motion from background environment, resulting in overall compactness while providing explicit embodied latent to support fine-grained action control. To further preserve the temporal consistency of learned robotic motion latent, we introduce an optimal-transport-based consistency module that explicitly enforces motion fidelity and inter-frame coherence. Extensive experiments demonstrate that our proposed EmbodiedVAE achieves superior reconstruction quality with high compression rate, while enabling more precise action control in robotic manipulation scenarios with an average of 2dB PSNR improvement over state-of-the-art video VAEs.",
  "published": "2026-08-04",
  "updated": "2026-08-04",
  "year": "2026",
  "authors": [
   "Jiayi Luo",
   "Hanxin Zhu",
   "Chen Gao",
   "Jiankun Wang",
   "Cong Wang",
   "Tianyu He",
   "Jianxin Li",
   "Zhibo Chen"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "ECCV 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The proposed EmbodiedVAE adopts a dual-encoder, single-decoder architecture with an asymmetric spatio-temporal compression module, which automatically disentangles the robot arm's motion from background environment, resulting in overall compactness while providing explicit embodied latent to support fine-grained action control.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jiayi Luo",
    "id": "2319302828",
    "h_index": 2,
    "papers": 28
   },
   {
    "name": "Hanxin Zhu",
    "id": "2282209914",
    "h_index": 7,
    "papers": 29
   },
   {
    "name": "Chen Gao",
    "id": "2384820291",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Jiankun Wang",
    "id": "2290024771",
    "h_index": 8,
    "papers": 40
   },
   {
    "name": "Cong Wang",
    "id": "2269795155",
    "h_index": 6,
    "papers": 30
   },
   {
    "name": "Tianyu He",
    "id": "2259935939",
    "h_index": 13,
    "papers": 30
   },
   {
    "name": "Jianxin Li",
    "id": "2318929344",
    "h_index": 1,
    "papers": 9
   },
   {
    "name": "Zhibo Chen",
    "id": "2276445327",
    "h_index": 3,
    "papers": 19
   }
  ],
  "comment": "ECCV 2026",
  "topics": [
   "world-models",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.02990v1",
  "pdf_url": "https://arxiv.org/pdf/2608.02990v1",
  "html_url": "https://arxiv.org/html/2608.02990v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.07566",
  "slug": "the-field-knows-cross-dimensional-geometry-from-navigation-to-black-ho",
  "title": "The Field Knows: Cross-Dimensional Geometry from Navigation to Black Holes",
  "abstract": "We introduce a continuous metric field framework trained by a single causal contrastive loss. The framework encodes a scene into coefficients of a fixed symmetric matrix basis, assembles them into a Lie algebra element, and exponentiates the result to a Riemannian or Lorentzian metric. Across dimensions, this field discovers the full spectrum of geometric structures: from obstacle-avoiding geodesics in robot navigation across planar and manipulator configuration spaces, to event horizons of black holes in Lorentzian spacetime. Extensive zero-shot generalization studies demonstrate that the field captures transferable geometric structure rather than memorizing specific configurations. In the black hole setting, the causal loss spontaneously evolves genuine black-hole-like structures with the correct Lorentzian signature. The same loss, the same architecture, and the same training protocol produce the full range of geometric phenomena across dimensions. The field knows geometry, and geometry knows physics.",
  "published": "2026-08-03",
  "updated": "2026-08-03",
  "year": "2026",
  "authors": [
   "Chenghao Xu"
  ],
  "author_count": 1,
  "categories": [
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.AI",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A continuous metric field framework trained by a single causal contrastive loss that discovers the full spectrum of geometric structures: from obstacle-avoiding geodesics in robot navigation across planar and manipulator configuration spaces, to event horizons of black holes in Lorentzian spacetime.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Chenghao Xu",
    "id": "2277225791",
    "h_index": 6,
    "papers": 32
   }
  ],
  "comment": "",
  "topics": [
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.07566v1",
  "pdf_url": "https://arxiv.org/pdf/2608.07566v1",
  "html_url": "https://arxiv.org/html/2608.07566v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.02958",
  "slug": "valueformer-a-causal-transformer-value-function-with-stage-aware-label",
  "title": "ValueFormer: A Causal Transformer Value Function with Stage-Aware Labels for Semi-Autonomous Vision-Language-Action Policies",
  "abstract": "Vision-Language-Action (VLA) policies trained by behavior cloning fail silently: from the action stream alone, a collapsing rollout looks much like one making clean progress, because imitation supplies no notion of progress. Reinforcement learning would supply one, but it is impractical here, where real-robot experience is costly and deformable food resists simulation. The cheap alternative, a terminal success / failure bit, is learnable in principle yet far too sparse to say when a rollout went wrong. We argue that the per-frame label, not the architecture, is the hard part: to be useful it must be dense, continuous, and correctly shaped. We present ValueFormer, a compact policy-agnostic causal transformer over a frozen DINOv3 backbone that emits two per-frame signals in one forward pass: a smooth Monte Carlo value, V_mc, for advantage estimation and a sharp binary value for online mistake detection, targets that pull in opposite directions by design. Failed episodes are labeled with a stage-aware, success-then-decay return that preserves the success curve before the failure stage, and detection is supervised from mistake intervals rather than a single failure time, so mistakes the policy recovers from also carry signal. On a real-robot bimanual sandwich-assembly task 1,427 episodes), a critic-derived per-frame training weight lifts task completion from 70% to 85% (within noise at n=20), and a batched bf16 encoder cuts the live serving cost 3~5 times so the critic runs at 2 Hz alongside the policy on a single GPU.",
  "published": "2026-08-03",
  "updated": "2026-08-03",
  "year": "2026",
  "authors": [
   "Inkyu Sa",
   "Konstantin Stulov",
   "Rajat Bhageria"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "ValueFormer is presented, a compact policy-agnostic causal transformer over a frozen DINOv3 backbone that emits two per-frame signals in one forward pass: a smooth Monte Carlo value, V_mc, for advantage estimation and a sharp binary value for online mistake detection, targets that pull in opposite directions by design.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Inkyu Sa",
    "id": "2249529212",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Konstantin Stulov",
    "id": "102848910",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Rajat Bhageria",
    "id": "2455446308",
    "h_index": 0,
    "papers": 1
   }
  ],
  "comment": "22 pages, 17 figures, 8 tables",
  "topics": [
   "vla",
   "imitation-diffusion",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.02958v1",
  "pdf_url": "https://arxiv.org/pdf/2608.02958v1",
  "html_url": "https://arxiv.org/html/2608.02958v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.02886",
  "slug": "exact-signed-distance-control-barrier-functions-via-minkowski-operatio",
  "title": "Exact Signed-Distance Control Barrier Functions via Minkowski Operations for Safe Navigation among Polytopes",
  "abstract": "Safely navigating polytopic environments while respecting the dynamics, control, and exact geometry of the underlying system is a challenge in robotics. Control barrier functions (CBFs) synthesize safe control policies by rendering the safe set forward invariant, but many existing CBF-based methods approximate polytopes using conservative smooth shapes, such as spheres or ellipsoids, to obtain explicit differentiable distance functions. In this article, we propose an exact Signed Distance Function (SDF) formulation for a {\\it polytopic} robot and {\\it polytopic} obstacles and integrate it with nonsmooth CBFs. Leveraging Minkowski operations, the proposed method computes the exact SDF via companion convex programs in both the collision-free (positive-sign) and in-collision (negative-sign) cases. Furthermore, by exploiting the convenient geometric properties of 2D Minkowski operations and the optimality conditions of the two companion convex programs, we derive a unified analytical expression for the gradient of the exact SDF via sensitivity analysis. The exact rotational gradient further reveals a previously masked class of local minima induced by the coupling between geometry and nonholonomic kinematics. We demonstrate the effectiveness of the proposed framework through a pure-translation case and three scenarios with unicycle models involving recovery from an unsafe initialization and single- and multiple-obstacle avoidance. Comparisons with baseline methods highlight how the proposed framework enables non-conservative maneuvers and safety recovery.",
  "published": "2026-08-03",
  "updated": "2026-08-19",
  "year": "2026",
  "authors": [
   "Yi-Hsuan Chen",
   "Shuo Liu",
   "Wei Xiao",
   "Calin Belta",
   "Michael Otte"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The effectiveness of the proposed framework is demonstrated through a pure-translation case and three scenarios with unicycle models involving recovery from an unsafe initialization and single- and multiple-obstacle avoidance and Comparisons with baseline methods highlight how the proposed framework enables non-conservative maneuvers and safety recovery.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yi-Hsuan Chen",
    "id": "2116614080",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Shuo Liu",
    "id": "2251151534",
    "h_index": 6,
    "papers": 16
   },
   {
    "name": "Wei Xiao",
    "id": "2249763989",
    "h_index": 6,
    "papers": 18
   },
   {
    "name": "C. Belta",
    "id": "1730719",
    "h_index": 60,
    "papers": 412
   },
   {
    "name": "Michael W. Otte",
    "id": "2249853163",
    "h_index": 3,
    "papers": 11
   }
  ],
  "comment": "16 pages, 13 figures. Expanded version of a paper published in IEEE CDC 2025. Demo video: https://youtu.be/D0zVswzyxaE",
  "topics": [
   "rl-control",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.02886v2",
  "pdf_url": "https://arxiv.org/pdf/2608.02886v2",
  "html_url": "https://arxiv.org/html/2608.02886v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.02834",
  "slug": "biconvex-optimization-for-smooth-minimum-time-trajectories-around-conv",
  "title": "Biconvex Optimization for Smooth Minimum-Time Trajectories around Convex Obstacles",
  "abstract": "We present a biconvex approach for minimum-time motion planning around convex obstacles that is guaranteed to converge, is anytime, and supports derivative constraints to arbitrary order. We jointly convexify the minimum-time objective and all derivative constraints through a change of variables, and handle collision avoidance via time-varying separating planes, reducing the problem to a biconvex program. This program is solved by alternating between computing maximum-margin separating planes and optimizing the trajectory. By only adding planes for obstacles that the current iterate collides with, the trajectory can jump around obstacles and escape local minima. The method is guaranteed to converge starting from a simple collision-free polygonal curve. In our experiments on drone navigation and dual-arm bin unloading, we find that the proposed method reliably produces high-quality trajectories with computation times comparable to state-of-the-art decomposition-based motion planners, while handling a larger class of problems and being substantially more robust to bad initialization. Project page:https://wernerpe.github.io/bmtp-website/",
  "published": "2026-08-03",
  "updated": "2026-08-03",
  "year": "2026",
  "authors": [
   "Peter Werner",
   "Tobia Marcucci",
   "Daniela Rus"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "In the experiments on drone navigation and dual-arm bin unloading, the proposed method reliably produces high-quality trajectories with computation times comparable to state-of-the-art decomposition-based motion planners, while handling a larger class of problems and being substantially more robust to bad initialization.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Peter Werner",
    "id": "2281965029",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Tobia Marcucci",
    "id": "9482298",
    "h_index": 12,
    "papers": 21
   },
   {
    "name": "Daniela Rus",
    "id": "2253476549",
    "h_index": 5,
    "papers": 10
   }
  ],
  "comment": "18 pages, 9 figures, 4 tables. Submitted to IEEE Transactions on Robotics. Project page: https://wernerpe.github.io/bmtp-website/ Code: https://github.com/wernerpe/pybmtp",
  "topics": [
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.02834v1",
  "pdf_url": "https://arxiv.org/pdf/2608.02834v1",
  "html_url": "https://arxiv.org/html/2608.02834v1",
  "code_url": "https://wernerpe.github.io/bmtp-website/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.3
 },
 {
  "id": "2608.02811",
  "slug": "staying-on-spec-real-time-monitoring-under-uncertainty-with-a-maritime",
  "title": "Staying on Spec: Real-Time Monitoring under Uncertainty with a Maritime Case Study",
  "abstract": "Robotic systems must operate under uncertainty while satisfying complex task and safety specifications. Monitoring such specifications under uncertainty remains challenging, as existing formulations typically require extensive data or explicit uncertainty distributions. In this paper, we propose a real-time monitoring framework that reduces data requirements by leveraging data-driven reachable sets for specification evaluation. We instantiate the framework for maritime navigation, where complex specifications arise from traffic rules. We develop a data-efficient pipeline for constructing reachable sets and derive a monitoring formulation suitable for real-time deployment. Simulation and hardware experiments demonstrate robust monitoring under realistic disturbances, achieving improved risk detection compared to state-of-the-art metrics.",
  "published": "2026-08-03",
  "updated": "2026-08-03",
  "year": "2026",
  "authors": [
   "Elizabeth Dietrich",
   "Hanna Krasowski",
   "Emir Cem Gezer",
   "Roger Skjetne",
   "Asgeir Johan S\u00f8rensen",
   "Murat Arcak"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper develops a data-efficient pipeline for constructing reachable sets and derives a monitoring formulation suitable for real-time deployment that reduces data requirements by leveraging data-driven reachable sets for specification evaluation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Elizabeth Dietrich",
    "id": "2309717726",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Hanna Krasowski",
    "id": "2042798834",
    "h_index": 9,
    "papers": 34
   },
   {
    "name": "Emir Cem Gezer",
    "id": "39456418",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "R. Skjetne",
    "id": "2213983",
    "h_index": 32,
    "papers": 186
   },
   {
    "name": "A. S\u00f8rensen",
    "id": "2360353506",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Murat Arcak",
    "id": "2335659565",
    "h_index": 1,
    "papers": 13
   }
  ],
  "comment": "",
  "topics": [
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.02811v1",
  "pdf_url": "https://arxiv.org/pdf/2608.02811v1",
  "html_url": "https://arxiv.org/html/2608.02811v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.02809",
  "slug": "toward-certified-functional-safety-for-industrial-humanoid-robots-the",
  "title": "Toward Certified Functional Safety for Industrial Humanoid Robots: The Fail-Passive Gap and a Feasibility Study",
  "abstract": "Industrial humanoid robots are constrained less by locomotion or manipulation capability than by the immaturity of functional safety certification for legged platforms. The root difficulty is that the safe state of a legged robot is an actively-controlled state, which violates the fail-passive assumption underlying ISO~13849-1 / EN~60204-1: removing power from a walking biped causes an uncontrolled fall, so classical de-energization is itself a hazard. We term this the fail-passive gap and use a certified external safety chain (light curtain, emergency stop, fail-safe input, fail-safe PLC, and wireless PROFIsafe) as an instrument to locate it precisely: because the external chain is closed and quantifiable with established methods (PFHD, DC, CCF, PL/SILCL), the residual uncertifiable element is pinpointed to the robot-side reaction chain. Using a Siemens fail-safe S7-1500 emergency-stop reference, we show its certifiable Reaction subsystem is contactor-based power removal (Stop Category~0)---exactly the element a balancing humanoid cannot have. We deliberately do not claim end-to-end certified PL~e / SIL~3. We validate the approach on a Unitree G1 EDU pick-and-place cell in a 3m x 1.5m semi-enclosed workspace, and contribute a humanoid-specific analysis of the active safe state (fall-as-hazard, single-support stop bounds, balancing-policy residual risk, ISO~13855 separation) and a provenance-labeled timing budget. Hosting an industrial software-defined automation (SDA) controller on the robot, co-located with the balancing policy, moves robot-side PROFINET/PROFIsafe reception onto a standardized IEC~61131-3 interface; because the G1's onboard compute is not safety-rated hardware, this endpoint is not a certified safety runtime, which reinforces rather than resolves the fail-passive gap and localizes it to the SDA-to-balancing-policy interface.",
  "published": "2026-08-03",
  "updated": "2026-08-03",
  "year": "2026",
  "authors": [
   "Caiwu Ding",
   "Tao Cui",
   "Lingyun Wang",
   "Chengtao Wen"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Caiwu Ding",
    "id": "52167128",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Tao Cui",
    "id": "2399041633",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Lingyu Wang",
    "id": "2335527672",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Chengtao Wen",
    "id": "2455446880",
    "h_index": 0,
    "papers": 1
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "safety-eval"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2608.02809v1",
  "pdf_url": "https://arxiv.org/pdf/2608.02809v1",
  "html_url": "https://arxiv.org/html/2608.02809v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2608.02780",
  "slug": "semantic-haptic-feedback-enhances-dexterous-robotic-teleoperation",
  "title": "Semantic Haptic Feedback Enhances Dexterous Robotic Teleoperation",
  "abstract": "In robot teleoperation, haptic feedback can be used to help human operators accomplish dexterous manipulation tasks. However, existing haptic feedback methods try to replicate high-fidelity sensory haptics that are felt in real world interactions, which are constrained by the sensing and feedback hardware capability and may lead to higher workload. To addresses these limitations, this work introduces semantic haptics for teleoperation, which uses abstract haptic patterns to convey critical information about robot states. We categorize robot states into \"Confirmations\" and \"Exceptions\", implement a modular haptic rendering pipeline in robot simulation, and deliver semantic haptic feedback to operators through pneumatic and vibrotactile wristbands. This simplifies hardware requirements and enables one-to-many mappings between haptic patterns and robot states. Through three evaluation studies, we identify the most effective semantic haptic design for a common pick and place teleoperation task and compare semantic haptics to other teleoperation feedback approaches including sensory haptics and visual feedback. Results suggest that while semantic haptics performs similarly as other feedback in unimanual tasks, it achieves superior performance in bimanual tasks, with reduced task workload, increased situational awareness, and overall preference.",
  "published": "2026-08-03",
  "updated": "2026-08-03",
  "year": "2026",
  "authors": [
   "Bingjian Huang",
   "Sahar Aseeri",
   "Jonas Schmidtler",
   "Joseph Zhang",
   "Sonny Chan",
   "Andrew Doxon",
   "Jom Preechayasomboon",
   "Evan Pezent",
   "Alberto Rigo",
   "Amir Memar",
   "Nicholas Colonnese",
   "Chase Tymms"
  ],
  "author_count": 12,
  "categories": [
   "cs.RO",
   "cs.HC"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Bingjian Huang",
    "id": "2278800618",
    "h_index": 4,
    "papers": 14
   },
   {
    "name": "Sahar A. Aseeri",
    "id": "2892331",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Jonas Schmidtler",
    "id": "49610749",
    "h_index": 8,
    "papers": 19
   },
   {
    "name": "Joseph Zhang",
    "id": "2325916578",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Sonny Chan",
    "id": "3214725",
    "h_index": 14,
    "papers": 49
   },
   {
    "name": "Andrew J. Doxon",
    "id": "2622373",
    "h_index": 9,
    "papers": 21
   },
   {
    "name": "Jom Preechayasomboon",
    "id": "2422648749",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Evan Pezent",
    "id": "28341608",
    "h_index": 9,
    "papers": 17
   },
   {
    "name": "Alberto Rigo",
    "id": "2264504913",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Amirhossein H. Memar",
    "id": "2273991344",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Nicholas Colonnese",
    "id": "2190760912",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Chase Tymms",
    "id": "2079394057",
    "h_index": 4,
    "papers": 6
   }
  ],
  "comment": "18 pages, 7 figures",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.02780v1",
  "pdf_url": "https://arxiv.org/pdf/2608.02780v1",
  "html_url": "https://arxiv.org/html/2608.02780v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.02713",
  "slug": "quo-vadis-world-modeling",
  "title": "Quo Vadis, World Modeling?",
  "abstract": "Continually improving agents require dynamic interaction feedback beyond static supervision, yet direct real-environment interaction is costly, slow, unsafe, and hard to parallelize. World modeling offers a natural intermediate proxy that allows agents to query lower-cost, more controllable feedback before committing to real actions. Classical world models instantiate this proxy primarily through future physical-state prediction, a formulation useful yet narrow for agents that require actionable feedback beyond raw state transitions. In this work, we conceptualize Agent-Centric Interactive World Proxies, shifting the fundamental paradigm from physical state transitions to agent-usable information transitions, such as execution outcomes, retrieved experiences or skills, and verification signals, broadening the scope of world modeling to provide versatile feedback for continually improving agents. To systematically map this design space, we organize world proxies into six functional forms based on their feedback modalities: dynamics, spatial, execution, memory/experience, skill, and reward/verification proxies, which together characterize the primary ways world modeling serves agent improvement. We further analyze how these proxies empower agents across three progressive levels: L.1 Inference-Time Guidance, where proxy outputs enrich in-context information for superior decisions; L.2 Training-Time Optimization, where proxy outputs yield rewards, critiques, or synthetic rollouts for policy learning; and L.3 Agent-Proxy Co-Evolution, where real-environment evidence continuously updates both the proxy and the agent for co-evolution. Ultimately, this work recasts world modeling into an agent-centric paradigm, establishing a roadmap for building world proxies that empower agents to plan better, learn faster, and evolve continually.",
  "published": "2026-08-03",
  "updated": "2026-08-03",
  "year": "2026",
  "authors": [
   "Yu Yang",
   "Xuemeng Yang",
   "Licheng Wen",
   "Lingdong Kong",
   "Xiaobin Hu",
   "Dongyue Lu",
   "Wei Chow",
   "Xiyan Huang",
   "Yuxiang Feng",
   "Yue Liao",
   "Jianbiao Mei",
   "Daocheng Fu",
   "Rong Wu",
   "Pinlong Cai",
   "Ran Yi",
   "Ying Tai",
   "Jiangning Zhang",
   "Botian Shi",
   "Yong Liu",
   "Shuicheng Yan"
  ],
  "author_count": 20,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work conceptualize Agent-Centric Interactive World Proxies, shifting the fundamental paradigm from physical state transitions to agent-usable information transitions, such as execution outcomes, retrieved experiences or skills, and verification signals, broadening the scope of world modeling to provide versatile feedback for continually improving agents.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yu Yang",
    "id": "2378747459",
    "h_index": 3,
    "papers": 13
   },
   {
    "name": "Xuemeng Yang",
    "id": "2265828805",
    "h_index": 12,
    "papers": 34
   },
   {
    "name": "Licheng Wen",
    "id": "153109152",
    "h_index": 15,
    "papers": 44
   },
   {
    "name": "Lingdong Kong",
    "id": "2335574081",
    "h_index": 9,
    "papers": 34
   },
   {
    "name": "Xiaobin Hu",
    "id": "2382924542",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Dongyue Lu",
    "id": "2334806165",
    "h_index": 8,
    "papers": 20
   },
   {
    "name": "Wei Chow",
    "id": "2282531255",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Xiyan Huang",
    "id": "2455559883",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yuxiang Feng",
    "id": "2317268399",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Yue Liao",
    "id": "2362056474",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Jianbiao Mei",
    "id": "2106504583",
    "h_index": 16,
    "papers": 47
   },
   {
    "name": "Daocheng Fu",
    "id": "150967272",
    "h_index": 20,
    "papers": 45
   },
   {
    "name": "Rong Wu",
    "id": "2363400631",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Pinlong Cai",
    "id": "26978261",
    "h_index": 19,
    "papers": 68
   },
   {
    "name": "Ran Yi",
    "id": "2447169047",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Ying Tai",
    "id": "2325161493",
    "h_index": 7,
    "papers": 18
   },
   {
    "name": "Jiang-She Zhang",
    "id": "2281104439",
    "h_index": 8,
    "papers": 33
   },
   {
    "name": "Botian Shi",
    "id": "2278899936",
    "h_index": 13,
    "papers": 36
   },
   {
    "name": "Yong Liu",
    "id": "2358455660",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Shuicheng Yan",
    "id": "2367752546",
    "h_index": 7,
    "papers": 22
   }
  ],
  "comment": "Technical Blog at https://worldbench.github.io/awesome-agentic-world-model GitHub Repo at https://github.com/worldbench/awesome-agentic-world-model",
  "topics": [
   "world-models",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.02713v1",
  "pdf_url": "https://arxiv.org/pdf/2608.02713v1",
  "html_url": "https://arxiv.org/html/2608.02713v1",
  "code_url": "https://worldbench.github.io/awesome-agentic-world-model",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2608.02580",
  "slug": "ego2robot-scalable-robot-data-synthesis-from-egocentric-human-data",
  "title": "Ego2Robot: Scalable Robot Data Synthesis from Egocentric Human Data",
  "abstract": "Learning generalizable robot manipulation policies requires large-scale and diverse demonstration data. Egocentric human manipulation videos offer rich scene and task diversity, and prior work has shown that retargeting and rendering such videos into robot-format data can yield effective per-task policies at small scale. However, whether this approach can provide pretraining benefits for vision-language-action models at scale remains unexplored. We present \\textbf{Ego2Robot}, a scalable pipeline that converts egocentric human manipulation videos into robot training data through action retargeting, robot-arm visual synthesis, and multi-level quality curation. Ego2Robot supports both curated datasets and in-the-wild videos, producing 18,561 hours of robot training data spanning 15 robot morphologies, making it the largest ego-to-robot dataset to date. To evaluate generalization, we extend RoboTwin2.0 with disentangled perturbation axes covering visual appearance, scene layout, embodiment morphology, and task semantics. Experiments show that joint pretraining on Ego2Robot-synthesized and robot data consistently improves out-of-distribution generalization across multiple perturbation types, with benefits validated on real-robot deployment. Project page: https://www-ye.github.io/ego2robot_blog/",
  "published": "2026-08-03",
  "updated": "2026-08-03",
  "year": "2026",
  "authors": [
   "Ye Wang",
   "Pei Lin",
   "Xiong-Hui Chen",
   "Haoqi Yuan",
   "Zhixuan Liang",
   "Yiyang Huang",
   "Anzhe Chen",
   "Zixing Lei",
   "Jie Zhang",
   "Tao Zhang",
   "Haoyang Li",
   "Tong Zhang",
   "Chenxi Xiao",
   "Ziyuan Jiao",
   "Qin Jin"
  ],
  "author_count": 15,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Experiments show that joint pretraining on Ego2Robot-synthesized and robot data consistently improves out-of-distribution generalization across multiple perturbation types, with benefits validated on real-robot deployment.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ye Wang",
    "id": "2291336339",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Peibin Lin",
    "id": "2187091585",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Xiong-hui Chen",
    "id": "2364976567",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Haoqi Yuan",
    "id": "1429192914",
    "h_index": 12,
    "papers": 37
   },
   {
    "name": "Zhixuan Liang",
    "id": "2257485304",
    "h_index": 10,
    "papers": 35
   },
   {
    "name": "Yiyang Huang",
    "id": "2443674360",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "An-Jen Chen",
    "id": "31187917",
    "h_index": 4,
    "papers": 18
   },
   {
    "name": "Zixing Lei",
    "id": "2176779762",
    "h_index": 6,
    "papers": 20
   },
   {
    "name": "Jie Zhang",
    "id": "2261470572",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Tao Zhang",
    "id": "2146342317",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Haoyang Li",
    "id": "2274084217",
    "h_index": 7,
    "papers": 35
   },
   {
    "name": "Tong Zhang",
    "id": "2362614855",
    "h_index": 4,
    "papers": 16
   },
   {
    "name": "Chenxi Xiao",
    "id": "2356920855",
    "h_index": 3,
    "papers": 14
   },
   {
    "name": "Ziyuan Jiao",
    "id": "2356946141",
    "h_index": 3,
    "papers": 13
   },
   {
    "name": "Qin Jin",
    "id": "2290783013",
    "h_index": 6,
    "papers": 16
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "egocentric-data",
   "foundation-pretraining",
   "hardware-codesign",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.02580v1",
  "pdf_url": "https://arxiv.org/pdf/2608.02580v1",
  "html_url": "https://arxiv.org/html/2608.02580v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.02571",
  "slug": "situation-aware-frontier-prioritization-for-quadruped-search-and-rescu",
  "title": "Situation Aware Frontier Prioritization for Quadruped Search and Rescue",
  "abstract": "Quadruped robots are a promising platform for search and rescue missions because they can navigate cluttered indoor environments that may be restrictive for wheeled systems. However, in unknown rescue scenarios, autonomous exploration must balance map expansion with the likelihood of finding victims, which is not explicitly addressed by clas- sical frontier selection strategies. This paper presents a situation aware frontier prioritization method for single robot quadruped search and rescue. The proposed approach preserves the frontier exploration framework, but extends frontier ranking with information gain, observation deficit, rescue relevance, terrain penalty, and travel cost. The method is eval- uated in Gazebo simulation with a quadruped robot in two indoor rescue scenarios with different levels of difficulty. The first scenario is used as a sanity check, while the second introduces stronger clutter and frontier ambiguity. Experimental results show that all methods perform reliably in a simple scenario, whereas in a complex scenario is different. In that setting, the proposed method achieves the highest completion rate and the highest victim recovery among the evaluated approaches. These results indicate that situation aware frontier prioritization is beneficial when frontier choice becomes nontrivial and rescue utility must be balanced against generic exploration objectives.",
  "published": "2026-08-03",
  "updated": "2026-08-03",
  "year": "2026",
  "authors": [
   "Kevin Farias",
   "Santiago Martin",
   "Barbara Flores",
   "Vinicio Melgar",
   "Igor Nunes",
   "Hiago Sodre",
   "Pablo Moraes",
   "Ricardo B. Grando"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A situation aware frontier prioritization method for single robot quadruped search and rescue that preserves the frontier exploration framework, but extends frontier ranking with information gain, observation deficit, rescue relevance, terrain penalty, and travel cost is presented.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Kevin Farias",
    "id": "2355348705",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Santiago Martin",
    "id": "2455443277",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Barbara Flores",
    "id": "2455416801",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Vinicio Melgar",
    "id": "2306251618",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Igor Nunes",
    "id": "2355348764",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Hiago Sodre",
    "id": "2306251592",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "P. Moraes",
    "id": "2306249192",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Ricardo B. Grando",
    "id": "2306259863",
    "h_index": 3,
    "papers": 15
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.02571v1",
  "pdf_url": "https://arxiv.org/pdf/2608.02571v1",
  "html_url": "https://arxiv.org/html/2608.02571v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.02497",
  "slug": "grounded-semantic-re-binding-for-robust-instruction-generalization-in",
  "title": "Grounded Semantic Re-Binding for Robust Instruction Generalization in Vision-Language-Action Models",
  "abstract": "Vision-Language-Action (VLA) models excel in robotic manipulation but suffer catastrophic performance drops when canonical instructions are simply paraphrased. Although this brittleness is typically addressed through costly data scaling, our probing reveals that the root cause is architectural rather than a lack of semantic understanding. Specifically, we demonstrate that current VLAs successfully retain the correct task identity internally. The failure actually stems from the joint encoding of dynamic visual observations and text, which introduces systematic feature shifts. Because the downstream action policy is highly vulnerable to these variations, it fails to translate the preserved semantics into correct control commands. To resolve this structural bottleneck, we propose Grounded Semantic Re-binding (GSR), an elegant intervention that bypasses unstable joint routing by explicitly fusing independently extracted task semantics with native visual features to train a completely re-initialized action expert from scratch. This targeted intervention dramatically restores paraphrastic invariance using only canonical demonstrations. On the LIBERO-Para benchmark, GSR improves success rates by up to 44.6 percent. It enables lightweight models to rival massively scaled baselines and pushes state-of-the-art models to a new record PRIDE score of 70.4, outperforming the recently introduced large-scale pretrained model Xiaomi-Robotics-0 in instruction generation capabilities. Building on these insights, we also introduce ParaVLA, a natively decoupled 0.33B-parameter model exhibiting near-perfect robustness to instruction rewording. Ultimately, our work proves that robust semantic grounding can be achieved through elegant structural design, bypassing the inefficient brute-force data scaling paradigm.",
  "published": "2026-08-03",
  "updated": "2026-08-03",
  "year": "2026",
  "authors": [
   "Zhaokai Yin",
   "Zhipeng Zhang"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work demonstrates that robust semantic grounding can be achieved through elegant structural design, bypassing the inefficient brute-force data scaling paradigm and introduces ParaVLA, a natively decoupled 0.33B-parameter model exhibiting near-perfect robustness to instruction rewording.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhaokai Yin",
    "id": "2455235839",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Zhipeng Zhang",
    "id": "2366555145",
    "h_index": 5,
    "papers": 16
   }
  ],
  "comment": "23 pages, 8 figures",
  "topics": [
   "vla",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.02497v1",
  "pdf_url": "https://arxiv.org/pdf/2608.02497v1",
  "html_url": "https://arxiv.org/html/2608.02497v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.02453",
  "slug": "certifying-plans-under-model-mismatch-a-trilemma-for-reachability-from",
  "title": "Certifying Plans under Model Mismatch: A Trilemma for Reachability from Scarce Data",
  "abstract": "Sim-to-real policies are designed under nominal dynamics, but target-system trials may yield only a few isolated one-step transitions. We study pre-execution certification of a fixed control sequence, such as an action chunk produced by a learned policy. If the sequence reaches an unobserved state-input region, the observations remain consistent with target systems whose trajectories separate along it by an arbitrarily large amount. Any deterministic certifier sound for all of them must then decline to certify or return a reachable tube with arbitrarily large projected width. For bounded smooth classes of the target-nominal model error, we derive a finite plan-dependent projected-width lower bound. These results expose a trilemma among uniform trajectory containment, finite projected width, and unrestricted model-error behavior beyond the observations. ForeReach requires a supplied componentwise Lipschitz bound on the model error. Observed transition pairs can refute this declaration but cannot establish it outside the observed locations. Conditional on a valid declaration, our method constructs a set-membership envelope for the model error, propagates a zonotopic reachable tube, and certifies only when propagation remains within the certification domain and every projected tube slice avoids the unsafe set. In two benchmark systems, calibration baselines may remain narrow after losing trajectory containment outside data support, whereas our method declines to certify unsupported sequences and recovers certification when relevant target data and sufficient obstacle clearance are available.",
  "published": "2026-08-03",
  "updated": "2026-08-03",
  "year": "2026",
  "authors": [
   "Yanliang Huang",
   "Zhen Zhang",
   "Ahmad Hafez",
   "Wenyuan Wu",
   "Peng Xie",
   "Zhuoqi Zeng",
   "Amr Alanwar"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work studies pre-execution certification of a fixed control sequence, such as an action chunk produced by a learned policy, and constructs a set-membership envelope for the model error, propagates a zonotopic reachable tube, and recovers certification when relevant target data and sufficient obstacle clearance are available.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yanliang Huang",
    "id": "2261875071",
    "h_index": 3,
    "papers": 16
   },
   {
    "name": "Zhen Zhang",
    "id": "2354233844",
    "h_index": 2,
    "papers": 19
   },
   {
    "name": "Ahmad Hafez",
    "id": "2293614450",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Wenyu Wu",
    "id": "2354982743",
    "h_index": 1,
    "papers": 10
   },
   {
    "name": "Peng Xie",
    "id": "2354320476",
    "h_index": 3,
    "papers": 21
   },
   {
    "name": "Zhuoqi Zeng",
    "id": "2358994946",
    "h_index": 3,
    "papers": 16
   },
   {
    "name": "Amr Alanwar",
    "id": "27706391",
    "h_index": 12,
    "papers": 75
   }
  ],
  "comment": "",
  "topics": [
   "sim2real"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.02453v1",
  "pdf_url": "https://arxiv.org/pdf/2608.02453v1",
  "html_url": "https://arxiv.org/html/2608.02453v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.02449",
  "slug": "moral-sensor-grounded-bev-reasoning-for-compact-vlms-toward-edge-orien",
  "title": "MoRAL: Sensor-Grounded BEV Reasoning for Compact VLMs toward Edge-Oriented Autonomous Driving",
  "abstract": "Deploying vision-language models (VLMs) for safety-critical spatial reasoning on resource-constrained autonomous driving platforms requires both compact model size and reliable metric grounding. We present MoRAL (Multimodal Reasoning for Autonomous Language Models), a two-stage fine-tuning pipeline that teaches Cosmos-Reason2-2B to first read a physics-encoded Bird's Eye View (BEV) representation and then reason over it for driving decisions. The BEV image encodes LiDAR metric distance as color bands, object class as cluster morphology, and radar Doppler velocity as directional wedge overlays, externalizing spatial perception into the input image so that no learned 3D backbone is required at inference. Stage 1 fine-tunes the vision encoder on 60,000 grounding records; zero-shot baselines produce no parseable BEV outputs, confirming the vocabulary requires explicit training. Stage 2 fine-tunes the full model (52M parameters, 2.4% of total) on 57,696 chain-of-thought records generated by Cosmos-Reason2-8B as teacher, spanning eight driving question types. On 2,304 held-out nuScenes frames evaluated by Gemma 4 (31B) calibrated against human review, MoRAL wins seven of eight question types over a zero-shot 8B baseline despite using four times fewer parameters, with the largest margins on question types requiring structured multi-step physics reasoning. Emergency braking recall improves from 10.8% to 47.8%, output degeneration falls from 94.1% to 20.8%, and the full pipeline fits a consumer 8 GB GPU at 42 tok/s without quantization. These results establish a reproducible foundation for compact, physics-grounded VLM reasoning on mobile edge platforms.",
  "published": "2026-08-03",
  "updated": "2026-08-03",
  "year": "2026",
  "authors": [
   "Ambarish Govindarajulu Kaliamurthi",
   "Kaikai Liu"
  ],
  "author_count": 2,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "MoRAL (Multimodal Reasoning for Autonomous Language Models), a two-stage fine-tuning pipeline that teaches Cosmos-Reason2-2B to first read a physics-encoded Bird's Eye View (BEV) representation and then reason over it for driving decisions, establishes a reproducible foundation for compact, physics-grounded VLM reasoning on mobile edge platforms.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ambarish Govindarajulu Kaliamurthi",
    "id": "2455413140",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Kai Liu",
    "id": "2452514777",
    "h_index": 0,
    "papers": 2
   }
  ],
  "comment": "7 pages, 5 figures, 6 tables. Accepted to the 14th IEEE International Conference on Intelligent Mobile Computing (IEEE IMC 2026), Fukuoka, Japan, July 27-30, 2026",
  "topics": [
   "spatial-3d",
   "navigation",
   "hardware-codesign",
   "video-generation",
   "safety-eval"
  ],
  "orgs": [
   "NVIDIA"
  ],
  "abs_url": "https://arxiv.org/abs/2608.02449v1",
  "pdf_url": "https://arxiv.org/pdf/2608.02449v1",
  "html_url": "https://arxiv.org/html/2608.02449v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2608.02395",
  "slug": "environmental-resilience-via-morphological-diversity-within-machines",
  "title": "Environmental resilience via morphological diversity within machines",
  "abstract": "Organisms contain diverse, sensorimotor parts across size scales and rapidly adapt to new environments, while machines contain only inert materials at smaller scales and struggle with surprise. We hypothesize that this agents-within-agents quality of organisms may aid their resilience: increasing experiences with internal physical adversity may pre-train organisms and machines to handle external adversity, such as encounters with new environments. Not only has this hypothesis not yet been articulated, mechanisms enabling this phenomenon have yet to be proposed. Here we show a mechanism by which this can occur: we found that physical connectors, in learning to restore behavior to previously independent, morphologically diverse agents they disrupted by tethering them together, trigger and tame sufficiently diverse disruptions that later encounters with new environments trigger disruptions that fall within this manageable range, enabling the collective to continue behaving properly without any additional learning or adaptation. Further, we found that building collectives from more agents, or more diverse agents, further increases the collective's resilience to new environments. This suggests that not just taming but intentionally creating internal physical adversity may indeed prepare organisms for external adversity, and could do so for machines, if they were built from smaller machines.",
  "published": "2026-08-03",
  "updated": "2026-08-03",
  "year": "2026",
  "authors": [
   "Alice Hein",
   "Josh Bongard"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Alice Hein",
    "id": "2158986489",
    "h_index": 4,
    "papers": 14
   },
   {
    "name": "J. Bongard",
    "id": "7373730",
    "h_index": 44,
    "papers": 218
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.02395v1",
  "pdf_url": "https://arxiv.org/pdf/2608.02395v1",
  "html_url": "https://arxiv.org/html/2608.02395v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.02385",
  "slug": "stablemimic-smooth-human-like-recovery-for-humanoid-motion-tracking-le",
  "title": "StableMimic: Smooth Human-Like Recovery for Humanoid Motion Tracking - Learning Beyond the Tracking Distribution for Structured Post-Fall Behavior",
  "abstract": "Humanoid motion trackers perform reliably within learned tracking distributions, but falls can move the robot into low-height, contact-rich states from which an advancing command is temporarily unreachable. Tracking-only policies may chase infeasible references, producing rapid, large-amplitude limb corrections that increase risk to the robot and its surroundings. We present StableMimic, a unified tracker trained beyond the nominal tracking distribution. Perturbed resets around multiple human get-up references expose prone, supine, off-balance, and intermediate ground-contact states, shaping structured recovery that returns the robot to the trackable region. Because tracking and recovery occupy markedly different state--action distributions, StableMimic uses dedicated experts for each regime and a proprioceptive gate that continuously blends their actions. A hidden successor-state objective teaches human-reference-shaped recovery without exposing reference identity or phase to the deployed Actor; deployment requires no get-up reference, recovery command, trajectory retrieval, or external policy switch. On the complete retargeted LAFAN1 dance subset, StableMimic achieves the lowest errors on all four tracking metrics among five methods. Across 100 matched push-to-fall trials per method, it recovers in 100/100 and attains the lowest values on six of seven post-fall motion and load measures, supporting improved interaction safety under this protocol. Real Unitree G1 dance and standing-reference deployments qualitatively demonstrate bounded limb motion, autonomous recovery, and command resumption.",
  "published": "2026-08-03",
  "updated": "2026-08-03",
  "year": "2026",
  "authors": [
   "Weihao Wu",
   "Ming Huang",
   "Ruofei Liu",
   "Jinglei Nie",
   "Shuxiang Guo",
   "Chunying Li"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "StableMimic is presented, a unified tracker trained beyond the nominal tracking distribution that achieves the lowest errors on all four tracking metrics among five methods and attains the lowest values on six of seven post-fall motion and load measures, supporting improved interaction safety under this protocol.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Weihao Wu",
    "id": "2376576469",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Mingzhe Huang",
    "id": "2449156186",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Ruofei Liu",
    "id": "2454707266",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Jinglei Nie",
    "id": "2316501687",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Shuxiang Guo",
    "id": "2316802006",
    "h_index": 3,
    "papers": 53
   },
   {
    "name": "Chunying Li",
    "id": "48161747",
    "h_index": 12,
    "papers": 68
   }
  ],
  "comment": "8 pages, 7 figures. Preprint, not formally peer-reviewed",
  "topics": [
   "humanoids",
   "tactile",
   "safety-eval"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2608.02385v1",
  "pdf_url": "https://arxiv.org/pdf/2608.02385v1",
  "html_url": "https://arxiv.org/html/2608.02385v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2608.02365",
  "slug": "faster-wam-do-world-action-models-need-deep-action-modules",
  "title": "Faster-WAM: Do World Action Models Need Deep Action Modules?",
  "abstract": "World Action Models (WAMs) couple robot action prediction with video world models. Existing WAMs with shared-backbone and Mixture-of-Transformers designs generally tie the depth of the action module to that of the video backbone, resulting in substantial computational overhead and high inference latency. To address this limitation, we introduce Dock of Transformer (DoT), a video-centric design principle that treats a pretrained video Transformer as a representation hub and connects lightweight output-heads through docking interfaces. This enables flexible output-head design while providing direct access to representations from all layers of the backbone. We then introduce \\textbf{Faster-WAM}, an instantiation of DoT for WAMs, which docks a single-layer action head onto a 30-layer video backbone. The docking interface fuses keys and values from all video layers and applies RoPE realignment. Without additional embodied pretraining, Faster-WAM achieves competitive performance on LIBERO and RoboTwin 2.0 while demonstrating strong out-of-distribution generalization on LIBERO-Plus. Faster-WAM also achieves the lowest end-to-end latency in our controlled comparison, requiring only 66.5 ms per inference --- a \\(3.2\\times\\) speedup over Fast-WAM. Overall, these results demonstrate that the video-centric DoT architecture supports flexible task-specific head design while delivering low inference latency, strong action-prediction performance, and robust generalization.",
  "published": "2026-08-03",
  "updated": "2026-08-03",
  "year": "2026",
  "authors": [
   "Liheng Ma",
   "Rui Heng Yang",
   "Zhanguang Zhang",
   "Mateo Clemente",
   "Ziwen Hu",
   "Tongtong Cao",
   "Yingxue Zhang"
  ],
  "author_count": 7,
  "categories": [
   "cs.AI",
   "cs.LG",
   "cs.RO"
  ],
  "primary_category": "cs.AI",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "Faster-WAM, an instantiation of DoT for WAMs, which docks a single-layer action head onto a 30-layer video backbone, achieves competitive performance on LIBERO and RoboTwin 2.0 while demonstrating strong out-of-distribution generalization on LIBERO-Plus.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Liheng Ma",
    "id": "1892081076",
    "h_index": 10,
    "papers": 24
   },
   {
    "name": "R. Yang",
    "id": "2346771941",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Zhanguang Zhang",
    "id": "2302452881",
    "h_index": 7,
    "papers": 24
   },
   {
    "name": "Mateo Clemente",
    "id": "2359259561",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Ziwen Hu",
    "id": "2455427364",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Tongtong Cao",
    "id": "2326975042",
    "h_index": 3,
    "papers": 16
   },
   {
    "name": "Yingxue Zhang",
    "id": "2260822531",
    "h_index": 4,
    "papers": 18
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "foundation-pretraining",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.02365v1",
  "pdf_url": "https://arxiv.org/pdf/2608.02365v1",
  "html_url": "https://arxiv.org/html/2608.02365v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2608.02326",
  "slug": "chainvla-chaining-vision-language-action-queries-through-a-unified-exe",
  "title": "ChainVLA: Chaining Vision-Language-Action Queries through a Unified Execution State for Long-Horizon Manipulation",
  "abstract": "Humans perform long-horizon manipulation by retaining knowledge of what earlier actions have established while continuously adapting the motion underway. By contrast, action-chunked vision-language-action (VLA) policies repeatedly replan from the current input at each query. Existing methods preserve either long-term task evidence through memory or short-term motion through action reuse and ensembling, leaving the cross-query handoff incomplete. We introduce ChainVLA, a 1.2B-parameter VLA policy that chains successive queries through a joint and revisable execution state. Progress Context combines a recurrent Working State with sparse event memory to carry observation-derived task progress, while Motion Tail feeds the preceding prediction's unexecuted continuation into state construction and action generation. Together, the two components condition a decoder that regenerates each action horizon under the latest observation, allowing the carried state to guide the next prediction without fixing it. ChainVLA reaches 62.8% average success on RMBench and 98.8% across four LIBERO suites, while removing Motion Tail or Progress Context reduces RMBench success to 11.2% and 3.0%, respectively. These asymmetric ablations are consistent with motion continuity helping preserve the observation stream from which task progress is inferred.",
  "published": "2026-08-03",
  "updated": "2026-08-04",
  "year": "2026",
  "authors": [
   "Yuzhi Huang",
   "Weijue Bu",
   "Ziyi Xiong",
   "Jie Wu",
   "Fanding Huang",
   "Jingyan Jiang",
   "Zhi Wang"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "ChainVLA is introduced, a 1.2B-parameter VLA policy that chains successive queries through a joint and revisable execution state that is consistent with motion continuity helping preserve the observation stream from which task progress is inferred.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yuzhi Huang",
    "id": "2329311852",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Wei Bu",
    "id": "2350047690",
    "h_index": 5,
    "papers": 15
   },
   {
    "name": "Ziyi Xiong",
    "id": "2186309854",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Jie Wu",
    "id": "2280913191",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Fanding Huang",
    "id": "2353326314",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Jingyan Jiang",
    "id": "2296747178",
    "h_index": 6,
    "papers": 26
   },
   {
    "name": "Zhi Wang",
    "id": "2305646994",
    "h_index": 4,
    "papers": 17
   }
  ],
  "comment": "13 pages (9 main + 4 appendix), 4 figures. Project page: https://muqy1818.github.io/chainvla-web/",
  "topics": [
   "vla"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.02326v2",
  "pdf_url": "https://arxiv.org/pdf/2608.02326v2",
  "html_url": "https://arxiv.org/html/2608.02326v2",
  "code_url": "https://muqy1818.github.io/chainvla-web/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2608.02320",
  "slug": "travkan-fast-and-interpretable-nonlinear-traversability-analysis-with",
  "title": "TravKAN: Fast and Interpretable Nonlinear Traversability Analysis with Kolmogorov-Arnold Networks",
  "abstract": "Traversability analysis is a fundamental capability for autonomous mobile robots operating in unstructured environments. While modern machine learning approaches such as deep neural networks and gradient-boosted trees achieve strong predictive performance, they lack interpretability and provide limited insight into the underlying terrain-robot interaction dynamics. In this paper, we propose TravKAN, a Kolmogorov-Arnold Network-based framework for fast, scalable, and interpretable traversability estimation. TravKAN represents multivariate decision functions through compositions of learnable univariate functions, enabling compact architectures and symbolic extraction of analytic expressions after training. In addition, we introduce a novel set of handcrafted features derived from the reflectivity channel of LiDAR sensors. To the best of our knowledge, reflectivity has not been systematically exploited for handcrafted traversability descriptors, despite its potential to capture material and surface properties complementary to geometric cues. We evaluate TravKAN on public, real-world urban and off-road datasets and compare it against strong baselines. TravKAN achieves strong performance across all metrics, outperforming conventional deep models and approaching the performance of XGBoost. TravKAN-Lite, i.e., TravKAN's symbolic representation, reveals meaningful nonlinear feature interactions and provides a compact, deployment-friendly, and fast analytic model. Ablation studies further show the robustness of our method to architectural variations and quantify the contribution of the proposed reflectivity-based features. These properties make TravKAN attractive for robotic systems requiring transparency, real-time computational efficiency, and interpretability in safety-critical decision-making.",
  "published": "2026-08-03",
  "updated": "2026-08-03",
  "year": "2026",
  "authors": [
   "Daniel Fusaro",
   "Simone Mosco",
   "Wanmeng Li",
   "Alberto Pretto"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "TravKAN achieves strong performance across all metrics, outperforming conventional deep models and approaching the performance of XGBoost, and TravKAN-Lite, i.e., TravKAN's symbolic representation, reveals meaningful nonlinear feature interactions and provides a compact, deployment-friendly, and fast analytic model.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Daniel Fusaro",
    "id": "2124733111",
    "h_index": 5,
    "papers": 21
   },
   {
    "name": "Simone Mosco",
    "id": "2299943701",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Wanmeng Li",
    "id": "2299953654",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Alberto Pretto",
    "id": "2268037616",
    "h_index": 4,
    "papers": 20
   }
  ],
  "comment": "This paper has been accepted for publication at the 2026 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS)",
  "topics": [
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.02320v1",
  "pdf_url": "https://arxiv.org/pdf/2608.02320v1",
  "html_url": "https://arxiv.org/html/2608.02320v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.02304",
  "slug": "trace-ergodic-trajectory-optimization-for-active-scene-reconstruction",
  "title": "TRACE: Ergodic Trajectory Optimization for Active Scene Reconstruction",
  "abstract": "Existing active reconstruction systems with Gaussian-splatting maps select observations greedily, optimizing a single next-best-view (NBV) at each step and connecting the chosen views by short-horizon path planning. This greedy decoupling disregards the global structure of scene information, producing inefficient trajectories that waste sensing capacity in transit between selected views. In this work, we study active reconstruction as an ergodic coverage problem: the time-averaged spatial statistics of the sensor trajectory should match a target information distribution induced by the current map. Our approach derives this target distribution online from uncertainty and visibility, and calculates ergodic trajectories via a kernel-ergodic horizon planner with gradient flow and footprint depletion, closing the loop between mapping and trajectory optimization. We thoroughly evaluate TRACE on the Replica dataset against the Next-Best-View (NBV) baselines, improving PSNR by 1.5 dB. Code: https://github.com/spikelab-jhu/trace-active-reconstruction.",
  "published": "2026-08-03",
  "updated": "2026-08-06",
  "year": "2026",
  "authors": [
   "Ziyue Zheng",
   "Linli Shi",
   "Bingkun He",
   "Wen Jiang",
   "Ziyun Wang"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work researches active reconstruction as an ergodic coverage problem: the time-averaged spatial statistics of the sensor trajectory should match a target information distribution induced by the current map, and calculates ergodic trajectories via a kernel-ergodic horizon planner with gradient flow and footprint depletion.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ziyue Zheng",
    "id": "2164759812",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Lin Shi",
    "id": "2443613474",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Bingkun He",
    "id": "2455254514",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Wen Jiang",
    "id": "2268758399",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Ziyun Wang",
    "id": "2269451841",
    "h_index": 6,
    "papers": 12
   }
  ],
  "comment": "11 pages, 7 figures, fixed a template bug in the Latex",
  "topics": [
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.02304v3",
  "pdf_url": "https://arxiv.org/pdf/2608.02304v3",
  "html_url": "https://arxiv.org/html/2608.02304v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.02257",
  "slug": "learning-panorama-aware-vla-for-mobile-manipulation-with-whole-body-te",
  "title": "Learning Panorama-Aware VLA for Mobile Manipulation with Whole-Body Teleoperation",
  "abstract": "Mobile manipulation is a key capability for embodied intelligence, enabling robots to accomplish complex multi-stage tasks in open-world environments. However, mobile manipulation poses two key challenges for vision-language-action (VLA) policies: At the data level, the efficient collection of high-quality whole-body demonstrations demands the coordinated control of both the mobile base and the robotic arms; at the model level, existing VLA models predominantly rely on local camera observations, whose limited field of view hinders global spatial understanding. To address these challenges, we develop a whole-body teleoperation system and a panoramic-aware VLA policy. The system enables coordinated control of a wheeled bimanual robot through a single VR interface and supports the acquisition of a real-world mobile manipulation dataset comprising 5.5 hours of multimodal demonstrations. Building upon this dataset, we propose PanoVLA, a panorama-aware vision-language-action policy for mobile bimanual manipulation. Built upon a Mixture-of-Transformers architecture, PanoVLA introduces global spatial context through dedicated panorama encoding and fusion modules, enabling effective integration of panoramic observations with language instructions and robot states for action generation. Evaluation on four real-world mobile manipulation tasks demonstrates that PanoVLA achieves an average stage completion rate of 91.3\\% and an end-to-end success rate of 73.4\\%, substantially outperforming local-view baselines. These results demonstrate that incorporating panoramic spatial context improves spatial understanding and closed-loop manipulation performance in mobile robots.",
  "published": "2026-08-03",
  "updated": "2026-08-03",
  "year": "2026",
  "authors": [
   "Donglin Yang",
   "Haoran Chen",
   "Xingyu Chen",
   "Lixing Liu",
   "Manyi Li",
   "Changhe Tu",
   "Ke Xu",
   "Xiaojian Ma",
   "Si Liu"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "PanoVLA introduces global spatial context through dedicated panorama encoding and fusion modules, enabling effective integration of panoramic observations with language instructions and robot states for action generation and demonstrates that incorporating panoramic spatial context improves spatial understanding and closed-loop manipulation performance in mobile robots.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Donglin Yang",
    "id": "2239165347",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Hao Chen",
    "id": "2446381246",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Xingyu Chen",
    "id": "2450424926",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Lixing Liu",
    "id": "2455433726",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Manyi Li",
    "id": "2243080602",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Changhe Tu",
    "id": "2285765172",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Ke Xu",
    "id": "2361735991",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Xiaojian Ma",
    "id": "121875989",
    "h_index": 18,
    "papers": 25
   },
   {
    "name": "Si Liu",
    "id": "2325537006",
    "h_index": 7,
    "papers": 13
   }
  ],
  "comment": "8 pages, 4 figures",
  "topics": [
   "vla",
   "humanoids",
   "navigation",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.02257v1",
  "pdf_url": "https://arxiv.org/pdf/2608.02257v1",
  "html_url": "https://arxiv.org/html/2608.02257v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.02197",
  "slug": "look-where-it-matters-adaptive-visual-refinement-for-vision-language-a",
  "title": "Look Where It Matters: Adaptive Visual Refinement for Vision-Language-Action Models",
  "abstract": "Visual representations of VLA models remain unreliable for spatially precise robotic manipulation. We uncover that vision encoders in VLAs also exhibit attention artifacts previously documented in generic Vision Transformers, and further show that, in embodied policies, these artifacts are closely associated with spatial perception capabilities acquired during post-training. As the encoder learns task-relevant information such as object location, depth ordering, and local geometry, limited global-token capacity causes part of this information to spill into low-information patch tokens. We introduce AtVLA, a framework that inserts learnable register tokens into the visual encoder. Trained end-to-end using only embodied data and the original action objective, these registers emerge as dedicated carriers of embodied spatial information, while the remaining patch tokens recover clean and spatially faithful attention distributions crucial for precise target localization and fine-grained contact. Clean attention restores reliable localization, but cannot recover geometric details lost in low-resolution observations. AtVLA therefore couples attention rectification with uncertainty-gated local refinement. The action expert samples multiple action chunks and estimates uncertainty from their disagreement; only for uncertain predictions, action-conditioned attention rollout identifies the task-relevant region, which is cropped, re-encoded at high resolution, and appended to the cached prefix for refined action generation. Across LIBERO, SimplerEnv, and a challenging single-view real-world benchmark, AtVLA improves the average LIBERO success rate from 94.2% to 98.4% and real-world success from 46.5% to 69.0%. The cropping is triggered on approximately 30% of replanning steps, resulting in only 1.4-1.6x the total computation of the base model under the representative deployment setting.",
  "published": "2026-08-03",
  "updated": "2026-08-03",
  "year": "2026",
  "authors": [
   "Jin Cui",
   "Yanbin Hu",
   "Xinyue Long",
   "Linkai Li",
   "Boran Zhao",
   "Pengju Ren"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "AtVLA, a framework that inserts learnable register tokens into the visual encoder and improves the average LIBERO success rate, is introduced, a framework that inserts learnable register tokens into the visual encoder and improves the average LIBERO success rate.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jin Cui",
    "id": "2399926288",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Yanbin Hu",
    "id": "2349546469",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Xinyue Long",
    "id": "2433818587",
    "h_index": 0,
    "papers": 5
   },
   {
    "name": "Linkai Li",
    "id": "2455559515",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Boran Zhao",
    "id": "2394250377",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Pengju Ren",
    "id": "2403071780",
    "h_index": 2,
    "papers": 11
   }
  ],
  "comment": "13 pages, 7 figures",
  "topics": [
   "vla",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.02197v1",
  "pdf_url": "https://arxiv.org/pdf/2608.02197v1",
  "html_url": "https://arxiv.org/html/2608.02197v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.02080",
  "slug": "toward-geometry-scalable-whole-body-touch-for-humanoids-a-3d-printed-c",
  "title": "Toward Geometry-Scalable Whole-Body Touch for Humanoids: A 3D-Printed Conformal EIT Skin",
  "abstract": "Whole-body tactile sensing is a prerequisite for humanoids that operate in contact-rich human environments, but conventional taxel arrays scale poorly with surface area, wiring complexity, and robot-specific curvature. We present a conformal electrical impedance tomography tactile skin fabricated through a geometry-adaptable additive-manufacturing workflow. A flexible conductive TPU layer forms a continuous sensing domain, while contact-induced coupling with conductive patches produces boundary voltage changes that are reconstructed using a one-step Gauss-Newton EIT solver. We first characterize the electromechanical design space of the layered structure and show that low-resistance contact-enhancement patches and a porous conductive TPU sensing layer improve sensitivity while preserving printability. We then validate contact localization on a planar prototype, a curved U-shaped prototype, and a qualitative iCub-face-shaped geometry. The curved sensor achieves a mean localization error of 6 mm over 18 contact positions without supervised post-processing. These results suggest that additively manufactured tomographic skins can reduce the morphology-specific redesign burden for humanoid tactile coverage and provide a practical route toward large-area contact sensing for human-centered deployment.",
  "published": "2026-08-03",
  "updated": "2026-08-03",
  "year": "2026",
  "authors": [
   "Haofeng Chen",
   "Carson Kohlbrenner",
   "Jiri Kubik",
   "Lukas Rustler",
   "Alexander Dickhans",
   "Karel Bartunek",
   "Alessandro Roncone",
   "Hyosang Lee",
   "Matej Hoffmann"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "Humanoids",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Haofeng Chen",
    "id": "2149051611",
    "h_index": 8,
    "papers": 34
   },
   {
    "name": "Carson Kohlbrenner",
    "id": "2333354233",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Jir\u00ed Kub\u00edk",
    "id": "2052085181",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Lukas Rustler",
    "id": "2105343640",
    "h_index": 7,
    "papers": 17
   },
   {
    "name": "Alexander Dickhans",
    "id": "2333354445",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Karel Bartunek",
    "id": "2376538464",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Alessandro Roncone",
    "id": "2365740226",
    "h_index": 5,
    "papers": 19
   },
   {
    "name": "Hyosang Lee",
    "id": "2378103521",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Matej Hoffmann",
    "id": "2350754412",
    "h_index": 1,
    "papers": 6
   }
  ],
  "comment": "Submitted to IEEE Humanoids",
  "topics": [
   "humanoids",
   "tactile",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.02080v1",
  "pdf_url": "https://arxiv.org/pdf/2608.02080v1",
  "html_url": "https://arxiv.org/html/2608.02080v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.02079",
  "slug": "tango-vio-triangulation-aware-navigation-with-guaranteed-feature-obser",
  "title": "TANGO-VIO: Triangulation-Aware Navigation with Guaranteed Feature-Observability for Visual-Inertial Odometry",
  "abstract": "In vision-aided navigation and visual-inertial odometry, the quality of triangulated three-dimensional feature positions is a fundamental prerequisite for state estimation accuracy. Triangulation becomes ill-conditioned or even impossible when a camera undergoes pure rotation without translation, or when the observed bearing vectors provide insufficient parallax. Even though visual-inertial odometry has been extensively studied, the active maintenance of feature-observability during navigation has not been sufficiently addressed in the literature. To address this gap, this study presents TANGO-VIO, a triangulation-aware navigation framework that embeds a log-determinant metric of the feature-wise stacked-bearing matrix into a control barrier function. In this proposed method, the observability guarantee is established in the feature-geometric sense by enforcing a lower bound on the aggregate triangulation-information metric through a nominal-direction-weighted minimum-deviation velocity correction. The proposed architecture is evaluated through software-inthe- loop simulations and real flight experiments. The results show improved triangulation conditioning under low-parallax motion, while the flight response closely reproduces the corresponding simulation behavior and confirms the practical realizability of the proposed safety filter. Supplementary materials are available on the project webpage.",
  "published": "2026-08-03",
  "updated": "2026-08-03",
  "year": "2026",
  "authors": [
   "Abd\u00fclbaki \u015eanlan",
   "Ege C. Altunkaya",
   "Hasan T. Ba\u011fci",
   "Emre Koyuncu",
   "\u0130brahim \u00d6zkol"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "TANGO-VIO is presented, a triangulation-aware navigation framework that embeds a log-determinant metric of the feature-wise stacked-bearing matrix into a control barrier function that establishes the observability guarantee in the feature-geometric sense by enforcing a lower bound on the aggregate triangulation-information metric through a nominal-direction-weighted minimum-deviation velocity correction.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "E. C. Altunkaya",
    "id": "2304461048",
    "h_index": 2,
    "papers": 13
   },
   {
    "name": "Abd\u00fclbaki Sanlan",
    "id": "2359463161",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Emre Koyuncu",
    "id": "2261540082",
    "h_index": 3,
    "papers": 21
   },
   {
    "name": "Ibrahim \u00d6zkol",
    "id": "2309717789",
    "h_index": 3,
    "papers": 18
   }
  ],
  "comment": "",
  "topics": [
   "rl-control",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.02079v1",
  "pdf_url": "https://arxiv.org/pdf/2608.02079v1",
  "html_url": "https://arxiv.org/html/2608.02079v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.02069",
  "slug": "open-diffloco-open-source-differentiable-learning-for-deployable-blind",
  "title": "Open-DiffLoco: Open-Source Differentiable Learning for Deployable Blind Quadruped Locomotion",
  "abstract": "Developing deployable locomotion policies through conventional reinforcement learning often requires complex reward engineering and expensive training times. While differentiable simulation offers a highly efficient alternative, open-source tools capable of end-to-end transfer of these policies to physical hardware remain limited. This paper introduces Open-DiffLoco, an open-source framework for training deployable blind quadruped locomotion policies with differentiable simulation. The framework implements the Short-Horizon Actor-Critic (SHAC) algorithm in MuJoCo XLA (MJX) and trains a proprioceptive policy that transfers to real-world hardware. The deployed policy removes privileged actor observations, including base linear velocity, and does not rely on reference trajectories. It also uses a substantially simplified reward function, enabling the robot to discover walking patterns without the complex auxiliary rewards typically used in conventional reinforcement learning pipelines. When deployed on physical hardware (a Unitree Go2 quadruped), the trained policy tracks omnidirectional velocity commands with root-mean-square error below 0.2 m/s, reaches speeds above 1 m/s, and remains robust to uneven terrain and external physical disturbances, such as lateral pushes. Across the reported configurations, training uses under 6 GB of VRAM on a single NVIDIA GeForce RTX 5080 GPU and completes in approximately 20-60 minutes. As an algorithmic extension to SHAC, we propose Jacobian-Augmented Value Estimation (JAVE), which supervises the critic Jacobians to improve early first-order policy-gradient training. To our knowledge, Open-DiffLoco is the first open-source framework for training deployable locomotion policies using differentiable simulation. Deployment videos and source code are available at: https://diffloco.martin-opat.com/",
  "published": "2026-08-03",
  "updated": "2026-08-03",
  "year": "2026",
  "authors": [
   "Martin Opat"
  ],
  "author_count": 1,
  "categories": [
   "cs.RO",
   "cs.LG",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Open-DiffLoco is the first open-source framework for training deployable blind quadruped locomotion policies with differentiable simulation and proposes Jacobian-Augmented Value Estimation (JAVE), which supervises the critic Jacobians to improve early first-order policy-gradient training.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Martin Opat",
    "id": "2385363842",
    "h_index": 0,
    "papers": 2
   }
  ],
  "comment": "8 pages, 5 figures, Project page, videos, and code available at: https://diffloco.martin-opat.com/",
  "topics": [
   "humanoids",
   "rl-control"
  ],
  "orgs": [
   "NVIDIA",
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2608.02069v1",
  "pdf_url": "https://arxiv.org/pdf/2608.02069v1",
  "html_url": "https://arxiv.org/html/2608.02069v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2608.02014",
  "slug": "mango-grasp-mahalanobis-fields-over-geometry-oriented-3d-gaussians-for",
  "title": "MANGO-Grasp: Mahalanobis Fields over Geometry-Oriented 3D Gaussians for Cross-Embodiment Dexterous Grasping",
  "abstract": "Cross-embodiment dexterous grasping aims to synthesize stable grasps across heterogeneous multi-fingered hands with little or no embodiment-specific tuning. Existing interaction-centric methods achieve promising results, but their object representations often underrepresent local surface geometry, while their robot descriptors do not explicitly encode both robot morphology and kinematics. We propose MANGO-Grasp, an anisotropic interaction framework that represents objects as geometry-oriented 3D Gaussian primitives and robot hands as surface keypoints encoded into morpho-kinematic descriptors. The object primitives are adaptively allocated by geometric complexity and shaped as surface-aligned plates with outward normals, encoding local geometry. Mahalanobis fields over keypoint--primitive pairs serve as interaction prediction targets during training and as optimization guidance for grasp realization at inference. These fields rise sharply for displacement along the surface normal but only gently within the tangent plane, matching the directional structure of contact. Grasps are realized with one shared optimization formulation and hyperparameter setting across all embodiments. On the CMAP and MultiGripperGrasp benchmarks, MANGO-Grasp outperforms the strongest seen-hand baseline by up to 8.24 percentage points in simulation. It also transfers zero-shot to the unseen SharpaWave hand, improving over the strongest zero-shot baseline by up to 16.57 percentage points, and achieves 86% success in real-world experiments. The code and additional materials will be made available upon publication at https://connor-zh.github.io/MANGO-Grasp/.",
  "published": "2026-08-03",
  "updated": "2026-08-03",
  "year": "2026",
  "authors": [
   "Heng Zhang",
   "Kevin Yuchen Ma",
   "Mike Zheng Shou",
   "Weisi Lin",
   "Yan Wu"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "MANGO-Grasp is proposed, an anisotropic interaction framework that represents objects as geometry-oriented 3D Gaussian primitives and robot hands as surface keypoints encoded into morpho-kinematic descriptors and achieves 86% success in real-world experiments.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Heng Zhang",
    "id": "2299917022",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "K. Ma",
    "id": "2324345511",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "M. Shou",
    "id": "2047358650",
    "h_index": 49,
    "papers": 277
   },
   {
    "name": "Weisi Lin",
    "id": "2290282651",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Yan Wu",
    "id": "2391837584",
    "h_index": 2,
    "papers": 7
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "foundation-pretraining",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.02014v1",
  "pdf_url": "https://arxiv.org/pdf/2608.02014v1",
  "html_url": "https://arxiv.org/html/2608.02014v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.01950",
  "slug": "fra-nbv-a-fast-and-reflectivity-aware-next-best-view-strategy",
  "title": "FRA-NBV: A Fast and Reflectivity-Aware Next-Best-View Strategy",
  "abstract": "Autonomous 3D reconstruction with depth sensors is strongly affected by reflective surfaces, which cause missing or unreliable measurements and reduce the effectiveness of conventional Next-Best-View (NBV) strategies. This limitation is particularly critical in industrial applications involving reflective components and low-cost, low-resolution depth sensing, where robustness to sensing failures is essential. This paper proposes a Fast Reflectivity-Aware Next-Best-View (FRA-NBV) strategy that explicitly addresses reflection-induced depth loss without relying on prior object models or assumptions on material reflectance, making it suitable for a wide range of industrial configurations. Reflective regions are identified from the spatial distribution of missing depth measurements and localized in three-dimensional space using an online ellipsoid-based representation of the object estimate. A recovery strategy then selects additional poses that modify the sensor's angle of incidence to improve the likelihood of reconstructing the affected regions. Experiments on objects with different geometric and reflective complexity demonstrate that the approach significantly improves reconstruction coverage under realistic industrial conditions.",
  "published": "2026-08-03",
  "updated": "2026-08-03",
  "year": "2026",
  "authors": [
   "G. F. Preziosa",
   "E. Setti",
   "M. Faroni",
   "A. M. Zanchettin",
   "P. Rocco"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A Fast Reflectivity-Aware Next-Best-View (FRA-NBV) strategy that explicitly addresses reflection-induced depth loss without relying on prior object models or assumptions on material reflectance, making it suitable for a wide range of industrial configurations.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "G. F. Preziosa",
    "id": "2186561201",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "E. Setti",
    "id": "2455408561",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "M. Faroni",
    "id": "2325401276",
    "h_index": 2,
    "papers": 12
   },
   {
    "name": "A. Zanchettin",
    "id": "2650655",
    "h_index": 32,
    "papers": 172
   },
   {
    "name": "P. Rocco",
    "id": "2273975260",
    "h_index": 5,
    "papers": 37
   }
  ],
  "comment": "8 pages, 5 figures",
  "topics": [
   "spatial-3d",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.01950v1",
  "pdf_url": "https://arxiv.org/pdf/2608.01950v1",
  "html_url": "https://arxiv.org/html/2608.01950v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.01855",
  "slug": "radar-perception-for-dynamic-obstacle-avoidance-onboard-small-scale-qu",
  "title": "RADAR Perception for Dynamic Obstacle Avoidance onboard small-scale Quadrotor UAVs",
  "abstract": "Fast dynamic obstacle avoidance (DOA) on uncrewed aerial vehicles (UAVs) demands not only low-latency control and actuation but also reliable perception with sufficient sensing range for accurate obstacle detection and speed estimation. This letter presents, to the best of our knowledge, the first mmWave RADAR-based perception-and-control system for fast onboard DOA. We derive and analyze latency and spatial bounds that relate sensing range, relative speed, and control delay, yielding sufficient conditions for successful avoidance. Our system adopts a lightweight tracker based on interacting multiple models and a controller based on control-barrier functions that directly outputs evasive accelerations. It achieves position errors of less than 0.15 m, 0.93 m, and 0.87 m in x, y, and z directions for 300 experiments with three different object sizes and varying visibility (light and dark), and a similar spread for 90 experiments in smoke. An onboard implementation on a Raspberry Pi 4B demonstrates real-time feasibility with an end-to-end sensing-to-command latency of approximately 14 ms. Code and the full dataset of 390 throws are available (https://tinyurl.com/radardoagit).",
  "published": "2026-08-03",
  "updated": "2026-08-03",
  "year": "2026",
  "authors": [
   "Dnyandeep Mandaokar",
   "Bernhard Rinner"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This letter presents the first mmWave RADAR-based perception-and-control system for fast onboard DOA, and derives and analyzes latency and spatial bounds that relate sensing range, relative speed, and control delay, yielding sufficient conditions for successful avoidance.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Dnyandeep Mandaokar",
    "id": "2375060687",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Bernhard Rinner",
    "id": "2237215993",
    "h_index": 2,
    "papers": 16
   }
  ],
  "comment": "This work has been submitted for publication. Copyright may be transferred without notice",
  "topics": [
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.01855v1",
  "pdf_url": "https://arxiv.org/pdf/2608.01855v1",
  "html_url": "https://arxiv.org/html/2608.01855v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.01851",
  "slug": "weights-or-skills-a-survey-of-robot-learning-techniques-from-action-pr",
  "title": "Weights or Skills? A Survey of Robot-Learning Techniques: from Action-Predicting Weights to Robots that Write their Own Skills",
  "abstract": "Robot learning is splitting into two bets: policies that bake competence into frozen weights (vision-language-action, or VLA, models), and agents that write and refine their own executable skills as code. This survey organises the field around that axis of weights versus skills. Its central analytical contribution is a deep-dive that arranges code-as-policy methods by their degree of self-improvement, from zero-shot program synthesis, through closed-loop self-repair and persistent skill memory, to the sparsely populated cell in which execution feedback, skill memory, and evolutionary search combine into one open-ended loop; only a few very recent systems (for example ASPIRE, ENPIRE, and RoboClaw) occupy that cell. We map the complementary \"skills\" pole, from unsupervised reinforcement-learning skill discovery to large-language-model skill libraries, and show that the word \"skill\" is used in at least five distinct senses, of which only the code sense self-improves without gradient updates. We then connect the taxonomy to the emerging skill economy: commercial robot-skill marketplaces now distribute one-tap skills across robots but ship only static playback, which surfaces open problems of adaptation, cross-embodiment portability, provenance, safety verification, composition, and standardisation. This is a deliberately focused survey. Rather than cataloguing the field exhaustively, it examines 77 representative systems across six technique families through one taxonomy and a set of contrast tables, and it supplies operational definitions of the self-improvement mechanisms together with a statement of what each family cannot do.",
  "published": "2026-08-03",
  "updated": "2026-08-03",
  "year": "2026",
  "authors": [
   "Gaytri Jena",
   "Kapil Wanaskar",
   "Vinija Jain",
   "Aman Chadha",
   "Vasu Sharma",
   "Amitava Das"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This survey organises the field around that axis of weights versus skills, and maps the complementary\"skills\"pole, from unsupervised reinforcement-learning skill discovery to large-language-model skill libraries, and shows that the word \"skill\" is used in at least five distinct senses.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Gaytri Jena",
    "id": "2359634616",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Kapil Wanaskar",
    "id": "100693087",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Vinija Jain",
    "id": "2212131028",
    "h_index": 15,
    "papers": 96
   },
   {
    "name": "Aman Chadha",
    "id": "2275226689",
    "h_index": 19,
    "papers": 161
   },
   {
    "name": "Vasu Sharma",
    "id": "2316591078",
    "h_index": 3,
    "papers": 16
   },
   {
    "name": "Amitava Das",
    "id": "2258322706",
    "h_index": 9,
    "papers": 72
   }
  ],
  "comment": "40 pages, 11 figures, 11 tables",
  "topics": [
   "vla",
   "foundation-pretraining",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.01851v1",
  "pdf_url": "https://arxiv.org/pdf/2608.01851v1",
  "html_url": "https://arxiv.org/html/2608.01851v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.01834",
  "slug": "teleopit-a-full-embodiment-humanoid-teleoperation-system",
  "title": "Teleopit: A Full-Embodiment Humanoid Teleoperation System",
  "abstract": "Humanoid teleoperation for demonstration collection requires coordinated whole-body motion, continuous dexterous hand control, and viewpoint control. Existing systems either simplify hand commands or depend on dedicated wearable sensors for fine-grained hand motion. We introduce Teleopit, a full-embodiment teleoperation system that maps body, hand, and head signals from VR to a humanoid body, configurable dexterous hands, and a 2-DoF active vision module. A history encoder and failure-aware rewind sampling improve the motion tracker on both motion-capture and live VR references. An optimization-based hand retargeter combines normalized finger directions, fingertip closure, and thumb-frame alignment to map human hand motion to different dexterous hands without tuning hand-specific objective or solver hyperparameters. Component experiments evaluate tracking success rate and retargeting behavior, while real-robot teleoperation demonstrates coordinated locomotion, manipulation, and viewpoint control. ACT and GR00T N1.7 policies trained on 96 successful demonstrations collected with Teleopit achieve task success rates of 90.0% and 95.0%, respectively, when deployed on the humanoid. The project page is available at https://botrunner64.github.io/teleopit-page.",
  "published": "2026-08-03",
  "updated": "2026-08-03",
  "year": "2026",
  "authors": [
   "Bingqian Wu",
   "Zicheng Xu",
   "Xianghui Fan",
   "Dayu Li",
   "Xiangru Huang"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Teleopit, a full-embodiment teleoperation system that maps body, hand, and head signals from VR to a humanoid body, configurable dexterous hands, and a 2-DoF active vision module, is introduced.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Bingqian Wu",
    "id": "2179628393",
    "h_index": 2,
    "papers": 14
   },
   {
    "name": "Zichen Xu",
    "id": "2449439286",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Xianghui Fan",
    "id": "2336792400",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Dayu Li",
    "id": "2455444263",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Xiangru Huang",
    "id": "2397630751",
    "h_index": 2,
    "papers": 11
   }
  ],
  "comment": "17 pages, 16 figures. Project page: https://botrunner64.github.io/teleopit-page",
  "topics": [
   "dexterous-manipulation",
   "humanoids",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.01834v1",
  "pdf_url": "https://arxiv.org/pdf/2608.01834v1",
  "html_url": "https://arxiv.org/html/2608.01834v1",
  "code_url": "https://botrunner64.github.io/teleopit-page",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2608.01826",
  "slug": "multi-view-unified-camera-fields-geometry-shaped-action-facing-represe",
  "title": "Multi-View Unified Camera Fields: Geometry-Shaped Action-Facing Representations for RGB-Only Multi-Camera VLA Policies",
  "abstract": "Vision-Language-Action (VLA) models have shown strong generalization in robotic manipulation, yet complex contact-rich tasks often benefit from multi-camera observations that jointly capture the end effector, objects, and targets under occlusion. Existing multi-camera VLAs usually concatenate view tokens, leaving action representations weak in metric depth and inconsistent across cameras. We introduce Multi-View Unified Camera Fields (MVUCF), a training-only framework that forms a shared action-facing latent field across views. A coordinate-query depth objective makes metric depth recoverable, while a preprocessing-aware correspondence objective aligns tokens observing the same physical point from different cameras. Both directly shape the hidden states consumed by the action module. After geometry injection, depth, camera calibration, and auxiliary heads are removed, so deployment uses the original RGB-only graph with no extra inference FLOPs. Held-out probes confirm stronger depth recovery and cross-view matching. Under matched GR00T-N1.6 settings, MVUCF reaches 98.9% on LIBERO, improves LIBERO-Plus by 22.4 points, and raises success by 23.3 points across six RoboTwin tasks spanning three action families: touch, move-and-place, and contact interaction. Real-world humanoid experiments further provide evidence of its practical effectiveness under RGB-only deployment.",
  "published": "2026-08-03",
  "updated": "2026-08-03",
  "year": "2026",
  "authors": [
   "Jiarui Yang",
   "Yehao Lu",
   "Yuning Su",
   "Yufeng Xie",
   "Yu Zhong",
   "Haiyu Lan",
   "Tianjing Hao",
   "Kaixiang Lu",
   "Peiwen Lin",
   "Chuang Wang",
   "Enyu Li",
   "Junwei Liang"
  ],
  "author_count": 12,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Multi-View Unified Camera Fields (MVUCF), a training-only framework that forms a shared action-facing latent field across views, and real-world humanoid experiments further provide evidence of its practical effectiveness under RGB-only deployment.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jiarui Yang",
    "id": "2447991900",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yehao Lu",
    "id": "2306078924",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "Yuning Su",
    "id": "2455452490",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yufeng Xie",
    "id": "2455448179",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yu Zhong",
    "id": "2366681207",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Haiyu Lan",
    "id": "2455273729",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Tianjing Hao",
    "id": "2455323467",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Kaixiang Lu",
    "id": "2455444426",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Pei Lin",
    "id": "2357017108",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Chuang Wang",
    "id": "2455563265",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Enyu Li",
    "id": "2455335642",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Junwei Liang",
    "id": "2268726427",
    "h_index": 10,
    "papers": 19
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "humanoids",
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.01826v1",
  "pdf_url": "https://arxiv.org/pdf/2608.01826v1",
  "html_url": "https://arxiv.org/html/2608.01826v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.01824",
  "slug": "retouch-empowering-contact-rich-dexterous-manipulation-with-online-ref",
  "title": "ReTouch: Empowering Contact-Rich Dexterous Manipulation with Online-Refined Tactile Prediction",
  "abstract": "Fusing tactile signals has proven effective for contact-rich manipulation, enabling robots to perceive contact states and adapt to rapidly changing physical interactions. Yet effectively integrating tactile feedback into dexterous manipulation remains underexplored. In this work, we introduce ReTouch, a vision-language-action model (VLA) that supports contact-rich dexterous manipulation through tactile predictions continually refined online using execution-time feedback. ReTouch builds on two main innovations for tactile representation and closed-loop action generation. First, its Tactile-Patch Encoder represents tactile observations as structured tactile patch features that preserve finger identity and local contact structure, providing contact cues for fine-grained dexterous control. Second, its high-frequency action module jointly predicts future tactile states and action chunks and refines both using incoming tactile feedback during execution. This closed-loop refinement keeps tactile predictions aligned with evolving physical interactions, enabling responsive action correction and improving robustness to contact changes and execution errors. We further introduce XHT-Dataset, comprising 900 real-world demonstrations across seven contact-rich tasks collected on an XHand--UR7e platform, and evaluate ReTouch through closed-loop real-robot experiments. ReTouch surpasses the strongest baseline by 18.4 and 23.8 percentage points in average success rate under standard and challenging conditions, respectively, demonstrating its effectiveness and robustness.",
  "published": "2026-08-03",
  "updated": "2026-08-18",
  "year": "2026",
  "authors": [
   "Shiqi Zhang",
   "Xin Zhang",
   "Yedong Shen",
   "Yao Li",
   "Yuxuan Gao",
   "Sha Zhang",
   "Yuan Zhang",
   "Kaixue Long",
   "Jiajia Wu",
   "Jia Pan",
   "Jiajun Deng",
   "Yanyong Zhang"
  ],
  "author_count": 12,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "ReTouch is introduced, a vision-language-action model (VLA) that supports contact-rich dexterous manipulation through tactile predictions continually refined online using execution-time feedback, and keeps tactile predictions aligned with evolving physical interactions, enabling responsive action correction and improving robustness to contact changes and execution errors.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Shiqi Zhang",
    "id": "2346482681",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Xin Zhang",
    "id": "2333419153",
    "h_index": 4,
    "papers": 22
   },
   {
    "name": "Yedong Shen",
    "id": "2346449141",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Yao Li",
    "id": "2268425991",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Yuxuan Gao",
    "id": "2369201549",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Sha Zhang",
    "id": "2283448371",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Y. Zhang",
    "id": "2439316277",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Kaixue Long",
    "id": "2455351843",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Jiajia Wu",
    "id": "46366066",
    "h_index": 8,
    "papers": 31
   },
   {
    "name": "Jia Pan",
    "id": "2390556259",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Jiajun Deng",
    "id": "2287933949",
    "h_index": 8,
    "papers": 22
   },
   {
    "name": "Yanyong Zhang",
    "id": "2406536524",
    "h_index": 2,
    "papers": 8
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "dexterous-manipulation",
   "tactile",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.01824v2",
  "pdf_url": "https://arxiv.org/pdf/2608.01824v2",
  "html_url": "https://arxiv.org/html/2608.01824v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.01802",
  "slug": "conav-uav-cooperative-dual-altitude-aerial-navigation-via-stackelberg",
  "title": "CoNav-UAV: Cooperative Dual-Altitude Aerial Navigation via Stackelberg Learning",
  "abstract": "Target-oriented vision-and-language navigation (VLN) on aerial platforms is attracting growing attention for missions such as disaster rescue, infrastructure inspection, and security patrol. In this task, an unmanned aerial vehicle (UAV) needs to locate targets given only a concise description of their appearance and surroundings. This requires global exploration and grounding as well as collision-free close-range approach, two interleaved processes difficult to reconcile within a single agent. Most existing methods transfer the ground VLN paradigm to a low-altitude UAV and compensate for its inefficient exploration with external assistance. A recent attempt deploys two UAVs at complementary altitudes yet still relies on privileged information and trains its two agents independently, precluding any mutual adaptation essential for cooperation. Here we propose CoNav-UAV, which explicitly models the task as a Stackelberg game between a high-altitude leader and a low-altitude follower, with the system operating on onboard visual and linguistic inputs alone. To solve this game, we introduce Iterative Stackelberg Learning. The leader's high-level vision-language reasoning is refined via memory-based in-context learning, while the follower's precise motion control is updated via DAgger-style expert distillation. The alternation drives both agents toward a Stackelberg equilibrium. CoNav-UAV consistently outperforms single- and dual-agent baselines across three high-fidelity urban scenes from the AerialVLN benchmark. Success rate improves by up to 30.8 points on the learning scene, and 9.0 points under cross-scene transfer while using about 3x less adaptation data. Further analyses validate the complementary gains of the leader and follower updates and reveal robust gains yet distinct learning dynamics across VLM backbones.",
  "published": "2026-08-03",
  "updated": "2026-08-03",
  "year": "2026",
  "authors": [
   "Junru Song",
   "Wenhao Zhang",
   "Yang Yang",
   "Xuekai Qiu",
   "Feifei Wang",
   "Weien Zhou",
   "Tingsong Jiang",
   "Ying Wen",
   "Yang Li",
   "Wen Yao"
  ],
  "author_count": 10,
  "categories": [
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.AI",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "CoNav-UAV is proposed, which explicitly models the target-oriented vision-and-language navigation task as a Stackelberg game between a high-altitude leader and a low-altitude follower, with the system operating on onboard visual and linguistic inputs alone.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Junru Song",
    "id": "2221337153",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Wenhao Zhang",
    "id": "2218844283",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Yang Yang",
    "id": "2293750643",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Xuekai Qiu",
    "id": "2326834785",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Feifei Wang",
    "id": "2148956712",
    "h_index": 7,
    "papers": 14
   },
   {
    "name": "Weien Zhou",
    "id": "2148943712",
    "h_index": 18,
    "papers": 69
   },
   {
    "name": "Tingsong Jiang",
    "id": "2114745862",
    "h_index": 16,
    "papers": 69
   },
   {
    "name": "Ying Wen",
    "id": "2155576705",
    "h_index": 8,
    "papers": 17
   },
   {
    "name": "Yang Li",
    "id": "2321328048",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Wen Yao",
    "id": "2283959173",
    "h_index": 5,
    "papers": 12
   }
  ],
  "comment": "",
  "topics": [
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.01802v1",
  "pdf_url": "https://arxiv.org/pdf/2608.01802v1",
  "html_url": "https://arxiv.org/html/2608.01802v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.01800",
  "slug": "hybrid-impedance-admittance-control-with-multi-link-aerial-robot-for-c",
  "title": "Hybrid Impedance-Admittance Control with Multi-Link Aerial Robot for Contact-Rich Surface Sliding Task",
  "abstract": "Multi-link aerial robots can actively deform their articulated structures during flight, giving them strong potential for aerial manipulation. However, they still face substantial challenges in contact-rich aerial manipulation tasks such as surface sliding, which requires both disturbance robustness and compliance to uncertain surface geometry. Force-control strategies such as impedance and admittance control are commonly employed to address these requirements. Although impedance control can provide disturbance-resistant interaction and admittance control can offer compliant adaptation, their opposite force--motion causalities prevent their simultaneous implementation when applied through the same actuation source, such as the rotor thrusts used by conventional aerial robots. To overcome this limitation, we propose a hybrid impedance--admittance control strategy for a multi-link aerial robot. The articulated morphology enables a functional separation of force and motion regulation across joint and rotor actuation sources. In this framework, admittance behavior is generated through joint angle regulation to enhance adaptive interaction, while impedance behavior is achieved by modulating rotor thrust to regulate the sliding motion. This structural coordination allows the robot to leverage the complementary strengths of both control paradigms. As a result, the multi-link aerial robot achieves resilient and adaptive surface sliding. Experimental results demonstrate robust and compliant sliding performance on unknown surfaces.",
  "published": "2026-08-03",
  "updated": "2026-08-03",
  "year": "2026",
  "authors": [
   "Zicheng Luo",
   "Maolin Lei",
   "Jinjie Li",
   "Yicheng Chen",
   "Zicen Xiong",
   "Moju Zhao"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zicheng Luo",
    "id": "2438558197",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Maolin Lei",
    "id": "30110242",
    "h_index": 7,
    "papers": 14
   },
   {
    "name": "Jinjie Li",
    "id": "2301504767",
    "h_index": 4,
    "papers": 25
   },
   {
    "name": "Yicheng Chen",
    "id": "2109311677",
    "h_index": 2,
    "papers": 11
   },
   {
    "name": "Zicen Xiong",
    "id": "2287018598",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Moju Zhao",
    "id": "3264557",
    "h_index": 18,
    "papers": 67
   }
  ],
  "comment": "",
  "topics": [
   "tactile",
   "hardware-codesign",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.01800v1",
  "pdf_url": "https://arxiv.org/pdf/2608.01800v1",
  "html_url": "https://arxiv.org/html/2608.01800v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.01733",
  "slug": "twins-a-tactile-wearable-isomorphic-arm-networked-system-for-contact-r",
  "title": "TWINS: A Tactile Wearable Isomorphic Arm Networked System for Contact-Rich Manipulation Learning",
  "abstract": "Recent advances in robot learning for manipulation have increased the importance of collecting real-world demonstration data. However, existing robotic systems primarily focus on end-effector manipulation, making it difficult to teach and execute manipulation tasks involving body-surface contact with the arms and chest. This paper presents TWINS (Tactile Wearable Isomorphic Arm Networked System), a robotic system for manipulation involving body-surface contact. TWINS consists of a Wearable Dual-Arm Device, which is worn by the operator, and an Isomorphic Robot with the same joint configuration and external dimensions. Distributed tactile sensors embedded in the chest and arms enable the measurement of body-surface contact synchronized with joint motion. Using the Wearable Dual-Arm Device, we collected demonstrations for four manipulation tasks involving body-surface contact. We then trained imitation learning policies using the collected demonstrations and deployed them on the Isomorphic Robot, enabling manipulation guided by body-surface tactile observations. Experimental results demonstrate that TWINS provides a unified robotic system for demonstration, learning, and execution of manipulation involving body-surface contact. https://mmurooka.github.io/twins-project-page/",
  "published": "2026-08-03",
  "updated": "2026-08-03",
  "year": "2026",
  "authors": [
   "Takahide Kitamura",
   "Masaki Murooka",
   "Natsuki Yamanobe",
   "Yukiyasu Domae"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Experimental results demonstrate that TWINS provides a unified robotic system for demonstration, learning, and execution of manipulation involving body-surface contact.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Takahide Kitamura",
    "id": "2328409977",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Masaki Murooka",
    "id": "2350756597",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "N. Yamanobe",
    "id": "1696822",
    "h_index": 17,
    "papers": 138
   },
   {
    "name": "Y. Domae",
    "id": "2512607",
    "h_index": 13,
    "papers": 128
   }
  ],
  "comment": "",
  "topics": [
   "tactile",
   "imitation-diffusion"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.01733v1",
  "pdf_url": "https://arxiv.org/pdf/2608.01733v1",
  "html_url": "https://arxiv.org/html/2608.01733v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.01697",
  "slug": "bridging-the-sim-to-real-gap-in-parallel-link-leg-mechanisms-via-simul",
  "title": "Bridging the Sim-to-Real Gap in Parallel-Link Leg Mechanisms via Simulator-Side Dynamics Normalization",
  "abstract": "This paper addresses the sim-to-real gap in dynamics arising when a parallel-link mechanism is represented by a serial-tree surrogate in simulation. Conventional Jacobian-based state and torque mappings preserve consistency with the kinematic and virtual-work relations but do not account for the coordinate-induced redistribution of actuator inertia and damping and the linkage inertia omitted during serial-tree reduction. To address this gap, Simulator-Side System Normalization (S3N) is proposed to normalize the serial-tree simulator's effective dynamics while preserving its tree topology. S3N-Act incorporates actuator inertia and damping into the serial-coordinate dynamics through coordinate transformation, whereas S3N-Full restores residual linkage inertia by separately identifying actuator- and leg-level frequency responses. In the 2-DoF validation, S3N-Full reduced the joint-position and torque RMSEs by 80.9% and 82.1%, respectively, relative to the Jacobian-mapping baseline. During pitch-in-place motion, S3N-Act and S3N-Full reduced the RMSE of the ground reaction force norm by 65.1% and 62.4%, respectively. During circular locomotion, S3N-Full reduced the phase-averaged, command-normalized sim-to-real gap from 17.3% to 9.9%. These results show that simulator-side normalization improves motion- and force-level sim-to-real consistency. It enables policy training in a serial-tree framework with hardware-consistent dynamics that better represent the physical parallel-link mechanism.",
  "published": "2026-08-03",
  "updated": "2026-08-06",
  "year": "2026",
  "authors": [
   "Jinsong Hong",
   "Jangho Kim",
   "Jihwan Lee",
   "Donghyun Kim",
   "Sehoon Oh"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Results show that simulator-side normalization improves motion- and force-level sim-to-real consistency and enables policy training in a serial-tree framework with hardware-consistent dynamics that better represent the physical parallel-link mechanism.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "J. Hong",
    "id": "2314531438",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Jangho Kim",
    "id": "2237950095",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Jihwan Lee",
    "id": "2401919311",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Donghyun Kim",
    "id": "2454699305",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Sehoon Oh",
    "id": "2244125248",
    "h_index": 1,
    "papers": 21
   }
  ],
  "comment": "10 pages, 12 figures",
  "topics": [
   "humanoids",
   "sim2real",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.01697v2",
  "pdf_url": "https://arxiv.org/pdf/2608.01697v2",
  "html_url": "https://arxiv.org/html/2608.01697v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.01690",
  "slug": "protoact-turning-wet-lab-protocols-into-embodied-robotic-actions",
  "title": "ProtoAct: Turning Wet-Lab Protocols into Embodied Robotic Actions",
  "abstract": "Biological wet-lab protocols are written for trained researchers and often leave routine operations, state-dependent conditions, and contextual parameters implicit, making them difficult to translate into robot-executable actions. We present ProtoAct, a structured protocol-grounding framework that converts free-form biological procedures into state-aware, embodiment-ready action sequences. ProtoAct uses ProtoRAG to retrieve manually annotated examples for context-sensitive parsing, employs RefineChecker to detect and revise missing or inconsistent steps, and applies ActSchema to map the refined procedure into constrained JSON function sequences. We further introduce BioP2E, for which we manually annotate 22 cell-culture protocols into 258 monitoring conditions, 910 executable subtasks, and 962 grounded action calls. Evaluation across seven large language models demonstrates that ProtoAct can be effectively instantiated with different backbones. Ablations confirm that retrieval, posterior checking, and schema constraints make complementary contributions. The parsed subtasks further support demonstration collection and VLA model training, enabling successful execution in both simulation and real-robot settings. ProtoAct thus provides a practical interface between biological protocol understanding and embodied robotic execution.",
  "published": "2026-08-03",
  "updated": "2026-08-03",
  "year": "2026",
  "authors": [
   "Zhe Liu",
   "Jiaming Gu",
   "Zhaohui Du",
   "Zhe Wang",
   "Huanbo Jin",
   "Quan Lu",
   "Qi Wang",
   "Ting Xiao",
   "Minting Pan",
   "Dongzhan Zhou"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "ProtoAct, a structured protocol-grounding framework that converts free-form biological procedures into state-aware, embodiment-ready action sequences, is presented, providing a practical interface between biological protocol understanding and embodied robotic execution.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhe Liu",
    "id": "2404004658",
    "h_index": 2,
    "papers": 13
   },
   {
    "name": "Jiaming Gu",
    "id": "2261096786",
    "h_index": 2,
    "papers": 13
   },
   {
    "name": "Zhaohui Du",
    "id": "2364085760",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Zhe Wang",
    "id": "2241483629",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Huanbo Jin",
    "id": "2442115636",
    "h_index": 0,
    "papers": 5
   },
   {
    "name": "Quan Lu",
    "id": "2311449124",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Qi Wang",
    "id": "2318419521",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Ting Xiao",
    "id": "2407742135",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Minting Pan",
    "id": "2027156855",
    "h_index": 5,
    "papers": 17
   },
   {
    "name": "Dongzhan Zhou",
    "id": "2359612649",
    "h_index": 6,
    "papers": 29
   }
  ],
  "comment": "15 pages, 13 figures",
  "topics": [
   "vla",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.01690v1",
  "pdf_url": "https://arxiv.org/pdf/2608.01690v1",
  "html_url": "https://arxiv.org/html/2608.01690v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.01636",
  "slug": "a-forward-inverse-dynamic-game-framework-for-enhanced-multi-agent-traj",
  "title": "A Forward-Inverse Dynamic Game Framework for Enhanced Multi-Agent Trajectory Planning",
  "abstract": "This paper studies feedback Nash equilibrium (FBNE) seeking for multi-agent trajectory planning in nonlinear dynamical systems with unknown agents' objectives and state-dependent inter-agent coupling. While dynamic game theory provides a principled framework for such problems, existing approaches typically assume fully rational agents with known objectives or rely on fixed regularization, limiting their ability to capture bounded rationality and spatially varying interaction intensity in safety-critical settings. To this end, we propose a KL-regularized dynamic game with a state-dependent weight that adaptively balances optimality and behavioral priors. To infer unknown cost parameters from demonstrated behaviors, we develop a context-aware inverse game module based on maximum-entropy inverse reinforcement learning with physics-informed regularization, ensuring structural consistency with the forward game. We establish per-iteration well-posedness of the regularized local game and show that the adaptive weighting function remains Lipschitz continuous under bounded nominal-trajectory updates. Numerical simulations and multi-robot experiments on cooperative navigation and merging scenarios validate the effectiveness of the proposed framework.",
  "published": "2026-08-03",
  "updated": "2026-08-03",
  "year": "2026",
  "authors": [
   "Tianle Liu",
   "Youcheng Niu",
   "Jing Zeng",
   "Shuo Li",
   "Jinming Xu"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A KL-regularized dynamic game with a state-dependent weight that adaptively balances optimality and behavioral priors is proposed for multi-agent trajectory planning in nonlinear dynamical systems with unknown agents'objectives and state-dependent inter-agent coupling.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Tianle Liu",
    "id": "2333461724",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Youcheng Niu",
    "id": "2276206993",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Jing Zeng",
    "id": "2179081503",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Shuo Li",
    "id": "2284733941",
    "h_index": 4,
    "papers": 18
   },
   {
    "name": "Jinming Xu",
    "id": "46372343",
    "h_index": 17,
    "papers": 102
   }
  ],
  "comment": "",
  "topics": [
   "rl-control",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.01636v1",
  "pdf_url": "https://arxiv.org/pdf/2608.01636v1",
  "html_url": "https://arxiv.org/html/2608.01636v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.01603",
  "slug": "affordtrajdp-dynamic-affordance-guided-visuomotor-policy-learning-for",
  "title": "AffordTrajDP: Dynamic Affordance-Guided Visuomotor Policy Learning for Robotic Manipulation",
  "abstract": "Affordance-guided imitation learning has shown impressive performance in robotic manipulation tasks by compressing visual perception into task-specific geometric constraints (e.g., fixed contact points). However, the commonly used static affordances can become inconsistent in precision-critical tasks or under object location perturbations, leading to post-contact trajectory drift. To address this issue, we propose AffordTrajDP, a dynamic framework that constructs affordance trajectories via object-centric temporal propagation to guide the progressive manipulation process. Specifically, given an RGB-D observation, our core insight is that a retrieved anchor affordance, which captures the desired contact point between the end-effector and the target object, can be propagated forward via affordance propagation, using the object's SE(3) pose as a natural propagation medium, to yield an affordance trajectory that provides temporally consistent, state-aware guidance throughout execution. AffordTrajDP achieves 70.0% average success rate on ManiSkill3, outperforming strong baselines by up to 17.8%. Real-world experiments on Galaxea A1 and UR7e robotic arms, covering StackCube, PickCup, AdapterInsertion, Ring-on-Peg, Put-in-Bowl, and USB Insertion, further validate robustness under object placement variations and appearance changes, with seen and unseen object instances evaluated on Galaxea A1, and ablations confirm the contribution of each proposed component.",
  "published": "2026-08-03",
  "updated": "2026-08-03",
  "year": "2026",
  "authors": [
   "Gaoyuan Wu",
   "Ziyu Shan",
   "Haoyang Du",
   "Yuyao Jiang",
   "Ziwei Wang"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The core insight is that a retrieved anchor affordance, which captures the desired contact point between the end-effector and the target object, can be propagated forward via affordance propagation, to yield an affordance trajectory that provides temporally consistent, state-aware guidance throughout execution.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Gaoyuan Wu",
    "id": "2168100474",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Ziyu Shan",
    "id": "2189373005",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Haoyang Du",
    "id": "2400497667",
    "h_index": 3,
    "papers": 20
   },
   {
    "name": "Yuyao Jiang",
    "id": "2445395263",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Ziwei Wang",
    "id": "2274572781",
    "h_index": 6,
    "papers": 16
   }
  ],
  "comment": "",
  "topics": [
   "imitation-diffusion",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.01603v1",
  "pdf_url": "https://arxiv.org/pdf/2608.01603v1",
  "html_url": "https://arxiv.org/html/2608.01603v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.01600",
  "slug": "perception-and-action-system-for-humanoid-robot-task-execution-in-cons",
  "title": "Perception-and-action system for humanoid robot task execution in construction",
  "abstract": "Humanoid robots, with their human-like shape and multi-tasking capabilities, are well-aligned with human-dominated workplaces, like those in civil and construction engineering, where they could collaborate with human workers or autonomously perform physically demanding and hazardous tasks. Despite this promise, limited research has explored how to endow these robots with the practical capabilities needed to perform construction tasks. To this end, this study proposes a novel perception-and-action system that enables humanoid robots to learn and perform construction tasks from worker demonstrations. This system contains two deep networks: Humanoid-PoseNet, which extracts human postures and translates them into mechanically feasible poses for a humanoid robot; and Humanoid-ActionNet, which learns robot-executable actions based on these translated poses. Experimental results demonstrate that the humanoid robot reliably executed eight construction-related actions, achieving an average motion-tracking error of 82.45 mm MPJPE (Mean Per Joint Position Error). This work provides an early step toward deploying humanoid collaborators in construction.",
  "published": "2026-08-03",
  "updated": "2026-08-03",
  "year": "2026",
  "authors": [
   "Yanxi Liu",
   "Yizhi Liu"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes a novel perception-and-action system that enables humanoid robots to learn and perform construction tasks from worker demonstrations, and demonstrates that the humanoid robot reliably executed eight construction-related actions.",
  "doi": "10.1016/j.cacaie.2026.100107",
  "oa_pdf": "https://doi.org/10.1016/j.cacaie.2026.100107",
  "s2_authors": [
   {
    "name": "Yanxi Liu",
    "id": "2368102405",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Yizhi Liu",
    "id": "2108204155",
    "h_index": 13,
    "papers": 59
   }
  ],
  "comment": "",
  "topics": [
   "humanoids"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.01600v1",
  "pdf_url": "https://arxiv.org/pdf/2608.01600v1",
  "html_url": "https://arxiv.org/html/2608.01600v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.01573",
  "slug": "uncovering-and-mitigating-positional-blind-spots-in-vision-language-ac",
  "title": "Uncovering and Mitigating Positional Blind Spots in Vision-Language-Action Models",
  "abstract": "Recent Vision-Language-Action (VLA) models achieve promising performance in robotic manipulation, typically measured by success rates aggregated over predefined object configurations, an evaluation that implicitly assumes spatially uniform competence across the workspace. However, this assumption does not hold: even with the instruction and every other scene factor held fixed, merely relocating a task-irrelevant distractor can sharply raise the failure probability within localized, spatially coherent regions, which we term Positional Blind Spots (PBS). In this paper, we propose a two-stage black-box framework to uncover and mitigate PBS. During the uncovering stage, we grid the workspace and apply a one-sided log-likelihood-ratio test to localize PBS cells with significantly elevated risk. During the mitigation stage, we fine-tune the policy via LoRA on demonstrations collected from these PBS regions, improving competence there while largely preserving performance across the rest of the workspace. We evaluate our framework on five state-of-the-art VLA policies across two benchmarks, and find that PBS are pervasive and spatially concentrated in all of them, with failure rates up to 0.58. Our search strategy achieves an average F1-score of 0.678, outperforming random search and adaptive sampling baselines by 0.268 and 0.178, respectively. Guided by the discovered regions, targeted fine-tuning reduces the overall failure rate by 40.00%--85.19%.",
  "published": "2026-08-03",
  "updated": "2026-08-03",
  "year": "2026",
  "authors": [
   "Dongdong An",
   "Pengjie Zhao",
   "Yihao Huang",
   "Wenbing Tang",
   "Ziming He",
   "Jiayi Zhu",
   "Jifeng Ning",
   "Qin Zhao"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper proposes a two-stage black-box framework to uncover and mitigate Positional Blind Spots, and evaluates its framework on five state-of-the-art VLA policies across two benchmarks, and finds that PBS are pervasive and spatially concentrated in all of them.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Dongdong An",
    "id": "2279075927",
    "h_index": 3,
    "papers": 16
   },
   {
    "name": "Pengjie Zhao",
    "id": "2455432983",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yihao Huang",
    "id": "2142367654",
    "h_index": 19,
    "papers": 67
   },
   {
    "name": "Wenbing Tang",
    "id": "2000923616",
    "h_index": 6,
    "papers": 14
   },
   {
    "name": "Ziming He",
    "id": "2264944954",
    "h_index": 3,
    "papers": 19
   },
   {
    "name": "Jiayi Zhu",
    "id": "2324704840",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Jifeng Ning",
    "id": "2455324310",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Qin Zhao",
    "id": "2455447571",
    "h_index": 0,
    "papers": 2
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.01573v1",
  "pdf_url": "https://arxiv.org/pdf/2608.01573v1",
  "html_url": "https://arxiv.org/html/2608.01573v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.07558",
  "slug": "learning-physical-interaction-a-survey-of-tactile-and-force-aware-robo",
  "title": "Learning Physical Interaction: A Survey of Tactile- and Force-aware Robot Learning",
  "abstract": "Physically grounded robot intelligence requires robots to perceive, reason about, and regulate their interactions with the physical world. This capability is particularly critical in contact-sensitive manipulation, where successful task execution depends not only on visual perception and motion generation, but also on force regulation and adaptive control. In this context, recent robot learning methods have made substantial progress by integrating force, tactile, vision, language, and proprioceptive sensing into learned manipulation policies. In parallel, many systems adopt multi-phase architectures that combine high-level policies, action-refinement modules, and low-level controllers to bridge semantic task understanding with reactive physical execution. Despite these advances, existing surveys have not explicitly reviewed force- and tactile-aware robot learning from a unified perspective that jointly captures multimodal sensing and multi-phase system design. This survey addresses this gap by proposing TF-ART, a Tactile/Force-Aware Robot learning Taxonomy for multimodal and multi-phase frameworks, which maps individual methods into a unified hierarchical structure. The framework characterizes how recent works organize observation modalities, encode and fuse heterogeneous sensory inputs, generate and refine actions across multiple phases, and connect learned policies to reactive robot-end control. Building on this methodological view, we further examine the task settings and infrastructure requirements of physical interaction, thereby integrating both algorithmic and practical perspectives on force- and tactile-aware robot learning.",
  "published": "2026-08-02",
  "updated": "2026-08-02",
  "year": "2026",
  "authors": [
   "Shilin Shan",
   "Chuhao Zhou",
   "Ruize Wang",
   "Xinyan Chen",
   "Xiangyu Chen",
   "Xinyu Zhou",
   "Boyu Ma",
   "Iris Yuxuan Hu",
   "Jingliang Li",
   "Celeste Yuxuan Hu",
   "Geng Li",
   "Guohao Chen",
   "Tianrui Zhu",
   "Zhe Li",
   "Yanjie Ze",
   "Haoran Geng",
   "Zhiyang Dou",
   "Jianxin Bi",
   "Yuejiang Liu",
   "Jianshu Zhou",
   "Jiachen Li",
   "Paul Liang",
   "Tatsuya Harada",
   "Robert Katzschmann",
   "Harold Soh",
   "Na Li",
   "Edward Johns",
   "Danica Kragic",
   "Jan Peters",
   "Wojciech Matusik",
   "Masayoshi Tomizuka",
   "Jitendra Malik",
   "Jianfei Yang"
  ],
  "author_count": 33,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "TF-ART is proposed, a Tactile/Force-Aware Robot learning Taxonomy for multimodal and multi-phase frameworks, which maps individual methods into a unified hierarchical structure and further examines the task settings and infrastructure requirements of physical interaction.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Shilin Shan",
    "id": "2203427813",
    "h_index": 4,
    "papers": 14
   },
   {
    "name": "Chuhao Zhou",
    "id": "2326548035",
    "h_index": 2,
    "papers": 17
   },
   {
    "name": "Ruize Wang",
    "id": "2360140309",
    "h_index": 2,
    "papers": 15
   },
   {
    "name": "Xinyan Chen",
    "id": "2150987892",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Xiangyu Chen",
    "id": "2388613119",
    "h_index": 1,
    "papers": 12
   },
   {
    "name": "Xinyu Zhou",
    "id": "2395637062",
    "h_index": 2,
    "papers": 13
   },
   {
    "name": "Boyu Ma",
    "id": "2275779097",
    "h_index": 4,
    "papers": 29
   },
   {
    "name": "Iris Yuxuan Hu",
    "id": "2456630988",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Jingliang Li",
    "id": "2409962931",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Celeste Yuxuan Hu",
    "id": "2456830071",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Gen Li",
    "id": "2306977914",
    "h_index": 2,
    "papers": 17
   },
   {
    "name": "Guohao Chen",
    "id": "2292407473",
    "h_index": 6,
    "papers": 17
   },
   {
    "name": "Tianrui Zhu",
    "id": "2456645458",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Zhe Li",
    "id": "2316522351",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Yanjie Ze",
    "id": "2325901084",
    "h_index": 10,
    "papers": 13
   },
   {
    "name": "Haoran Geng",
    "id": "2287929608",
    "h_index": 7,
    "papers": 15
   },
   {
    "name": "Zhiyang Dou",
    "id": "2292386296",
    "h_index": 8,
    "papers": 24
   },
   {
    "name": "Jianxin Bi",
    "id": "2190428643",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Yuejiang Liu",
    "id": "2319285921",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Jianshu Zhou",
    "id": "2299188652",
    "h_index": 3,
    "papers": 24
   },
   {
    "name": "Jiachen Li",
    "id": "2449464309",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "P. Liang",
    "id": "2351606035",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Tatsuya Harada",
    "id": "2253762492",
    "h_index": 4,
    "papers": 20
   },
   {
    "name": "Robert Katzschmann",
    "id": "2456630868",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Harold Soh",
    "id": "2286883070",
    "h_index": 7,
    "papers": 16
   },
   {
    "name": "Na Li",
    "id": "2300488405",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Edward Johns",
    "id": "2253752957",
    "h_index": 10,
    "papers": 13
   },
   {
    "name": "Danica Kragic",
    "id": "2283846373",
    "h_index": 3,
    "papers": 15
   },
   {
    "name": "Jan Peters",
    "id": "2298968504",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Wojciech Matusik",
    "id": "2309691812",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Masayoshi Tomizuka",
    "id": "2261974717",
    "h_index": 7,
    "papers": 27
   },
   {
    "name": "Jitendra Malik",
    "id": "2242761335",
    "h_index": 15,
    "papers": 32
   },
   {
    "name": "Jianfei Yang",
    "id": "2404007795",
    "h_index": 2,
    "papers": 26
   }
  ],
  "comment": "53 pages, 7 figures",
  "topics": [
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.07558v1",
  "pdf_url": "https://arxiv.org/pdf/2608.07558v1",
  "html_url": "https://arxiv.org/html/2608.07558v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.07557",
  "slug": "aerodpo-unleashing-lightweight-uav-navigation-with-high-fidelity-perce",
  "title": "AeroDPO: Unleashing Lightweight UAV Navigation with High-Fidelity Perception and Automated Preference Optimization",
  "abstract": "Vision-Language Navigation for Unmanned Aerial Vehicles (UAV-VLN) requires rapid and reactive control in complex 3D environments. Recent minimalist end-to-end paradigms show great promise but typically rely on massive language models containing billions of parameters, incurring prohibitive latency for real-world edge deployment. In this paper, we challenge this parameter-heavy reliance. Comprehensive cross-scale evaluations reveal the critical insight that perception quality fundamentally outweighs language reasoning capacity. We demonstrate that a lightweight 2B model equipped with high-fidelity visual inputs completely matches the overall success rates of massive 7B baselines. However, this minimalist policy exposes a fundamental robustness flaw inherent to pure Behavior Cloning (BC). Lacking explicit negative feedback, the agent fails to internalize robust spatial constraints and exhibits alarming collision rates in out-of-distribution (OOD) scenarios. To overcome this vulnerability without relying on unscalable human annotations, we propose AeroDPO, a zero-cost automated Direct Preference Optimization pipeline driven by deterministic physical simulation state rollback. Upon detecting collisions, the system autonomously rewinds the environment to extract causal reasoning errors as rejected actions, applies decoupled privileged interventions to synthesize collision-avoidance preferred maneuvers, and leverages an offline vision language inspector to filter visual ambiguities. By equipping our 2B model with this automated data flywheel, AeroDPO boosts success rates to 49.16% on unmapped scenarios while drastically suppressing collision rates, establishing a new SOTA for autonomous aerial agents.",
  "published": "2026-08-02",
  "updated": "2026-08-02",
  "year": "2026",
  "authors": [
   "Peng Xu",
   "Chengcheng Wang",
   "Shaohua Wan"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper proposes AeroDPO, a zero-cost automated Direct Preference Optimization pipeline driven by deterministic physical simulation state rollback and boosts success rates to 49.16% on unmapped scenarios while drastically suppressing collision rates, establishing a new SOTA for autonomous aerial agents.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Peng Xu",
    "id": "2153917579",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Chengcheng Wang",
    "id": "2449438113",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Shaohua Wan",
    "id": "2404167588",
    "h_index": 2,
    "papers": 7
   }
  ],
  "comment": "7 pages, 3 figures, 4 tables",
  "topics": [
   "imitation-diffusion",
   "navigation",
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.07557v1",
  "pdf_url": "https://arxiv.org/pdf/2608.07557v1",
  "html_url": "https://arxiv.org/html/2608.07557v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.07555",
  "slug": "you-don-t-need-to-stay-in-the-loop-an-agentic-robotics-loop-for-robot",
  "title": "You Don't Need To Stay in The Loop: An Agentic Robotics Loop for Robot-Policy Improvement",
  "abstract": "Coding agents such as Claude Code and Codex close the software loop: a main agent manages the loop, subagents analyze and execute, tools do the work. We port this architecture to robot-policy improvement, where one difference dominates the design: robotic tools---trained policies, training pipelines, data collection---fail routinely, so a tool's quality must be measured, recorded at every call, and expired when the artifact behind it changes. AgenticRobotics is a backend-independent control plane in which an LLM controller drives disposable workers through durable train--evaluate--improve transactions: an immutable objective, controller-owned measurement, commit-keyed crash recovery, an evidence-graded skill library, and a tool registry with a standardized, recorded call surface. The title is an operational claim, not a selection claim: the operator can leave because promotion is evidence-gated, state is recoverable, and capability quality is derived from records---not because the loop picks better checkpoints than a human; on the one lineage we measured, it does not. The gates measurably buy false-promotion control (0.001 per run hardened versus 0.005--0.021 shipped), anytime-valid decisions under optional stopping, zero lost or duplicate effects under kill injection, and six of six artifact-tampering classes caught by a signed verifier.",
  "published": "2026-08-02",
  "updated": "2026-08-02",
  "year": "2026",
  "authors": [
   "Hang Yu"
  ],
  "author_count": 1,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "AgenticRobotics is a backend-independent control plane in which an LLM controller drives disposable workers through durable train--evaluate--improve transactions: an immutable objective, controller-owned measurement, commit-keyed crash recovery, an evidence-graded skill library, and a tool registry with a standardized, recorded call surface.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hang Yu",
    "id": "2110985387",
    "h_index": 3,
    "papers": 9
   }
  ],
  "comment": "Agentic Robotics Preview Version",
  "topics": [
   "data-teleop"
  ],
  "orgs": [
   "MIT"
  ],
  "abs_url": "https://arxiv.org/abs/2608.07555v1",
  "pdf_url": "https://arxiv.org/pdf/2608.07555v1",
  "html_url": "https://arxiv.org/html/2608.07555v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2608.01535",
  "slug": "star-vlm-spatiotemporal-grounding-vision-language-models-for-motion-an",
  "title": "STAR-VLM: Spatiotemporal Grounding Vision-Language Models for Motion and Velocity Estimation via Automotive Radar Supervision",
  "abstract": "Vision-language models (VLMs) are emerging as a key component of embodied intelligence, with growing applications in auto-labeling and end-to-end autonomous driving. However, existing approaches for improving spatiotemporal reasoning in VLMs often rely on complex preprocessing pipelines, expensive human annotations, or synthetic data, which limit scalability and introduce potential sim-to-real gaps. Moreover, although these methods have improved spatiotemporal understanding, they still lack strong metric reasoning capabilities for dynamic scenes, such as estimating object motion in real-world units. Prior work has explored LiDAR-based metric depth supervision to enhance spatial perception, but it does not directly address temporal reasoning. We introduce STAR-VLM, an automotive radar-supervised framework that enhances spatiotemporal VLMs with motion reasoning and metric velocity estimation for autonomous driving. Automotive radar is a low-cost and widely deployed sensor that provides complementary spatiotemporal supervision through range and Doppler measurements. By leveraging these measurements as label-free ground truth during training, STAR-VLM improves the metric spatiotemporal reasoning ability of VLMs. Through experiments on driving scenarios, we show that STAR-VLM achieves state-of-the-art performance on both motion classification and metric velocity estimation, outperforming even task-specific methods designed for each task. These results highlight automotive radar as a scalable and cost-effective source of supervision for building metric-aware spatiotemporal VLMs for real-world autonomous driving.",
  "published": "2026-08-02",
  "updated": "2026-08-02",
  "year": "2026",
  "authors": [
   "Pou-Chun Kung",
   "Aryaman Rao",
   "Utkrisht Sahai",
   "Hemanth Murali",
   "Yi Liu",
   "Rui-Yu Lin",
   "Katherine A. Skinner"
  ],
  "author_count": 7,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "STAR-VLM is introduced, an automotive radar-supervised framework that enhances spatiotemporal VLMs with motion reasoning and metric velocity estimation for autonomous driving, and achieves state-of-the-art performance on both motion classification and metric velocity estimation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Pou-Chun Kung",
    "id": "2053817685",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Aryaman Rao",
    "id": "2191078339",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "U. Sahai",
    "id": "2148755625",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Hemanth Murali",
    "id": "2455410329",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yi Liu",
    "id": "2321101598",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Ruijie Lin",
    "id": "2445678424",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Katherine A. Skinner",
    "id": "2241279092",
    "h_index": 7,
    "papers": 35
   }
  ],
  "comment": "",
  "topics": [
   "sim2real",
   "navigation",
   "data-teleop",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.01535v1",
  "pdf_url": "https://arxiv.org/pdf/2608.01535v1",
  "html_url": "https://arxiv.org/html/2608.01535v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.01506",
  "slug": "rapid-embodiment-adaptation-for-quadrupedal-locomotion",
  "title": "Rapid Embodiment Adaptation for Quadrupedal Locomotion",
  "abstract": "Humans readily adapt their movements as their bodies change through aging, injury, or load carrying, but learning-based robot policies often break when hardware properties shift. We introduce an online embodiment adaptation framework for quadrupedal locomotion that infers embodiment parameters from short interaction histories and conditions control on the inferred hardware state. Our method pairs a generalist policy trained under embodiment randomization with a lightweight adaptation module that identifies physical changes within half a second. We evaluate two representative forms of embodiment variation: joint-range constraints and trunk-mass changes, corresponding to joint-level kinematic degradation and body-level dynamic variation. In simulation, the module accurately estimates these changes and enables closed-loop control that substantially outperforms policies conditioned directly on interaction history. On a real Unitree Go2 robot, our system maintains stable locomotion under severe instances of the evaluated changes, including a fully locked leg and a 5 kg payload, where non-adaptive methods fail. These results demonstrate the practicality of explicit online embodiment identification for rapid adaptation to joint-limit and payload-mass changes, and provide a step toward handling broader forms of uncertain, degraded, or changing robot hardware.",
  "published": "2026-08-02",
  "updated": "2026-08-05",
  "year": "2026",
  "authors": [
   "Dichen Li",
   "Bo Ai",
   "Nico Bohlinger",
   "Jan Peters",
   "Hao Su",
   "Henrik I. Christensen"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work introduces an online embodiment adaptation framework for quadrupedal locomotion that infers embodiment parameters from short interaction histories and conditions control on the inferred hardware state to demonstrate the practicality of explicit online embodiment identification for rapid adaptation to joint-limit and payload-mass changes.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Dichen Li",
    "id": "2238904846",
    "h_index": 12,
    "papers": 54
   },
   {
    "name": "Bo Ai",
    "id": "2352099808",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Nico Bohlinger",
    "id": "2261085918",
    "h_index": 4,
    "papers": 14
   },
   {
    "name": "Jan Peters",
    "id": "2285252911",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Hao Su",
    "id": "2382452551",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Henrik I. Christensen",
    "id": "2290489704",
    "h_index": 5,
    "papers": 6
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "foundation-pretraining"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2608.01506v2",
  "pdf_url": "https://arxiv.org/pdf/2608.01506v2",
  "html_url": "https://arxiv.org/html/2608.01506v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2608.01410",
  "slug": "gentrack-physical-alignment-for-robot-native-motion-generation-and-zer",
  "title": "GenTrack: Physical Alignment for Robot-Native Motion Generation and Zero-Shot Humanoid Tracking",
  "abstract": "General-purpose humanoid trackers can execute diverse references, but their zero-shot coverage depends on large embodied corpora that are costly to extend. Text-to-motion generators offer scalable supervision, yet models trained on human motion or retargeted data inherit a gap between kinematic plausibility and robot executability. Existing one-way pipelines fix either the generated corpus or the reward tracker. We introduce GenTrack, an online generator--tracker framework that alternates execution-grounded, group-relative generator alignment with tracker training on newly generated references; anchoring and rehearsal constrain drift. On Unitree G1, we evaluate GenTrack with ProtoMotions and SONIC backbones across three zero-shot tracking splits including public AMASS and LAFAN benchmarks, and a private out-of-distribution test set of 1,024 prompt-motion pairs in the wild. The online co-training strategy consistently produces generators that output more robot-executable motions with strong semantic alignment, and trackers with markedly broader zero-shot coverage and improved tracking accuracy, especially on out-of-distribution references. These results demonstrate that joint online post-training effectively narrows the executability gap between retargeted references and robot-native motion, advancing zero-shot humanoid control without additional data collection and beyond the limitations of a static reference pool.",
  "published": "2026-08-02",
  "updated": "2026-08-05",
  "year": "2026",
  "authors": [
   "Zeyu Ling",
   "Xinyao Yu",
   "Renye Yan",
   "Jikang Cheng",
   "Zhanke Wang",
   "Qing Shuai",
   "Changqing Zou"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "GenTrack is introduced, an online generator--tracker framework that alternates execution-grounded, group-relative generator alignment with tracker training on newly generated references; anchoring and rehearsal constrain drift and demonstrates that joint online post-training effectively narrows the executability gap between retargeted references and robot-native motion.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zeyu Ling",
    "id": "2237987392",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Xinyao Yu",
    "id": "2267397071",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Renye Yan",
    "id": "2147420753",
    "h_index": 4,
    "papers": 18
   },
   {
    "name": "Jikang Cheng",
    "id": "2148467637",
    "h_index": 5,
    "papers": 23
   },
   {
    "name": "Zhanke Wang",
    "id": "2455268645",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Qing Shuai",
    "id": "2066148547",
    "h_index": 20,
    "papers": 34
   },
   {
    "name": "Changqing Zou",
    "id": "2332354089",
    "h_index": 1,
    "papers": 5
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2608.01410v2",
  "pdf_url": "https://arxiv.org/pdf/2608.01410v2",
  "html_url": "https://arxiv.org/html/2608.01410v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2608.01402",
  "slug": "demystifying-when-and-why-vlas-fail-in-contact-rich-tasks-and-how-to-f",
  "title": "Demystifying When and Why VLAs Fail in Contact-Rich Tasks and How to Fix Them",
  "abstract": "We address the problem of understanding when and why Vision-Language-Action models struggle with contact-rich manipulation tasks that require precise physical interaction. Prior work has primarily focused on addressing contact failures through force-augmented architectures and training-time regularizers, yet the root causes of these failures remain underexplored. We identify two distinct failure modes underlying this gap. Precision failures are rooted in a flow-matching policy training mismatch, and force failures arise from the distinctive structure of force signals. We address each failure mode with a targeted mechanism and combine them into FACT, which achieves 66% average success rate across five contact-rich tasks against 41% for the best prior baseline, in an evaluation spanning almost 2,500 real-world rollouts.",
  "published": "2026-08-02",
  "updated": "2026-08-02",
  "year": "2026",
  "authors": [
   "Carlota Par\u00e9s-Morlans",
   "Nils Kuhn",
   "Isabel Liu",
   "Alberta Longhini",
   "Jeannette Bohg"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work addresses the problem of understanding when and why Vision-Language-Action models struggle with contact-rich manipulation tasks that require precise physical interaction with a targeted mechanism and identifies two distinct failure modes underlying this gap.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Carlota Par\u00e9s-Morlans",
    "id": "2152053979",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Nils Kuhn",
    "id": "2455376311",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Isabel Liu",
    "id": "2455334732",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "A. Longhini",
    "id": "2052021416",
    "h_index": 7,
    "papers": 18
   },
   {
    "name": "Jeannette Bohg",
    "id": "2300236693",
    "h_index": 6,
    "papers": 13
   }
  ],
  "comment": "16 pages",
  "topics": [
   "vla",
   "tactile",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.01402v1",
  "pdf_url": "https://arxiv.org/pdf/2608.01402v1",
  "html_url": "https://arxiv.org/html/2608.01402v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.01381",
  "slug": "dreamtrajectory-trajectory-guided-action-generation-with-world-model-a",
  "title": "DreamTrajectory: Trajectory-Guided Action Generation with World Model Alignment for Mobile Manipulation",
  "abstract": "Mobile manipulation requires a robot to coordinate base and arm motion under continuously changing viewpoints and contact conditions, within an action space far larger than that of fixed-base manipulation. Existing Vision-Language-Action (VLA) policies are limited in two respects. (i)They map observations directly to whole-body action chunks, searching this large action space without an explicit task-space motion plan, which makes coordinated base--arm prediction imprecise. (ii)They execute the predicted chunk open-loop, without checking whether the actions can realize the motion the policy intended, so control errors and unmodeled contacts accumulate into a gap between planned and realized motion. We present DreamTrajectory, a trajectory-guided framework for language-conditioned mobile manipulation that introduces one component for each limitation. Addressing(i), DreamTrajectory jointly predicts an intention-level end-effector trajectory and a whole-body action chunk in a single action expert, so that the trajectory explicitly guides base--arm action generation instead of remaining implicit. Addressing(ii), a lightweight trajectory world model predicts the trajectory that a candidate action chunk would induce, and a test-time search--predict--score procedure selects the candidate best aligned with the planned trajectory. On MS-HAB, trajectory guidance raises average success from 32.3% to 47.5% and test-time refinement further to 54.8%, with the largest gains on contact-rich articulated-object tasks. On three real-world mobile manipulation tasks, the corresponding average success rates are 63.3%, 81.7%, and 90.0%.",
  "published": "2026-08-02",
  "updated": "2026-08-02",
  "year": "2026",
  "authors": [
   "Zheng Yang",
   "Wenjie Zhang",
   "Xiangyu Chen",
   "Wenxuan Song",
   "Xianpeng Wang",
   "Yihang Kang",
   "Wen Chen",
   "Lujia Wang",
   "Renjing Xu",
   "Xiaowen Chu"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "DreamTrajectory is presented, a trajectory-guided framework for language-conditioned mobile manipulation that introduces one component for each limitation of existing Vision-Language-Action policies, and jointly predicts an intention-level end-effector trajectory and a whole-body action chunk in a single action expert.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zheng Yang",
    "id": "2455495509",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Wenjie Zhang",
    "id": "2321872992",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Xiangyu Chen",
    "id": "2388613119",
    "h_index": 1,
    "papers": 12
   },
   {
    "name": "Wenxuan Song",
    "id": "2293142288",
    "h_index": 14,
    "papers": 46
   },
   {
    "name": "Xianpeng Wang",
    "id": "2455488317",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yihan Kang",
    "id": "2214764272",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Wen Chen",
    "id": "2444700942",
    "h_index": 0,
    "papers": 5
   },
   {
    "name": "Lujia Wang",
    "id": "2440936661",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Renjing Xu",
    "id": "2385455909",
    "h_index": 2,
    "papers": 21
   },
   {
    "name": "Xiaowen Chu",
    "id": "2307015152",
    "h_index": 4,
    "papers": 9
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "vla",
   "humanoids",
   "tactile",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.01381v1",
  "pdf_url": "https://arxiv.org/pdf/2608.01381v1",
  "html_url": "https://arxiv.org/html/2608.01381v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.01332",
  "slug": "rake-compress-riccati-recursions-for-parallel-scenario-tree-model-pred",
  "title": "Rake-Compress Riccati Recursions for Parallel Scenario-Tree Model Predictive Control",
  "abstract": "Scenario-tree model predictive control (MPC) represents future information by a rooted tree and optimizes a nonanticipative policy over that tree. Numerical methods for solving the resulting nonlinear program typically compute their search directions through a sequence of branched linear-quadratic regulator (LQR) subproblems. The standard tree Riccati recursion requires linear work but has a dependency chain proportional to tree height. We present an algebraically exact parallel solver based on rake-compress tree contraction. After independent local control condensation, its two operations act on node and edge data that represent conditional quadratic functions. A rake eliminates a leaf and its parent edge, adding their reduced contribution to the parent-node data. A compress eliminates a unary node and replaces its two adjacent edges by one edge, using the same conditional-value composition as parallel Riccati methods on a chain. Together they contract an arbitrary rooted tree to its root; reversing the contraction recovers every Riccati coefficient, state, control, and multiplier. Given a reusable topology plan, a solve with $N$ nodes and fixed state and control dimensions has $O(N)$ arithmetic work and storage and $O(\\log N)$ span, independently of tree height, balance, and maximum out-degree. The formulation allows positive-semidefinite dual regularization, including the unregularized case, and an exact linear-size lifting covers the standard scenario-MPC convention of one control per information node. We prove the contraction identities and equivalence to the Karush-Kuhn-Tucker (KKT) system. Three MIT-licensed JAX packages implement the bidirectional contraction, the dual-regularized LQR solver, and a user-facing primal-dual interior-point solver for tree-structured optimal control.",
  "published": "2026-08-02",
  "updated": "2026-08-02",
  "year": "2026",
  "authors": [
   "Jo\u00e3o Sousa-Pinto"
  ],
  "author_count": 1,
  "categories": [
   "math.OC",
   "cs.RO"
  ],
  "primary_category": "math.OC",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The contraction identities and equivalence to the Karush-Kuhn-Tucker (KKT) system are proved and three MIT-licensed JAX packages implement the bidirectional contraction, the dual-regularized LQR solver, and a user-facing primal-dual interior-point solver for tree-structured optimal control.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jo\u00e3o Sousa-Pinto",
    "id": "2366067347",
    "h_index": 1,
    "papers": 3
   }
  ],
  "comment": "16 pages",
  "topics": [
   "rl-control"
  ],
  "orgs": [
   "MIT"
  ],
  "abs_url": "https://arxiv.org/abs/2608.01332v1",
  "pdf_url": "https://arxiv.org/pdf/2608.01332v1",
  "html_url": "https://arxiv.org/html/2608.01332v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2608.01265",
  "slug": "hermite-curves-as-trajectory-priors-for-vision-language-action-models",
  "title": "Hermite Curves as Trajectory Priors for Vision-Language-Action Models",
  "abstract": "Despite recent progress in Vision-Language-Action (VLA) models for robotic manipulation, the action chunk remains a weakly structured interface. Existing work typically flatten each chunk into per-timestep controls, relying on implicit data learning that manifests as jagged motion and boundary discontinuities during physical execution. To address these limitations, we introduce Hermite trajectory priors, parameterizing the chunk trajectory as a piecewise cubic Hermite curve defined by endpoint positions and velocities to explicitly enforce smoothness and continuity. We instantiate this fixed operator across discrete autoregressive and continuous generative paradigms via three variants: (1) Hermite Tokens, which predict quantized boundary variables autoregressively; (2) Hermite Scaffold, which decomposes clean actions into a base scaffold and residuals; and (3) Hermite Regularization, which applies the prior strictly as an auxiliary training objective. Across simulation benchmarks and real-robot platforms, Hermite Regularization achieves superior performance among these three variants, improving \u03c00.5 baseline success rates from 95.9% to 98.7% on LIBERO, 85.7% to 90.9% on LIBERO-plus, and 63.4% to 90.0% across four real-robot tasks without additional inference overhead. Trajectory analyses reveal that explicitly structuring trajectory priors serves most effectively as a learning inductive bias rather than a runtime constraint.",
  "published": "2026-08-02",
  "updated": "2026-08-09",
  "year": "2026",
  "authors": [
   "Qi Lv",
   "Jianming Xing",
   "Zhao Yang",
   "Mingyuan Yao",
   "Yinan Shi",
   "Yawei Jueluo",
   "Mike Zheng Shou",
   "Xiang Deng"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Hermite trajectory priors are introduced, parameterizing the chunk trajectory as a piecewise cubic Hermite curve defined by endpoint positions and velocities to explicitly enforce smoothness and continuity in VLA models.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Qi Lv",
    "id": "66962290",
    "h_index": 7,
    "papers": 16
   },
   {
    "name": "Jianming Xing",
    "id": "2445689943",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Zhao Yang",
    "id": "2327890025",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Mingyuan Yao",
    "id": "2449884698",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yinan Shi",
    "id": "2348741875",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Yawei Jueluo",
    "id": "2311117886",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "M. Shou",
    "id": "2047358650",
    "h_index": 49,
    "papers": 277
   },
   {
    "name": "Xiang Deng",
    "id": "2267916805",
    "h_index": 6,
    "papers": 18
   }
  ],
  "comment": "Project page is available at https://aopolin-lv.github.io/Hermite/",
  "topics": [
   "vla",
   "sim2real"
  ],
  "orgs": [
   "Physical Intelligence"
  ],
  "abs_url": "https://arxiv.org/abs/2608.01265v2",
  "pdf_url": "https://arxiv.org/pdf/2608.01265v2",
  "html_url": "https://arxiv.org/html/2608.01265v2",
  "code_url": "https://aopolin-lv.github.io/Hermite/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.01221",
  "slug": "endowam-a-grounded-world-action-model-for-generalizable-endoscopic-nav",
  "title": "EndoWAM: A Grounded World-Action Model for Generalizable Endoscopic Navigation",
  "abstract": "Autonomous endoscopic navigation can reduce clinicians' operational burden, yet robust control remains challenging due to tissue deformation, transient occlusions, and rapidly changing viewpoints. Existing learning-based policies typically predict actions from current observations without explicitly modeling future dynamics, limiting their robustness and reliability in safety-critical settings. World Action Models (WAMs) offer a promising alternative by coupling predictive visual dynamics with action generation, but extending them to robotic endoscopy remains challenging due to limited training data, restricted viewpoint diversity, deformable anatomy, and high inference latency. We present EndoWAM, which is, to our knowledge, the first WAM for generalizable robotic endoscopic navigation. EndoWAM introduces future grounding, which predicts task-relevant target regions in future observations from intermediate denoising features of a video world model. Specifically, EndoWAM couples a lightweight diffusion transformer for future target-region prediction with a discrete action expert through a shared predictive representation. This design injects target-aware supervision into predictive dynamics modeling, improving robustness to visual degradation and viewpoint changes while enabling real-time control in a single denoising pass. We further introduce EndoMotion, a robotic endoscopic motion dataset spanning three anatomically distinct procedures: ureteroscopy, esophagoscopy, and endoscopic retrograde cholangiopancreatography (ERCP). EndoWAM consistently outperforms all baselines and alternative grounding strategies, while demonstrating strong zero-shot generalization to unseen viewpoints, environments, and targets. These results establish EndoWAM as a predictive, target-grounded framework for accurate, generalizable, and long-horizon navigation in visually constrained endoscopic environments.",
  "published": "2026-08-02",
  "updated": "2026-08-02",
  "year": "2026",
  "authors": [
   "Jinsong Lin",
   "Zikang Pan",
   "Wanhao Liu",
   "Chi Kit Ng",
   "Liangjing Shao",
   "Zihang Yu",
   "Ziyu Wang",
   "Yin Wang",
   "Jiaxi Wang",
   "Jeremy Yuen-Chun Teoh",
   "Zhiyong Xiong",
   "Huxin Gao",
   "Hongliang Ren"
  ],
  "author_count": 13,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "EndoWAM is presented, which is, to the authors' knowledge, the first WAM for generalizable robotic endoscopic navigation and introduces future grounding, which predicts task-relevant target regions in future observations from intermediate denoising features of a video world model.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jinsong Lin",
    "id": "2445481374",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Zikang Pan",
    "id": "1720879996",
    "h_index": 9,
    "papers": 17
   },
   {
    "name": "Wanhao Liu",
    "id": "2455435040",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Chikit Ng",
    "id": "2307447792",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Liangjing Shao",
    "id": "2228022503",
    "h_index": 2,
    "papers": 21
   },
   {
    "name": "Zihang Yu",
    "id": "2455430441",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Ziyu Wang",
    "id": "2276664873",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Yin Wang",
    "id": "2455440174",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Jiaxi Wang",
    "id": "2455432840",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "J. Teoh",
    "id": "2201993686",
    "h_index": 8,
    "papers": 15
   },
   {
    "name": "Zhiyong Xiong",
    "id": "47845657",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Huxin Gao",
    "id": "1653073415",
    "h_index": 13,
    "papers": 49
   },
   {
    "name": "Hongliang Ren",
    "id": "2260612957",
    "h_index": 13,
    "papers": 53
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.01221v1",
  "pdf_url": "https://arxiv.org/pdf/2608.01221v1",
  "html_url": "https://arxiv.org/html/2608.01221v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.01102",
  "slug": "caat-contact-aware-attention-scaling-and-tactile-masking-for-data-effi",
  "title": "CAAT: Contact-Aware Attention Scaling and Tactile Masking for Data-Efficient Contact-Rich Manipulation",
  "abstract": "In contact-rich manipulation, visual observations primarily guide motion in free space, whereas tactile observations become particularly informative during contact. However, standard Transformer-based visuo-tactile policies typically rely on either token concatenation or learnable gating. These approaches lack explicit contact-aware priors, making it difficult to efficiently learn effective cross-modal representations from demonstrations. To address this limitation, we propose CAAT, a lightweight contact-aware framework that explicitly incorporates contact priors through attention scaling and dynamic tactile masking. Specifically, CAAT emphasizes visual information before contact and tactile information during contact. It also suppresses static background tokens by comparing the current tactile observation with a non-contact reference. CAAT can be integrated into commonly used Transformer-based policies without modifying their action decoders. In simulation, integrating CAAT with ACT improves the average success rate by 18.0 percentage points over direct visuo-tactile fusion and by 10.0 percentage points over gated fusion. In real-world experiments using a visuo-tactile UMI platform, CAAT achieves an average success rate of 60.0% across ACT, Diffusion Policy, and $\u03c0_0$, outperforming the strongest baseline by an average of 21.1 percentage points. These results demonstrate that explicit contact priors and dynamic tactile masking are effective in improving visuo-tactile policy learning and task performance of diverse policy architectures. https://mrjiangjm.github.io/caat/",
  "published": "2026-08-02",
  "updated": "2026-08-02",
  "year": "2026",
  "authors": [
   "Jiaming Jiang",
   "Yuzhe Huang",
   "Hao Liang",
   "Pei Lin",
   "Shengcheng Luo",
   "Fanrong Dong",
   "Jiaping Wu",
   "Chenxi Xiao",
   "Wanlin Li",
   "Ziyuan Jiao"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "CAAT is a lightweight contact-aware framework that explicitly incorporates contact priors through attention scaling and dynamic tactile masking and demonstrates that explicit contact priors and dynamic tactile masking are effective in improving visuo-tactile policy learning and task performance of diverse policy architectures.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jiaming Jiang",
    "id": "2167229335",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Yuzhe Huang",
    "id": "2356910381",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Haoxiang Liang",
    "id": "2455420369",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Pei Lin",
    "id": "2357017108",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Shengcheng Luo",
    "id": "2309215693",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Fanrong Dong",
    "id": "2455291899",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Jiaping Wu",
    "id": "2445223101",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Chenxi Xiao",
    "id": "2356920855",
    "h_index": 3,
    "papers": 14
   },
   {
    "name": "Wanlin Li",
    "id": "2357100311",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Ziyuan Jiao",
    "id": "2356946141",
    "h_index": 3,
    "papers": 13
   }
  ],
  "comment": "11 pages, 6 figures",
  "topics": [
   "tactile",
   "imitation-diffusion",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.01102v1",
  "pdf_url": "https://arxiv.org/pdf/2608.01102v1",
  "html_url": "https://arxiv.org/html/2608.01102v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.01035",
  "slug": "wam-diff2-hierarchical-ar-to-diffusion-distillation-for-highly-efficie",
  "title": "WAM-Diff2: Hierarchical AR-to-Diffusion Distillation for Highly Efficient Autonomous Driving VLA",
  "abstract": "Vision-Language-Action (VLA) models have emerged as a prominent paradigm for end-to-end autonomous driving; however, their efficient deployment is severely constrained by high computational latency and exposure bias arising from sequential autoregressive decoding. Conversely, while specialized diffusion policies enable low-latency, parallel execution, training them from scratch typically yields narrow, single-task architectures that lack holistic visual-linguistic reasoning. Successfully transforming pre-trained autoregressive generalists into parallel diffusion models could combine multi-task cognitive intelligence with execution efficiency, yet this transition presents a formidable architectural challenge due to mismatched attention patterns (causal versus bidirectional) and divergent optimization objectives. To bridge this divide, we introduce WAM-Diff2, a multi-task discrete diffusion VLA framework powered by a three-stage hierarchical distillation strategy. By structuring the architectural shift through progressive block-wise adaptation, block-wise distillation, and model-wise cross-scale distillation, WAM-Diff2 preserves the underlying semantic foundations of the base model while accelerating inference. Extensive evaluations across driving understanding, perception, and planning benchmarks demonstrate that WAM-Diff2 effectively mitigates exposure bias and achieves performance parity with autoregressive baselines. Crucially, the autoregressive-to-diffusion transition yields a 2.8x decoding speedup, which scales to an ultimate 15.1x acceleration when combined with system-level optimizations including FlashInfer and CUDA Graphs.",
  "published": "2026-08-02",
  "updated": "2026-08-18",
  "year": "2026",
  "authors": [
   "Zhihao Zhu",
   "Hanlin Shang",
   "Mingwang Xu",
   "Feipeng Cai",
   "Zhuolin He",
   "Yaoyi Li",
   "Jianhua Han",
   "Hang Xu",
   "Siyu Zhu"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "WAM-Diff2 is introduced, a multi-task discrete diffusion VLA framework powered by a three-stage hierarchical distillation strategy that effectively mitigates exposure bias and achieves performance parity with autoregressive baselines.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhihao Zhu",
    "id": "2397618745",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Hanlin Shang",
    "id": "2306124581",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Mingwang Xu",
    "id": "2306097377",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Feipeng Cai",
    "id": "2397378213",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Zhuolin He",
    "id": "2308221747",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Yaoyi Li",
    "id": "2397481875",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Jianhua Han",
    "id": "47180442",
    "h_index": 29,
    "papers": 86
   },
   {
    "name": "Hang Xu",
    "id": "2237071248",
    "h_index": 9,
    "papers": 12
   },
   {
    "name": "Siyu Zhu",
    "id": "2288876720",
    "h_index": 7,
    "papers": 20
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "navigation",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.01035v4",
  "pdf_url": "https://arxiv.org/pdf/2608.01035v4",
  "html_url": "https://arxiv.org/html/2608.01035v4",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.01013",
  "slug": "rl-bootstrapping-of-openvla-oft-for-a-novel-robot-embodiment",
  "title": "RL Bootstrapping of OpenVLA-OFT for a Novel Robot Embodiment",
  "abstract": "Adapting a pretrained vision-language-action (VLA) policy to a new robot usually assumes embodiment-specific demonstrations. This assumption is especially restrictive for custom robots whose morphology differs strongly from the manipulators seen in large robot datasets. We study a harder setting: zero-demo embodiment alignment of OpenVLA-OFT on a cable-driven parallel robot (CDPR) with a simple gripper and a previously unseen control interface. Instead of supervised fine-tuning, we use reinforcement learning in simulation with dense geometric rewards computed from simulator state. The training is performed in two stages: a PPO stage for directional motion primitives, followed by GRPO continuation from the PPO checkpoint with an expanded instruction space that includes object-conditioned commands. On the four shared directional instructions, the average held-out success rate improves from 34.25\\% after PPO to 53.50\\% after PPO$\\rightarrow$GRPO, with especially large gains on \\texttt{move left} and \\texttt{move backward}. In the GRPO stage we additionally introduce \\texttt{move to <object>} over eight target objects and obtain 39/400 = 9.75\\% strict success, while qualitative rollouts frequently show correct target-directed approach behavior before late-stage instability. Compared with prior OpenVLA and OpenVLA-OFT results, which rely on demonstration datasets and mostly standard rigid-arm embodiments, our method uses no embodiment-specific dataset at all. The results do not yet establish robust manipulation, but they provide stronger evidence that RL-only bootstrapping can create the first usable language-conditioned controller for a genuinely novel embodiment.",
  "published": "2026-08-02",
  "updated": "2026-08-02",
  "year": "2026",
  "authors": [
   "Damir Nurtdinov",
   "Alexei Kornaev",
   "Alexander Maloletov"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The results provide stronger evidence that RL-only bootstrapping can create the first usable language-conditioned controller for a genuinely novel embodiment, and use reinforcement learning in simulation with dense geometric rewards computed from simulator state.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Damir Nurtdinov",
    "id": "2384761643",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "A. Kornaev",
    "id": "2330867666",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "A. Maloletov",
    "id": "2384762305",
    "h_index": 2,
    "papers": 6
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "sim2real",
   "rl-control",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.01013v1",
  "pdf_url": "https://arxiv.org/pdf/2608.01013v1",
  "html_url": "https://arxiv.org/html/2608.01013v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.00970",
  "slug": "freqnav-stage-wise-frequency-routing-for-object-oriented-aerial-vision",
  "title": "FreqNav: Stage-Wise Frequency Routing for Object-Oriented Aerial Vision-Language Navigation",
  "abstract": "Object-oriented aerial vision-and-language navigation (VLN) requires searching for a described target and landing on it precisely, under long-horizon and closed-loop control. Guided by a target-descriptive instruction during navigation, perceptual priorities dynamically evolve: early-stage exploration prioritizes low-frequency spatial layout, and then shifts to high-frequency target details. Existing VLN methods model the varying perceptual requirements across navigation stages with identical visual tokens, leading to interference from irrelevant objects and background clutter. To this end, we therefore formulate long-horizon aerial navigation as a frequencypreference shift from spatial structure to local detail and propose FreqNav, a lightweight frequency-routing adaptive perception framework. Under a fixed computational budget, FreqNav dynamically reallocates visual tokens across frequency components according to the current navigation stage. A Frequency Token Router selects stage-relevant visual representations from dual-view observations, while a Phase-dependent Grounding Module anchors visual evidence through explicit supervision. A Diffusion Transformer then predicts smooth trajectories for continuous control. Experiments show that FreqNav outperforms strong baselines while achieving approximately 3x faster inference. Real-world deployment further demonstrates its effectiveness, efficiency, and practical potential for long-horizon aerial autonomy.",
  "published": "2026-08-02",
  "updated": "2026-08-02",
  "year": "2026",
  "authors": [
   "Yin Tang",
   "Jiawei Ma",
   "Jiahao Li",
   "Hao Zhang",
   "Zhemin Sun",
   "Jianqiao Sun",
   "Deyu Zhang"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "FreqNav, a lightweight frequency-routing adaptive perception framework, which dynamically reallocates visual tokens across frequency components according to the current navigation stage and outperforms strong baselines while achieving approximately 3x faster inference.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yin Tang",
    "id": "2291062877",
    "h_index": 3,
    "papers": 13
   },
   {
    "name": "Jiawei Ma",
    "id": "2361814641",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Jiahao Li",
    "id": "2455476319",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Hao Zhang",
    "id": "2304963754",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Zhemin Sun",
    "id": "2455444668",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Jianqiao Sun",
    "id": "2278823838",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Deyu Zhang",
    "id": "2290946554",
    "h_index": 2,
    "papers": 12
   }
  ],
  "comment": "",
  "topics": [
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.00970v1",
  "pdf_url": "https://arxiv.org/pdf/2608.00970v1",
  "html_url": "https://arxiv.org/html/2608.00970v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.00950",
  "slug": "swimm3r-splatting-with-medium-aware-sfm-for-underwater-3d-reconstructi",
  "title": "Swimm3R: Splatting with Medium-aware SfM for Underwater 3D Reconstruction",
  "abstract": "We propose Swimm3R, a unified framework that combines medium-aware structure-from-motion (SfM) with Underwater Beta Splatting to address scattering- and attenuation-induced failures in underwater 3D reconstruction. Swimm3R distills in-air geometric priors into a feed-forward backbone and uses a physics head to regress underwater image-formation parameters, camera poses, and restored point clouds. Additionally, we introduce Underwater Beta Splatting, which extends Gaussian splatting with Beta primitives and scattering-aware geometric gradients for stable underwater geometry representation. We further establish the Barbados underwater video dataset to demonstrate the effectiveness of our method in challenging underwater environments. On this dataset, Swimm3R robustly recovers underwater scene structure under challenging scattering conditions, yielding coherent seafloor geometry. Using these predicted point clouds, the proposed Underwater Beta Splatting improves average PSNR by $1.47$ dB over WaterSplatting while increasing downstream localization performance by $2.0$ and $2.4$ percentage points in RRA@15 and RTA@15, respectively.",
  "published": "2026-08-02",
  "updated": "2026-08-02",
  "year": "2026",
  "authors": [
   "Minseong Kweon",
   "Junaed Sattar"
  ],
  "author_count": 2,
  "categories": [
   "cs.CV",
   "cs.RO",
   "eess.IV"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Underwater Beta Splatting is introduced, which extends Gaussian splatting with Beta primitives and scattering-aware geometric gradients for stable underwater geometry representation and robustly recovers underwater scene structure under challenging scattering conditions, yielding coherent seafloor geometry.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Minseong Kweon",
    "id": "2338413044",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Junaed Sattar",
    "id": "1765604",
    "h_index": 28,
    "papers": 105
   }
  ],
  "comment": "Project Page: https://mnseong.github.io/swimm3r.github.io",
  "topics": [
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.00950v1",
  "pdf_url": "https://arxiv.org/pdf/2608.00950v1",
  "html_url": "https://arxiv.org/html/2608.00950v1",
  "code_url": "https://mnseong.github.io/swimm3r.github.io",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2608.00945",
  "slug": "vertiakd-adaptive-off-road-kinodynamics-on-vertically-challenging-terr",
  "title": "VertiAKD: Adaptive Off-Road Kinodynamics on Vertically Challenging Terrain",
  "abstract": "Off-road mobility requires autonomous mobile robots to generalize across heterogeneous vehicle fleets and continuously changing terrain conditions. Existing cross-vehicle adaptation approaches generally assume flat terrain, while terrain-aware kinodynamic models often require platform-specific data collection and retraining. To this end, we propose VertiAKD, a unified framework for transferring and adapting off-road kinodynamic knowledge across diverse vehicles on geometrically and semantically complex terrain simultaneously. VertiAKD learns a shared mobility representation that jointly encodes vehicle configurations, trajectory transitions, and local elevation and semantic terrain features. Given limited data from a novel vehicle operating on unseen terrain, VertiAKD identifies the most relevant mobility descriptors and transfers their knowledge to initialize a terrain-aware kinodynamic model via function encoders, which is then periodically refined online from streaming observations without gradient-based retraining. We evaluate VertiAKD in the Verti-Bench simulator, built on the Chrono multi-physics engine, and on five physical configurations of the Verti-4-Wheeler platform. With only one minute of new trajectory data and associated terrain features, VertiAKD reduces long-horizon prediction error by up to 34.52% over direct mobility descriptor transfer across diverse unseen vehicle configurations and 94.43% over competing baselines. We further demonstrate robust closed-loop trajectory tracking in both simulation and physical experiments, highlighting the effectiveness of terrain-aware cross-vehicle knowledge transfer for accurate modeling and reliable off-road navigation.",
  "published": "2026-08-02",
  "updated": "2026-08-18",
  "year": "2026",
  "authors": [
   "Tong Xu",
   "Chenhui Pan",
   "Francesco Cancelliere",
   "Xuesu Xiao"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "VertiAKD, a unified framework for transferring and adapting off-road kinodynamic knowledge across diverse vehicles on geometrically and semantically complex terrain simultaneously, is proposed and robust closed-loop trajectory tracking is demonstrated in both simulation and physical experiments, highlighting the effectiveness of terrain-aware cross-vehicle knowledge transfer.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Tong Xu",
    "id": "2320264790",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "Chenhui Pan",
    "id": "2084643982",
    "h_index": 8,
    "papers": 16
   },
   {
    "name": "Francesco Cancelliere",
    "id": "2375384741",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Xuesu Xiao",
    "id": "2320187442",
    "h_index": 6,
    "papers": 18
   }
  ],
  "comment": "",
  "topics": [
   "sim2real",
   "navigation",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.00945v3",
  "pdf_url": "https://arxiv.org/pdf/2608.00945v3",
  "html_url": "https://arxiv.org/html/2608.00945v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.00931",
  "slug": "stipple-real-time-incremental-gaussian-splatting-with-visual-inertial",
  "title": "Stipple: Real-Time Incremental Gaussian Splatting with Visual-Inertial Tracking",
  "abstract": "3D Gaussian Splatting (3DGS) provides efficient rendering of photo-realistic scenes, but its heavy preprocessing and training steps make it a poor fit for applications that require real-time reconstruction in robotics or XR. This capability is important since it allows immediate feedback and interaction with new environments. Visual-inertial odometry (VIO) and simultaneous localization and mapping (VI-SLAM) systems, on the other hand, specifically target these real-time applications, which makes them a good choice for integration with 3DGS. We propose a new method that tracks and reconstructs simultaneously in real-time by leveraging an efficient visual-inertial tracking system based on Basalt together with a novel incremental method built on top of Brush, an efficient Rust-based GPU-vendor-agnostic implementation of 3D Gaussian Splatting. We show that many of the heavy preprocessing and training steps of 3DGS can be replaced with a more efficient incremental training strategy that has direct access to the information generated by the visual-inertial tracking system. Furthermore, we propose and combine multiple practical improvements to increase the efficiency of the training pipeline and adapt it to run in real-time, parallel to the tracking thread. This work highlights the value of exploiting the complementary nature of SLAM and 3DGS, and how that can lead to promising results for real-time 3D reconstruction.",
  "published": "2026-08-02",
  "updated": "2026-08-02",
  "year": "2026",
  "authors": [
   "Kilian Northoff",
   "Mateo de Mayo",
   "Daniel Cremers"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes a new method that tracks and reconstructs simultaneously in real-time by leveraging an efficient visual-inertial tracking system based on Basalt together with a novel incremental method built on top of Brush, an efficient Rust-based GPU-vendor-agnostic implementation of 3D Gaussian Splatting.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Kilian Northoff",
    "id": "2375057565",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Mateo de Mayo",
    "id": "2374403979",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Daniel Cremers",
    "id": "2332092495",
    "h_index": 5,
    "papers": 21
   }
  ],
  "comment": "",
  "topics": [
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.00931v1",
  "pdf_url": "https://arxiv.org/pdf/2608.00931v1",
  "html_url": "https://arxiv.org/html/2608.00931v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.07553",
  "slug": "self-supervised-learning-from-automatically-generated-demonstrations-f",
  "title": "Self Supervised Learning from Automatically Generated Demonstrations for Visual Robotic Manipulation",
  "abstract": "Robotic manipulation often requires object specific programming, manual data annotation, or calibrated perception pipelines, which limits rapid deployment in practical settings. Learning from demonstration offers a more direct alternative, but collecting demonstrations can still demand human teleoperation or kinesthetic teaching. This paper presents a self supervised visual manipulation method in which a robot automatically generates demonstrations around a target pose and learns relative pose corrections directly from wrist mounted RGB images. The proposed pipeline uses ROS~2 and Isaac Sim to collect labeled image-pose pairs without requiring explicit camera to robot extrinsic calibration. Separate datasets are generated for planar refinement and coarse three dimensional approach, and a convolutional network is trained to regress relative translation and rotation from single frame RGB observations. During execution, a coarse to fine controller first approaches the object using models trained with height variation and then refines the final alignment using planar data. The method is evaluated both in simulation and on a real UR5e collaborative robot equipped with a gripper and a monocular camera. In simulation, the refinement stage reduces the final planar dispersion from 9.69 mm to 5.38 mm. In real world experiments, the system performs end to end grasp attempts on three physical objects and reaches success rates of 66.6% and 63.6% for two objects without object rotation, while still maintaining partial robustness under rotated conditions. These results show that automatically generated demonstrations can support practical visual manipulation with limited setup effort, while also exposing remaining challenges in depth prediction and object dependent generalization.",
  "published": "2026-08-01",
  "updated": "2026-08-01",
  "year": "2026",
  "authors": [
   "Andres Rivas",
   "Anselmo R. Cukla",
   "Rodrigo S. Guerra",
   "Bruna V. Guterres",
   "Ricardo B. Grando"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The results show that automatically generated demonstrations can support practical visual manipulation with limited setup effort, while also exposing remaining challenges in depth prediction and object dependent generalization.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "A. Rivas",
    "id": "123923823",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "A. Cukla",
    "id": "9093628",
    "h_index": 5,
    "papers": 46
   },
   {
    "name": "Rodrigo S. Guerra",
    "id": "2394472886",
    "h_index": 0,
    "papers": 6
   },
   {
    "name": "B. Guterres",
    "id": "2306258966",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "R. B. Grando",
    "id": "103883886",
    "h_index": 11,
    "papers": 32
   }
  ],
  "comment": "Paper accepted at the ICCAS 2026",
  "topics": [
   "dexterous-manipulation",
   "data-teleop",
   "hri",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.07553v1",
  "pdf_url": "https://arxiv.org/pdf/2608.07553v1",
  "html_url": "https://arxiv.org/html/2608.07553v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.07548",
  "slug": "sc-2-wm-a-self-correcting-world-model-with-closed-loop-feedback-for-vi",
  "title": "SC$^{2}$-WM: A Self-Correcting World Model with Closed-Loop Feedback for Vision-and-Language Navigation in Continuous Environments",
  "abstract": "Vision-and-Language Navigation in Continuous Environments (VLN-CE) requires agents to make fine-grained navigation decisions under partial observability. However, most existing methods rely on open-loop execution, lacking mechanisms to detect and correct internal state drift during inference. We propose SC$^{2}$-WM, a self-correcting world model framework that introduces internal feedback for closed-loop decision making in VLN-CE. Our method derives feedback from world-model foresight to perform state-level plan refinement before action execution. To handle challenging scenarios, we further introduce conditional world-aware adaptation, which enables model-level correction by selectively updating the world model at test time when feedback indicates model capacity insufficiency. Experiments on standard VLN-CE benchmarks demonstrate improved navigation robustness and generalization. Our code is available at https://github.com/sunrise-ikun/SC2_WM.",
  "published": "2026-08-01",
  "updated": "2026-08-01",
  "year": "2026",
  "authors": [
   "Xuan Yao",
   "Yuze Zhu",
   "Junyu Gao",
   "Zongmeng Wang",
   "Changsheng Xu"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "ICML 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes SC-2-WM, a self-correcting world model framework that introduces internal feedback for closed-loop decision making in VLN-CE, and introduces conditional world-aware adaptation, which enables model-level correction by selectively updating the world model at test time when feedback indicates model capacity insufficiency.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xuan Yao",
    "id": "2261894061",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Yuze Zhu",
    "id": "2455656339",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Junyu Gao",
    "id": "46930271",
    "h_index": 28,
    "papers": 76
   },
   {
    "name": "Zongmeng Wang",
    "id": "2367428977",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Changsheng Xu",
    "id": "2237947504",
    "h_index": 9,
    "papers": 16
   }
  ],
  "comment": "Accepted by ICML 2026",
  "topics": [
   "world-models",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.07548v1",
  "pdf_url": "https://arxiv.org/pdf/2608.07548v1",
  "html_url": "https://arxiv.org/html/2608.07548v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.02653",
  "slug": "light-loco-parkour-versatile-perceptive-whole-body-locomotion-via-mult",
  "title": "Light-Loco-Parkour: Versatile Perceptive Whole-Body Locomotion via Multi-Skill Distillation",
  "abstract": "Existing humanoid whole-body control systems still fall short of the way humans move through cluttered terrain: they either track expressive whole-body references without terrain generalization, or react to terrain online while leaving the arms, torso, and knees largely unused. We present \\texttt{Light-Loco-Parkour} (LLP), an end-to-end perceptive whole-body locomotion system that closes this gap with a single deployable policy. Conditioned only on onboard depth and a velocity command, the policy decides when to walk, balance, climb, step down, or vault, with no reference input, skill label, hand-coded gate, or runtime motion graph. Compared with prior humanoid systems, LLP makes three contributions. First, it introduces a whole-body perceptive-control pipeline that extends an RL-trained, velocity-tracking locomotion policy with parkour skills learned from object-interacting motions, so the same policy tracks velocity in open terrain, executes whole-body traversal at obstacles, and resumes locomotion afterward. Second, it acquires terrain-conditioned skills from sparse seeds by expanding a single motion into dynamically feasible, terrain-paired references across obstacle geometry, rather than relying on a large motion corpus. Third, it learns autonomous skill transitions from reward, letting the policy decide when and which whole-body skill to invoke from depth and command alone, with no one-hot skill label, hand-coded state machine, or runtime motion generator. Simulation and real-world experiments show high success across both benchmarked terrains and unseen obstacle variations, and the same policy transfers zero-shot to indoor and outdoor hardware experiments. These results demonstrate autonomous perceptive whole-body locomotion on a humanoid in outdoor settings, using only onboard sensing and a single deployable policy.",
  "published": "2026-08-01",
  "updated": "2026-08-01",
  "year": "2026",
  "authors": [
   "Hongming Chen",
   "Zhuoran Li",
   "Hongxi Wang",
   "Jiangpeng Hu",
   "Ziliang Li",
   "Peize Liu",
   "QingRui Zhao",
   "Xuhao Liu",
   "Liang Pan",
   "Ximin Lyu",
   "Yuntao Ma",
   "Tingxiang Fan"
  ],
  "author_count": 12,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "These results demonstrate autonomous perceptive whole-body locomotion on a humanoid in outdoor settings, using only onboard sensing and a single deployable policy.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hongming Chen",
    "id": "2339713019",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Zhuoran Li",
    "id": "2449971374",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Hongxi Wang",
    "id": "2302212963",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Jiangpeng Hu",
    "id": "2318032274",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Ziliang Li",
    "id": "2366602747",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Peize Liu",
    "id": "2189605465",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Qingrui Zhao",
    "id": "2298936138",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Xuhao Liu",
    "id": "2455452157",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Liang Pan",
    "id": "2297133591",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Ximin Lyu",
    "id": "2395912543",
    "h_index": 5,
    "papers": 23
   },
   {
    "name": "Yuntao Ma",
    "id": "2109268030",
    "h_index": 7,
    "papers": 12
   },
   {
    "name": "Tingxiang Fan",
    "id": "26336089",
    "h_index": 16,
    "papers": 24
   }
  ],
  "comment": "https://light-loco-parkour.github.io/",
  "topics": [
   "humanoids"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.02653v1",
  "pdf_url": "https://arxiv.org/pdf/2608.02653v1",
  "html_url": "https://arxiv.org/html/2608.02653v1",
  "code_url": "https://light-loco-parkour.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2608.00880",
  "slug": "bicycle-acrobatics-with-reinforcement-learning",
  "title": "Bicycle Acrobatics with Reinforcement Learning",
  "abstract": "Bicycle robots are fast and energy efficient, but their simple mechanical design and their underactuated and non-holonomic dynamics make highly agile maneuvers difficult to achieve. Here, we use Reinforcement Learning (RL) to enable a bicycle robot to learn and compose a diverse repertoire of dynamic acrobatic stunts. Using different RL formulations such as waypoint following, pose reaching, twist tracking, guided tracking, and motion imitation, the robot acquires autonomous single and multi-table forward and lateral jumps, steerable jumps, front flips, kip-ups, kip-downs, driving, wheelies, bunny hops, and three-point turns. To coordinate these behaviors, we introduce an orchestrator that transitions between policies using state-dependent triggers, enabling robust long-horizon acrobatic stunts. We validate the approach on the Ultra Mobility Vehicle (UMV), a custom bicycle robot, in simulation and hardware. The robot repeatedly traverses tables up to 1 m high, performs more than 15 consecutive autonomous jumps while following waypoints, handles previously unseen multi-table configurations, executes continuous repertoires of kipups, jumps, flips, kip-downs, over more than 20 consecutive trials, and performs more than 10 consecutive autonomous and steerable repertoires of wheelies, lateral jumps, and single-wheel jump downs. These results demonstrate that RL can endow bicycle robots with levels of agility previously associated primarily with legged platforms while preserving the speed and efficiency of wheeled locomotion, establishing a foundation for bicycle acrobatics.",
  "published": "2026-08-01",
  "updated": "2026-08-01",
  "year": "2026",
  "authors": [
   "Shamel Fahmi",
   "Arianna Ilvonen",
   "Samuel Zapolsky",
   "Yu-Ming Chen",
   "Ravi Boggavarapu",
   "Paul Drews",
   "Ashwin Khadke",
   "Dean Molinaro",
   "Kaiyu Zheng",
   "Alfred Rizzi",
   "Gabriel Nelson"
  ],
  "author_count": 11,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Reinforcement Learning can endow bicycle robots with levels of agility previously associated primarily with legged platforms while preserving the speed and efficiency of wheeled locomotion, establishing a foundation for bicycle acrobatics.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Shamel Fahmi",
    "id": "51196778",
    "h_index": 9,
    "papers": 19
   },
   {
    "name": "Arianna Ilvonen",
    "id": "2154624563",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Samuel Zapolsky",
    "id": "3046216",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Yu-Ming Chen",
    "id": "2109307419",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "R. Boggavarapu",
    "id": "10662219",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Paul Drews",
    "id": "2281993663",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Ashwin Khadke",
    "id": "30623582",
    "h_index": 3,
    "papers": 17
   },
   {
    "name": "Dean D. Molinaro",
    "id": "101552178",
    "h_index": 13,
    "papers": 23
   },
   {
    "name": "Kaiyu Zheng",
    "id": "50443937",
    "h_index": 9,
    "papers": 20
   },
   {
    "name": "A. Rizzi",
    "id": "1775644",
    "h_index": 38,
    "papers": 118
   },
   {
    "name": "G. Nelson",
    "id": "27707857",
    "h_index": 14,
    "papers": 27
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "rl-control",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.00880v1",
  "pdf_url": "https://arxiv.org/pdf/2608.00880v1",
  "html_url": "https://arxiv.org/html/2608.00880v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.00854",
  "slug": "minute-scale-training-for-microrobot-navigation",
  "title": "Minute-Scale Training for Microrobot Navigation",
  "abstract": "Microrobots hold significant potential for various applications, where targeted navigation is a basic requirement. Deep reinforcement learning (DRL) has recently emerged as a powerful paradigm for fully autonomous microrobot navigation. Yet, current DRL-based approaches pay limited attention to learning efficiency and effectiveness, requiring hours to days for model training. Consequently, this impedes both rapid practical deployment and parameter optimization. To address these challenges, we present a learning framework that enables effective microrobot navigation policies to be trained within minutes. In the proposed framework, we develop a fully vectorized simulator with more than 10,000 artificial vascular environments, parallelizing dynamics, LiDAR-inspired perception, and feasibility checks across thousands of environments to achieve roughly 190,000 transitions per second. To achieve effectiveness in fast training, we propose a task-shaping-regularization (TSR) reward framework. The TSR framework accelerates convergence, improves final performance, reduces action variation by at least 33.7%, and increases obstacle clearance by at least 2.1% across all evaluated scenarios. Results show that the proposed learning framework reduces training time to under 10 minutes, while supporting zero-shot deployment across distinct microrobot types and navigation scenarios. Collectively, this framework can substantially shorten the design loop and accelerate the deployment of autonomous microrobots.",
  "published": "2026-08-01",
  "updated": "2026-08-09",
  "year": "2026",
  "authors": [
   "Yinghan Sun",
   "Aoji Zhu",
   "Xiang Ji",
   "Yamei Li",
   "Jiachi Zhao",
   "Yun Wang",
   "Li Zhang",
   "Huijun Gao",
   "Lidong Yang"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Results show that the proposed learning framework reduces training time to under 10 minutes, while supporting zero-shot deployment across distinct microrobot types and navigation scenarios, can substantially shorten the design loop and accelerate the deployment of autonomous microrobots.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yinghan Sun",
    "id": "2267731587",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Aoji Zhu",
    "id": "2212961373",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Xiang Ji",
    "id": "2376466840",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Yamei Li",
    "id": "2311300097",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Jiachi Zhao",
    "id": "2213616587",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Yun Wang",
    "id": "2395039826",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Li Zhang",
    "id": "2268770695",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Huijun Gao",
    "id": "2112621462",
    "h_index": 18,
    "papers": 58
   },
   {
    "name": "Lidong Yang",
    "id": "2325782117",
    "h_index": 2,
    "papers": 10
   }
  ],
  "comment": "",
  "topics": [
   "sim2real",
   "rl-control",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.00854v2",
  "pdf_url": "https://arxiv.org/pdf/2608.00854v2",
  "html_url": "https://arxiv.org/html/2608.00854v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.00820",
  "slug": "loopermuscle-fast-and-stable-learning-of-humanoid-whole-body-tracking",
  "title": "LooperMuscle: Fast and Stable Learning of Humanoid Whole-Body Tracking via Structured Mixture-of-Experts",
  "abstract": "FastSAC-style methods significantly reduce humanoid motion training time but often suffer from notable performance degradation compared with PPO in whole-body tracking tasks. We target this speed-performance gap by introducing LooperMuscle, a composed expert policy learning framework that restores tracking quality while preserving high training efficiency. LooperMuscle combines a semantically structured mixture-of-experts actor, an expert-aware distributional critic, and contribution-routed replay with deferred curriculum scheduling. These three components form a closed training loop in which expert contributions guide data routing, routed data shape value learning, and value gradients in turn refine expert specialization. Empirically, our approach substantially outperforms vanilla FastSAC in motion tracking accuracy while requiring far less wall-clock time than PPO: where FastSAC trains in about 15 minutes but underperforms, and PPO achieves stronger results but requires about 6 hours, LooperMuscle recovers a substantial fraction of the remaining gap to PPO in roughly 45 minutes of simulation training, delivering practical efficiency for rapid policy iteration. The code will be released to benefit the research community at https://loopermuscle.github.io/.",
  "published": "2026-08-01",
  "updated": "2026-08-01",
  "year": "2026",
  "authors": [
   "Boyi Liu",
   "Qijin Li",
   "Tianqi Yu",
   "Qinrui Yan",
   "Xingxing Zuo"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "LooperMuscle is introduced, a composed expert policy learning framework that restores tracking quality while preserving high training efficiency, and substantially outperforms vanilla FastSAC in motion tracking accuracy while requiring far less wall-clock time than PPO.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Boyi Liu",
    "id": "2453508540",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Qijin Li",
    "id": "2455424409",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Tianqi Yu",
    "id": "2451110441",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Qinrui Yan",
    "id": "2455443203",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Xingxing Zuo",
    "id": "2272795526",
    "h_index": 3,
    "papers": 8
   }
  ],
  "comment": "",
  "topics": [
   "humanoids"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.00820v1",
  "pdf_url": "https://arxiv.org/pdf/2608.00820v1",
  "html_url": "https://arxiv.org/html/2608.00820v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.00792",
  "slug": "stochsipp-safe-interval-path-planning-in-stochastic-dynamic-environmen",
  "title": "StochSIPP: Safe Interval Path Planning in Stochastic Dynamic Environments",
  "abstract": "Safe navigation under uncertain time-dependent blockage requires anticipating observations before committing to motion. We present StochSIPP, an exact contingent planner for temporal roadmaps with uncertain edge and vertex statuses revealed locally during execution. StochSIPP uses SIPP to generate certified-safe macro-actions that terminate at the next observation or the goal, and bounded AND/OR search over a cached action--observation graph to select actions for every reachable observation outcome. Optimistic and robust SIPP relaxations provide admissible lower and upper bounds for bounded AND/OR search. When every interval declared deterministically safe is truly safe, sensing is exact, and execution follows the planned timing, the resulting policy is provably collision-free. With correct independent probabilities and complete action and outcome generation, it minimizes expected arrival time within the roadmap and horizon. Experiments on controlled roadmap instances show that StochSIPP preserves the observed success of safe fixed-path baselines while reducing arrival time, and solves gated scenarios in which conservative fixed-path planners return no plan. A scalability study further reveals rapid growth as the number of simultaneously observed uncertain statuses increases.",
  "published": "2026-08-01",
  "updated": "2026-08-01",
  "year": "2026",
  "authors": [
   "Ajith Kemisetti",
   "Shahaf S. Shperberg",
   "Yoonchang Sung"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "StochSIPP is presented, an exact contingent planner for temporal roadmaps with uncertain edge and vertex statuses revealed locally during execution that preserves the observed success of safe fixed-path baselines while reducing arrival time, and solves gated scenarios in which conservative fixed-path planners return no plan.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ajith Kemisetti",
    "id": "2455405092",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Shahaf S. Shperberg",
    "id": "10732333",
    "h_index": 8,
    "papers": 69
   },
   {
    "name": "Yoonchang Sung",
    "id": "40276020",
    "h_index": 11,
    "papers": 29
   }
  ],
  "comment": "16 pages, 4 figures",
  "topics": [
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.00792v1",
  "pdf_url": "https://arxiv.org/pdf/2608.00792v1",
  "html_url": "https://arxiv.org/html/2608.00792v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.00775",
  "slug": "orcestra-vlm-driven-visual-robot-programming-in-mixed-reality",
  "title": "ORCESTRA: VLM-driven Visual Robot programming in Mixed Reality",
  "abstract": "ORCESTRA is a mixed-reality system for programming robot digital twins through no-code waypoint teaching and language-guided control. In a passthrough mixed-reality workspace, users place robot twins on real surfaces, teach trajectories, save robot-relative episodes, or issue spoken/typed commands that a vision-language model converts into structured digital-twin plans. Both interaction modes share a backend for metric grounding, embodiment-aware validation, preview, confirmation, and digital-twin execution. The system supports heterogeneous robot embodiments, including fixed-base manipulators, a mobile base, and a humanoid robot, demonstrating MR validation as a safety layer for language-guided robot programming before physical deployment.",
  "published": "2026-08-01",
  "updated": "2026-08-01",
  "year": "2026",
  "authors": [
   "Ivan Snegirev",
   "Elizaveta Semenyakina",
   "Mikhail Konenkov",
   "Artem Lykov",
   "Miguel Altamirano Cabrera",
   "Dzmitry Tsetserukou"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.CV",
   "cs.HC"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The ORCESTRA system supports heterogeneous robot embodiments, including fixed-base manipulators, a mobile base, and a humanoid robot, demonstrating MR validation as a safety layer for language-guided robot programming before physical deployment.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ivan Snegirev",
    "id": "2406042504",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "E.M. Semenyakina",
    "id": "2163874391",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Mikhail Konenkov",
    "id": "2226258666",
    "h_index": 8,
    "papers": 14
   },
   {
    "name": "Artem Lykov",
    "id": "2189477580",
    "h_index": 13,
    "papers": 33
   },
   {
    "name": "Miguel Altamirano Cabrera",
    "id": "144548970",
    "h_index": 10,
    "papers": 62
   },
   {
    "name": "D. Tsetserukou",
    "id": "48470616",
    "h_index": 27,
    "papers": 298
   }
  ],
  "comment": "4 page, 3 figures, 1 table, ISMAR 2026",
  "topics": [
   "humanoids",
   "video-generation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.00775v1",
  "pdf_url": "https://arxiv.org/pdf/2608.00775v1",
  "html_url": "https://arxiv.org/html/2608.00775v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.00730",
  "slug": "push-wiper-toward-general-purpose-robotic-cleaning-across-varied-stain",
  "title": "Push-Wiper: Toward General-Purpose Robotic Cleaning across Varied Stains and Surfaces with Segmented Pushing Trajectories",
  "abstract": "Viscous stains, characterized by high viscosity and complex rheological properties, remain a major challenge for robotic surface cleaning. Conventional wiping often spreads the stain, while scrubbing provides stronger friction but risks damaging the surface. In this paper, we propose Push-Wiper, a framework that reformulates viscous stain cleaning as an aggregation problem. Push-Wiper employs a sponge to progressively gather stains through segmented pushing trajectories, followed by a post-processing phase that detaches the aggregated material and enables sponge self-cleaning. We adopt a stepwise strategy for stain gathering and leverage Diffusion Policy to generate adaptive pushing action sequences. These sequences are executed through our Arbitrary Surface Pose Interpolator (ASPI) and a hybrid force-position controller, allowing the method to generalize to stains with diverse spatial distributions. Push-Wiper achieves a cleaning score (CS), defined as the percentage of stain area removed, up to 130% higher than baseline methods. Without additional training, Push-Wiper also transfers in a zero-shot manner to solid residues, liquid spills, unseen viscous stains, and curved surfaces with varying geometries. Our experiments demonstrate the cleaning effectiveness of Push-Wiper and its strong generalization ability. The project website is available at https://push-wiper.github.io/.",
  "published": "2026-08-01",
  "updated": "2026-08-01",
  "year": "2026",
  "authors": [
   "Renhao Lu",
   "Mingxin Wang",
   "Chenyang Cao",
   "Yang Yang",
   "Guoping Pan",
   "Kangkang Dong",
   "Yi Cheng",
   "Houde Liu"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Renhao Lu",
    "id": "2290024323",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Mingxin Wang",
    "id": "2115447018",
    "h_index": 1,
    "papers": 13
   },
   {
    "name": "Chenyang Cao",
    "id": "2290442582",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Yang Yang",
    "id": "2363933376",
    "h_index": 4,
    "papers": 16
   },
   {
    "name": "Guoping Pan",
    "id": "2293396602",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Kangkang Dong",
    "id": "51053421",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Yi Cheng",
    "id": "2243263128",
    "h_index": 4,
    "papers": 17
   },
   {
    "name": "Houde Liu",
    "id": "2293440971",
    "h_index": 3,
    "papers": 26
   }
  ],
  "comment": "8 pages, 8 figures. Accepted to the 2026 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS 2026)",
  "topics": [
   "imitation-diffusion",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.00730v1",
  "pdf_url": "https://arxiv.org/pdf/2608.00730v1",
  "html_url": "https://arxiv.org/html/2608.00730v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.00652",
  "slug": "assistant-placement-aria-a-benchmark-for-egocentric-placement-assistan",
  "title": "Assistant Placement Aria: A Benchmark for Egocentric Placement Assistance",
  "abstract": "Human assistance in robotics spans around several tasks such as navigation, object manipulation, and placement, where a key challenge is selecting target destinations that align with human intentions or preferences. We focus on this challenge in the context of Virtual Placement (VP), the task of identifying all plausible target locations given scene context and human-centric constraints. This differs from traditional placement tasks that typically focus on a single, predefined target location. The VP problem is complex, as it requires both global and local reasoning about the scene's geometry, semantics, and plausibility. To address this gap, we introduce {\\bf Assistant Placement Aria}, the first benchmark to explore diverse aspects of VP, including global, local, and human-centric constraints. It contains both synthetic and real indoor scenes annotated for three tasks: (i)~2D Panel Placement, (ii)~Sitting Suggestion, and (iii)~TV Placement. Each scene includes 2D images, a 3D point cloud, and a textual description of the objects within the scene. By contributing this benchmark, we aim to encourage further research in this underexplored and challenging field that is critically dependent on relevant data. We also evaluate several foundation models for object detection and segmentation on our benchmark.",
  "published": "2026-08-01",
  "updated": "2026-08-01",
  "year": "2026",
  "authors": [
   "Amir Belder",
   "Gon\u00e7alo Dias Pais",
   "Refael Vivanti",
   "Omri Carmi",
   "Daniel DeTone",
   "Oren Shrout",
   "Ido Gattegno",
   "Ayellet Tal"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "This work introduces {\\bf Assistant Placement Aria}, the first benchmark to explore diverse aspects of VP, including global, local, and human-centric constraints, and evaluates several foundation models for object detection and segmentation on this benchmark.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Amir Belder",
    "id": "2154574281",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Goncalo Dias Pais",
    "id": "144498495",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "R. Vivanti",
    "id": "2419573",
    "h_index": 7,
    "papers": 12
   },
   {
    "name": "Omri Carmi",
    "id": "2455190223",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Daniel DeTone",
    "id": "3422291",
    "h_index": 13,
    "papers": 24
   },
   {
    "name": "Oren Shrout",
    "id": "2181874457",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Ido Gattegno",
    "id": "2455189979",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "A. Tal",
    "id": "3226509",
    "h_index": 47,
    "papers": 157
   }
  ],
  "comment": "",
  "topics": [
   "egocentric-data",
   "spatial-3d",
   "navigation",
   "foundation-pretraining",
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.00652v1",
  "pdf_url": "https://arxiv.org/pdf/2608.00652v1",
  "html_url": "https://arxiv.org/html/2608.00652v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2608.00625",
  "slug": "learning-based-motion-planning-for-dynamic-environments-from-foundatio",
  "title": "Learning-Based Motion Planning for Dynamic Environments: From Foundational Algorithms to Emerging Paradigms",
  "abstract": "Motion planning in dynamic environments is a fundamental problem in robotics, aiming to generate safe and efficient paths, trajectories, or control actions in the presence of moving obstacles, uncertain predictions, and multi-agent interactions. It has broad applications in autonomous driving, service robotics, warehouse logistics, human-robot collaboration, crowd navigation, and multi-robot systems. This survey reviews representative works published primarily between 2015 and 2025, with a particular focus on how recent learning-based advances extend, complement, or interact with classical planning foundations. We first revisit classical planning methods as algorithmic foundations and reference frameworks for learning-based extensions. We then propose a role-of-learning taxonomy that categorizes existing methods according to how learning participates in the planning pipeline, including direct policy learning, learning-augmented classical planning, hybrid planning, and training enhancement methods. For each category, we summarize the main problem settings, representative algorithms, key ideas, integration mechanisms, strengths, and limitations. We further analyze how observation representations, prediction uncertainty, interaction modeling, planner integration, safety constraints, and training strategies shape learning-based motion planning in dynamic environments. Finally, we discuss open challenges and future directions, including sim-to-real gap, safe and certifiable planning, dense crowd navigation, perception-planning coupling, and embodied AI.",
  "published": "2026-08-01",
  "updated": "2026-08-01",
  "year": "2026",
  "authors": [
   "Zongyuan Shen",
   "Shalabh Gupta",
   "Shancheng Zhao",
   "Dehua Zhou",
   "Gao Wang",
   "Rui Cheng",
   "Yaming Ou",
   "Zhongqiang Ren",
   "Yikui Zhai",
   "C. L. Philip Chen"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A role-of-learning taxonomy is proposed that categorizes existing methods according to how learning participates in the planning pipeline, including direct policy learning, learning-augmented classical planning, hybrid planning, and training enhancement methods.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zongyuan Shen",
    "id": "31355028",
    "h_index": 7,
    "papers": 23
   },
   {
    "name": "Shalabh Gupta",
    "id": "2257275993",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Shancheng Zhao",
    "id": "2335425525",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "De-qiang Zhou",
    "id": "2287751589",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Gaotian Wang",
    "id": "2290064206",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Ruizhe Cheng",
    "id": "2444025723",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yaming Ou",
    "id": "2135222291",
    "h_index": 9,
    "papers": 26
   },
   {
    "name": "Zhongqiang Ren",
    "id": "2310235114",
    "h_index": 2,
    "papers": 11
   },
   {
    "name": "Yikui Zhai",
    "id": "2205040886",
    "h_index": 9,
    "papers": 78
   },
   {
    "name": "C. L. P. Chen",
    "id": "2152857279",
    "h_index": 11,
    "papers": 23
   }
  ],
  "comment": "",
  "topics": [
   "sim2real",
   "navigation",
   "hri",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.00625v1",
  "pdf_url": "https://arxiv.org/pdf/2608.00625v1",
  "html_url": "https://arxiv.org/html/2608.00625v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.00613",
  "slug": "from-failures-to-supervision-dynamicenvplan-for-robust-long-horizon-em",
  "title": "From Failures to Supervision: DynamicEnvPlan for Robust Long-Horizon Embodied Planning",
  "abstract": "Physical-world interaction is inherently dynamic, as environments can evolve during execution, requiring agents to adapt their plans under non-stationary conditions. We study this challenge through long-horizon embodied planning under environment deviations and execution uncertainty. Existing embodied-task benchmarks can expose such failures, but these failures are usually treated as evaluation outcomes instead of learnable signals for training agents to recover. In this work, we introduce DynamicEnvPlan, a closed-loop framework for high-level planning in dynamic environments. It extends embodied task execution with humanoid agents, high-level primitive skills, structured semantic memory, and controllable perturbations. Our data synthesis design consists of planning, perturbation, and guarded correction modules that turn dynamic execution states into recovery-oriented traces. The resulting traces are used for staged supervised fine-tuning, enabling the planner to learn from both nominal execution and perturbed recovery trajectories. Using 104 task-scene combinations spanning i.i.d., compositional generalization, and out-of-distribution settings for fine-tuning and evaluation, DynamicEnvPlan boosts success rate from 33.3% for the base planner to 76.2%, while improving across all seven evaluation metrics critical to physical-world interaction, including safety and affordance compliance.",
  "published": "2026-08-01",
  "updated": "2026-08-17",
  "year": "2026",
  "authors": [
   "Hao Yuan",
   "Yuxin Wang",
   "Lei Ji",
   "Zhiwei Yu"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work introduces DynamicEnvPlan, a closed-loop framework for high-level planning in dynamic environments that extends embodied task execution with humanoid agents, high-level primitive skills, structured semantic memory, and controllable perturbations.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hao Yuan",
    "id": "2455208172",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yuxin Wang",
    "id": "2452192927",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Lei Ji",
    "id": "2364932488",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Zhiwei Yu",
    "id": "2336248434",
    "h_index": 3,
    "papers": 13
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.00613v2",
  "pdf_url": "https://arxiv.org/pdf/2608.00613v2",
  "html_url": "https://arxiv.org/html/2608.00613v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.00554",
  "slug": "dexmani-human-derived-manipulability-guidance-for-dexterous-rotation",
  "title": "DexMani: Human-Derived Manipulability Guidance for Dexterous Rotation",
  "abstract": "Dexterous object rotation is a sequential contact problem: each support, release, and re-contact decision must both produce the desired object motion, and prepare the hand configuration for continued rotation. Existing reinforcement learning methods discover such movement patterns through trial and error on specific robotic hand embodiments, without explicitly accounting for how each contact transition affects the hand's ability to sustain object rotation in subsequent steps. We introduce DexMani, a framework that transfers human demonstrations as contact-conditioned manipulability evolution. This prior captures how successful human contact transitions reshape the object-rotation directions available to the hand. DexMani then learns this manipulability evolution and uses it to guide downstream reinforcement learning, enabling rotation skills to be acquired across robot embodiments with distinct kinematics and active-contact configurations. Across the Shadow Hand, Allegro Hand, and XHand, DexMani achieves the highest success rates in every evaluated setting for both seen and unseen objects. DexMani reaches an average success rate of 57.5% on LEAP Hand, outperforming other baselines and producing smoother rotatory motions. Project site: https://dexmani.github.io",
  "published": "2026-08-01",
  "updated": "2026-08-01",
  "year": "2026",
  "authors": [
   "Xiaoyang Chen",
   "Shengcheng Luo",
   "Haoran Guo",
   "Jiaming Jiang",
   "Wanlin Li",
   "Ziyuan Jiao",
   "Chenxi Xiao"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work introduces DexMani, a framework that transfers human demonstrations as contact-conditioned manipulability evolution and uses it to guide downstream reinforcement learning, enabling rotation skills to be acquired across robot embodiments with distinct kinematics and active-contact configurations.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xiaoyang Chen",
    "id": "2455265909",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Shengcheng Luo",
    "id": "2309215693",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Haoran Guo",
    "id": "2324074648",
    "h_index": 1,
    "papers": 11
   },
   {
    "name": "Jiaming Jiang",
    "id": "2167229335",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Wanlin Li",
    "id": "2357100311",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Ziyuan Jiao",
    "id": "2356946141",
    "h_index": 3,
    "papers": 13
   },
   {
    "name": "Chenxi Xiao",
    "id": "2356920855",
    "h_index": 3,
    "papers": 14
   }
  ],
  "comment": "16 pages, 17 figures",
  "topics": [
   "dexterous-manipulation",
   "egocentric-data",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.00554v1",
  "pdf_url": "https://arxiv.org/pdf/2608.00554v1",
  "html_url": "https://arxiv.org/html/2608.00554v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.00547",
  "slug": "disentangling-visuo-tactile-foresight-oracle-guided-interface-discover",
  "title": "Disentangling Visuo-Tactile Foresight: Oracle-Guided Interface Discovery for World Action Models",
  "abstract": "Contact-rich manipulation remains challenging because successful control depends on physical interaction cues that are often weakly observable from vision alone. Recent tactile world action models jointly model future visual observations and tactile signals to guide action generation, but how such futures should be structured for effective use by the action expert remains underexplored. Directly studying this question with learned world action models is difficult because end-to-end behavior entangles physically invalid visual futures, unreliable predictions, inaccurate or cross-modally inconsistent tactile forecasts, and an unreadable future-to-action interface. To make this interface independently studyable, we introduce Oracle Visuo-Tactile Foresight (OVTF), a controlled framework that supplies paired RGB and tactile futures from successful trajectories verified in simulation. By fixing the future provider, OVTF isolates the interface and asks a cleaner question: if the future is successful and physically executable, what representation allows the action expert to absorb its benefit? Within OVTF, we propose Asymmetric Phase-Local Future Memory (AFM), in which visual memory reads future vision, each tactile memory jointly attends to its own tactile stream and phase-aligned future vision, and cross-tactile access is blocked. We compare AFM with Modality-Isolated Future Memory (IFM), which removes visual-to-tactile access and processes each future modality independently. Across seven tasks on the UniVTAC simulation benchmark, AFM achieves 32.0% average success, compared with 23.7% for IFM and 14.9% for UniVTAC-ACT. This controlled comparison shows that selective phase-aligned visual-tactile routing provides a more actionable future-to-action bridge than complete modality isolation.",
  "published": "2026-08-01",
  "updated": "2026-08-01",
  "year": "2026",
  "authors": [
   "Zihang Yao",
   "Chaoyue Ding",
   "Yingying Yu"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes Asymmetric Phase-Local Future Memory (AFM), in which visual memory reads future vision, each tactile memory jointly attends to its own tactile stream and phase-aligned future vision, and cross-tactile access is blocked, and compares it with Modality-Isolated Future Memory (IFM), which removes visual-to-tactile access and processes each future modality independently.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zihang Yao",
    "id": "2455207084",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Chaoyue Ding",
    "id": "2258948671",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Yingying Yu Brigham Young University",
    "id": "2455187845",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Beijing Academy of Science",
    "id": "2455187770",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Technology",
    "id": "2453999600",
    "h_index": 0,
    "papers": 4
   }
  ],
  "comment": "6 pages, 3 figures, 2 tables",
  "topics": [
   "tactile",
   "sim2real"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.00547v1",
  "pdf_url": "https://arxiv.org/pdf/2608.00547v1",
  "html_url": "https://arxiv.org/html/2608.00547v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.00500",
  "slug": "a-change-of-frame-makes-balance-observable-distillation-free-humanoid",
  "title": "A Change of Frame Makes Balance Observable: Distillation-Free Humanoid Single-Leg Stance",
  "abstract": "Unified humanoid policies handle agile whole-body motion, yet stumble on a simple demand: staying balanced on one leg. On our single-leg-balance benchmark, eight released state-of-the-art general policies hold a clean single-leg stance on 0 of 90 test motions; they stay up only by stepping or hopping, recovering from imbalance rather than preventing it. Prevention needs the capture point (xCoM), the center of mass (CoM) extrapolated by its velocity, which has never driven a learned hardware policy because it requires a base linear velocity that no on-board sensor measures directly. A change of frame makes it observable: expressed relative to the support foot, that velocity cancels exactly, leaving an observation reconstructible from encoders and IMU alone. We put this first deployable dynamic-CoM observation directly into the actor that runs on hardware, and pair it with a reward library translated term by term from human postural control, under one principle: prevention over repair. Trained via asymmetric FastSAC without distillation, the resulting policy, DDC (Deployable Dynamic-CoM), holds clean single-leg balance on 89 of 90 held-out motions across nine stratified pose classes and transfers to a real Unitree G1; in ablation, the dynamic-CoM observation is the single largest driver: removing it alone costs 43 points of clean single-leg balance. We release the full stack with the first method-agnostic, reproducible sim2sim benchmark for humanoid single-leg balance, scoring each policy in a simulator distinct from the one it was trained in, to help turn balance from a per-task trick into a capability the field can measure and build in.",
  "published": "2026-08-01",
  "updated": "2026-08-14",
  "year": "2026",
  "authors": [
   "Yikai Zhou",
   "Xingyun Wang",
   "Jieming Cui",
   "Bozhou Chen",
   "Yikai Fan",
   "Yixin Zhu",
   "Wenxin Li"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work releases the full stack with the first method-agnostic, reproducible sim2sim benchmark for humanoid single-leg balance, scoring each policy in a simulator distinct from the one it was trained in, to help turn balance from a per-task trick into a capability the field can measure and build in.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yikai Zhou",
    "id": "2455461692",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Xingyun Wang",
    "id": "2453028782",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Jieming Cui",
    "id": "2217941702",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Bozhou Chen",
    "id": "2375096313",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Yikai Fan",
    "id": "2455426733",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yixin Zhu",
    "id": "2261513442",
    "h_index": 11,
    "papers": 20
   },
   {
    "name": "Wenxin Li",
    "id": "2455420008",
    "h_index": 0,
    "papers": 2
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "sim2real"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2608.00500v2",
  "pdf_url": "https://arxiv.org/pdf/2608.00500v2",
  "html_url": "https://arxiv.org/html/2608.00500v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2608.00484",
  "slug": "from-digital-to-physical-reservoir-computing-co-optimizing-soft-roboti",
  "title": "From Digital to Physical Reservoir Computing: Co-Optimizing Soft Robotic Reservoirs via Dynamics Matching",
  "abstract": "Soft robotic substrates are promising for Physical Reservoir Computing (PRC) because their compliant nonlinear dynamics can provide temporal memory, high-dimensional state transformations, and efficient inference. However, physical reservoirs are often adopted as-is rather than pretrained or co-optimized, potentially limiting soft robotic PRC performance relative to digital reservoirs. We investigate whether a physical reservoir can instead be pretrained against high-performing digital reference dynamics. Our formulation jointly optimizes physical parameters, a diffeomorphic physical-reference state map, and feedforward-feedback control using a differentiable physical model and an acceleration-level equation-error objective that avoids temporal integration. As a proof of concept, we instantiate the formulation with simulated soft robots, a Random Oscillators Network (RON) reference, and parallel multi-start gradient descent. We evaluate the optimized reservoirs on classification (sMNIST and ADIAC) and forecasting (Mackey-Glass and Lorenz96) tasks across four reservoir dimensions. Compared with unoptimized soft robot reservoirs, the optimized reservoirs achieve a mean relative improvement of 33.7% across all tasks and datasets, while remaining close to the digital reference. These results demonstrate the feasibility of dynamics-level co-optimization for the simulated soft robotic reservoirs considered here.",
  "published": "2026-08-01",
  "updated": "2026-08-01",
  "year": "2026",
  "authors": [
   "Nicola Visentin",
   "Maximilian St\u00f6lzle",
   "Mariano Ram\u00edrez Montero",
   "Francesco Braghin",
   "Daniela Rus",
   "Cosimo Della Santina"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The feasibility of dynamics-level co-optimization for the simulated soft robotic reservoirs considered here is demonstrated, with results demonstrating the feasibility of dynamics-level co-optimization for the simulated soft robotic reservoirs considered here.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Nicola Visentin",
    "id": "2455187125",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Maximilian St\u00f6lzle",
    "id": "2127775868",
    "h_index": 8,
    "papers": 18
   },
   {
    "name": "Mariano Ram\u00edrez Montero",
    "id": "2186054078",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Francesco Braghin",
    "id": "2305617504",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Daniela Rus",
    "id": "2261287511",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "C. D. Santina",
    "id": "35178897",
    "h_index": 25,
    "papers": 177
   }
  ],
  "comment": "",
  "topics": [
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.00484v1",
  "pdf_url": "https://arxiv.org/pdf/2608.00484v1",
  "html_url": "https://arxiv.org/html/2608.00484v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.07546",
  "slug": "generalizing-deep-reinforcement-learning-across-cable-driven-parallel",
  "title": "Generalizing deep reinforcement learning across cable-driven parallel robot configurations with actuator-level policies",
  "abstract": "Cable-driven parallel robots (CDPRs) present diverse configurations and complex control challenges, which can be addressed by deep reinforcement learning (DRL) by learning their nonlinear dynamics. However, DRL methods often require extensive training time, and the resulting policies do not generalize well to different robot configurations or varying numbers of actuators. In this article, we introduce a novel DRL approach for controlling CDPRs that does not depend on the specific robot configuration. Our method trains an actuator-level policy that controls each motor to achieve its target cable length, in contrast to conventional DRL approaches that learn to control the entire robot to reach a desired end-effector position. To the best of our knowledge, this is the first work to apply DRL to control CDPRs using an actuator-level policy. This approach offers two main advantages: (i) a single shared policy can be applied to any CDPR configuration, regardless of actuator count, and (ii) reliance on inverse kinematics, avoiding the more challenging forward kinematics problem. Training is performed in simulation, and the learned policy is successfully transferred to a real CDPR. Experimental results show that the actuator-level policy (ALP) surpasses traditional reinforcement learning methods in both robustness and precision. We further control a real 8-motor CDPR with 3D motion using a policy trained on a simulated 4-motor planar CDPR operating in 2D. This illustrates that the proposed method is applicable to any CDPR configuration, independent of actuator number or placement.",
  "published": "2026-07-31",
  "updated": "2026-07-31",
  "year": "2026",
  "authors": [
   "Abir Bouaouda",
   "Mohamed Boutayeb",
   "Fran\u00e7ois Charpillet",
   "Dominique Martinez",
   "R\u00e9mi Pannequin"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This article introduces a novel DRL approach for controlling CDPRs that does not depend on the specific robot configuration, and shows that the actuator-level policy (ALP) surpasses traditional reinforcement learning methods in both robustness and precision.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Abir Bouaouda",
    "id": "2267984540",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Mohamed Boutayeb",
    "id": "2257071856",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "F. Charpillet",
    "id": "1731714",
    "h_index": 31,
    "papers": 260
   },
   {
    "name": "Dominique Martinez",
    "id": "2257126187",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "R\u00e9mi Pannequin",
    "id": "2267981849",
    "h_index": 1,
    "papers": 3
   }
  ],
  "comment": "",
  "topics": [
   "rl-control",
   "hardware-codesign",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.07546v1",
  "pdf_url": "https://arxiv.org/pdf/2608.07546v1",
  "html_url": "https://arxiv.org/html/2608.07546v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.00337",
  "slug": "action-chunk-scheduling-for-batched-robot-policy-serving",
  "title": "Action Chunk Scheduling for Batched Robot Policy Serving",
  "abstract": "Deploying robot foundation models at scale is the next step towards realizing the potential of general-purpose robots. However, Vision-Language-Action (VLA) and other foundation models are computationally demanding, and on-device compute is constrained by power and space. In this paper, we introduce the problem of serving a robot policy to multiple robots from a remote GPU and formulate it as a scheduling problem. We build Armory, a serving system validated on fleets of both simulated and real robots. Our experiments show that naive scheduling heuristics perform well when all robots are the same, but fall short when robots consume action chunks at different rates, uncovering a mismatch between conventional batching methods and the closed-loop requirements of robot policy execution. To address this, we propose a scheduling algorithm that accounts for this heterogeneity and improves overall system throughput by up to $18\\%$ in real-world experiments. Additional details are available at https://gatech-rl2.github.io/actionchunkscheduling.",
  "published": "2026-07-31",
  "updated": "2026-07-31",
  "year": "2026",
  "authors": [
   "Rohan Bansal",
   "David He",
   "Nadun Ranawaka Arachchige",
   "Zhenyang Chen",
   "Soobum Kim",
   "Kexin Rong",
   "Danfei Xu"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper introduces the problem of serving a robot policy to multiple robots from a remote GPU and forms it as a scheduling problem, and proposes a scheduling algorithm that accounts for this heterogeneity and improves overall system throughput by up to $18\\% in real-world experiments.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Rohan Bansal",
    "id": "2054868734",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "David He",
    "id": "2374288725",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "N. R. Arachchige",
    "id": "2141127741",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Zhenyang Chen",
    "id": "2367302562",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Soobum Kim",
    "id": "2300899978",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Kexin Rong",
    "id": "7804757",
    "h_index": 10,
    "papers": 49
   },
   {
    "name": "Danfei Xu",
    "id": "2322756258",
    "h_index": 5,
    "papers": 12
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.00337v1",
  "pdf_url": "https://arxiv.org/pdf/2608.00337v1",
  "html_url": "https://arxiv.org/html/2608.00337v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.00284",
  "slug": "hybrid-attention-estimation-pipeline-for-adaptive-hri-using-an-express",
  "title": "Hybrid Attention Estimation Pipeline for Adaptive HRI Using an Expressive Robotic Head",
  "abstract": "This paper presents an applied case study on hybrid visual attention estimation for human-robot interaction using an expressive robotic head based on the InMoov ecosystem. The proposed pipeline combines a fast geometric perception layer with an independent semantic perception layer based on a vision-language model. The geometric layer provides high-frequency face and head-pose information for temporal regulation, while the semantic layer receives only raw egocentric camera frames and produces contextual attention labels related to attention toward the robot, phone use, or attention elsewhere. These signals are integrated through a finite state machine that regulates adaptive interaction behavior, including activation, waiting, interaction resumption, and return to rest. The system was evaluated with 10 participants across 40 trials covering baseline and adaptive interaction conditions. Results show reliable interaction start across all trials, consistent pause behavior in the adaptive distraction condition, and non-redundant semantic information between the geometric and semantic outputs.",
  "published": "2026-07-31",
  "updated": "2026-07-31",
  "year": "2026",
  "authors": [
   "Pablo Moraes",
   "Monica Rodriguez",
   "Christopher Peters",
   "Hiago Sodre",
   "Tobias Doernbach",
   "Bruna Guterres",
   "Ricardo Grando"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Results show reliable interaction start across all trials, consistent pause behavior in the adaptive distraction condition, and non-redundant semantic information between the geometric and semantic outputs.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "P. Moraes",
    "id": "2306249192",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "M\u00f3nica Rodr\u00edguez",
    "id": "2355360346",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Christopher Peters",
    "id": "2306112049",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Hiago Sodre",
    "id": "2306251592",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Tobias Doernbach",
    "id": "2292201091",
    "h_index": 0,
    "papers": 5
   },
   {
    "name": "B. Guterres",
    "id": "2306258966",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Ricardo B. Grando",
    "id": "2306259863",
    "h_index": 3,
    "papers": 15
   }
  ],
  "comment": "",
  "topics": [
   "egocentric-data",
   "hri",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.00284v1",
  "pdf_url": "https://arxiv.org/pdf/2608.00284v1",
  "html_url": "https://arxiv.org/html/2608.00284v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.00208",
  "slug": "developing-combined-manipulation-and-locomotion-skills-with-interactio",
  "title": "Developing Combined Manipulation and Locomotion Skills with Interaction Representation and Skill Composition",
  "abstract": "This paper addresses how to enable a humanoid robot to learn motion policies based on developmental principles and combine policies to create more sophisticated and useful behaviors. Specifically, we present an approach to (1) learning a whole-body reaching and grasping policy and (2) combining it and a standing-up and walking policy to compose a more complex policy of manipulation and locomotion: grasping, standing up, and walking. In (1), our method draws inspiration from harmonic analysis and adopts cubic harmonics as weights to represent the hand-object spatial relationship via spatial convolution. Utilizing an intra-episode finger joint decoupling curriculum based on developmental principles, a robot can autonomously learn a generalizable grasping policy without relying on external datasets or pretrained models. In (2), our method combines the grasping policy with a separately learned getting-up policy by providing both policies with their respective observation vectors and using hand-object interaction scores to determine when each policy should control which robot joints. Our results show a 93% zero-shot success rate for grasping unseen objects and a 96-100% success rate for standing up while holding the object. Our work also demonstrates that combining different policies is only effective if each policy learning happens on the same whole humanoid body even if a policy (such as for locomotion) does not seem to need all the body parts (such as fingers).",
  "published": "2026-07-31",
  "updated": "2026-07-31",
  "year": "2026",
  "authors": [
   "Fanxing Meng",
   "Jing Xiao"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "Humanoids 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper presents an approach to learning a whole-body reaching and grasping policy and combining it and a standing-up and walking policy to compose a more complex policy of manipulation and locomotion: grasping, standing up, and walking.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Fanxing Meng",
    "id": "46413175",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Jing Xiao",
    "id": "2210810311",
    "h_index": 1,
    "papers": 5
   }
  ],
  "comment": "8 pages, 5 figures. Submitted to Humanoids 2026. Video available at https://youtu.be/x-7x89fSJWY",
  "topics": [
   "dexterous-manipulation",
   "humanoids"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.00208v1",
  "pdf_url": "https://arxiv.org/pdf/2608.00208v1",
  "html_url": "https://arxiv.org/html/2608.00208v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2608.00206",
  "slug": "motion-planning-for-mobile-manipulators-navigating-doorways-via-model",
  "title": "Motion Planning for Mobile Manipulators Navigating Doorways via Model Predictive Control",
  "abstract": "Navigating doorways is a fundamental capability for mobile manipulators operating in human environments, requiring coordinated motion between the mobile base and manipulator arm. This paper presents a motion planning framework that generates dynamically feasible and collision-free trajectories for autonomously opening and traversing both push and pull doors. The proposed method formulates the robot and door as a coupled dynamical system within a nonlinear Model Predictive Control (MPC) optimization framework. Manipulation feasibility is enforced through a penalty-based constraint, avoiding explicit arm kinematic modeling in the planner. Simulations and a hardware experiment demonstrate that the approach successfully plans feasible trajectories for door traversal.",
  "published": "2026-07-31",
  "updated": "2026-07-31",
  "year": "2026",
  "authors": [
   "Kasra Sinaei",
   "Kasun Weerakoon",
   "Christopher Bradley",
   "Seyed Abolfazl Fakoorian",
   "Donald Ebeigbe"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The proposed method formulates the robot and door as a coupled dynamical system within a nonlinear Model Predictive Control (MPC) optimization framework, and plans feasible trajectories for door traversal.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "K. Sinaei",
    "id": "2149075788",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Kasun Weerakoon",
    "id": "123689410",
    "h_index": 12,
    "papers": 38
   },
   {
    "name": "Christopher Bradley",
    "id": "10289493",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "S. Fakoorian",
    "id": "2317839201",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Donald Ebeigbe",
    "id": "3420796",
    "h_index": 6,
    "papers": 23
   }
  ],
  "comment": "",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.00206v1",
  "pdf_url": "https://arxiv.org/pdf/2608.00206v1",
  "html_url": "https://arxiv.org/html/2608.00206v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2608.00113",
  "slug": "track-guided-hierarchical-reinforcement-learning-for-autonomous-vehicl",
  "title": "Track-Guided Hierarchical Reinforcement Learning for Autonomous Vehicle Drifting with Minimum-Lap-Time Planning",
  "abstract": "In Formula 1, drivers optimize racing lines within tire grip limits to minimize lap times; however, in rally racing, drivers intentionally break traction to drift on loose surfaces. This maneuver rapidly aligns the vehicle for corner exits, ultimately reducing lap time. Autonomously executing such maneuvers formulates a complex dual-objective control problem: stabilizing highly nonlinear drift dynamics while strictly minimizing lap time. Addressing this challenge motivates the development of advanced Minimum-Lap-Time (MLT) drift control architectures. This paper proposes a planning-control framework specifically designed for MLT drifting scenario. First, we formulate an optimal control problem to generate a MLT drift planning trajectory, which is used as prior data to train a deep reinforcement learning drift controller. Given that drifting involves extremely large sideslip angles and is therefore challenging to learn directly, a Track-guided Reinforcement Learning (TgRL) drift control method is proposed to enable progressive training in a step-by-step manner, from drift control policy, to drift corner policy, and finally to a comprehensive drift race policy. The reward function incorporates both an instant reward term and an end reward term derived from the Minimum-Lap-Time objective. Simulation results demonstrate that the proposed framework enables the agent to learn a drift racing policy that not only ensures vehicle motion control performance but also effectively reduces lap time.",
  "published": "2026-07-31",
  "updated": "2026-07-31",
  "year": "2026",
  "authors": [
   "Sheng Zhao",
   "Bolin Zhao",
   "Xiaodong Wu",
   "Chen Lv"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A planning-control framework specifically designed for MLT drifting scenario and demonstrated that the proposed framework enables the agent to learn a drift racing policy that not only ensures vehicle motion control performance but also effectively reduces lap time.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Sheng Zhao",
    "id": "2311074187",
    "h_index": 2,
    "papers": 14
   },
   {
    "name": "Bolin Zhao",
    "id": "2179641325",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Xiaodong Wu",
    "id": "2266807802",
    "h_index": 5,
    "papers": 25
   },
   {
    "name": "Chen Lv",
    "id": "2323024988",
    "h_index": 3,
    "papers": 13
   }
  ],
  "comment": "",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2608.00113v1",
  "pdf_url": "https://arxiv.org/pdf/2608.00113v1",
  "html_url": "https://arxiv.org/html/2608.00113v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.29687",
  "slug": "diagnosing-compositional-generalization-in-sequential-robot-tasks",
  "title": "Diagnosing Compositional Generalization in Sequential Robot Tasks",
  "abstract": "Sequential robot manipulation requires policies to execute novel combinations of familiar instruction components. However, collecting demonstrations for all possible instruction tuples is combinatorially expensive, while sparsely covered datasets often fail under out-of-distribution recombination. This paper studies compositional generalization through the lens of instruction-space coverage. We decompose the generalization gap into three sources: \\textit{marginal instruction shift}, \\textit{instruction-compositional shift}, and \\textit{context--action shift}. This decomposition allows us to diagnose when sparse training coverage is sufficient, and what structure the training set must preserve for reliable action prediction. Our results show that exhaustive tuple enumeration is unnecessary: a structured subset, as small as one quarter of the full task space, can recover strong out-of-distribution performance when it covers action-relevant dependencies. We further find that sparse training often fails due to instruction steering rather than missing low-level skills; finetuning only one demonstration per task improves OOD success from \\(0.4\\%\\) to \\(54.7\\%\\). For semantically dependent tasks, effective coverage must capture relational structure rather than only factor diversity. These findings suggest that efficient robot data collection should prioritize dependency coverage in instruction space over exhaustive task expansion. More results are available in the supplementary material. Project website: https://yixiaowang7.github.io/Diagnosing_Compositional_Generalization_Robot_Page/.",
  "published": "2026-07-31",
  "updated": "2026-07-31",
  "year": "2026",
  "authors": [
   "Yixiao Wang",
   "Cheng-En Wu",
   "Lingfeng Sun",
   "Pengcheng Wang",
   "Xiang Ji",
   "Boyuan Liang",
   "Guojian Zhan",
   "Masayoshi Tomizuka"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "These findings suggest that efficient robot data collection should prioritize dependency coverage in instruction space over exhaustive task expansion, and that efficient robot data collection should prioritize dependency coverage in instruction space over exhaustive task expansion.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yixiao Wang",
    "id": "2294387767",
    "h_index": 7,
    "papers": 16
   },
   {
    "name": "Chengen Wu",
    "id": "2446369714",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Lingfeng Sun",
    "id": "3435176",
    "h_index": 14,
    "papers": 22
   },
   {
    "name": "Pengcheng Wang",
    "id": "2309198916",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Xiang Ji",
    "id": "2385451344",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Boyuan Liang",
    "id": "2292198026",
    "h_index": 4,
    "papers": 14
   },
   {
    "name": "Guojian Zhan",
    "id": "2134672998",
    "h_index": 6,
    "papers": 38
   },
   {
    "name": "Masayoshi Tomizuka",
    "id": "2293316662",
    "h_index": 8,
    "papers": 20
   }
  ],
  "comment": "",
  "topics": [
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.29687v1",
  "pdf_url": "https://arxiv.org/pdf/2607.29687v1",
  "html_url": "https://arxiv.org/html/2607.29687v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.29640",
  "slug": "bootstrapping-self-supervised-learning-of-binary-classification-using",
  "title": "Bootstrapping Self-Supervised Learning of Binary Classification Using Error Bounds: A Case Study on a Robotic Insertion Task",
  "abstract": "Flexible manufacturing requires rapid deployment of solutions and minimal setup time to remain competitive. An essential attribute is the ability to control error levels, as failures can range from minor performance degradation to severe equipment damage. However, conventional deployment often involves extensive setup, data collection, model training or parameter tuning, and system testing, resulting in significant delays that hinder commercial feasibility. We propose a data engine which gathers data and improves its performance while executing the task. The data engine consists of two classifiers, a fast model prediction and expensive verification. First, a model prediction is performed and based on the confidence level of the prediction, the expensive verification can be used. By adjusting the confidence level, users can control the level of tolerable error. Our method is implemented on a real-world robotic insertion task, which uses force data for the model prediction. The system applies UMAP dimensionality reduction and uses Wilson-Score to compute the confidence bounds of the prediction. Results demonstrate the ability to learn and reduce the need for expensive verifications over time, while staying within the set error-rate. The results highlight the potential of confidence bounds in self-improving models to enhance reliability in robotic classification task.",
  "published": "2026-07-31",
  "updated": "2026-07-31",
  "year": "2026",
  "authors": [
   "Zebin Duan",
   "Norbert Kr\u00fcger",
   "Juan Heredia",
   "Thorbj\u00f8rn Mosekj\u00e6r Iversen",
   "Frederik Hagelskj\u00e6r"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A data engine which gathers data and improves its performance while executing the task, and demonstrates the ability to learn and reduce the need for expensive verifications over time, while staying within the set error-rate.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zebin Duan",
    "id": "2403954956",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Norbert Kr\u00fcger",
    "id": "2195767051",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Juan Heredia",
    "id": "2248320329",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Thorbj\u00f8rn Mosekj\u00e6r Iversen",
    "id": "2020492",
    "h_index": 8,
    "papers": 23
   },
   {
    "name": "Frederik Hagelskj\u00e6r",
    "id": "10712026",
    "h_index": 9,
    "papers": 32
   }
  ],
  "comment": "8 pages, 7 figures, 2 tables",
  "topics": [
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.29640v1",
  "pdf_url": "https://arxiv.org/pdf/2607.29640v1",
  "html_url": "https://arxiv.org/html/2607.29640v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.29625",
  "slug": "balancing-of-humanoid-with-object-mass-trade-off-analyses-and-lifting",
  "title": "Balancing of Humanoid with Object Mass: Trade-off Analyses and Lifting Control",
  "abstract": "The demand for humanoid loco-manipulation tasks with an object has recently increased, and most existing control approaches for stability in such tasks rely on heuristics or machine-learning techniques. This study rigorously analyzes and exploits the dynamic effects of the object mass on balance stability. By formulating the object mass parameters in the whole-body dynamics with distributed contact wrenches and centers of pressure at the stance contacts, their nonlinear effects on the system momenta and constraints are quantified. The dynamic models and constraints are incorporated into the construction of the balanced state basin/boundary (BSB), a partition of the center-of-mass state space for a biped system to maintain balance in its desired contacts. The implications of the BSB for prediction and control are highlighted using a humanoid robot and an analytically tractable reduced-order mechanism. The BSBs under different conditions of base of support, actuation capacity, and pose provide systematic analyses of the effects of object mass on the balancing capability of a system. In particular, the trade-off relationships between momentum regulation and limiting factors in balancing are characterized, introducing two key quantities of the object: the critical mass, at which the system's balancing capability is maximum, and the transition mass, which activates different limiting factors. In addition, sufficient conditions for imposing balanced states on a trajectory are established and implemented with BSBs as explicit threshold constraints in the whole-body trajectory optimization for stable object-lifting control of the humanoid, demonstrating the lift-and-hold and lift-and-release tasks with distinct mass properties in simulations and experiments.",
  "published": "2026-07-31",
  "updated": "2026-07-31",
  "year": "2026",
  "authors": [
   "Hyunjong Song",
   "William Z. Peng",
   "Joo H. Kim"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hyunjong Song",
    "id": "2580982",
    "h_index": 6,
    "papers": 22
   },
   {
    "name": "William Z. Peng",
    "id": "51515990",
    "h_index": 5,
    "papers": 20
   },
   {
    "name": "Joo H. Kim",
    "id": "2267790100",
    "h_index": 2,
    "papers": 12
   }
  ],
  "comment": "22 pages, 13 figures, 1 table",
  "topics": [
   "humanoids"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.29625v1",
  "pdf_url": "https://arxiv.org/pdf/2607.29625v1",
  "html_url": "https://arxiv.org/html/2607.29625v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.29622",
  "slug": "rayvit-ray-conditioned-visual-representations-for-viewpoint-robust-imi",
  "title": "RayViT: Ray-Conditioned Visual Representations for Viewpoint-Robust Imitation Learning",
  "abstract": "Visual imitation learning enables robots to acquire visuomotor skills directly from images, yet RGB observations lack explicit geometric cues, making learned policies brittle to camera perturbations. To address this, we propose \\textbf{Ray-conditioned Vision Transformer Encoder (RayViT)}, a lightweight architecture that injects camera geometry into pretrained ViT backbones. RayViT represents camera geometry as a Pl\u00fccker ray map, patchifies it into ray features, and uses gated cross-attention to produce a ray-conditioned class token. These ray features are added as dense positional embeddings, while the ray class token replaces the original ViT class token to provide a geometry-aware summary representation. We combine this approach with an auxiliary cosine similarity loss to consistently improve the performance and robustness for geometry-aware tokens. Experiments on sim- and real-robot tasks demonstrate that RayViT improves robustness by approximately 13 percentage points under camera perturbations in multi-task RoboCasa benchmark and by 1.78 average completed stages in real-world multi-task success rate compared to baselines.",
  "published": "2026-07-31",
  "updated": "2026-07-31",
  "year": "2026",
  "authors": [
   "Qian Wang",
   "Longrui Chen",
   "Peiran Sun",
   "Aleksandar Taranovic",
   "Niklas Freymuth",
   "Ge Li",
   "Weiran Liao",
   "C. F. Maximilian Nagy",
   "Yucheng Tan",
   "Tao Chen",
   "Gerhard Neumann"
  ],
  "author_count": 11,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Ray-conditioned Vision Transformer Encoder is proposed, a lightweight architecture that injects camera geometry into pretrained ViT backbones and combines this approach with an auxiliary cosine similarity loss to consistently improve the performance and robustness for geometry-aware tokens.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Qian Wang",
    "id": "2305814350",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "Longrui Chen",
    "id": "2188774806",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Peiran Sun",
    "id": "2455160181",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Aleksandar Taranovic",
    "id": "2275603011",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "N. Freymuth",
    "id": "1443836920",
    "h_index": 8,
    "papers": 29
   },
   {
    "name": "Ge Li",
    "id": "2186951194",
    "h_index": 10,
    "papers": 19
   },
   {
    "name": "Weiran Liao",
    "id": "2331177608",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "C. F. M. Nagy",
    "id": "2455133356",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Y. Tan",
    "id": "2452009646",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Tao Chen",
    "id": "2447731442",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Gerhard Neumann",
    "id": "2263395917",
    "h_index": 10,
    "papers": 22
   }
  ],
  "comment": "",
  "topics": [
   "imitation-diffusion",
   "video-generation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.29622v1",
  "pdf_url": "https://arxiv.org/pdf/2607.29622v1",
  "html_url": "https://arxiv.org/html/2607.29622v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.29613",
  "slug": "wcm-a-world-critic-model-for-vision-language-action-reinforcement-lear",
  "title": "WCM: A World Critic Model for Vision-Language-Action Reinforcement Learning",
  "abstract": "Reinforcement learning (RL) post-training of Vision-Language-Action (VLA) models has shown strong promise for robotic manipulation. Among RL methods, critic-based approaches rely on a value estimator that predominantly operates on single-frame observations or single-frame VLM backbone latents, which is a fundamental mismatch with the partially observable nature of robot control. A naive approach to incorporate observation history into the critic incurs exponential complexity with high-dimensional visual space, and still fails because pure scalar-return regression provides insufficient supervision for learning cross-temporal dynamics. We identify the root cause as a state approximation problem: without an explicit world modeling objective, the critic's representation cannot capture the temporal structure needed for accurate value estimation. To address this, we propose the World Critic Model (WCM), built on a lightweight LeJEPA architecture; WCM jointly predicts future latent state and estimates values, such that the critic's representation is explicitly trained to capture temporal dynamics rather than merely regress scalar returns. WCM integrates seamlessly into both on-policy and off-policy training pipelines and is compatible with state-of-the-art VLA backbones including Pi0, Pi0.5, and OpenVLA-OFT. Extensive experiments on 149 tasks across four benchmarks demonstrate that WCM consistently achieves state-of-the-art performance in both in-distribution and out-of-distribution settings, with particularly strong generalization gains. We further validate WCM on seven real-world manipulation tasks using OpenVLA-OFT and Pi0.5 with off-policy RL, confirming stable deployment across diverse settings.",
  "published": "2026-07-31",
  "updated": "2026-07-31",
  "year": "2026",
  "authors": [
   "Senyu Fei",
   "Xiaopeng Yu",
   "Siyin Wang",
   "Xianzhong Zhao",
   "Jingjing Gong",
   "Xipeng Qiu"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.CL",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The World Critic Model is proposed, built on a lightweight LeJEPA architecture; WCM jointly predicts future latent state and estimates values, such that the critic's representation is explicitly trained to capture temporal dynamics rather than merely regress scalar returns.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Senyu Fei",
    "id": "2385785349",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Xiaopeng Yu",
    "id": "2337038918",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Siyin Wang",
    "id": "2182224120",
    "h_index": 9,
    "papers": 24
   },
   {
    "name": "Xianzhong Zhao",
    "id": "2352990513",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Jingjing Gong",
    "id": "2371292918",
    "h_index": 6,
    "papers": 20
   },
   {
    "name": "Xipeng Qiu",
    "id": "2350155100",
    "h_index": 9,
    "papers": 20
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "vla",
   "rl-control",
   "safety-eval"
  ],
  "orgs": [
   "Physical Intelligence"
  ],
  "abs_url": "https://arxiv.org/abs/2607.29613v1",
  "pdf_url": "https://arxiv.org/pdf/2607.29613v1",
  "html_url": "https://arxiv.org/html/2607.29613v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2607.29569",
  "slug": "safe-vision-language-action-models-via-barrier-enhanced-flow-matching",
  "title": "Safe Vision Language Action Models via Barrier Enhanced Flow Matching",
  "abstract": "This article presents a modular inference framework that integrates Flow Matching generative models with formal Control Barrier Function (CBF) safety guarantees. Unlike existing methods that apply external safety filters to a model's final output, our approach modifies the Flow Matching denoising process within the model to inherently generate safe trajectories. By employing a smooth Log-Sum-Exponential aggregate barrier, we enforce safety over entire action chunks. This aggregate barrier ensures a minimal increase in computational overhead and does not alter the semantic intent of the model. We show that, within the proposed framework, the 2-Wasserstein distance between the generated distribution and the target distribution remains bounded. Our method eliminates the need for safety-specific datasets or costly model retraining, providing a versatile solution for safe inference. We validate the approach on two robotic manipulation platforms and a 2D navigation benchmark, verifying that our framework achieves reliable safety without degrading the success rate of the model.",
  "published": "2026-07-31",
  "updated": "2026-07-31",
  "year": "2026",
  "authors": [
   "Kasra Sinaei",
   "Hung-Chieh Wu",
   "Donald Ebeigbe"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This article presents a modular inference framework that integrates Flow Matching generative models with formal Control Barrier Function (CBF) safety guarantees, and modifies the Flow Matching denoising process within the model to inherently generate safe trajectories using a smooth Log-Sum-Exponential aggregate barrier.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "K. Sinaei",
    "id": "2149075788",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Hung-Chieh Wu",
    "id": "2376521887",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Donald Ebeigbe",
    "id": "3420796",
    "h_index": 6,
    "papers": 23
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "rl-control",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.29569v1",
  "pdf_url": "https://arxiv.org/pdf/2607.29569v1",
  "html_url": "https://arxiv.org/html/2607.29569v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.29567",
  "slug": "transgraspnet-physically-and-geometrically-consistent-manipulation-of",
  "title": "TransGraspNet: Physically and Geometrically Consistent Manipulation of Transparent Labware",
  "abstract": "Manipulating transparent laboratory glassware that contains liquid is inherently safety-critical: even small geometric errors can cause unstable grasps and hazardous spillage. Although recent progress has been made in transparent object perception and robotic grasping, most existing systems optimize detection, depth reconstruction, and grasp planning independently, which leads to cross-stage inconsistency imperfect boundaries induce depth bleeding, distorted surfaces corrupt normal estimation, and task agnostic grasp scoring yields tilted or off-center grasps that fail under dynamic motion. In this paper, we propose TransGraspNet, a geometry physics consistent framework that explicitly enforces consistency from perception to execution through three coupled principles: boundary consistency to produce structurally reliable object contours as downstream priors, surface consistency to preserve geometric fidelity and surface normal accuracy during depth reconstruction, and physics consistency to refine grasp selection with centroid alignment and wrench-space stability for upright and dynamically robust manipulation. We evaluate TransGraspNet on public benchmarks, a dedicated transparent glassware dataset, and a real robotic platform. The results show improved boundary quality and surface normal fidelity, and demonstrate strong task-level performance in cluttered transparent scenes. Most importantly, the proposed system achieves reliable real-world operation, including high grasp success rates in clutter and zero spillage during high speed liquid transport, highlighting the effectiveness of our method.",
  "published": "2026-07-31",
  "updated": "2026-07-31",
  "year": "2026",
  "authors": [
   "Hailing Hu",
   "Mingyi Zhu",
   "Yiquan An",
   "Yifei Tian",
   "Tianyou Zuo",
   "Lifeng Zhou"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper proposes TransGraspNet, a geometry physics consistent framework that explicitly enforces consistency from perception to execution through three coupled principles: boundary consistency to produce structurally reliable object contours as downstream priors, surface consistency to preserve geometric fidelity and surface normal accuracy during depth reconstruction, and physics consistency to refine grasp selection.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hailing Hu",
    "id": "2455457754",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Mingyi Zhu",
    "id": "2455184428",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yiquan An",
    "id": "2405412843",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Yifei Tian",
    "id": "2455177287",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Tianyou Zuo",
    "id": "2446846003",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Lifeng Zhou",
    "id": "2248357482",
    "h_index": 2,
    "papers": 7
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.29567v1",
  "pdf_url": "https://arxiv.org/pdf/2607.29567v1",
  "html_url": "https://arxiv.org/html/2607.29567v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.29559",
  "slug": "lemur-learning-to-align-with-multi-objective-reinforcement-learning-fr",
  "title": "LEMUR: Learning to Align with Multi-Objective Reinforcement Learning from Preference Feedback",
  "abstract": "Reinforcement Learning (RL) systems are typically trained using a single, well-specified scalar reward function. However, real-world decision-making tasks often involve multiple, competing objectives, such as performance versus efficiency, where ground-truth reward functions are difficult to specify or inaccessible. While Multi-Objective RL (MORL) addresses such trade-offs by modeling rewards as vectors, existing approaches typically assume access to a well-specified reward function for each objective, inheriting the same challenges faced by single-objective RL. Meanwhile, Preference-based RL (PbRL) has shown great potential in solving complex tasks without access to a pre-defined reward function through reward learning from human feedback, yet has largely been studied in single-objective settings. In this work, we bridge this gap with LEMUR: Learning to Align with Multi-Objective Reinforcement Learning with Preference feedback, a novel framework where an agent interactively learns from the preferences of multiple humans to learn optimal multi-objective policies. Our approach jointly learns policies and multiple objective-specific reward models from human feedback, enabling agents to effectively balance competing objectives during learning. We evaluate LEMUR on a variety of benchmark multi-objective tasks, and empirical results demonstrate its superior performance over baseline methods. Our method presents a promising direction for solving multi-objective decision-making tasks without pre-defined reward functions.",
  "published": "2026-07-31",
  "updated": "2026-07-31",
  "year": "2026",
  "authors": [
   "Manith Adikari",
   "Bei Peng",
   "Samuele Vinanzi",
   "Angelo Cangelosi"
  ],
  "author_count": 4,
  "categories": [
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.AI",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work develops LEMUR: Learning to Align with Multi-Objective Reinforcement Learning with Preference feedback, a novel framework where an agent interactively learns from the preferences of multiple humans to learn optimal multi-objective policies.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Manith Adikari",
    "id": "2261082276",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Bei Peng",
    "id": "2323268",
    "h_index": 19,
    "papers": 38
   },
   {
    "name": "Samuele Vinanzi",
    "id": "19291807",
    "h_index": 6,
    "papers": 24
   },
   {
    "name": "Angelo Cangelosi",
    "id": "2257001079",
    "h_index": 4,
    "papers": 34
   }
  ],
  "comment": "",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.29559v1",
  "pdf_url": "https://arxiv.org/pdf/2607.29559v1",
  "html_url": "https://arxiv.org/html/2607.29559v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.29517",
  "slug": "stage-style-controllable-action-generation-for-personalized-autonomous",
  "title": "STAGE: STyle-controllable Action GEneration for personalized autonomous driving",
  "abstract": "Driving style refers to the behavioral preferences that drivers maintain during driving, shaped by their diverse experiences, habits, and needs, and is typically reflected in varying levels of aggressiveness. If humans choose to use autonomous driving systems, they would expect the driving style of the systems to closely resemble their own habit. However, this is challenging for current industrial autonomous driving systems. To address this, we developed a style controllable action generation method, STAGE, for driving tasks. Its training process is based on imitation learning, incorporating both style value and latent value action modality encoding. Preference learning is then used to identify the user's driving style as a continuous, monotonic style value. And to reduce the cost of human involvement in the preference training process, we also developed a set of rules to compare driving style in data pairs. Then, during inference, the user inputs the style value to control the generated action patterns, dynamically meeting the user's expectations. Using the STAGE method, we verified that the style-controlled action generation results in several typical road scenarios significantly align with human expectations. Furthermore, through comparisons between the STAGE method and various other approaches, we reveal the unique functionalities of STAGE, including its style controllability, style continuity, driving style alignment capability and driving safety. The code for this work is available at: https://github.com/CarlDegio/STAGE",
  "published": "2026-07-31",
  "updated": "2026-07-31",
  "year": "2026",
  "authors": [
   "Zihao Liu",
   "Xing Liu",
   "Yizhai Zhang",
   "Panfeng Huang"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.HC"
  ],
  "primary_category": "cs.RO",
  "venue": "RA-L",
  "venue_source": "semantic-scholar",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "Through comparisons between the STAGE method and various other approaches, the unique functionalities of STAGE are revealed, including its style controllability, style continuity, driving style alignment capability and driving safety.",
  "doi": "10.1109/LRA.2025.3640974",
  "oa_pdf": "https://arxiv.org/pdf/2607.29517",
  "s2_authors": [
   {
    "name": "Zihao Liu",
    "id": "2267905159",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Xing Liu",
    "id": "2146036470",
    "h_index": 6,
    "papers": 27
   },
   {
    "name": "Yizhai Zhang",
    "id": "2652234",
    "h_index": 20,
    "papers": 95
   },
   {
    "name": "Panfeng Huang",
    "id": "2244154121",
    "h_index": 5,
    "papers": 20
   }
  ],
  "comment": "Accepted for publication in IEEE Robotics and Automation Letters",
  "topics": [
   "imitation-diffusion",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.29517v1",
  "pdf_url": "https://arxiv.org/pdf/2607.29517v1",
  "html_url": "https://arxiv.org/html/2607.29517v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.8
 },
 {
  "id": "2607.29482",
  "slug": "temporal-policy-history-initialized-action-generation-for-robotic-lear",
  "title": "Temporal Policy: History-Initialized Action Generation for Robotic Learning from Demonstration",
  "abstract": "By relying on independent couplings from uninformative Gaussian priors, standard diffusion and flow matching models are forced to learn complex, high-cost vector fields to reach the physical action space. Generative models excel at capturing multimodal behaviors for robotic Learning from Demonstration (LfD), but often suffer from high inference cost. This paper introduces Temporal Policy, a generative framework based on stochastic interpolants that formulates action generation as a temporally coupled transport problem. By initializing the generative flow at the robot's recent history, we explicitly couple past states to future action sequences. This data-dependent coupling reduces transport cost and produces straight vector fields. We validate Temporal Policy across visuomotor simulation benchmarks and on a physical Barrett WAM 2x 7DoF teleoperation platform. Our approach reduces transport costs by nearly an order of magnitude compared to noise-initialized baselines, achieving a 19.1 ms inference latency on a single NVIDIA RTX 4080. Crucially, these geometric and computational efficiencies are achieved while matching the success rates of state-of-the-art baselines. This simplified transport geometry bypasses the computational bottleneck of independent Gaussian priors, helping enable high-frequency, closed-loop control. The code is publicly available at https://github.com/dmiller12/TemporalPolicy.",
  "published": "2026-07-31",
  "updated": "2026-07-31",
  "year": "2026",
  "authors": [
   "Dylan Miller",
   "Martin Jagersand"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Temporal Policy is introduced, a generative framework based on stochastic interpolants that formulates action generation as a temporally coupled transport problem and bypasses the computational bottleneck of independent Gaussian priors, helping enable high-frequency, closed-loop control.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Dylan Miller",
    "id": "2446316035",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Martin Jagersand",
    "id": "4089056",
    "h_index": 11,
    "papers": 48
   }
  ],
  "comment": "Accepted to the 2026 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS 2026)",
  "topics": [
   "sim2real",
   "data-teleop"
  ],
  "orgs": [
   "NVIDIA"
  ],
  "abs_url": "https://arxiv.org/abs/2607.29482v1",
  "pdf_url": "https://arxiv.org/pdf/2607.29482v1",
  "html_url": "https://arxiv.org/html/2607.29482v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.0
 },
 {
  "id": "2607.29302",
  "slug": "bwm-a-low-cost-high-fidelity-world-simulator-for-robot-learning",
  "title": "BWM: A Low-Cost High-Fidelity World Simulator for Robot Learning",
  "abstract": "Reliable robot learning requires a world simulator that can predict action consequences before execution on physical hardware, including risky and failure-prone outcomes. Existing physics simulators require substantial asset construction and calibration and still face a sim-to-real gap, while video generators often lack precise control over their responses to fine-grained robot actions. In this paper, we present the Boundless World Model (BWM), an open-source, low-cost, high-fidelity world simulator for robot manipulation. BWM is an action-conditioned world model that combines initial-environment guidance, dynamic visual history, and temporally aligned robot-action conditioning for stateful autoregressive prediction of future observations. We construct action-aligned training clips through trajectory replay, overlapping clip sampling, and initial-observation enhancement. BWM serves as a data engine that augments imitation-learning data with action-aligned rollouts, and as a policy evaluator for closed-loop assessment, risk anticipation, and policy ranking. Experiments on the WorldArena benchmark and physical robots demonstrate improved simulator fidelity and functional utility across the data-engine and policy-evaluator settings. BWM ranks first overall in the WorldArena Challenge across Track 1 and its two Track 2 applications. We release the BWM open-source ecosystem, including model checkpoints, training and inference code, and interfaces for data generation and policy evaluation.",
  "published": "2026-07-31",
  "updated": "2026-07-31",
  "year": "2026",
  "authors": [
   " BWM Team"
  ],
  "author_count": 1,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "BWM is an action-conditioned world model that combines initial-environment guidance, dynamic visual history, and temporally aligned robot-action conditioning for stateful autoregressive prediction of future observations and is released as an open-source, low-cost, high-fidelity world simulator for robot manipulation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Bwm Team",
    "id": "2455114304",
    "h_index": 0,
    "papers": 1
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "sim2real",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.29302v1",
  "pdf_url": "https://arxiv.org/pdf/2607.29302v1",
  "html_url": "https://arxiv.org/html/2607.29302v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.29285",
  "slug": "tract-temporally-routed-action-chunks-with-chronological-phase-authori",
  "title": "TRACT: Temporally Routed Action Chunks with Chronological Phase Authority for Contact-Rich Manipulation",
  "abstract": "Action chunking shortens the effective decision horizon of robot imitation learning by predicting multiple future actions, while conventional phase conditioning describes the current control instant. When a predicted horizon crosses a procedural boundary, assigning the current phase to the entire chunk creates a structural temporal mismatch. We present TRACT, which factorizes phase-structured action chunking into an accepted current phase and a single CURRENT-to-NEXT boundary inside the future horizon. A task-local graph constrains chronological phase authority, and a cumulative boundary distribution monotonically routes future queries through phase-specific query and action paths. For contact execution, a causal response-deficit integrator compares policy intent with ACK-eligible subsequent motion, accumulates arm compensation when directional response is suppressed, and decays after confirmed recovery. Across six real-robot variants with ten trials each, full TRACT achieves 10/10 full-sequence success, 99.00 [88.75, 100.00]% median [min, max] wipe completion, zero observed phase ambiguity, and zero stalls. Under the current complete method package and evaluation setting, the routed representation obtains better observed task results than the flat package (6/10 vs. 3/10 success; 77.08% vs. 8.03% median wipe completion). Chronological authority reduces observed phase ambiguity from 8/10 to 0/10, and response integration reduces stalls from 4/10 to 0/10. The package comparison does not isolate routing from other generator-package differences.",
  "published": "2026-07-31",
  "updated": "2026-08-06",
  "year": "2026",
  "authors": [
   "Jiahao Liu",
   "Kento Kawaharazuka",
   "Tasuku Makabe",
   "Kei Okada"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "TRACT is presented, which factorizes phase-structured action chunking into an accepted current phase and a single CURRENT-to-NEXT boundary inside the future horizon inside the future horizon, and obtains better observed task results than the flat package.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jiahao Liu",
    "id": "2386490351",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Kento Kawaharazuka",
    "id": "8308607",
    "h_index": 17,
    "papers": 220
   },
   {
    "name": "Tasuku Makabe",
    "id": "32031054",
    "h_index": 7,
    "papers": 55
   },
   {
    "name": "Kei Okada",
    "id": "2248244895",
    "h_index": 5,
    "papers": 68
   }
  ],
  "comment": "6 pages, 3 figures",
  "topics": [
   "tactile",
   "imitation-diffusion"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.29285v2",
  "pdf_url": "https://arxiv.org/pdf/2607.29285v2",
  "html_url": "https://arxiv.org/html/2607.29285v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.29271",
  "slug": "mdir-a-task-manifold-impedance-retargeting-method-for-contact-rich-tel",
  "title": "MDIR: A Task-Manifold Impedance Retargeting Method for Contact-Rich Teleoperation",
  "abstract": "Fixed Cartesian impedance makes contact-rich teleoperation demonstrations practical, but gains that secure progress and contact support also determine impact and force variability. We study single-demonstration controller-to-controller impedance retargeting. Given one fixed Cartesian impedance command sequence {K0, D0, xcmd}, Manifold-Decomposed Impedance Retargeting (MDIR) deterministically reparameterizes the recorded controller into an executable task-channel variable-impedance command. MDIR targets this local retargeting problem by preserving projected task-channel responses near the demonstrated trajectory. It represents the source response in operational work, exertion, and support channels with a passive residual complement under a control-chain metric, computes an executable Cartesian-to-Manifold Retargeting (C2M) baseline, and applies Manifold-Constrained Parameter Optimization (MPO) to select a feasible representative with lower wrist-force peaks, impulse, force variability, and nominal controller power. Across planar wiping, pick-and-place, and pushing on a Franka Panda, the full MDIR controller passes Task Check in all 15 closed-loop executions and reduces all four aggressiveness metrics relative to the fixed-impedance demonstrations.",
  "published": "2026-07-31",
  "updated": "2026-08-06",
  "year": "2026",
  "authors": [
   "Liu Jiahao",
   "Kento Kawaharazuka",
   "Tasuku Makabe",
   "Kei Okada"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Manifold-Decomposed Impedance Retargeting deterministically reparameterizes the recorded controller into an executable task-channel variable-impedance command, and applies Manifold-Constrained Parameter Optimization to select a feasible representative with lower wrist-force peaks, impulse, force variability, and nominal controller power.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jiahao Liu",
    "id": "2386490351",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Kento Kawaharazuka",
    "id": "8308607",
    "h_index": 17,
    "papers": 220
   },
   {
    "name": "Tasuku Makabe",
    "id": "32031054",
    "h_index": 7,
    "papers": 55
   },
   {
    "name": "Kei Okada",
    "id": "2248244895",
    "h_index": 5,
    "papers": 68
   }
  ],
  "comment": "8 pages, 6 figures",
  "topics": [
   "tactile",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.29271v2",
  "pdf_url": "https://arxiv.org/pdf/2607.29271v2",
  "html_url": "https://arxiv.org/html/2607.29271v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.29231",
  "slug": "tacprint-a-wearable-fingertip-tactile-sensor-for-human-to-robot-contac",
  "title": "TacPrint: A Wearable Fingertip Tactile Sensor for Human-to-Robot Contact Reproduction",
  "abstract": "Human-centric data collection is emerging as a significant paradigm for robot skill acquisition, but seamlessly integrating low-cost, scalable tactile sensing systems that capture fine-grained fingertip interactions without compromising natural operation remains a key challenge. This reduces the reliability of human-to-robot transfer in contact-rich tasks. In this work, we present TacPrint, a wearable fingertip tactile sensor, where protrusions on the inner surface of the silicone skin are aligned one-to-one with 24 capacitive taxels to enable localized capacitive responses. A real-to-sim-to-real pipeline estimates a 35 $\\times$ 26 contact-depth map from 24-channel capacitive signals. Against simulation-generated labels, the model achieved a contact-region RMSE of 0.223 $\\pm$ 0.161 mm, a weighted-centroid error of 1.213 $\\pm$ 2.379 pixels, and an IoU of 0.829 $\\pm$ 0.169. With measured capacitive inputs, the network-predicted depth evaluated at the guide-calibrated contact center showed a mean absolute error of 0.085 $\\pm$ 0.057 mm across all 40 controlled trials, while the mean contact-position error was 0.250 $\\pm$ 0.208 mm across the 37 trials whose reference contact regions were not truncated by the sensing boundary. In human-to-robot replay, tactile-guided compensation increased grasping and wiping success rates from 0% to 91.67% and 90%, respectively. In closed-loop grasping, dense-depth feedback achieved success rates of 87.5% over all tested positions and 85% under edge-contact conditions, compared with 67.5% and 45% for raw-taxel feedback.",
  "published": "2026-07-31",
  "updated": "2026-07-31",
  "year": "2026",
  "authors": [
   "Yongxi Liu",
   "Chaofan Zhang",
   "Xingyu Zhang",
   "Xiangyin Bao",
   "Boyue Zhang",
   "Shaowei Cui",
   "Shuo Wang"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "TacPrint, a wearable fingertip tactile sensor, where protrusions on the inner surface of the silicone skin are aligned one-to-one with 24 capacitive taxels to enable localized capacitive responses is presented, where the model estimates a 35 $\\times$ 26 contact-depth map from 24-channel capacitive signals.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yongxi Liu",
    "id": "2455173323",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Chaofan Zhang",
    "id": "2256775583",
    "h_index": 7,
    "papers": 19
   },
   {
    "name": "Xingyu Zhang",
    "id": "2382422624",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Xiangyin Bao",
    "id": "2405023102",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Boyue Zhang",
    "id": "2202061746",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Shaowei Cui",
    "id": "1853836031",
    "h_index": 17,
    "papers": 57
   },
   {
    "name": "Shuo Wang",
    "id": "2360886366",
    "h_index": 3,
    "papers": 13
   }
  ],
  "comment": "8 pages, 11 figures",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "sim2real",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.29231v1",
  "pdf_url": "https://arxiv.org/pdf/2607.29231v1",
  "html_url": "https://arxiv.org/html/2607.29231v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.29227",
  "slug": "event-based-upper-body-humanoid-teleoperation-under-challenging-illumi",
  "title": "Event-Based Upper-Body Humanoid Teleoperation Under Challenging Illumination",
  "abstract": "We present a real-time upper-body human-to-humanoid motion imitation framework driven by neuromorphic event-based vision. This work addresses practical perceptual bottlenecks of conventional frame-based RGB sensors, specifically their difficulty in high dynamic range (HDR) scenes and rapid motions due to fixed integration times. By leveraging the Prophesee EVK4 event camera, which operates asynchronously with high temporal resolution and a dynamic range exceeding 120 dB, our system supports stable tracking in conditions where standard vision pipelines degrade, such as severe backlighting and very low light environments below 5 lux. The architecture integrates a low-latency Perception Module, utilizing optimized event accumulation and gravity-aligned inertial fusion, with a causal Motion Module (TWIST) that performs online kinematic retargeting. We validate the system on an embedded NVIDIA Booster T1 platform and an 18-DoF humanoid upper-body setup, demonstrating an end-to-end photon-to-action latency of 23-34 ms and advantages over RGB baselines under our experimental setup. The results indicate a practical trade-off: events can be preferable for fast or poorly lit upper-body teleoperation, whereas well-lit static scenes may favor RGB or hybrid sensing.",
  "published": "2026-07-31",
  "updated": "2026-07-31",
  "year": "2026",
  "authors": [
   "Haoyu Fu",
   "Zhou Ge",
   "Chengze Li",
   "Chenzhao Sun",
   "Ze Cui",
   "Wenjing Zhou",
   "Xulei Qin"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The results indicate a practical trade-off: events can be preferable for fast or poorly lit upper-body teleoperation, whereas well-lit static scenes may favor RGB or hybrid sensing.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Haoyu Fu",
    "id": "2455130246",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Zhou Ge",
    "id": "2247881794",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Chengze Li",
    "id": "2447569582",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Chenzhao Sun",
    "id": "2455175408",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Ze Cui",
    "id": "2276181409",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Wenjing Zhou",
    "id": "2281370196",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Xulei Qin",
    "id": "2455122179",
    "h_index": 0,
    "papers": 1
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "data-teleop"
  ],
  "orgs": [
   "NVIDIA"
  ],
  "abs_url": "https://arxiv.org/abs/2607.29227v1",
  "pdf_url": "https://arxiv.org/pdf/2607.29227v1",
  "html_url": "https://arxiv.org/html/2607.29227v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2607.29172",
  "slug": "clift-turning-gemini-robotics-on-device-into-humanoid-specialists-via",
  "title": "CLIFT: Turning Gemini Robotics On-Device into Humanoid Specialists via Non-Invasive Closed-Loop Iterative Fine-Tuning",
  "abstract": "While robot foundation models are growing increasingly capable, the strongest models are typically trained on proprietary data and remain closed-source, limiting downstream users' ability to adapt them to new tasks, embodiments, and deployment settings. Following the LLM community, an emerging access paradigm for closed-weight robot foundation models is the managed supervised fine-tuning (SFT) API, where users submit training data and receive a tuned policy without access to model weights, gradients, or training internals. While such APIs let downstream users leverage powerful proprietary foundation models, they restrict policy improvement to pure imitation, ruling out reinforcement learning and other closed-loop methods that rely on internal training signals. This limitation is particularly acute for agile, contact-rich humanoid manipulation, where the gap between policy outputs and deployed behavior is large due to novel states, action tracking dynamics, latency, and controller-specific failure modes. We study how effective this managed-API regime is for humanoid adaptation, and how closed-loop improvement can be realized within it to push policies toward task mastery. We conduct one of the first empirical studies of managed-API adaptation on a real humanoid, instantiated on Gemini Robotics On-Device (GROD). We find that direct SFT through the API substantially outperforms a leading open-weight VLA trained on the same demonstrations, yet still falls short of deployment-level mastery on agile, contact-rich tasks. To close this gap, we introduce CLIFT: Closed-Loop Iterative Fine-Tuning, which turns deployment-time reward feedback into API-compatible supervised data and enables closed-loop policy improvement without accessing weights, gradients, likelihoods, or losses-pushing GROD to near-perfect success after two flywheel cycles, all without \"opening the model box.\"",
  "published": "2026-07-31",
  "updated": "2026-07-31",
  "year": "2026",
  "authors": [
   "Yuxin Chen",
   "Hari Srikanth",
   "Nathan Jew",
   "Menglin Wu",
   "Pengcheng Wang",
   "Junli Ren",
   "Masayoshi Tomizuka",
   "Peng Xu",
   "Jinyu Xie",
   "Thomas Tian"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "CLIFT: Closed-Loop Iterative Fine-Tuning is introduced, which turns deployment-time reward feedback into API-compatible supervised data and enables closed-loop policy improvement without accessing weights, gradients, likelihoods, or losses-pushing GROD to near-perfect success after two flywheel cycles, all without opening the model box.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yuxin Chen",
    "id": "2257096146",
    "h_index": 4,
    "papers": 19
   },
   {
    "name": "H. Srikanth",
    "id": "2305817831",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "N. Jew",
    "id": "38766961",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Mengling Wu",
    "id": "2452937783",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Pengcheng Wang",
    "id": "2309198916",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Junli Ren",
    "id": "2331744206",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "Masayoshi Tomizuka",
    "id": "2293316662",
    "h_index": 8,
    "papers": 20
   },
   {
    "name": "Peng Xu",
    "id": "2315280138",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Jinyu Xie",
    "id": "2375047508",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "T. Tian",
    "id": "2253675978",
    "h_index": 4,
    "papers": 7
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "humanoids",
   "tactile",
   "rl-control",
   "foundation-pretraining",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.29172v1",
  "pdf_url": "https://arxiv.org/pdf/2607.29172v1",
  "html_url": "https://arxiv.org/html/2607.29172v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.29102",
  "slug": "vstai-design-and-characterization-of-variable-stiffness-tactile-interf",
  "title": "VSTaI: Design and Characterization of Variable-Stiffness Tactile Interfaces Based on 3D-Printed Structured Fabrics",
  "abstract": "Realistic palpation training requires reliable rendering of soft tissue stiffness changes in real time, which is difficult to achieve with conventional simulators. This paper presents a compact, variable-stiffness tactile interface (VSTaI) based on vacuum-induced jamming of 3D-printed structured fabrics. A vacuum-sealed fabric layer is sandwiched between two silicone layers, and stiffness is tuned by regulating internal pressure. Four fabric patterns with different geometric parameters were fabricated and evaluated using force-indentation tests under atmospheric and vacuum conditions. Across the tested pattern and geometry combinations, vacuum jamming increased stiffness significantly, producing an effective modulus from sub-megapascal to megapascal levels. Specifically, one configuration exhibited a stiffness increase of up to 140% under the jammed state. Circular chainmail patterns provided the most spatially uniform distribution of tactile stiffness, while denser geometries reached higher peak stiffness. VSTaI was also shown to exhibit excellent conformability to the underlying geometry. These results support structured-fabric jamming as a practical approach for shape-conformable, tunable-stiffness displays aimed at physical examination training.",
  "published": "2026-07-31",
  "updated": "2026-07-31",
  "year": "2026",
  "authors": [
   "Yiting Mo",
   "Xinyuan Mao",
   "Jashan Preet Singh",
   "Fernando Bello"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "10.1007/978-3-032-32350-7_38",
  "oa_pdf": "https://arxiv.org/pdf/2607.29102",
  "s2_authors": [
   {
    "name": "Yiting Mo",
    "id": "115656144",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Xinyuan Mao",
    "id": "2455119508",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Jashan Preet Singh",
    "id": "2455184215",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Fernando Bello",
    "id": "2300487229",
    "h_index": 1,
    "papers": 3
   }
  ],
  "comment": "",
  "topics": [
   "tactile",
   "sim2real"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.29102v1",
  "pdf_url": "https://arxiv.org/pdf/2607.29102v1",
  "html_url": "https://arxiv.org/html/2607.29102v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.29031",
  "slug": "auto-jepa-a-latent-world-model-of-continuous-intent-for-end-to-end-aut",
  "title": "Auto-JEPA: A Latent World Model of Continuous Intent for End-to-End Autonomous Driving",
  "abstract": "Existing autonomous-driving world models typically perform dense prediction of future videos, occupancy states, BEV representations, or agent motion. We argue that planning need not reconstruct the complete future world, but only focus on scene features that affect future ego action. Based on this perspective, we propose Auto-JEPA, an action-oriented latent world model that learns continuous future driving intent through joint-embedding prediction. Given visual observations, egomotion history, and navigation commands, Auto-JEPA predicts an intent embedding aligned with the latent representation of the future ego trajectory. The predicted intent retrieves executable trajectories from a fixed trajectory memory, which are then ranked by a scene-conditioned candidate selection module. Auto-JEPA keeps the visual encoder frozen, requires no explicit perception annotations, and uses no learned trajectory generator. By optimizing only task-specific modules for trajectory representation, intent prediction, and candidate selection, Auto-JEPA achieves 91.3 PDMS on NAVSIM v1 and 89.1 EPDMS on NAVSIM v2. Semantic occlusion experiments show that masking dynamic-agent regions induces an average intent change 2.97x that of equal-area random masking. Moreover, occluding vehicles that affect future driving substantially changes the predicted intent and selected trajectory, whereas both remain essentially unchanged when non-influential vehicles are occluded. These results show that future-intent prediction encourages the model to focus on planning-relevant visual features and supports high-quality planning without dense future-world modeling.",
  "published": "2026-07-31",
  "updated": "2026-07-31",
  "year": "2026",
  "authors": [
   "Jiwei Yang",
   "Zhengxian Chen",
   "Chaosheng Huang",
   "Jun Li"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "Results show that future-intent prediction encourages the model to focus on planning-relevant visual features and supports high-quality planning without dense future-world modeling, and shows that future-intent prediction encourages the model to focus on planning-relevant visual features.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jiwei Yang",
    "id": "2452600301",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Zhengxian Chen",
    "id": "2346918271",
    "h_index": 2,
    "papers": 15
   },
   {
    "name": "Chaosheng Huang",
    "id": "2311377751",
    "h_index": 4,
    "papers": 17
   },
   {
    "name": "Jun Li",
    "id": "2346539626",
    "h_index": 2,
    "papers": 13
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "spatial-3d",
   "navigation",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.29031v1",
  "pdf_url": "https://arxiv.org/pdf/2607.29031v1",
  "html_url": "https://arxiv.org/html/2607.29031v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2607.28952",
  "slug": "advances-challenges-and-opportunities-for-legged-robots",
  "title": "Advances, challenges, and opportunities for legged robots",
  "abstract": "Humanoid and quadrupedal robots have the potential to revolutionize the way we work, interact, and coexist with intelligent machines. To understand their effects on society and how they can enable scientific discovery, we assess the current capabilities of these systems along hardware, locomotion, autonomy, data, and applications. We identify recent advances and key open challenges that must be overcome to enable widespread adoption and new use cases for legged robots. Last, we provide an outlook on the future of legged robots, exploring their ethical considerations, economic potential, policy implications, and broader societal effects.",
  "published": "2026-07-31",
  "updated": "2026-07-31",
  "year": "2026",
  "authors": [
   "Jonas Frey",
   "Mat\u00edas Mattamala",
   "Hae-Won Park",
   "Mayank Mittal",
   "Georg Martius",
   "Maike Osborne",
   "Robert Sparrow",
   "Marco Hutter"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "Science Robotics",
  "venue_source": "semantic-scholar",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The current capabilities of humanoid and quadrupedal robots are assessed along hardware, locomotion, autonomy, data, and applications, and an outlook on the future of legged robots is provided.",
  "doi": "10.1126/scirobotics.aee0787",
  "oa_pdf": "https://arxiv.org/pdf/2607.28952",
  "s2_authors": [
   {
    "name": "Jonas Frey",
    "id": "2249531943",
    "h_index": 10,
    "papers": 29
   },
   {
    "name": "Mat\u00edas Mattamala",
    "id": "2347864",
    "h_index": 13,
    "papers": 35
   },
   {
    "name": "Hae-Won Park",
    "id": "2249136281",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Mayank Mittal",
    "id": "2061780867",
    "h_index": 13,
    "papers": 30
   },
   {
    "name": "G. Martius",
    "id": "144247521",
    "h_index": 34,
    "papers": 137
   },
   {
    "name": "Michael A. Osborne",
    "id": "144484861",
    "h_index": 39,
    "papers": 141
   },
   {
    "name": "R. Sparrow",
    "id": "52504962",
    "h_index": 30,
    "papers": 132
   },
   {
    "name": "Marco Hutter",
    "id": "2365116683",
    "h_index": 1,
    "papers": 5
   }
  ],
  "comment": "Accepted for publication in Science Robotics. This is the author's version of the work. It is posted here by permission of the AAAS for personal use, not for redistribution. The definitive version was published in Science Robotics on July 29, 2026, DOI: 10.1126/scirobotics.aee0787",
  "topics": [
   "humanoids"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.28952v1",
  "pdf_url": "https://arxiv.org/pdf/2607.28952v1",
  "html_url": "https://arxiv.org/html/2607.28952v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2607.28737",
  "slug": "mirror-learning",
  "title": "Mirror Learning",
  "abstract": "We investigate imitation learning through the lens of third-person observation and propose a framework for mirror learning: acquiring actionable policies from passive observation. While behavior cloning (BC) excels under dense, well-aligned first-person data, it fundamentally fails to leverage the rich observational signals arising from third-person demonstrations that humans and animals routinely exploit. We introduce a method that composes (i) a learned perspective transformation that places learners in demonstrators' shoes using a fine-tuned video diffusion model and (ii) an inverse dynamics model that infers action trajectories in the learners' control space. This enables the synthesis of mirror data, pseudo first-person expert data generated from third-person observations of demonstrator behavior. Empirically, we show that mirror data alone can train effective policies, and that augmenting first-person BC training with mirror data further improves downstream policy performance. Our results suggest that modern generative world models implicitly encode sufficient structure to enable a scalable and safe alternative to teleoperation-heavy data collection.",
  "published": "2026-07-30",
  "updated": "2026-07-30",
  "year": "2026",
  "authors": [
   "Yunpeng Liu",
   "Matthew Niedoba",
   "Oluwanifemi A. Adekanye",
   "Jason Yoo",
   "Yingchen He",
   "Berend Zwartsenberg",
   "Frank Wood"
  ],
  "author_count": 7,
  "categories": [
   "cs.LG",
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "It is shown that mirror data alone can train effective policies, and that augmenting first-person BC training with mirror data further improves downstream policy performance, and suggest that modern generative world models implicitly encode sufficient structure to enable a scalable and safe alternative to teleoperation-heavy data collection.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yunpeng Liu",
    "id": "2242387232",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "M. Niedoba",
    "id": "41032454",
    "h_index": 8,
    "papers": 20
   },
   {
    "name": "O. A. Adekanye",
    "id": "2293492455",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Jason Yoo",
    "id": "2115662753",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Yingchen He",
    "id": "2333403142",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Berend Zwartsenberg",
    "id": "2242252496",
    "h_index": 7,
    "papers": 21
   },
   {
    "name": "Frank Wood",
    "id": "2283931804",
    "h_index": 6,
    "papers": 14
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "egocentric-data",
   "imitation-diffusion",
   "data-teleop",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.28737v1",
  "pdf_url": "https://arxiv.org/pdf/2607.28737v1",
  "html_url": "https://arxiv.org/html/2607.28737v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.28623",
  "slug": "pac-man-perception-aware-cbf-rl-for-whole-body-safety-in-humanoid-dodg",
  "title": "PAC-MAN: Perception-Aware CBF-RL for Whole-Body Safety in Humanoid Dodgeball",
  "abstract": "We present PAC-MAN, a perception-aware CBF-RL framework that couples control-barrier safety with deployment-realistic onboard sensing for whole-body humanoid dodgeball. The deployed policy sees the ball only as segmentation-masked depth from a head-mounted camera, while training-time CBF guidance represents clearance to every body link, and an adversarial motion prior regularizes the resulting evasive reflexes. We evaluate on a controlled any-link contact benchmark with seeded throws in two regimes: single throws and a deployment loop in which the robot walks back to its station and recovers between throws. On this benchmark, the policy comes within a few points of a privileged state oracle: a fixed onboard camera alone is adequate for evasion. We find that usable barrier structure depends on perceptual observability: Joint-CBF gives the best performance with accurate ball states, degrades under fixed-camera observations when used only as training guidance, and recovers with a ball-tracking gimbal or privileged runtime filter. We therefore deploy a lightweight Link-CBF policy zero-shot on the Unitree G1 in the real world, where it tolerates imperfect perception, succeeds on 95% of throws, and uses semantic segmentation to dodge different balls.",
  "published": "2026-07-30",
  "updated": "2026-07-30",
  "year": "2026",
  "authors": [
   "Lizhi Yang",
   "Junheng Li",
   "Aaron D. Ames"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Lizhi Yang",
    "id": "2362153740",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Junheng Li",
    "id": "2108988997",
    "h_index": 7,
    "papers": 24
   },
   {
    "name": "Aaron D. Ames",
    "id": "2338277217",
    "h_index": 4,
    "papers": 23
   }
  ],
  "comment": "Website at https://lzyang2000.github.io/perceptive_cbf_rl/",
  "topics": [
   "humanoids",
   "safety-eval"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2607.28623v1",
  "pdf_url": "https://arxiv.org/pdf/2607.28623v1",
  "html_url": "https://arxiv.org/html/2607.28623v1",
  "code_url": "https://lzyang2000.github.io/perceptive_cbf_rl/",
  "presented": [
   {
    "club": "saturday-robotics",
    "club_name": "Saturday Robotics",
    "session_number": 30,
    "session_title": "\ud83c\udf7e IROS 2026 x Saturday Robotics \u2014 Robotics Research Night | Reading Club 30. Pittsburgh 9/28",
    "date_text": "",
    "city": "Pittsburgh, PA",
    "url": "https://lu.ma/tzbw7n61",
    "listed_as": ""
   }
  ],
  "club_note": "PAC-MAN: Perception-Aware CBF-RL for Whole-Body Safety in Humanoid Dodgeball presents a perception-aware reinforcement learning framework designed to improve the safety and robustness of humanoid robots operating in dynamic environments. The work combines Control Barrier Functions (CBFs) with reinforcement learning and realistic onboard perception, addressing a key challenge in robot learning: policies can achieve impressive performance but may behave unsafely when exposed to unexpected disturbances or imperfect observations.",
  "featured": true,
  "signal": 4.5
 },
 {
  "id": "2607.28596",
  "slug": "fa-rdp-a-frequency-adaptive-reactive-diffusion-policy-for-contact-rich",
  "title": "FA-RDP: A Frequency-Adaptive Reactive Diffusion Policy for Contact-Rich Manipulation",
  "abstract": "In contact-rich manipulation, action multimodality and reactivity dominate different stages of a single episode. Before contact, multiple trajectories might be equally valid, making it important to preserve diverse action modes. After contact, geometric constraints and force limits narrow the solution space, while successful execution demands rapid responses to force feedback. However, standard diffusion policies use a fixed inference frequency and sampling steps throughout the episode, forcing a fundamental compromise: low-frequency, multi-step sampling better preserves pre-contact multimodality but responds slowly to force feedback, whereas high-frequency sampling improves reactivity but tends to collapse distinct pre-contact modes. To resolve this tradeoff, we present FA-RDP, a frequency-adaptive reactive diffusion policy. A shared multi-frequency visual-force Transformer predicts action chunks at both low and high frequencies, while a learned multimodality indicator dynamically selects multi-step low-frequency sampling before contact and one-step high-frequency sampling as action ambiguity decreases. We further introduce Manifold Consistency Distillation (MCD), which reparameterizes the diffusion network to predict actions on the robot action manifold while retaining DDPM-based residual supervision. Experiments on three contact-rich manipulation tasks show that FA-RDP achieves the highest success rate while preserving diverse pre-contact trajectory modes. Code and videos are available at https://fa-rdp.github.io.",
  "published": "2026-07-30",
  "updated": "2026-07-30",
  "year": "2026",
  "authors": [
   "Lifeng Zhuo",
   "Wendi Chen",
   "Han Xue",
   "Shirun Tang",
   "Jun Lv",
   "Cewu Lu",
   "Chuan Wen"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "FA-RDP is presented, a frequency-adaptive reactive diffusion policy that reparameterizes the diffusion network to predict actions on the robot action manifold while retaining DDPM-based residual supervision, and Manifold Consistency Distillation is introduced.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Lifeng Zhuo",
    "id": "2393140176",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Wendi Chen",
    "id": "2326063487",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Han Xue",
    "id": "2351107813",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Shirun Tang",
    "id": "2397957257",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Jun Lv",
    "id": "2054671126",
    "h_index": 11,
    "papers": 26
   },
   {
    "name": "Cewu Lu",
    "id": "2301174899",
    "h_index": 7,
    "papers": 29
   },
   {
    "name": "Chuan Wen",
    "id": "2381956964",
    "h_index": 5,
    "papers": 20
   }
  ],
  "comment": "Project page: https://fa-rdp.github.io",
  "topics": [
   "tactile",
   "imitation-diffusion"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.28596v1",
  "pdf_url": "https://arxiv.org/pdf/2607.28596v1",
  "html_url": "https://arxiv.org/html/2607.28596v1",
  "code_url": "https://fa-rdp.github.io",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2607.28560",
  "slug": "x-navdp-generalizing-navigation-diffusion-policy-to-novel-behavior-and",
  "title": "X-NavDP: Generalizing Navigation Diffusion Policy to Novel Behavior and Embodiments with Group Q-score Reweighted Matching",
  "abstract": "Pretraining navigation diffusion policies rely on large-scale expert demonstrations. These data are typically generated by a fully-informed oracle planner suited to a single nominal robot. This limits the policy's generalization to diverse embodiments and challenging scenarios (e.g., escaping dead ends or detouring long obstacles) that demand diverse local reactive behaviors with only onboard local observations. Post-training the policy with reinforcement learning (RL) offers a principled remedy. However, previous RL for diffusion approaches lead to only marginal improvements. This is because the intractable likelihood of diffusion policies renders policy gradients unstable in addition to inefficient policy exploration. To address these challenges, we propose a data-efficient diffusion RL post-training framework - GQRM (Group Q-score Reweighted Matching). Our framework introduces two complementary designs: (i) a self-bootstrapped exploration strategy with behavior perturbation that preserves the pretrained policy prior, and (ii) a group Q-score normalization mechanism that computes per-trajectory values on each state for efficient reweighted score matching. By conducting distributed online RL training across heterogeneous embodiments, the resulting fine-tuned policy, X-NavDP, achieves state-of-the-art cross-embodiment visual navigation performance, improving the overall success rate from 61.20% to 84.28% in simulation and 10% to 65% in real-world hard cases. The code and model are publicly available at https://yty-sky.github.io/x-navdp-project-page.",
  "published": "2026-07-30",
  "updated": "2026-08-11",
  "year": "2026",
  "authors": [
   "Tianyu Yang",
   "Yiming Zeng",
   "Wenzhe Cai",
   "Yuqiang Yang",
   "Jiaqi Peng",
   "Hui Cheng",
   "Jiangmiao Pang",
   "Tai Wang"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A data-efficient diffusion RL post-training framework - GQRM (Group Q-score Reweighted Matching), which achieves state-of-the-art cross-embodiment visual navigation performance and introduces two complementary designs: a self-bootstrapped exploration strategy with behavior perturbation that preserves the pretrained policy prior, and a group Q-score normalization mechanism that computes per-trajectory values on each state for efficient reweighted score matching.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Tianyu Yang",
    "id": "2454470093",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yiming Zeng",
    "id": "2269772544",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Wenzhe Cai",
    "id": "2362197963",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Yuqiang Yang",
    "id": "2360919920",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Jiaqi Peng",
    "id": "2357982673",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Hui Cheng",
    "id": "2269708435",
    "h_index": 4,
    "papers": 16
   },
   {
    "name": "Jiangmiao Pang",
    "id": "2277447920",
    "h_index": 24,
    "papers": 61
   },
   {
    "name": "Tai Wang",
    "id": "2359108866",
    "h_index": 11,
    "papers": 26
   }
  ],
  "comment": "20 pages, 4 figures",
  "topics": [
   "imitation-diffusion",
   "rl-control",
   "navigation",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.28560v2",
  "pdf_url": "https://arxiv.org/pdf/2607.28560v2",
  "html_url": "https://arxiv.org/html/2607.28560v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.28451",
  "slug": "machines-that-know-they-are-aging-a-framework-for-hardware-aware-auton",
  "title": "Machines that know they are aging: a framework for hardware-aware autonomous intelligence",
  "abstract": "Autonomous systems inevitably age, yet their artificial intelligence typically assumes hardware remains in its original condition. Batteries degrade, sensors drift, processors accumulate timing errors, and memory reliability declines, creating a growing mismatch between assumed and actual capability. This can lead to agnostic collapse, where mission failure arises from accumulated hardware degradation rather than a single component fault. We propose Aging-Aware Autonomous Intelligence (AAAI), a framework that integrates hardware health directly into reasoning, planning, and mission execution. AAAI is built on three pillars: hardware self-awareness, which continuously estimates the health of power, sensing, memory, and computation subsystems using physics-of-failure models; self-adaptive reasoning, which adjusts inference complexity, planning horizon, and task priorities according to remaining hardware capability; and survival-centric intelligence, which allocates remaining operational life across mission objectives through performance optimization, resource conservation, and graceful degradation. Rather than introducing new hardware, AAAI unifies prognostics, lifecycle management, and hardware-aware computing into a closed-loop cognitive architecture. We argue that such integration is essential for autonomous systems operating in inaccessible or safety-critical environments, including space missions, marine robotics, and implantable medical devices. By enabling machines to recognize and respond to their own aging, AAAI improves resilience, extends operational lifetime, and supports safer, more graceful mission completion.",
  "published": "2026-07-30",
  "updated": "2026-07-30",
  "year": "2026",
  "authors": [
   "Cheng Siong Chin",
   "Jianhua Zhang",
   "Mohan Venkateshkumar"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "By enabling machines to recognize and respond to their own aging, Aging-Aware Autonomous Intelligence improves resilience, extends operational lifetime, and supports safer, more graceful mission completion.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Cheng Siong Chin",
    "id": "2269230699",
    "h_index": 5,
    "papers": 19
   },
   {
    "name": "Jianhua Zhang",
    "id": "2144214797",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "M. Venkateshkumar",
    "id": "2287839080",
    "h_index": 4,
    "papers": 20
   }
  ],
  "comment": "1 figure, 8 pages",
  "topics": [
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.28451v1",
  "pdf_url": "https://arxiv.org/pdf/2607.28451v1",
  "html_url": "https://arxiv.org/html/2607.28451v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.28416",
  "slug": "fastac-a-curved-multispectral-vision-based-tactile-sensor-for-high-spe",
  "title": "FasTac: A Curved Multispectral Vision-Based Tactile Sensor for High-Speed High-Precision 3D Shape and Force Perception",
  "abstract": "Curved tactile fingertips for dexterous manipulation must resolve fine contact geometry, distinguish normal and tangential loads, and capture transient signals. Existing curved vision-based tactile sensors struggle to combine accurate 3D reconstruction, three-axis force estimation, and high-speed processing in a compact form. This article presents FasTac, a curved vision-based tactile sensor integrating multispectral photometric stereo, dynamic-convolution force estimation, and hardware acceleration on a field-programmable gate array (FPGA). Single-image-sensor simultaneous multispectral imaging provides spatially aligned observations for robust surface normal estimation, followed by boundary-prior fast Poisson depth reconstruction. HyperForce uses position-aware dynamic convolution to model the spatially nonuniform mechanical response of curved elastomers and estimate three-axis forces. The complete image-to-normal-force pipeline is deployed on an FPGA. Experiments show that near-infrared (NIR) illumination and the boundary prior decrease depth mean absolute error (MAE) from 0.2730 mm to 0.0415 mm; HyperForce achieves normalized mean absolute error (NMAE) values of 2.74% and 2.39% for normal and shear forces, respectively; and FPGA deployment shortens processing latency from 3.26 ms on the GPU to 1.09 ms. Multi-object reconstruction, feedback grasping, and vibration measurement validate fine geometric perception, stable force feedback, and dynamic contact sensing.",
  "published": "2026-07-30",
  "updated": "2026-07-30",
  "year": "2026",
  "authors": [
   "Xiaofan Lu",
   "Kaiji Huang",
   "Jiahui Chen",
   "Yuankai Lin",
   "Hua Yang",
   "Zhouping Yin"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "FasTac, a curved vision-based tactile sensor integrating multispectral photometric stereo, dynamic-convolution force estimation, and hardware acceleration on a field-programmable gate array (FPGA), and HyperForce, a curved vision-based tactile sensor integrating position-aware dynamic convolution, are presented.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xiaofan Lu",
    "id": "2369210501",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Kaiji Huang",
    "id": "2112768853",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Jiahui Chen",
    "id": "2362293540",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Yuankai Lin",
    "id": "2242604918",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Hua Yang",
    "id": "2362349621",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Zhouping Yin",
    "id": "2069535081",
    "h_index": 11,
    "papers": 29
   }
  ],
  "comment": "13 pages, 11 figures, including 2 pages of supplementary material. Submitted to IEEE/ASME Transactions on Mechatronics",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "spatial-3d",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.28416v1",
  "pdf_url": "https://arxiv.org/pdf/2607.28416v1",
  "html_url": "https://arxiv.org/html/2607.28416v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2607.28415",
  "slug": "qqworld-quantile-quantile-matching-for-world-model-regularization",
  "title": "QQWorld: Quantile-Quantile Matching for World Model Regularization",
  "abstract": "Latent world models enable efficient planning by predicting future states in a compact representation space, but their performance depends critically on the quality of the learned latent distribution. LeWorldModel (LeWM) regularizes its latents toward an isotropic Gaussian using the Epps-Pulley (EP) objective. We show that the corrective gradients of EP rapidly vanish for isolated tail samples, leaving heavy-tailed deviations insufficiently controlled. To address this limitation, we propose QQWorld, which replaces EP with a quantile-quantile matching objective that directly aligns projected latent samples with rank-matched Gaussian quantiles, thereby maintaining effective corrective gradients in the tails. We further develop cross-batch QQ, which enlarges the effective ranking pool using detached samples from previous batches, and characterize its bias-variance trade-off. Across four control environments, QQWorld effectively improves the average planning success rate of LeWM, while consistently yielding better Gaussian alignment and thinner latent tails.",
  "published": "2026-07-30",
  "updated": "2026-07-30",
  "year": "2026",
  "authors": [
   "Zhoushun Yu",
   "Xiaoyu Hu",
   "Xiangyu Xu"
  ],
  "author_count": 3,
  "categories": [
   "cs.LG",
   "cs.AI",
   "cs.CV",
   "cs.MM",
   "cs.RO"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "QWorld is proposed, which replaces EP with a quantile-quantile matching objective that directly aligns projected latent samples with rank-matched Gaussian quantiles, thereby maintaining effective corrective gradients in the tails.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhou Yu",
    "id": "2449143045",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Xiaoyu Hu",
    "id": "2336256310",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Xiangyu Xu",
    "id": "2454424989",
    "h_index": 0,
    "papers": 2
   }
  ],
  "comment": "",
  "topics": [
   "world-models"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.28415v1",
  "pdf_url": "https://arxiv.org/pdf/2607.28415v1",
  "html_url": "https://arxiv.org/html/2607.28415v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.28391",
  "slug": "tacwam-anchor-guided-world-action-model-with-mechanics-aware-tactile-p",
  "title": "TacWAM: Anchor-Guided World Action Model with Mechanics-Aware Tactile Prediction",
  "abstract": "World Action Models (WAMs) combine future-state prediction with robot action generation, but existing approaches largely rely on visual futures. Visual prediction captures scene structure and object motion, yet provides limited supervision for force, deformation, shear, and slip during contact-rich manipulation. This creates two design requirements: tactile futures should carry meaningful physical information, and they should not become privileged cues for action generation. We present TacWAM, a mechanics-aware tactile WAM that addresses this challenge in three steps. First, a Spatially Aligned Fusion (SAF) Tactile Encoder maps tactile appearance, dense force fields, and deformation flow into a shared latent prediction space, with bilateral force and torque reconstruction preserving global contact information. Second, a tactile history encoder provides temporal context so future tactile prediction reflects how force and deformation change beyond the current tactile observation. Third, Anchor-Guided Tri-Modal (AGT) Attention separates current visual and tactile anchors, future prediction tokens, and action tokens, allowing future tactile states to supervise training without being directly read by the action branch. We evaluate TacWAM on four real-world contact-rich manipulation tasks covering fragile grasping, sustained surface contact, and dynamic in-hand manipulation. TacWAM achieves an average success rate of 75.0%, exceeding the strongest evaluated baseline by 37.5 percentage points. Staged ablations show consistent degradation when tactile history is removed and access to future prediction targets is relaxed. These results indicate that future tactile supervision can improve contact-aware action learning when combined with informative tactile representations and deployment-consistent information constraints.",
  "published": "2026-07-30",
  "updated": "2026-07-30",
  "year": "2026",
  "authors": [
   "Lei Jin",
   "Yiding Ma",
   "Xin Zhang",
   "Chen Gao",
   "Wei Wu",
   "Yong Li"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "TacWAM is presented, a mechanics-aware tactile WAM that addresses the challenge of contact-aware action learning when combined with informative tactile representations and deployment-consistent information constraints, and indicates that future tactile supervision can improve contact-aware action learning.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Lei Jin",
    "id": "2371321484",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Yiding Ma",
    "id": "2382924880",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Xin Zhang",
    "id": "2333419153",
    "h_index": 4,
    "papers": 22
   },
   {
    "name": "Chen Gao",
    "id": "2384820291",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Wei Wu",
    "id": "2302827202",
    "h_index": 5,
    "papers": 15
   },
   {
    "name": "Yong Li",
    "id": "2367365672",
    "h_index": 2,
    "papers": 4
   }
  ],
  "comment": "8 pages, 4 figures, 2 tables",
  "topics": [
   "dexterous-manipulation",
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.28391v1",
  "pdf_url": "https://arxiv.org/pdf/2607.28391v1",
  "html_url": "https://arxiv.org/html/2607.28391v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2607.28198",
  "slug": "unicross-unified-cross-skill-dexterous-manipulation-synthesis",
  "title": "UniCross: Unified Cross-Skill Dexterous Manipulation Synthesis",
  "abstract": "Many dexterous manipulation tasks require the object to remain securely held throughout the interaction. From the perspective of hand-object relational motion, such manipulation comprises four canonical skills: grasping, relocation, in-hand rotation, and in-hand translation. Human hands flexibly compose these skills to accomplish complex tasks. Existing approaches, however, model these skills separately with skill-specific action constraints, objectives, or even dedicated hand morphologies, which breaks the compatibility and continuity required for long-horizon composition. In this work, we present a unified framework that models all four skills in a single formulation that shares the same state and action spaces and a common objective structure. This formulation enables straightforward distillation of a single cross-skill policy that performs strongly on every skill, generalizes to unseen objects, stays robust to disturbances, and chains skills seamlessly into long-horizon manipulation. The framework also transfers effectively across different hand morphologies. Overall, our results suggest that different dexterous manipulation skills can be viewed as instantiations of a shared task formulation, revealing the intrinsic consistency across different behaviors.",
  "published": "2026-07-30",
  "updated": "2026-07-30",
  "year": "2026",
  "authors": [
   "Hui Zhang",
   "Julian Ferchow",
   "Jie Song",
   "Mirko Meboldt"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work presents a unified framework that models all four skills in a single formulation that shares the same state and action spaces and a common objective structure, and suggests that different dexterous manipulation skills can be viewed as instantiations of a shared task formulation, revealing the intrinsic consistency across different behaviors.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hui Zhang",
    "id": "2238389387",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Julian Ferchow",
    "id": "100579091",
    "h_index": 7,
    "papers": 29
   },
   {
    "name": "Jie Song",
    "id": "2353205118",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Mirko Meboldt",
    "id": "2255003178",
    "h_index": 4,
    "papers": 13
   }
  ],
  "comment": "Project page: https://zdchan.github.io/UniCross/",
  "topics": [
   "dexterous-manipulation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.28198v1",
  "pdf_url": "https://arxiv.org/pdf/2607.28198v1",
  "html_url": "https://arxiv.org/html/2607.28198v1",
  "code_url": "https://zdchan.github.io/UniCross/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2607.27924",
  "slug": "odeworld-a-continuous-predictive-architecture-via-physical-time-flow",
  "title": "ODEWorld: A Continuous Predictive Architecture via Physical-Time Flow",
  "abstract": "In the physical world we inhabit, space and time are fundamentally continuous. However, existing machine learning paradigms for world modeling are largely confined to discrete-time prediction, thereby exhibiting significant inefficiency in capturing the dynamics of physical world. We introduce Physical-Time Flow (PT-Flow), a novel approach that learns a continuous latent velocity field operating in physical time. Crucially, the underlying dynamics of sequential data are parameterized by an ordinary differential equation (ODE) embedded in a well-structured representation space. Under this paradigm, the prediction of future can be recast as temporal integration via an ODE solver in the compressed latent space. Building upon PT-Flow, we construct ODEWorld, a continuous-time latent world model that is both efficient and versatile. By extracting time-variant features and enforcing ODE properties on both the dynamical representation space and the latent velocity field, ODEWorld effectively addresses the long-standing representation collapse issue in latent world model literature. This also enables high-quality image reconstruction even after long-horizon prediction. Moreover, its continuous nature allows for arbitrary temporal resolution and even backward prediction, which is impossible for most discrete-time models. Lastly, ODEWorld can provide rich planning-oriented information to facilitate downstream policy learning. Comprehensive experiments demonstrate that ODEWorld successfully reconciles planning-conducive dynamics abstraction with visual realism, excelling in both video generation and robotic control. Project page: https://dstate.github.io/odeworld_website/.",
  "published": "2026-07-30",
  "updated": "2026-08-14",
  "year": "2026",
  "authors": [
   "Dongxiu Liu",
   "Haoyi Niu",
   "Peng Cheng",
   "Yuan Gao",
   "Xirui Kang",
   "Sangli Teng",
   "Koushil Sreenath",
   "Xianyuan Zhan"
  ],
  "author_count": 8,
  "categories": [
   "cs.LG",
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "ODEWorld is constructed, a continuous-time latent world model that is both efficient and versatile and successfully reconciles planning-conducive dynamics abstraction with visual realism, excelling in both video generation and robotic control.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Dongxiu Liu",
    "id": "2340937401",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Haoyi Niu",
    "id": "122919426",
    "h_index": 10,
    "papers": 25
   },
   {
    "name": "Peng Cheng",
    "id": "2167325163",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Yuan Gao",
    "id": "2445256696",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Xirui Kang",
    "id": "2385783820",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Sangli Teng",
    "id": "1657710391",
    "h_index": 11,
    "papers": 29
   },
   {
    "name": "K. Sreenath",
    "id": "144116765",
    "h_index": 55,
    "papers": 230
   },
   {
    "name": "Xianyuan Zhan",
    "id": "2242851906",
    "h_index": 21,
    "papers": 55
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.27924v3",
  "pdf_url": "https://arxiv.org/pdf/2607.27924v3",
  "html_url": "https://arxiv.org/html/2607.27924v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.27922",
  "slug": "learning-social-robot-navigation-by-sensing-human-legs",
  "title": "Learning Social Robot Navigation By Sensing Human Legs",
  "abstract": "Robots navigating among pedestrians typically sense their surroundings with a 2D LiDAR mounted close to the ground. At that height, the sensor mostly sees moving legs rather than whole people, yet most learning-based navigation methods still treat pedestrians as simple shapes like circles. This paper addresses that gap with CALF (Convolutional Attention for Leg Features), an end-to-end neural architecture that combines convolutional layers, attention, and MLP to interpret leg motion directly from LiDAR scans and produce safe navigation commands. The CALF policy is trained using deep reinforcement learning algorithms within LegNav, a custom lightweight 2D simulator that combines 2D LiDAR ray tracing with a novel pedestrian gait model. The resulting policy is compared against classical and learning-based baselines in terms of navigation performance and social compliance. The approach is validated through real-world experiments via zero-shot deployment on a TurtleBot 4, yielding smooth and socially compliant trajectories. Written in JAX, the LegNav simulator enables the training of a deployment-ready CALF policy in under an hour on a single consumer GPU.",
  "published": "2026-07-30",
  "updated": "2026-07-30",
  "year": "2026",
  "authors": [
   "Alberto Vaglio",
   "Andrea Garulli",
   "Antonio Giannitrapani",
   "Renato Quartullo",
   "Tommaso Van Der Meer"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "CALF (Convolutional Attention for Leg Features), an end-to-end neural architecture that combines convolutional layers, attention, and MLP to interpret leg motion directly from LiDAR scans and produce safe navigation commands is addressed.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Alberto Vaglio",
    "id": "2454271543",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "A. Garulli",
    "id": "1709237",
    "h_index": 39,
    "papers": 253
   },
   {
    "name": "Antonio Giannitrapani",
    "id": "2588622",
    "h_index": 23,
    "papers": 103
   },
   {
    "name": "Renato Quartullo",
    "id": "122185163",
    "h_index": 8,
    "papers": 21
   },
   {
    "name": "T. V. D. Meer",
    "id": "2061656927",
    "h_index": 0,
    "papers": 5
   }
  ],
  "comment": "",
  "topics": [
   "sim2real",
   "rl-control",
   "navigation",
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.27922v1",
  "pdf_url": "https://arxiv.org/pdf/2607.27922v1",
  "html_url": "https://arxiv.org/html/2607.27922v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.27881",
  "slug": "robobridge-a-modular-framework-for-bridging-policies-to-robust-real-wo",
  "title": "RoboBRIDGE: A Modular Framework for Bridging Policies to Robust Real-World Robotic Agents",
  "abstract": "Vision-Language-Action (VLA) models have attracted growing interest as a scalable approach to robotic manipulation. While these models are effective action predictors, deploying them as robotic agents exposes critical gaps: no mechanism for failure recovery, inconsistent execution over long horizons, and limited robustness to shifts in observations, tasks, or embodiments. Existing solutions address these limitations individually through model retraining or environment-specific modules, yet what is needed is a general framework that systematically transforms a pretrained VLA into a robotic agent. We present RoboBRIDGE, a modular framework that provides an orchestration layer over five coordinated modules, namely Monitor, Perceptor, Planner, Controller, and Robot Interface, to compose robust robotic agents from off-the-shelf components, including pretrained VLAs. The Monitor pairs rapid failure detection with hierarchical recovery to correct errors before they cascade. When the environment diverges from the current plan, the Planner triggers replanning while the Perceptor updates scene understanding asynchronously, avoiding execution stalls. Within the Controller, primitive skill fine-tuning factors manipulation into domain-invariant primitives with dedicated LoRA adapters, reducing sensitivity to domain shifts when a VLA is used. Across LIBERO, RoboCasa, and real-world case studies spanning multiple robot platforms and VLA backbones, RoboBRIDGE consistently outperforms both standalone policies and prior augmented VLA deployments. These results suggest that reliable robotic agency does not arise from scaling action predictors alone, but from structured orchestration around them.",
  "published": "2026-07-30",
  "updated": "2026-07-30",
  "year": "2026",
  "authors": [
   "Sihyung Yoon",
   "Minjong Yoo",
   "Sanghyun Ahn",
   "Seojeong Choi",
   "Honguk Woo"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "RoboBRIDGE is presented, a modular framework that provides an orchestration layer over five coordinated modules, namely Monitor, Perceptor, Planner, Controller, and Robot Interface, to compose robust robotic agents from off-the-shelf components, including pretrained VLAs.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Sihyung Yoon",
    "id": "2379062881",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Minjong Yoo",
    "id": "1381642501",
    "h_index": 6,
    "papers": 17
   },
   {
    "name": "Sanghyun Ahn",
    "id": "2349305973",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Seojeong Choi",
    "id": "2454381509",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Honguk Woo",
    "id": "2283845059",
    "h_index": 6,
    "papers": 25
   }
  ],
  "comment": "Accepted to IROS 2026. 8 pages, 6 figures",
  "topics": [
   "vla",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.27881v1",
  "pdf_url": "https://arxiv.org/pdf/2607.27881v1",
  "html_url": "https://arxiv.org/html/2607.27881v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.8
 },
 {
  "id": "2607.27784",
  "slug": "dexdirect-direct-kinesthetic-arm-guidance-for-efficient-dexterous-demo",
  "title": "DexDirect: Direct Kinesthetic Arm Guidance for Efficient Dexterous Demonstration Collection",
  "abstract": "Scalable collection of dexterous manipulation demonstrations remains a major bottleneck for robot learning. High-fidelity interfaces often require costly hardware and extensive setup, while low-setup, low cost alternatives tend to provide less precise control and impose greater cognitive workload on operators. We present DexDirect, a direct kinesthetic arm guidance for efficient dexterous demonstration collection. The operator drags a 6-DoF gravity-compensated robot arm directly by a handle, while a single webcam retargets operator's other hand onto a 16 joints 13-DoF dexterous robot hand. User studies suggest DexDirect collects 17.2x and 3.2x more successful demonstrations compared to purely vision (AnyTeleop) and pose-tracking (TeleDex) baselines. An adapted NASA-TLX shows DexDirect greatly reduces mental demand, effort, and frustration, despite raising physical demand. A diffusion policy trained on DexDirect demonstrations reaches a 90% success rate on a cube pick-and-place task. These results suggest that direct kinesthetic arm guidance combined with vision-based hand retargeting provides an efficient low-setup and scalable interface for collecting dexterous manipulation demonstrations",
  "published": "2026-07-30",
  "updated": "2026-07-30",
  "year": "2026",
  "authors": [
   "Beom Jun Kim",
   "Shiu-Jen Wang",
   "Jonathan Liu",
   "Alvin Zhu",
   "Quanyou Wang",
   "Hanzhang Fang",
   "Feng Xu",
   "Mingzhang Zhu",
   "Yuchen Cui",
   "Dennis W. Hong"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The results suggest that direct kinesthetic arm guidance combined with vision-based hand retargeting provides an efficient low-setup and scalable interface for collecting dexterous manipulation demonstrations.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Beomdo Kim",
    "id": "2345408475",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Shiu-Jen Wang",
    "id": "2454361511",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Jonathan Liu",
    "id": "2454456528",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Alvin Zhu",
    "id": "2359447907",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Quanyou Wang",
    "id": "2359580930",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Han Fang",
    "id": "2358818018",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Feng Xu",
    "id": "2152480936",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Mingzhang Zhu",
    "id": "2350497013",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Yuchen Cui",
    "id": "2238151901",
    "h_index": 9,
    "papers": 23
   },
   {
    "name": "Dennis W. Hong",
    "id": "2376201986",
    "h_index": 1,
    "papers": 8
   }
  ],
  "comment": "8pages, 6 figures",
  "topics": [
   "dexterous-manipulation",
   "imitation-diffusion",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.27784v1",
  "pdf_url": "https://arxiv.org/pdf/2607.27784v1",
  "html_url": "https://arxiv.org/html/2607.27784v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.27782",
  "slug": "redflow-redirect-failure-into-action-level-corrections-for-flow-matchi",
  "title": "RedFlow: Redirect Failure into Action-Level Corrections for Flow-matching VLA Policy",
  "abstract": "Flow-matching Vision-Language-Action (VLA) policies have shown strong potential for robotic manipulation but often suffer from compounding errors caused by distribution shifts during deployment. While offline reinforcement learning (RL) provides a practical way to improve deployed policies using rollout data, existing methods either ignore failure data or exploit it only at the trajectory level, resulting in low learning efficiency and persistent errors. We propose **RedFlow**, a fine-grained offline RL framework that redirects failure experiences into action-level corrective supervision for flow-matching VLA policies. RedFlow consists of two key components: (1) a **Context-Aware Corrective Matching** mechanism that identifies failure-inducing actions and retrieves successful alternatives from similar contexts as corrective targets, and (2) an **Adaptive Redirection Objective** that jointly reinforces successful actions, suppresses undesirable ones, and redirects recoverable failures toward corrective targets. By converting both successful and failed experiences into dense supervision, RedFlow enables robust recovery learning from mixed-quality data. Experiments on the LIBERO benchmark and three real-world manipulation tasks show that RedFlow consistently outperforms state-of-the-art offline RL baselines, improving the real-world success rate from 56.7% to 74.7%. It also matches strong on-policy methods (PPO, GRPO, and DDPO) while requiring roughly an order of magnitude fewer training samples.",
  "published": "2026-07-30",
  "updated": "2026-07-30",
  "year": "2026",
  "authors": [
   "Zhengyang Yan",
   "Junhao Li",
   "Fangqi Zhu",
   "Zijun Wang",
   "Quanxin Shou",
   "Yikun Miao",
   "Xiaoyi Pang",
   "Zicong Hong",
   "Song Guo"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "RedFlow is proposed, a fine-grained offline RL framework that redirects failure experiences into action-level corrective supervision for flow-matching VLA policies and enables robust recovery learning from mixed-quality data.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhengyang Yan",
    "id": "2398982194",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Junhao Li",
    "id": "2447886873",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Fangqi Zhu",
    "id": "2307561705",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Zijun Wang",
    "id": "2445486601",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Quanxin Shou",
    "id": "2391961599",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Yikun Miao",
    "id": "2221010820",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Xiaoyi Pang",
    "id": "2338828906",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Zicong Hong",
    "id": "89600513",
    "h_index": 19,
    "papers": 72
   },
   {
    "name": "Song Guo",
    "id": "2307558383",
    "h_index": 3,
    "papers": 10
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.27782v1",
  "pdf_url": "https://arxiv.org/pdf/2607.27782v1",
  "html_url": "https://arxiv.org/html/2607.27782v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.27627",
  "slug": "arm2air-cross-embodiment-skeleton-transfer-for-3d-relay-formation",
  "title": "Arm2Air: Cross-Embodiment Skeleton Transfer for 3D Relay Formation",
  "abstract": "Unmanned aerial vehicle (UAV) relay networks can restore connectivity after communication infrastructure is damaged. Urban relay placement is difficult because line-of-sight blockage, communication range, altitude, and three-dimensional obstacles must be considered jointly. Arm2Air transfers obstacle-avoidance skeletons from robot arms to UAV relay placement through cross-embodiment transfer. Source-domain robot-arm motions from a pretrained Neural MP model are converted into ordered skeletons that pretrain a transformer-based transfer platform, which is then adapted to the UAV domain using limited target data and Low-Rank Adaptation. The transferred skeleton initializes a relay chain that is refined for connectivity, bottleneck capacity, delay, and movement cost. On nine held-out high-clutter 3D urban maps, Arm2Air reduced median end-to-end planning runtime by 64.9 percent relative to the fastest conventional planner. On the high-obstruction group of a separate 30-map dense urban holdout, it increased bottleneck capacity by 32.6 percent, reduced capacity variance by 74.7 percent, reduced maximum hop distance by 13.2 percent, reduced hop-distance variance by 75.2 percent, and reduced relay displacement by 16.9 percent relative to IMPC-MD. With only three target-domain training maps, Arm2Air reduced relay-position root mean square error by 53.6 percent relative to training from scratch while updating 0.134 million parameters, compared with 1.383 million for Scratch and Full Fine-tuning. These results demonstrate computationally and data-efficient UAV relay placement and suggest a broader principle for transferring ordered structural priors across heterogeneous embodied tasks.",
  "published": "2026-07-30",
  "updated": "2026-07-30",
  "year": "2026",
  "authors": [
   "Dohun Lee",
   "Kyeonghyun Yoo",
   "Seokmin Kim",
   "Byongho Lee",
   "Seungjoo Oh",
   "Hwangnam Kim"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Arm2Air transfers obstacle-avoidance skeletons from robot arms to UAV relay placement through cross-embodiment transfer and suggests a broader principle for transferring ordered structural priors across heterogeneous embodied tasks.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Dohun Lee",
    "id": "2324833903",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Kyeonghyun Yoo",
    "id": "2341363807",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Seokmin Kim",
    "id": "2454264037",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Byongho Lee",
    "id": "101668227",
    "h_index": 1,
    "papers": 29
   },
   {
    "name": "S. Oh",
    "id": "2449965930",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Hwangnam Kim",
    "id": "2292293650",
    "h_index": 6,
    "papers": 37
   }
  ],
  "comment": "9 pages, 4 figures",
  "topics": [
   "navigation",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.27627v1",
  "pdf_url": "https://arxiv.org/pdf/2607.27627v1",
  "html_url": "https://arxiv.org/html/2607.27627v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.27599",
  "slug": "world-action-planner-generalizable-decision-making-with-action-conditi",
  "title": "World Action Planner: Generalizable Decision-Making with Action-Conditioned World Models",
  "abstract": "Building generalizable agents for diverse applications remains a fundamental challenge. While imitation learning-based policies succeed in specific training environments, they often fail to generalize to novel scenes and tasks. In this work, we propose World Action Planner, a robot planning system that leverages the reasoning capabilities of Vision-Language Models (VLMs) and the physical grounding of a multi-task pose-image conditioned world model. Our system enables an agent to propose initial action plans and iteratively refine them via optimization and search, reasoning over imagined world model rollouts. We demonstrate that our approach achieves superior performance across compositional tasks, new layouts, and zero-shot generalization scenarios, significantly outperforming state-of-the-art end-to-end policy models such as VLAs and WAMs. Project website at worldactionplanner.github.io",
  "published": "2026-07-30",
  "updated": "2026-07-30",
  "year": "2026",
  "authors": [
   "Xiangcheng Zhang",
   "Yilun Du"
  ],
  "author_count": 2,
  "categories": [
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.AI",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes World Action Planner, a robot planning system that leverages the reasoning capabilities of Vision-Language Models (VLMs) and the physical grounding of a multi-task pose-image conditioned world model, significantly outperforming state-of-the-art end-to-end policy models such as VLAs and WAMs.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xiangcheng Zhang",
    "id": "2205711469",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Yilun Du",
    "id": "2364083747",
    "h_index": 3,
    "papers": 5
   }
  ],
  "comment": "Project page at worldactionplanner.github.io",
  "topics": [
   "world-models",
   "imitation-diffusion",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.27599v1",
  "pdf_url": "https://arxiv.org/pdf/2607.27599v1",
  "html_url": "https://arxiv.org/html/2607.27599v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.27549",
  "slug": "cross-embodiment-transfer-via-behavior-aligned-representations",
  "title": "Cross-Embodiment Transfer via Behavior-Aligned Representations",
  "abstract": "Recent progress in large-scale imitation learning for robot manipulation has been driven by leveraging datasets across a wide range of robot embodiments. However, achieving significant cross-embodiment transfer is often still challenging. In this work, we study the role of using behavior-aligned representations (e.g., object bounding boxes, language motions, end-effector traces of robot motion) in vision-language-action (VLA) models to promote cross-embodiment transfer. We hypothesize that by possessing invariances across embodiments while being predictive of robot actions, these representations can help unify large-scale cross-embodiment data to enhance transfer. To assess our hypothesis, we develop a simulation-based benchmark designed to assess transfer with diverse cross-embodiment data to new embodiments. Using this benchmark, we compare different representations and ways of incorporating them. We identify that end-effector traces can be particularly beneficial for transfer, representations are generally more useful with larger prior datasets, and can be used to benefit from action-free data. We also demonstrate that they can enhance sim-to-real cross-embodiment transfer, improving task completion progress of real robot policies pre-trained on simulation data by 28%. We provide videos of our evaluations at our website: https://ajaysridhar.com/barx/.",
  "published": "2026-07-30",
  "updated": "2026-07-30",
  "year": "2026",
  "authors": [
   "Ajay Sridhar",
   "Jensen Gao",
   "Jonathan Yang",
   "Jean Mercat",
   "Suneel Belkhale",
   "Dorsa Sadigh"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The role of using behavior-aligned representations in vision-language-action models to promote cross-embodiment transfer is studied, finding that end-effector traces can be particularly beneficial for transfer, representations are generally more useful with larger prior datasets, and can be used to benefit from action-free data.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "A. Sridhar",
    "id": "52515248",
    "h_index": 8,
    "papers": 14
   },
   {
    "name": "Jensen Gao",
    "id": "2238154243",
    "h_index": 8,
    "papers": 10
   },
   {
    "name": "Jonathan Yang",
    "id": "2366057808",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Jean-Pierre Mercat",
    "id": "72847120",
    "h_index": 10,
    "papers": 33
   },
   {
    "name": "Suneel Belkhale",
    "id": "69879999",
    "h_index": 18,
    "papers": 25
   },
   {
    "name": "Dorsa Sadigh",
    "id": "1779671",
    "h_index": 69,
    "papers": 225
   }
  ],
  "comment": "Project page: https://ajaysridhar.com/barx/",
  "topics": [
   "vla",
   "sim2real",
   "imitation-diffusion",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.27549v1",
  "pdf_url": "https://arxiv.org/pdf/2607.27549v1",
  "html_url": "https://arxiv.org/html/2607.27549v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.27511",
  "slug": "failure-detection-for-surgical-robot-imitation-policies-via-flow-match",
  "title": "Failure Detection for Surgical Robot Imitation Policies via Flow-Matching World Modeling",
  "abstract": "Imitation learning has shown increasing promise for autonomous robotic surgery, yet safe deployment remains challenging due to the safety-critical nature of surgical tasks and the complexity and variability of surgical environments. Failure detection is therefore an essential safeguard, but its development remains difficult due to the challenges of scarce failure data, highly variable manipulation dynamics, and the need to balance missed detections against disruptive false alarms. To address these challenges, we introduce FoMo-FD (Flow-Matching World Model for Failure Detection), a failure detection method that learns nominal short-horizon visual dynamics with an action-conditioned flow-matching world model. FoMo-FD scores the inverse-transport nonconformity of observed endpoint latents, enabling window-level detection of visual-action inconsistencies without requiring failure demonstrations. Detection thresholds are obtained by conformal calibration on successful executions, yielding task-specific alarms without assuming future failure types. We evaluate FoMo-FD on four surgically relevant manipulation tasks with twenty failure modes across simulation and real-world experiments using the da Vinci Research Kit (dVRK). Results show that FoMo-FD outperforms observation-level anomaly baselines and a prediction-error variant of the same world model, with the wrist-camera view achieving the strongest performance, including a 96.6% failure detection rate (FDR) at a 1.3% false alarm rate (FAR).",
  "published": "2026-07-29",
  "updated": "2026-07-29",
  "year": "2026",
  "authors": [
   "Zhefeng Huang",
   "Yilin Cai",
   "Ankit Patel",
   "Mohammad Hajiha",
   "Brendan Browne",
   "Yue Chen"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "FoMo-FD (Flow-Matching World Model for Failure Detection), a failure detection method that learns nominal short-horizon visual dynamics with an action-conditioned flow-matching world model, outperforms observation-level anomaly baselines and a prediction-error variant of the same world model.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhefeng Huang",
    "id": "2257435976",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Yilin Cai",
    "id": "2295803354",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Ankit B. Patel",
    "id": "2307987492",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Mohammad Hajiha",
    "id": "2454273469",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Brendan Browne",
    "id": "2296877089",
    "h_index": 2,
    "papers": 12
   },
   {
    "name": "Yue Chen",
    "id": "2242992002",
    "h_index": 3,
    "papers": 15
   }
  ],
  "comment": "9 pages, 6 figures. Submitted to IEEE Robotics and Automation Letters (RA-L)",
  "topics": [
   "world-models",
   "imitation-diffusion",
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.27511v1",
  "pdf_url": "https://arxiv.org/pdf/2607.27511v1",
  "html_url": "https://arxiv.org/html/2607.27511v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.27494",
  "slug": "simulation-of-surgical-suturing-using-position-based-dynamics-and-the",
  "title": "Simulation of Surgical Suturing Using Position-Based Dynamics and the Material Point Method for Robot Reinforcement Learning",
  "abstract": "Recent advances in robotics research have created a strong demand for high-performance simulators. Surgical robotics simulation faces unique challenges due to the need to model diverse objects, such as rigid instruments, soft tissue, and fluids. While many studies simulate sutures or soft tissue independently, only a few have considered the complete soft-tissue suturing scenario, including the contact between sutures and deformable tissue during suture insertion. Building on previous work, this paper presents a novel suturing simulation environment using sutures modelled by position-based dynamics (PBD) and soft bodies modelled by the material point method (MPM) while considering two-way contact with frictional and drag forces. We introduce a contact coupling method between the PBD suture and the MPM soft tissue, enabling visually plausible suture-tissue interactions. The simulator is optimized for GPU execution with parallel scenes using multiple CUDA streams, and we present a Reinforcement Learning (RL) environment for autonomous suturing sub-tasks, including needle insertion, driving, and extraction. Using ML-Agents, RL agents trained in the simulator show stable learning and achieve 80% and 68% success rates in needle insertion and extraction, respectively, under the strictest distance threshold.",
  "published": "2026-07-29",
  "updated": "2026-07-29",
  "year": "2026",
  "authors": [
   "Tleukhan Mussin",
   "Yafei Ou",
   "Mahdi Tavakoli"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A contact coupling method is introduced between the PBD suture and the MPM soft tissue, enabling visually plausible suture-tissue interactions and presenting a novel suturing simulation environment using sutures modelled by position-based dynamics and soft bodies modelled by the material point method.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Tleukhan Mussin",
    "id": "2199437366",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Yafei Ou",
    "id": "2258631799",
    "h_index": 5,
    "papers": 17
   },
   {
    "name": "Mahdi Tavakoli",
    "id": "2258641926",
    "h_index": 4,
    "papers": 10
   }
  ],
  "comment": "7 pages, 9 figures, accepted for the IEEE RAS/EMBS 11th International Conference on Biomedical Robotics and Biomechatronics (BioRob 2026)",
  "topics": [
   "sim2real",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.27494v1",
  "pdf_url": "https://arxiv.org/pdf/2607.27494v1",
  "html_url": "https://arxiv.org/html/2607.27494v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.27261",
  "slug": "it-s-not-just-more-demos-counterfactual-action-sensitivity-coverage-fo",
  "title": "It's Not Just More Demos: Counterfactual Action Sensitivity Coverage for Data-Efficient Robust Robot Imitation",
  "abstract": "Visuomotor imitation learning has demonstrated success for manipulation tasks. However, the trained policies remain brittle to visual `nuisances', with even minor task-preserving variations such as lighting, distractions or changes in colour result in heavy degradation of the trained policy's performance. While increasing data diversity can improve robustness, it is unclear which additional demonstrations are informative for a particular trained policy. We propose Counterfactual Nuisance Behaviour Cloning (CFNBC), an offline data-selection framework for targeted robustness repair. Starting from a nominal policy trained on `clean' demonstrations, CFNBC generates paired clean and nuisance observations that preserve the expert action, then measures \\emph{action drift}: the change in the policy's predicted action under a nuisance that should not alter the desired behaviour. This provides a policy-specific sensitivity signal for selecting a compact, response-diverse repair set from a larger candidate pool, without requiring rollout success labels or online policy execution. We show in MuJoCo bimanual cube transfer and SimplerEnv cube stacking that action drift correlates with nuisance-induced failure, and that response-guided repair with only $20$--$30$ selected candidates substantially outperforms matched-budget random selection while approaching the performance of much larger random repair budgets. These results support a data-centric view of robustness repair: the most useful data are not necessarily the most numerous, visually diverse, or obviously difficult, but the examples that cover fragile response modes of the current policy.",
  "published": "2026-07-29",
  "updated": "2026-07-29",
  "year": "2026",
  "authors": [
   "Giovanni D'urso",
   "Kaushik Roy",
   "Nicholas Lawrance",
   "Brendan Tidd"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "RSS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "It is shown in MuJoCo bimanual cube transfer and SimplerEnv cube stacking that action drift correlates with nuisance-induced failure, and that response-guided repair with only $20$--$30$ selected candidates substantially outperforms matched-budget random selection while approaching the performance of much larger random repair budgets.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Giovanni D'Urso",
    "id": "1400691211",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Kaushik Roy",
    "id": "2268120688",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Nicholas Lawrance",
    "id": "2281035155",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Brendan Tidd",
    "id": "2325093088",
    "h_index": 3,
    "papers": 9
   }
  ],
  "comment": "Workshop paper accepted at the RSS 2026 Workshop on Data-Centric Robotics: What Data Do Robots Really Need?",
  "topics": [
   "imitation-diffusion",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.27261v1",
  "pdf_url": "https://arxiv.org/pdf/2607.27261v1",
  "html_url": "https://arxiv.org/html/2607.27261v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2607.27180",
  "slug": "humanclaw-can-vision-language-models-act-through-a-body",
  "title": "HumanCLAW: Can Vision-Language Models Act Through a Body?",
  "abstract": "Evaluating whether a vision-language model (VLM) can act through a physical body is challenging. The outcome of an action couples the VLM's decision with motor control. When a task fails, it is hard to tell whether the VLM made a bad choice or the motor controller simply failed to execute it, e.g., losing balance and falling. In this work, we introduce HumanCLAW, an evaluation framework that decouples action decision-making from low-level execution. At every step, a harnessed, off-the-shelf VLM issues an atomic skill command, and the command is translated into a sub-second chunk of continuous full-body motion with real physical consequences, including gravity and collisions. The body can therefore act freely in the physical world, while execution-side disturbances, balance and motor errors, are factored out. What remains measurable is the model's action intelligence: its moment-to-moment choice of what the body should execute next. Based on this framework, we build HumanCLAW-Bench: 1,218 long-horizon, egocentric find-navigate-interact episodes across 41 indoor scenes. We test nine state-of-the-art VLMs and find that none solves the benchmark; the best model reaches only a 16.8% success rate. Recognizing the target is not the bottleneck. What current VLMs lack is embodied self-awareness: they lose track of their own body, failing to tell where it is, whether it has reached the goal, or whether it has hit an obstacle.",
  "published": "2026-07-29",
  "updated": "2026-08-03",
  "year": "2026",
  "authors": [
   "Li Siyao",
   "Jiawei Gu",
   "Shuai Liu",
   "Kairui Hu",
   "Zekun Li",
   "Linjie Li",
   "Chengcheng Tang",
   "Po-Chen Wu",
   "Ivan Shugurov",
   "Lingni Ma",
   "Michael Zollhoefer",
   "Sizhe An",
   "Abhay Mittal",
   "Amy Zhao",
   "Ranjay Krishna",
   "Manling Li",
   "Ziwei Liu",
   "Chuan Guo"
  ],
  "author_count": 18,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work introduces HumanCLAW, an evaluation framework that decouples action decision-making from low-level execution in a vision-language model (VLM), and builds HumanCLAW-Bench, a database of 1,218 long-horizon, egocentric find-navigate-interact episodes across 41 indoor scenes.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Siyao Li",
    "id": "2454469768",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Jiawei Gu",
    "id": "2216587705",
    "h_index": 9,
    "papers": 17
   },
   {
    "name": "Shuai Liu",
    "id": "2257375680",
    "h_index": 5,
    "papers": 15
   },
   {
    "name": "Kairui Hu",
    "id": "2312180931",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Zekun Li",
    "id": "2268496034",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Linjie Li",
    "id": "2363415109",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Chengcheng Tang",
    "id": "2330561874",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Po-Chen Wu",
    "id": "3239304",
    "h_index": 8,
    "papers": 14
   },
   {
    "name": "Ivan Shugurov",
    "id": "2378128352",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Lingni Ma",
    "id": "2284187142",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Michael Zollh\u00f6fer",
    "id": "2342277746",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Sizhe An",
    "id": "2378504570",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Abhay Mittal",
    "id": "2378128807",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Amy Zhao",
    "id": "2239549780",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Ranjay Krishna",
    "id": "2262217505",
    "h_index": 22,
    "papers": 61
   },
   {
    "name": "Manling Li",
    "id": "2386131426",
    "h_index": 3,
    "papers": 18
   },
   {
    "name": "Ziwei Liu",
    "id": "2321147297",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Chuan Guo",
    "id": "2391164156",
    "h_index": 5,
    "papers": 15
   }
  ],
  "comment": "Project page: https://human-claw.github.io/",
  "topics": [
   "egocentric-data",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.27180v2",
  "pdf_url": "https://arxiv.org/pdf/2607.27180v2",
  "html_url": "https://arxiv.org/html/2607.27180v2",
  "code_url": "https://human-claw.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2607.27138",
  "slug": "dlam-distributional-latent-actions-with-temporal-constraints",
  "title": "DLAM: Distributional Latent Actions with Temporal Constraints",
  "abstract": "Vision-language-action (VLA) models remain constrained by scarce action-labeled robot data, whereas action-free videos offer abundant observations of physical change. Latent action models can extract such priors, but reconstruction-trained codes may predict future observations without the structure required for joint generation with robot actions. Existing structured methods add temporal constraints but retain deterministic transition points, so residual errors in locally inferred transitions may propagate and compound under recursive composition. We introduce DLAM, a distributional latent-action model that represents each transition as a diagonal Gaussian. Reconstruction conditioned on the reference frame grounds the mean in observed visual change, while normalized composition and reversal over equal-gap triplets constrain both the mean and dimension-wise variance. Variance composition uses a lightweight shared-correlation coefficient to account for dependence between adjacent transitions that share an intermediate frame, whereas reversal negates the mean and preserves the variance. For downstream policy learning, we freeze the encoder and train a flow-matching policy to jointly generate mean transition sequences and robot actions. On held-out transitions, DLAM learns more temporally consistent latent dynamics than existing latent-action baselines and achieves stronger direct and cumulative reconstruction on held-out videos. Under the same controlled $\u03c0_0$ transfer protocol, it also improves policy performance on MetaWorld MT50, LIBERO, and real-world manipulation tasks. Controlled ablations show that normalized mean constraints account for most of the reconstruction gain, while learned variance and correlation-aware composition provide complementary improvements in downstream control.",
  "published": "2026-07-29",
  "updated": "2026-07-29",
  "year": "2026",
  "authors": [
   "Zuojin Tang",
   "Feifan Luo",
   "Haoyun Liu",
   "Botai Yuan",
   "Dekang Qi",
   "Ronghan Chen",
   "Yandan Yang",
   "Tong Lin",
   "Xinyuan Chang",
   "Mu Xu",
   "Bin Liu",
   "De Ma",
   "Zhiheng Ma"
  ],
  "author_count": 13,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "DLAM, a distributional latent-action model that represents each transition as a diagonal Gaussian, is introduced, a distributional latent-action model that learns more temporally consistent latent dynamics than existing latent-action baselines and achieves stronger direct and cumulative reconstruction on held-out videos.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zuojin Tang",
    "id": "2330552170",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Feifan Luo",
    "id": "2454108996",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Haoyun Liu",
    "id": "2280484160",
    "h_index": 5,
    "papers": 20
   },
   {
    "name": "B. Yuan",
    "id": "2446037360",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Dekang Qi",
    "id": "2382761853",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Ronghan Chen",
    "id": "2383246949",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Yandan Yang",
    "id": "2382929955",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Tong Lin",
    "id": "2238905623",
    "h_index": 4,
    "papers": 15
   },
   {
    "name": "Xinyuan Chang",
    "id": "2320818428",
    "h_index": 8,
    "papers": 25
   },
   {
    "name": "Mu Xu",
    "id": "2363407946",
    "h_index": 7,
    "papers": 25
   },
   {
    "name": "Bing Liu",
    "id": "143830417",
    "h_index": 21,
    "papers": 46
   },
   {
    "name": "De Ma",
    "id": "2284589308",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Zhiheng Ma",
    "id": "2249623432",
    "h_index": 9,
    "papers": 36
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "vla"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.27138v1",
  "pdf_url": "https://arxiv.org/pdf/2607.27138v1",
  "html_url": "https://arxiv.org/html/2607.27138v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.27017",
  "slug": "what-can-latent-world-models-know-physical-parameter-identifiability-i",
  "title": "What Can Latent World Models Know? Physical Parameter Identifiability in Multimodal Predictive Representations",
  "abstract": "A central premise of latent world models is that predicting the future forces a representation to internalize the physics of its environment. Which physical quantities does a trained latent actually contain, and what decides this? We answer with controlled interventions in POKEWORLD, an interactive environment whose visually identical objects hide mass, drag, and contact stiffness. A certificate-gated protocol first certifies each parameter as recoverable from raw observations, then measures whether it enters the latent, so a null result can be attributed to the objective rather than to the environment. The resulting identifiability map has two organizing mechanisms and one frontier. Inputs limit what can be known, while prediction targets decide what is retained. Stiffness enters the latent only when touch is forecast ($R^2=0.50$, compared with $-0.02$ when the same signal is merely fused into the input), and under single-step prediction a vision-only latent discards even perfectly visible object state. Drag marks the frontier. It carries a recoverability certificate of 0.89 yet plateaus near 0.13 under every deterministic prediction objective we test, while a supervised head on the same trunk reaches 0.45. Parameters whose readout is slow and ratio-type under the sensed coordinates fall outside what these objectives acquire. On RH20T, an input-target factorial across scaling curves reproduces both mechanisms across two robots and 4,258 episodes. Every arm missing information or prediction pressure stays flat over a fivefold data range, and only the full multimodal objective forecasts force beyond a persistence baseline, with held-out gains that grow with scale. Objective structure determines which physical parameters a latent acquires, and additional data improves only the parameters it already acquires.",
  "published": "2026-07-29",
  "updated": "2026-07-31",
  "year": "2026",
  "authors": [
   "Kaizhen Tan",
   "Xin Xu",
   "Siru Tao",
   "Yixiao Li",
   "Hanzhe Hong",
   "Yang Feng",
   "Heqing Du"
  ],
  "author_count": 7,
  "categories": [
   "cs.LG",
   "cs.RO"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 2,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Kaizhen Tan",
    "id": "2380624774",
    "h_index": 1,
    "papers": 14
   },
   {
    "name": "Xin Xu",
    "id": "2454075825",
    "h_index": 0,
    "papers": 5
   },
   {
    "name": "Siru Tao",
    "id": "2362506658",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Yixiao Li",
    "id": "2455182273",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Han Hong",
    "id": "2093217886",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Yang Feng",
    "id": "2261199971",
    "h_index": 14,
    "papers": 37
   },
   {
    "name": "Heqing Du",
    "id": "2430741201",
    "h_index": 0,
    "papers": 4
   }
  ],
  "comment": "",
  "topics": [
   "world-models"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.27017v3",
  "pdf_url": "https://arxiv.org/pdf/2607.27017v3",
  "html_url": "https://arxiv.org/html/2607.27017v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.48
 },
 {
  "id": "2607.26991",
  "slug": "rl-2-vla-adaptive-rl-latent-compositional-steering-with-test-time-scal",
  "title": "RL$^2$-VLA: Adaptive RL Latent Compositional Steering with Test-Time Scaling for Vision-Language-Action Models",
  "abstract": "Despite the impressive visuomotor capabilities enabled by Vision-Language-Action (VLA) models, their performance often degrades on challenging and out-of-domain tasks. Recent test-time steering and scaling methods improve performance without extensive data collection and retraining, but action samples often remain concentrated around similar behaviors and therefore inherit correlated failure modes. Moreover, existing methods apply the same intervention strategy at every timestep, regardless of whether the base policy is already likely to succeed. To address these limitations, we introduce $RL^2$, an adaptive inference-time steering framework that leverages Reinforcement Learning on VLA Latents. First, we train a lightweight offline RL policy conditioned on expressive latents extracted from the VLA action expert and compose its flow velocity with that of the frozen VLA during inference. This compositional steering strategy combines the behavioral priors of large-scale imitation learning with the action diversity induced by offline RL beyond dominant demonstration modes. We further discover that inference-time steering follows fundamentally different scaling laws under success and failure states, revealing that action diversity is most beneficial when the base VLA is likely to fail, but can unnecessarily perturb already-accurate actions when success is likely. Building on this insight, $RL^2$ activates compositional steering only when failure is predicted. Across the SIMPLER and PolaRiS benchmarks, $RL^2$ improves success rates by up to +17.3% in out-of-domain settings, while ablations and scaling studies demonstrate the importance of latent representations and RL training. Finally, real-world experiments demonstrate that these gains transfer beyond simulation, establishing $RL^2$ as a practical and modular steering framework for VLA deployment.",
  "published": "2026-07-29",
  "updated": "2026-07-30",
  "year": "2026",
  "authors": [
   "Derek Ming Siang Tan",
   "Shailesh Shailesh",
   "Srikrishna Iyer",
   "William Wei Jie Teo",
   "Yuanliang Ju",
   "Qiao Gu",
   "Guillaume Sartoretti"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work introduces an adaptive inference-time steering framework that leverages Reinforcement Learning on VLA Latents, and discovers that inference-time steering follows fundamentally different scaling laws under success and failure states, revealing that action diversity is most beneficial when the base VLA is likely to fail, but can unnecessarily perturb already-accurate actions when success is likely.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Derek Ming Siang Tan",
    "id": "2313640644",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Shailesh Shailesh",
    "id": "2454107953",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Srikrishna Iyer",
    "id": "2372288188",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "William Wei Jie Teo",
    "id": "2454107436",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yuanliang Ju",
    "id": "2389706336",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Qiao Gu",
    "id": "2284863276",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "G. Sartoretti",
    "id": "2292917033",
    "h_index": 11,
    "papers": 67
   }
  ],
  "comment": "Code and models are available at https://rl2-vla.github.io",
  "topics": [
   "vla",
   "imitation-diffusion",
   "rl-control",
   "foundation-pretraining",
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.26991v2",
  "pdf_url": "https://arxiv.org/pdf/2607.26991v2",
  "html_url": "https://arxiv.org/html/2607.26991v2",
  "code_url": "https://rl2-vla.github.io",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2607.26985",
  "slug": "symmgrid-super-scaling-on-robot-learning-with-parallelized-symmetries",
  "title": "SymmGrid: Super-Scaling On-Robot Learning with Parallelized Symmetries and Egocentric-Exocentric Visual Perception",
  "abstract": "Deep reinforcement policy learning directly in physical robots (on-robot learning) remains bottlenecked by slow wall-clock training times. We present SymmGrid, a trajectory level augmentation framework inspired by parallelized symmetries that super-scales group transformations to significantly accelerate on-robot learning in both egocentric and exocentric visual setups. We model a Markov Decision Process (MDP) under a symmetry tree, in which state-action pairs have admissible parallelized invariant transformations that yield a geometric grid structure. The state is modelled with ego- or exocentric images and proprioception information. The latter require special treatment, in the form of homographies, to warp visual scenes in line with their corresponding spatial transformations. These parallelized transformations produce a large set of unique symmetric equivalences that populate the replay buffer with diverse and consistent experiences that speed up learning and improve performance. We present extensive training and evaluations performed directly on real robot manipulation contact tasks including peg-insertions, cable routing, and object relocations. Relative to SOTA, SymmGrid achieved wall-clock training convergence speed-ups of 1.37-2.17x, evaluation success rate improvements of 1.09x-1.27x, fastest training convergence times of 16.6, 10.9, and 79.3 minutes respectively. For trajectory wide assessments, we used normalized area under the curve (nAUC) ratios. SymmGrid achieved improvements of up to 2.59x. These results confirm that simple branch symmetries can have an outsized result due to super-scaling and bring us closer to sub-10 minute on-robot learning training in manipulation tasks suitable for arms and humanoids. The project page is available at symmgrid-robot.github.io",
  "published": "2026-07-29",
  "updated": "2026-07-29",
  "year": "2026",
  "authors": [
   "Gabe Everett",
   "Brice Gunter",
   "Ryan Vander Stelt",
   "Cleiver Ruiz-Martinez",
   "Blake Hull",
   "Juan Rojas"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "SymmGrid is presented, a trajectory level augmentation framework inspired by parallelized symmetries that super-scales group transformations to significantly accelerate on-robot learning in both egocentric and exocentric visual setups and confirms that simple branch symmetries can have an outsized result due to super-scaling.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Gabe Everett",
    "id": "2454108294",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Brice Gunter",
    "id": "2454108666",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Ryan Vander Stelt",
    "id": "2454108581",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Cleiver Ruiz-Martinez",
    "id": "2446569304",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Blake Hull",
    "id": "2446569589",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "J. Rojas",
    "id": "2303466306",
    "h_index": 1,
    "papers": 6
   }
  ],
  "comment": "9 pages, 7 figures, 1 table",
  "topics": [
   "humanoids",
   "egocentric-data"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.26985v1",
  "pdf_url": "https://arxiv.org/pdf/2607.26985v1",
  "html_url": "https://arxiv.org/html/2607.26985v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.26924",
  "slug": "temporally-centered-sigreg-improves-multi-task-leworldmodel-learning-f",
  "title": "Temporally Centered SIGReg Improves Multi-Task LeWorldModel Learning: From Analysis to Method",
  "abstract": "Recent work on LeWorldModel (LeWM) has shown that the Sketched Isotropic Gaussian Regularizer (SIGReg) enables stable end-to-end world-model learning from pixels by regularizing the latent marginal distribution toward an isotropic Gaussian, thereby preventing representation collapse. While effective and elegant in single-task settings, this recipe does not extend reliably to multi-task training, leading to substantially worse downstream behavior-cloning performance. In this paper, we show that marginal Gaussianization compresses the separation between task-dependent latent clusters relative to within-cluster variation. This compression introduces representation aliasing across tasks and states, and makes the learned representations highly sensitive to small visual perturbations. To address this problem, we apply SIGReg to temporally centered residuals rather than to the latent marginal distribution. This surrogate target places no direct regularization pressure on the separation among cluster centers, removes the requirement that the full latent follow a single isotropic Gaussian, and retains the anti-collapse effect of SIGReg. On the LIBERO benchmark, our method improves downstream success on the long-horizon suite by 1.7x and raises the average success rate across four suites from 53.2% to 73.6%. Without external pretraining, it slightly outperforms Diffusion Policy trained from scratch and approaches the performance of large-scale pretrained policy baselines. These results reveal a structural incompatibility between marginal Gaussian priors and multi-task latent structure, and provide a simple route toward stable and scalable end-to-end multi-task world-model learning.",
  "published": "2026-07-29",
  "updated": "2026-07-31",
  "year": "2026",
  "authors": [
   "Chang Liu",
   "Fei Suo",
   "Yanzhou Jin",
   "Yusuke Iwasawa",
   "Yutaka Matsuo",
   "Yaonan Zhu"
  ],
  "author_count": 6,
  "categories": [
   "cs.LG",
   "cs.RO"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "It is shown that marginal Gaussianization compresses the separation between task-dependent latent clusters relative to within-cluster variation, and introduces representation aliasing across tasks and states, and makes the learned representations highly sensitive to small visual perturbations.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Chang Liu",
    "id": "2453766468",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Feiya Suo",
    "id": "2451145328",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yanzhou Jin",
    "id": "2454215353",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yusuke Iwasawa",
    "id": "1715282",
    "h_index": 23,
    "papers": 171
   },
   {
    "name": "Yutaka Matsuo",
    "id": "2153732825",
    "h_index": 14,
    "papers": 42
   },
   {
    "name": "Yaonan Zhu",
    "id": "2363506887",
    "h_index": 2,
    "papers": 10
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "imitation-diffusion",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.26924v2",
  "pdf_url": "https://arxiv.org/pdf/2607.26924v2",
  "html_url": "https://arxiv.org/html/2607.26924v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.26914",
  "slug": "biovln-a-simulation-platform-for-visual-language-navigation-in-biomedi",
  "title": "BioVLN: A Simulation Platform for Visual Language Navigation in Biomedical Laboratories",
  "abstract": "Biomedical laboratory robots must navigate to instruments before performing experimental procedures. Existing embodied navigation platforms are designed for household environments and treat a target as an object center or an arbitrary nearby position. This representation is inadequate for laboratory instruments, which must be approached from their operating side while maintaining safe clearance from surrounding equipment. We introduce BioVLN, a simulation platform for developing and evaluating visual-language navigation agents in biomedical laboratories. BioVLN represents each instrument with three regions: its physical body, a surrounding clearance region, and an operation area in front of the usable side. This model is applied consistently to scene generation, target placement, navigation evaluation, and safety analysis, so success depends on reaching a position from which the instrument can be accessed. BioVLN supports procedural scene generation and manually designed environments, producing 47 scenes and 1667 episodes. Standardized navigation and reinforcement-learning interfaces enable trajectory collection and policy training. Experiments show that geometric exploration reaches 74.4--87.5% success, while sampling multiple valid positions in the operation area improves success to 83.3--92.5% and reduces unsafe proximity.",
  "published": "2026-07-29",
  "updated": "2026-07-29",
  "year": "2026",
  "authors": [
   "Zhe Liu",
   "Quan Lu",
   "Zhaohui Du",
   "Zhe Wang",
   "Huanbo Jin",
   "Jiaming Gu",
   "Qi Wang",
   "Ting Xiao",
   "Minting Pan",
   "Dongzhan Zhou"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work introduces BioVLN, a simulation platform for developing and evaluating visual-language navigation agents in biomedical laboratories and shows that geometric exploration reaches 74.4--87.5% success, while sampling multiple valid positions in the operation area improves success and reduces unsafe proximity.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhe Liu",
    "id": "2404004658",
    "h_index": 2,
    "papers": 13
   },
   {
    "name": "Quan Lu",
    "id": "2311449124",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Zhaohui Du",
    "id": "2364085760",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Zhe Wang",
    "id": "2241483629",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Huanbo Jin",
    "id": "2442115636",
    "h_index": 0,
    "papers": 5
   },
   {
    "name": "Jiaming Gu",
    "id": "2261096786",
    "h_index": 2,
    "papers": 13
   },
   {
    "name": "Qi Wang",
    "id": "2318419521",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Ting Xiao",
    "id": "2407742135",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Minting Pan",
    "id": "2027156855",
    "h_index": 5,
    "papers": 17
   },
   {
    "name": "Dongzhan Zhou",
    "id": "2359612649",
    "h_index": 6,
    "papers": 29
   }
  ],
  "comment": "17 pages, 4 figures",
  "topics": [
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.26914v1",
  "pdf_url": "https://arxiv.org/pdf/2607.26914v1",
  "html_url": "https://arxiv.org/html/2607.26914v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.26903",
  "slug": "from-passive-video-to-editable-experience-physically-grounded-experien",
  "title": "From Passive Video to Editable Experience: Physically Grounded Experience Synthesis for Embodied Intelligence",
  "abstract": "The key bottleneck in embodied AI is not model architecture but data. Although billions of human manipulation videos exist online, robots cannot directly learn from them due to the embodiment gap between human morphology and robot hardware. We introduce Pegasus, a low-resource framework that bridges this gap by translating human demonstrations into robot-learnable data through structured knowledge transfer. Instead of relying on raw video prompts, Pegasus constructs a graph-based intermediate representation: a Task Graph extracted from human videos is transformed through Affordance and Constraint Graphs into a Robot Planning Graph for robot-conditioned video generation. A hierarchical affordance latent space models the relationship between object states, affordances, and tasks, enabling generalization beyond object identities. A closed-loop physics verifier further filters invalid generations using kinematic feasibility, collision constraints, and joint limits. We evaluate Pegasus across a range of egocentric manipulation benchmarks, including GTEA Gaze+ and EPIC-KITCHENS-100, and diverse robot embodiments, assessing Task Correctness, Executability, State Consistency, and Learnability. Results demonstrate reliable cross-embodiment translation and show that robot data generation can be reframed from a hardware collection problem into a scalable, low-resource knowledge transfer problem.",
  "published": "2026-07-29",
  "updated": "2026-07-29",
  "year": "2026",
  "authors": [
   "Jia Luo"
  ],
  "author_count": 1,
  "categories": [
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.AI",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Results demonstrate reliable cross-embodiment translation and show that robot data generation can be reframed from a hardware collection problem into a scalable, low-resource knowledge transfer problem.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jia Luo",
    "id": "2339235957",
    "h_index": 2,
    "papers": 6
   }
  ],
  "comment": "",
  "topics": [
   "egocentric-data",
   "foundation-pretraining",
   "hardware-codesign",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.26903v1",
  "pdf_url": "https://arxiv.org/pdf/2607.26903v1",
  "html_url": "https://arxiv.org/html/2607.26903v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.26889",
  "slug": "structuregs-structure-aware-gaussian-splatting-for-articulated-object",
  "title": "StructureGS: Structure-aware Gaussian Splatting for Articulated Object Reconstruction",
  "abstract": "Reconstructing articulated objects with multiple movable parts is essential for understanding object structure and enabling physical interaction. However, this reconstruction task poses significant challenges due to the entanglement of geometry, appearance, and motion parameters during optimization. Existing methods rely primarily on photometric supervision, which commonly fails to disentangle these interdependent components, resulting in poor part decomposition with blurred boundaries and geometric artifacts. To address this limitation, we introduce StructureGS, a reconstruction framework for articulated objects that integrates structure-aware guidance into 3D Gaussian Splatting. Our approach leverages oriented bounding boxes of object parts to enforce two key structural properties: spatial coherence, which constrains each part's geometry to remain compact and spatially coherent within its designated region, and structural connectivity, which enforces physically plausible contact relationships between adjacent parts. These properties are realized through structure-aware losses that inject explicit structural constraints into the optimization process. Extensive experiments demonstrate that our method achieves state-of-the-art performance in articulated object reconstruction, producing high-quality results with well-defined part geometries.",
  "published": "2026-07-29",
  "updated": "2026-07-29",
  "year": "2026",
  "authors": [
   "Gahye Lee",
   "Gyoonseo Kim",
   "Wonjong Jang",
   "Jooeun Son",
   "Seungyong Lee"
  ],
  "author_count": 5,
  "categories": [
   "cs.GR",
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.GR",
  "venue": "ECCV 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work introduces StructureGS, a reconstruction framework for articulated objects that integrates structure-aware guidance into 3D Gaussian Splatting and achieves state-of-the-art performance in articulated object reconstruction, producing high-quality results with well-defined part geometries.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Gahye Lee",
    "id": "2351002784",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Gyoo-Sik Kim",
    "id": "2213854332",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "W. Jang",
    "id": "115626480",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Jooeun Son",
    "id": "2373643286",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Seungyong Lee",
    "id": "2294927346",
    "h_index": 2,
    "papers": 10
   }
  ],
  "comment": "accepted at ECCV 2026",
  "topics": [
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.26889v1",
  "pdf_url": "https://arxiv.org/pdf/2607.26889v1",
  "html_url": "https://arxiv.org/html/2607.26889v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2607.26817",
  "slug": "from-uncertainty-to-determinism-coarse-to-fine-visual-floorplan-locali",
  "title": "From Uncertainty to Determinism: Coarse-to-Fine Visual Floorplan Localization without Ray Matching",
  "abstract": "Visual Floorplan Localization (FLoc) has emerged as a promising solution for indoor localization by matching egocentric images against minimalist structural maps. However, due to cross-modal information asymmetry and repetitive indoor layouts, visual FLoc is fundamentally challenged by multimodal pose distributions, where visually identical observations map to distinct, spatially separated locations. Existing ray-matching-based methods tackle this by explicitly predicting sparse geometric or semantic rays, which inherently incur information loss and demand resource-intensive preprocessing alongside exhaustive matching during inference. In this paper, we bypass the intermediate ray-matching paradigm and propose a coarse-to-fine visual FLoc framework that progresses from uncertainty to determinism. In the coarse stage, we design an image-conditioned pose diffusion model to parameterize the continuous multimodal pose distribution, effectively routing stochastically initialized pose particles toward distinct candidate modes. In the refinement stage, we propose a localized refiner that predicts bounded sub-meter pose residuals from candidate-centered floorplan crops, where structural ambiguities are largely eliminated. Our method effectively balances global multi-hypothesis tracking and local sub-meter refinement without requiring any offline map preprocessing or test-time lookup tables. Comprehensive results on the S3D (full) and ZInD benchmarks demonstrate that our approach achieves state-of-the-art accuracy and robustness.",
  "published": "2026-07-29",
  "updated": "2026-08-02",
  "year": "2026",
  "authors": [
   "Shiyong Meng",
   "Bolei Chen",
   "Ping Zhong",
   "Yang Wan",
   "Rongzhi Wang",
   "Jiazhi Xia",
   "Jianxin Wang"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper bypasses the intermediate ray-matching paradigm and proposes a coarse-to-fine visual FLoc framework that progresses from uncertainty to determinism, and effectively balances global multi-hypothesis tracking and local sub-meter refinement without requiring any offline map preprocessing or test-time lookup tables.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Shiyong Meng",
    "id": "152847451",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Bolei Chen",
    "id": "2152692482",
    "h_index": 8,
    "papers": 36
   },
   {
    "name": "Ping Zhong",
    "id": "2242594347",
    "h_index": 7,
    "papers": 29
   },
   {
    "name": "Yang Wan",
    "id": "2454113067",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Rongzhi Wang",
    "id": "2454432587",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Jiazhi Xia",
    "id": "2407150825",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Jianxin Wang",
    "id": "2188020750",
    "h_index": 5,
    "papers": 15
   }
  ],
  "comment": "",
  "topics": [
   "egocentric-data",
   "video-generation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.26817v2",
  "pdf_url": "https://arxiv.org/pdf/2607.26817v2",
  "html_url": "https://arxiv.org/html/2607.26817v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.26809",
  "slug": "practice-makes-policies-bootstrapping-and-consolidating-robotic-capabi",
  "title": "Practice Makes Policies: Bootstrapping and Consolidating Robotic Capabilities from Zero Human Demonstrations",
  "abstract": "General-purpose robotic manipulation requires robots to perform diverse tasks in open-world environments while improving their skills over time. Despite recent progress in robotic manipulation, existing systems still primarily acquire manipulation skills in a static manner, where capabilities are learned for specific tasks or settings rather than adaptively evolving through physical interaction. Resembling how repeated practice enables humans to develop muscle memory, advanced manipulation proficiency requires an autonomous capability evolution mechanism that allows robots to progressively transform interaction experiences into increasingly effective manipulation abilities. To this end, we propose HERO, a self-improving hierarchical embodied agent that enables autonomous capability evolution from zero human demonstrations. HERO organizes heuristic reasoning, exemplar reuse, and reflexive execution into a unified orchestration framework, allowing robots to autonomously bootstrap manipulation experience, rapidly accumulate reusable behaviors through experience transfer, and progressively consolidate recurring interactions into efficient closed-loop visuomotor policies. By tightly coupling autonomous data collection with task execution, HERO continuously expands and dynamically schedules manipulation capabilities according to different stages of experience accumulation and execution requirements. Extensive experiments demonstrate that HERO substantially reduces human intervention during robotic data collection while achieving robust manipulation across diverse tasks, providing a promising path toward self-improving robotic systems.",
  "published": "2026-07-29",
  "updated": "2026-07-29",
  "year": "2026",
  "authors": [
   "Jialiang Li",
   "Yuhan Wang",
   "Haojun Li",
   "Gaojing Zhang",
   "Yangtian Ye",
   "Qipeng Liu",
   "Haotian Liang",
   "Wenzhao Lian"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "HELP, a self-improving hierarchical embodied agent that enables autonomous capability evolution from zero human demonstrations, is proposed, allowing robots to autonomously bootstrap manipulation experience, rapidly accumulate reusable behaviors through experience transfer, and progressively consolidate recurring interactions into efficient closed-loop visuomotor policies.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jialiang Li",
    "id": "2382821886",
    "h_index": 0,
    "papers": 5
   },
   {
    "name": "Yuhan Wang",
    "id": "2446324131",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Haojun Li",
    "id": "2454161138",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Gaojing Zhang",
    "id": "2382841296",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Yangtian Ye",
    "id": "2454105785",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Qipeng Liu",
    "id": "2282237503",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Haotian Liang",
    "id": "2365384725",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Wenzhao Lian",
    "id": "2323497285",
    "h_index": 3,
    "papers": 16
   }
  ],
  "comment": "",
  "topics": [
   "egocentric-data",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.26809v1",
  "pdf_url": "https://arxiv.org/pdf/2607.26809v1",
  "html_url": "https://arxiv.org/html/2607.26809v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.26789",
  "slug": "checkvla-execution-time-verification-with-action-conditioned-world-mod",
  "title": "CheckVLA: Execution-Time Verification with Action-Conditioned World Model for Long-Horizon Mobile Manipulation",
  "abstract": "Vision-language-action (VLA) policies commonly execute long-horizon mobile manipulation through open-loop action chunks, issuing multiple actions without receiving new high-level visual input. A committed chunk therefore implies how observations should evolve, but accidental deviations can violate this expectation while the remaining actions continue to propagate the error: commit-time policy confidence cannot react to a deviation that occurs after dispatch, and observation-only anomaly scores lack an action-conditioned reference for separating expected effects from unexplained changes. We propose CheckVLA, which verifies execution with a separately trained, frozen action-conditioned world model. A conformally calibrated risk threshold bounds the episode-level probability of an unnecessary first intervention and determines when to intervene, its exceedance controls how strongly the rewritten suffix retains the superseded chunk, latency-aware hard prefixing restricts replacement to actions that remain deployable, and an event-driven keyframe bank preserves evidence of prior progress across repairs. On RoboCasa365, under a common training recipe and a matched invocation budget, CheckVLA attains a 36.1% average success rate against 27.6% for periodic replanning (+8.5 points). At a matched 5% episode-level false-alarm target, action conditioning raises timely recall to 77.9%, against 48.6% for an observation-only control and 37.9% for an action-shuffled control. These simulation results support action-conditioned verification as a way to restore feedback during chunked execution while keeping the repair consistent with inference latency.",
  "published": "2026-07-29",
  "updated": "2026-07-29",
  "year": "2026",
  "authors": [
   "Yushan Liu",
   "Peibo Sun",
   "Xintao Chao",
   "Zhenyang Yang",
   "Yifan Xie",
   "Lingfeng Zhang",
   "Shoujie Li",
   "Chenyu Tang",
   "Fang Chen",
   "Xiao-Ping Zhang",
   "Wenbo Ding"
  ],
  "author_count": 11,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "CheckVLA is proposed, which verifies execution with a separately trained, frozen action-conditioned world model, and results support action-conditioned verification as a way to restore feedback during chunked execution while keeping the repair consistent with inference latency.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yushan Liu",
    "id": "2348411949",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Peibo Sun",
    "id": "2156281203",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Xintao Chao",
    "id": "2348439323",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Zhen Yang",
    "id": "2453312416",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yifan Xie",
    "id": "2393037213",
    "h_index": 1,
    "papers": 11
   },
   {
    "name": "Lingfeng Zhang",
    "id": "2327199957",
    "h_index": 5,
    "papers": 16
   },
   {
    "name": "Shoujie Li",
    "id": "2155443623",
    "h_index": 11,
    "papers": 49
   },
   {
    "name": "Chenyu Tang",
    "id": "2454309084",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Fangda Chen",
    "id": "2353341501",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Xiao-Ping Zhang",
    "id": "2271214144",
    "h_index": 5,
    "papers": 23
   },
   {
    "name": "Wenbo Ding",
    "id": "2375173478",
    "h_index": 5,
    "papers": 21
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "vla",
   "navigation",
   "safety-eval"
  ],
  "orgs": [
   "MIT"
  ],
  "abs_url": "https://arxiv.org/abs/2607.26789v1",
  "pdf_url": "https://arxiv.org/pdf/2607.26789v1",
  "html_url": "https://arxiv.org/html/2607.26789v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2607.26712",
  "slug": "actswm-action-sensitive-world-models-for-long-horizon-planning-in-open",
  "title": "ActSWM: Action-Sensitive World Models for Long-Horizon Planning in Open-World Games",
  "abstract": "Latent world models support efficient model-predictive control by optimizing future control sequences in latent space and replanning in a receding-horizon manner. However, existing latent predictors often lack stable long-horizon rollout ability, and prediction accuracy alone does not ensure that rollouts remain responsive to the actions being planned. We identify Context Collapse, a failure mode in which autoregressive latent predictors maintain high similarity to future states while producing nearly indistinguishable futures under different action sequences. To address this issue, we propose ActSWM, an action-sensitive latent world model grounded in a transition-separation principle: a planning-useful latent dynamics model should keep alternative-action futures distinguishable and make the action associated with each local transition recoverable. Under this principle, action sensitivity is enforced as a constraint on latent rollouts rather than treated only as an auxiliary prediction target, encouraging predicted futures to preserve action-dependent differences over long horizons. Across step-drift analysis, closed-loop Minecraft planning, and cross-game local action recovery, ActSWM preserves larger action-dependent rollout gaps than existing baselines, improves task success in long-horizon interactive settings, and enables world-model-based action recovery from offline gameplay videos.",
  "published": "2026-07-29",
  "updated": "2026-08-15",
  "year": "2026",
  "authors": [
   "Zhenfeng Gan",
   "ZiTong Zeng",
   "Jiajun Cheng",
   "Yeke Song",
   "Yongyi Tang",
   "Xueqian Wang"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes ActSWM, an action-sensitive latent world model grounded in a transition-separation principle, which preserves larger action-dependent rollout gaps than existing baselines, improves task success in long-horizon interactive settings, and enables world-model-based action recovery from offline gameplay videos.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhe Gan",
    "id": "2268495128",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "Zitong Zeng",
    "id": "2367714139",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Jiajun Cheng",
    "id": "2454257891",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yeke Song",
    "id": "2454215135",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yongyi Tang",
    "id": "2454366264",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Xueqian Wang",
    "id": "2275989943",
    "h_index": 2,
    "papers": 7
   }
  ],
  "comment": "10 pages, 5 figures",
  "topics": [
   "world-models",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.26712v2",
  "pdf_url": "https://arxiv.org/pdf/2607.26712v2",
  "html_url": "https://arxiv.org/html/2607.26712v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.26579",
  "slug": "contactflow-a-video-action-conditioning-that-transfers-across-embodime",
  "title": "ContactFlow: A video action conditioning that transfers across embodiments",
  "abstract": "World models offer a promising route toward robot planning by enabling agents to imagine and verify the consequences of actions before execution. However, current video-based world models often struggle to capture the physical constraints that govern manipulation, particularly contact. Further, their action conditioning is often constrained to specific embodiments such as parallel grippers. We propose \\emph{Contact Flow}, an embodiment-agnostic action representation that encodes manipulation through the trajectory of 3D contact points between an actor and a target object. By discarding actor-specific appearance and kinematics, Contact Flow provides a shared conditioning signal for both human demonstrations and robotic execution. Therefore, we can train a large-scale video generative model on both human and robotic object interaction videos conditioned on Contact Flow, yielding a world model that predicts physically plausible manipulation outcomes. We integrate this model into a propose-imagine-verify-act pipeline, where generated rollouts are assessed by a vision-language model before execution. Experiments on the DROID dataset and real-world tabletop manipulation tasks demonstrate that Contact Flow enables transfer between human demonstrations and different robotic embodiments.",
  "published": "2026-07-29",
  "updated": "2026-07-29",
  "year": "2026",
  "authors": [
   "Sami Azirar",
   "Enrico Pallotta",
   "Jan Nogga",
   "J\u00fcrgen Gall",
   "Sven Behnke",
   "Hermann Blum"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "This work proposes Contact Flow, an embodiment-agnostic action representation that encodes manipulation through the trajectory of 3D contact points between an actor and a target object, and integrates this model into a propose-imagine-verify-act pipeline, where generated rollouts are assessed by a vision-language model before execution.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Sami Azirar",
    "id": "2303408209",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Enrico Pallotta",
    "id": "2265490751",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Jan Nogga",
    "id": "2090609885",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Jurgen Gall",
    "id": "1576263143",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Sven Behnke",
    "id": "2312401022",
    "h_index": 3,
    "papers": 16
   },
   {
    "name": "Hermann Blum",
    "id": "2349540902",
    "h_index": 2,
    "papers": 5
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "egocentric-data",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.26579v1",
  "pdf_url": "https://arxiv.org/pdf/2607.26579v1",
  "html_url": "https://arxiv.org/html/2607.26579v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2607.26567",
  "slug": "speech2grasp-data-efficient-transfer-of-text-conditioned-grasp-detecti",
  "title": "Speech2Grasp: Data-Efficient Transfer of Text-Conditioned Grasp Detection to Speech in Humanoid Robots",
  "abstract": "Humanoid robots increasingly require multi-modal understanding for natural interaction with humans. Despite the prominence of vision-language models, they generally assume textual rather than the more natural speech inputs. In this paper, we investigate whether a well-established text-conditioned model can be transferred to speech in a data-efficient manner. Using ALBEF as a case study, we conduct diagnostic analyses showing that a lightweight MLP-based projector effectively adapts it to speech, while preserving semantic discrimination and robustness. Motivated by these findings, we introduce Speech2Grasp, a framework for data-efficient transfer of text-conditioned grasp detection to speech. Real-world humanoid robot experiments show that Speech2Grasp outperforms cascaded ASR-based pipeline, while reducing inference latency. Our findings suggest a practical paradigm for extending established text-conditioned systems to speech.",
  "published": "2026-07-29",
  "updated": "2026-07-29",
  "year": "2026",
  "authors": [
   "Hung Nguyen",
   "Kim Nhat Minh Nguyen",
   "Van Duc Vu",
   "Van-Danh Le",
   "Hoang Huy Le",
   "Dinh Tuan Nguyen",
   "Pham Tuyen Le",
   "Van-Truong Nguyen",
   "Quan Nguyen"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Speech2Grasp is introduced, a framework for data-efficient transfer of text-conditioned grasp detection to speech, and its findings suggest a practical paradigm for extending established text-conditioned systems to speech.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "H. Nguyen",
    "id": "2380985970",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Kim Nhat Minh Nguyen",
    "id": "2454105227",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Van Duc Vu",
    "id": "2454079185",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "V. L\u00ea",
    "id": "2407508853",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Hoang H. Le",
    "id": "2238572888",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Dinh Tuan Nguyen",
    "id": "2454104313",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Pham-Binh Le",
    "id": "2443734904",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "V. Nguyen",
    "id": "2448463078",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Quan Nguyen",
    "id": "2453644861",
    "h_index": 0,
    "papers": 2
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "humanoids",
   "video-generation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.26567v1",
  "pdf_url": "https://arxiv.org/pdf/2607.26567v1",
  "html_url": "https://arxiv.org/html/2607.26567v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.26513",
  "slug": "explicit-kinematic-guidance-from-analytic-concepts-for-vision-language",
  "title": "Explicit Kinematic Guidance from Analytic Concepts for Vision-Language-Action Models",
  "abstract": "Current Vision-Language-Action (VLA) models rely mainly on 2D inputs, neglecting the rich object structural information and commonsense knowledge inherent in the 3D physical world. This deficiency restricts their spatial awareness and adaptability for complex, high-precision manipulation. To bridge this crucial gap, we construct a Concept Expert module for VLA to build executable Analytic Concepts that represent objects as explicit, programmatic blueprints. Our mechanism operates in two synergistic phases: First, prior to VLA inference, the Concept Expert leverages 3D information from Vision Foundation Models (VFMs) to estimate the initial kinematic and structural parameters. Second, throughout the manipulation process, the VLA model utilizes its inherent capability to dynamically track the dynamic concept parameters, continuously aligning them with observational changes to ensure persistent accuracy. Once established, the Analytic Concepts provide explicit, high-quality guidance for VLA fine-tuning through (1) dense, programmatic manipulation rewards and (2) precise spatial guidance. This formulation allows VLA models to learn physically grounded interaction behaviors while maintaining end-to-end learning flexibility. Our experimental results show consistent improvements in success rate and learning efficiency across supervised and reinforcement learning settings, demonstrating the effectiveness of structured, concept-based guidance for VLA post-training.",
  "published": "2026-07-29",
  "updated": "2026-07-29",
  "year": "2026",
  "authors": [
   "Mingyang Sun",
   "Jiude Wei",
   "Xiujian Liang",
   "Qichen He",
   "Donglin Wang",
   "Cewu Lu",
   "Jianhua Sun"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A Concept Expert module is constructed for VLA to build executable Analytic Concepts that represent objects as explicit, programmatic blueprints that provide explicit, high-quality guidance for VLA fine-tuning through dense, programmatic manipulation rewards and precise spatial guidance.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Mingyang Sun",
    "id": "2334782875",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "Jiude Wei",
    "id": "2321886708",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Xiujian Liang",
    "id": "2395756679",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Qichen He",
    "id": "2387756186",
    "h_index": 0,
    "papers": 8
   },
   {
    "name": "Donglin Wang",
    "id": "2384798031",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Cewu Lu",
    "id": "1830034",
    "h_index": 22,
    "papers": 66
   },
   {
    "name": "Jianhua Sun",
    "id": "2152148778",
    "h_index": 8,
    "papers": 23
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "rl-control",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.26513v1",
  "pdf_url": "https://arxiv.org/pdf/2607.26513v1",
  "html_url": "https://arxiv.org/html/2607.26513v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.26434",
  "slug": "reinforcement-learning-on-cost-constrained-quadrupedal-hardware",
  "title": "Reinforcement Learning on Cost-Constrained Quadrupedal Hardware",
  "abstract": "Deploying learned control policies on low-cost robotic platforms introduces transport latencies and noisy motor feedback that systematically widens the sim-to-real gap. The chasm of simulation to deployment in hardware lies in the delay of the actuator reaching the commanded position. On platforms such as the Mini Pupper 2, a measured >50 ms transport delay transforms the locomotion task from a standard Markov decision process into a partially observable one. In this paper, we take a biologically inspired approach of handling noisy and delayed feedback to close the sim-to-real gap, thereby expanding the capability of reinforcement learning on cost-constrained hardware. Using a low-cost quadrupedal hardware platform, we find that using a forward model of the average actuator delay, paired with a time-aware neural network results in robust locomotion. Additionally, our time-aware neural network learned a central pattern generator (CPG): a self-sustaining rhythmic gait that is robust to +320 ms latency perturbations, mirroring the CPGs found in the spinal cords of vertebrates. We posit that temporal self-organization may be a general strategy for cost-constrained locomotion.",
  "published": "2026-07-29",
  "updated": "2026-07-30",
  "year": "2026",
  "authors": [
   "Javier C. Weddington",
   "Bence P. \u00d6lveczky",
   "Stephen A. Baccus"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper finds that using a low-cost quadrupedal hardware platform, using a forward model of the average actuator delay, paired with a time-aware neural network results in robust locomotion, and posit that temporal self-organization may be a general strategy for cost-constrained locomotion.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Javier C. Weddington",
    "id": "2044961396",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "B. \u00d6lveczky",
    "id": "5452377",
    "h_index": 32,
    "papers": 61
   },
   {
    "name": "S. Baccus",
    "id": "3152203",
    "h_index": 32,
    "papers": 75
   }
  ],
  "comment": "Sim-to-real transfer, locomotion, reinforcement learning, central pattern generator",
  "topics": [
   "humanoids",
   "sim2real",
   "rl-control",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.26434v2",
  "pdf_url": "https://arxiv.org/pdf/2607.26434v2",
  "html_url": "https://arxiv.org/html/2607.26434v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.26336",
  "slug": "learning-implicit-causal-world-models-from-multi-agent-demonstrations",
  "title": "Learning Implicit Causal World Models from Multi-Agent Demonstrations",
  "abstract": "In model-based reinforcement learning, world models exist as internal simulators, but their training often conflates statistical correlations with causal mechanisms. This problem is exacerbated in multi-agent systems where physical transitions are intertwined with strategic agent intents, causing world models to fail under distribution shift. We introduce Implicit Causal World Models to recover environmental dynamics from offline demonstrations without requiring pre-defined causal graphs. By incorporating policy variance, we render world models discoverable via the sequential backdoor condition. Evaluations across coordination tasks (Two-Door, Navigation, and Giveway) demonstrate that these models provide interpretable causal representations under both full and partial observability, with model accuracy scaling directly with interventional strength.",
  "published": "2026-07-28",
  "updated": "2026-07-28",
  "year": "2026",
  "authors": [
   "Jasorsi Ghosh"
  ],
  "author_count": 1,
  "categories": [
   "cs.LG",
   "cs.MA",
   "cs.RO"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Evaluations across coordination tasks demonstrate that Implicit Causal World Models provide interpretable causal representations under both full and partial observability, with model accuracy scaling directly with interventional strength.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jasorsi Ghosh",
    "id": "1379535320",
    "h_index": 3,
    "papers": 5
   }
  ],
  "comment": "Preprint",
  "topics": [
   "world-models",
   "sim2real",
   "rl-control",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.26336v1",
  "pdf_url": "https://arxiv.org/pdf/2607.26336v1",
  "html_url": "https://arxiv.org/html/2607.26336v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.26315",
  "slug": "momo-dial-motion-mode-in-robot-manipulation-with-spatiotemporal-action",
  "title": "MoMo: Dial Motion Mode in Robot Manipulation with Spatiotemporal Action Tokenization",
  "abstract": "To operate effectively across diverse contexts, robots must not only perform manipulation tasks accurately but also adapt how their actions unfold to the task, object, and interaction setting. We ask whether this execution-level variation can be learned as a reusable behavioral factor shared across tasks. We present \\textbf{MoMo}, a two-stage imitation-learning framework consisting of a spatiotemporal action tokenizer and a behavior-cloning transformer that takes task and a continuous motion-mode condition as inputs. Across six real-robot manipulation tasks, varying this condition produces steady, dynamic, and intermediate behaviors that human raters can distinguish and that differ in joint speed, acceleration, and end-effector approach pitch. On tasks demonstrated in only one mode, MoMo transfers the unseen requested mode while largely preserving task success. Together, these results provide evidence of compositional generalization to unseen task--mode combinations and show that motion mode can be reused across tasks to control how a manipulation skill is performed.",
  "published": "2026-07-28",
  "updated": "2026-07-28",
  "year": "2026",
  "authors": [
   "Yuhan Hu",
   "Hugues Thomas",
   "Peide Huang",
   "Mouli Sivapurapu",
   "Benoit Landry",
   "Arto Kivila"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A two-stage imitation-learning framework consisting of a spatiotemporal action tokenizer and a behavior-cloning transformer that takes task and a continuous motion-mode condition as inputs is presented, showing that motion mode can be reused across tasks to control how a manipulation skill is performed.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yuhan Hu",
    "id": "49994805",
    "h_index": 11,
    "papers": 26
   },
   {
    "name": "Hugues Thomas",
    "id": "30535797",
    "h_index": 11,
    "papers": 18
   },
   {
    "name": "Peide Huang",
    "id": "2328617675",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Mouli Sivapurapu",
    "id": "3317431",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Benoit Landry",
    "id": "2005292398",
    "h_index": 10,
    "papers": 15
   },
   {
    "name": "Arto Kivila",
    "id": "2454080245",
    "h_index": 0,
    "papers": 1
   }
  ],
  "comment": "",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.26315v1",
  "pdf_url": "https://arxiv.org/pdf/2607.26315v1",
  "html_url": "https://arxiv.org/html/2607.26315v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.26056",
  "slug": "intact-isomorphic-intent-to-action-learning-for-search-free-world-mode",
  "title": "INTACT: Isomorphic Intent-to-Action Learning for Search-Free World Models",
  "abstract": "Forward latent world models predict how actions change a scene, but recover actions for a desired change only through expensive test-time search. We introduce INTACT (INtent-To-ACTion), an end-to-end JEPA that turns action-labeled, reward-free trajectories into a deployable intent-to-action interface. Each transition supplies physical intent $z_{t+1}-z_t$, while a future goal supplies deployment intent $\\operatorname{sg}(z_g)-z_t$. The architecture is isomorphic between the local and goal motion-intent backbone-input graphs through an identical four-slot grammar and shared parameters, and between supported local and goal motion-intent families through action-law semantics induced by the same predictor rather than pointwise latent equality. INTACT also provides intact transfer from RGB evidence to action-effective latent intent coordinates and from intent families to their corresponding action-law families. Asymmetric endpoint gradients ground physical successors and fix future goals as anchors, joining representation learning and control without pointwise latent matching or globally linear dynamics. The resulting coordinates support a robust distributional action law: its conditional mean serves directly as a search-free policy, while sampling remains available for diversity or optional verification. On the four official LeWM tasks, one-epoch, zero-search models reach 85.78\\%, 100.00\\%, 97.67\\%, and 97.89\\% success. Optional local CEM centered on the Direct plan reaches 96.86\\% macro success using 384 instead of 9,000 candidate sequences, reducing sampling by $23.44\\times$ while improving pure CEM by 16.00 points. One shared four-task encoder reaches 89.39\\% E5 Direct macro and improves every task over jointly trained LeWM, while predicted--expert action-family kNN tracks Direct success at $r=0.954$. Direct inference takes 2.9--5.5 ms.",
  "published": "2026-07-28",
  "updated": "2026-07-28",
  "year": "2026",
  "authors": [
   "Junhan Sun",
   "Hao Zhao",
   "Guofeng Zhang"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "INTACT (INtent-To-ACTion), an end-to-end JEPA that turns action-labeled, reward-free trajectories into a deployable intent-to-action interface is introduced, an end-to-end JEPA that serves directly as a search-free policy, while sampling remains available for diversity or optional verification.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jun Sun",
    "id": "2349194828",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Hao Zhao",
    "id": "2326064308",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Guofeng Zhang",
    "id": "2375855711",
    "h_index": 1,
    "papers": 6
   }
  ],
  "comment": "28 pages, 11 figures, including appendices",
  "topics": [
   "world-models",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.26056v1",
  "pdf_url": "https://arxiv.org/pdf/2607.26056v1",
  "html_url": "https://arxiv.org/html/2607.26056v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2607.26047",
  "slug": "s2a2-audio-visual-imitation-learning-for-manipulation-tasks-using-acou",
  "title": "S2A2: Audio-Visual Imitation Learning for Manipulation Tasks Using Acoustic Spatial Information",
  "abstract": "Acoustic information provides rich cues about object location, material properties, and changes caused by contact or motion. This paper introduces a new set of acoustic-aware manipulation tasks for imitation learning, in which robots must use auditory cues to determine manipulation targets. These tasks require sound source localization and identification for active exploration in robotic manipulation. Also, we propose a multimodal imitation learning framework, Spatial-Spectral Audio Action (S2A2), that integrates visual features with acoustic spatial and acoustic signal information for the acoustic-aware manipulation tasks. We implemented S2A2 models that integrates policies such as ACT, Diffusion Policy, VQ-BeT, and $\u03c0_0$, into our framework. Simulation experiments showed that the proposed method is the most effective for tasks requiring both position and timbre. Furthermore, real-robot experiments confirm the applicability of the proposed tasks and framework to real-world manipulation.",
  "published": "2026-07-28",
  "updated": "2026-07-28",
  "year": "2026",
  "authors": [
   "Kaneyoshi Hiratsuka",
   "Benjamin Yen",
   "Ryosuke Kojima"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A new set of acoustic-aware manipulation tasks for imitation learning, in which robots must use auditory cues to determine manipulation targets, is introduced and a multimodal imitation learning framework, Spatial-Spectral Audio Action (S2A2), is proposed that integrates visual features with acoustic spatial and acoustic signal information for the acoustic-aware manipulation tasks.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Kaneyoshi Hiratsuka",
    "id": "2453999249",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Benjamin Yen",
    "id": "2269269202",
    "h_index": 3,
    "papers": 30
   },
   {
    "name": "Ryosuke Kojima",
    "id": "145627842",
    "h_index": 11,
    "papers": 42
   }
  ],
  "comment": "Project page: https://azuma413.github.io/projects/s2a2",
  "topics": [
   "imitation-diffusion",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.26047v1",
  "pdf_url": "https://arxiv.org/pdf/2607.26047v1",
  "html_url": "https://arxiv.org/html/2607.26047v1",
  "code_url": "https://azuma413.github.io/projects/s2a2",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2607.26005",
  "slug": "pictura-perspective-view-self-play-at-scale-for-driving",
  "title": "Pictura: Perspective-View Self-Play at Scale for Driving",
  "abstract": "Self-play in simulation produces robust driving policies at scale. Demonstrations of such behavior have been made using privileged vectorized observations such as exact poses and velocities, even for occluded agents. This assumes that perception is solved and introduces a representation gap with the partial observation of a deployed agent driving from the perspective view of egocentric cameras. A common fix, distilling the privileged policy into a camera-input student, leaves the student imitating decisions its own view cannot justify. Instead, we establish perspective-view self-play as a practical training regime. We introduce Pictura, a GPU-accelerated multi-agent driving simulator that renders each agent's egocentric view at every step, mitigating the representation gap at its source. Pictura sustains up to 500K agent-steps/s (2M images/s) on a single H100. Using Pictura, we train Alberti by self-play with plain PPO. It is the first large-scale driving self-play policy trained directly from perspective images, without privileged observations. Training spans 50B agent steps for ~35M km of driving. It approaches the driving performance of its privileged vectorized counterpart, and transfers zero-shot to Waymo Open Motion Dataset layouts re-rendered in Pictura, where it outperforms privileged vectorized agents. Project page: https://valeoai.github.io/Pictura/",
  "published": "2026-07-28",
  "updated": "2026-07-28",
  "year": "2026",
  "authors": [
   "Yuan Yin",
   "Elias Ramzi",
   "Marc Lafon",
   "Valentin Charraut",
   "Victor Bares",
   "Yihong Xu",
   "\u00c9loi Zablocki",
   "Alexandre Boulch",
   "Thibault Buhet",
   "Andrei Bursuc",
   "Matthieu Cord"
  ],
  "author_count": 11,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Pictura, a GPU-accelerated multi-agent driving simulator that renders each agent's egocentric view at every step, mitigating the representation gap at its source, is introduced, establishing perspective-view self-play as a practical training regime.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yuan Yin",
    "id": "2320822162",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Elias Ramzi",
    "id": "2321572133",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "M. Lafon",
    "id": "2184275262",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Valentin Charraut",
    "id": "2282539495",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "V. Bares",
    "id": "2111671368",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Yihong Xu",
    "id": "2257442712",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "\u00c9. Zablocki",
    "id": "39541096",
    "h_index": 13,
    "papers": 33
   },
   {
    "name": "A. Boulch",
    "id": "2394072288",
    "h_index": 2,
    "papers": 13
   },
   {
    "name": "Thibault Buhet",
    "id": "52163354",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Andrei Bursuc",
    "id": "2370933469",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Matthieu Cord",
    "id": "2238525173",
    "h_index": 12,
    "papers": 34
   }
  ],
  "comment": "",
  "topics": [
   "egocentric-data",
   "sim2real"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.26005v1",
  "pdf_url": "https://arxiv.org/pdf/2607.26005v1",
  "html_url": "https://arxiv.org/html/2607.26005v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.25989",
  "slug": "untangling-co-drift-proactive-multi-intent-failure-prediction-and-root",
  "title": "Untangling Co-Drift: Proactive Multi-Intent Failure Prediction and Root-Cause Disambiguation for Self-Driving Networks",
  "abstract": "The vision of self-driving networks that monitor, reason, and act upon themselves with minimal human intervention relies on tightly coupled monitoring, analytics, and actuation functions. In this work, we treat these functions as three operational macro-intents: continuous telemetry, real-time analytics, and programmatic actuation, and formalize the health of each function as an intent that the network must continuously satisfy. A critical, yet underexplored, challenge stems from the causal coupling among these intents, where a singular fault within one macro-intent propagates as a co-drift and subsequently triggers cascading, symptomatic anomalies across the remaining intents. This ambiguity makes it exceedingly difficult for existing, reactive approaches to distinguish the true root-cause intent from symptomatic victim intents, and their reliance on threshold-crossing detection leaves insufficient time for proactive remediation. We introduce MILD, a novel framework that reformulates intent assurance from reactive drift detection to proactive failure prediction. Grounded in our three-macro-intent formulation of the self-driving control loop, MILD employs a teacher-augmented Mixture-of-Experts architecture with a hybrid objective that jointly optimizes intent failure prediction and root-cause attribution. MILD enables KPI-level diagnostics via SHAP explainability and dynamic intent failure urgency estimation via multi-horizon modeling. Our extensive evaluation of MILD across three environments of increasing realism, from a controlled statistical benchmark, to a microservices application, to an SDN-based edge-to-cloud testbed, demonstrates that MILD achieves high failure detection rates, strong remediation lead times, and accurate intent-level root-cause disambiguation. This positions MILD as a practical enabler of closed-loop assurance in next-generation autonomous networks.",
  "published": "2026-07-28",
  "updated": "2026-07-28",
  "year": "2026",
  "authors": [
   "Md. Kamrul Hossain",
   "Walid Aljoby"
  ],
  "author_count": 2,
  "categories": [
   "cs.NI",
   "cs.LG",
   "cs.RO"
  ],
  "primary_category": "cs.NI",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work introduces MILD, a novel framework that reformulates intent assurance from reactive drift detection to proactive failure prediction, and employs a teacher-augmented Mixture-of-Experts architecture with a hybrid objective that jointly optimizes intent failure prediction and root-cause attribution.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "M. Hossain",
    "id": "2370471192",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "W. Aljoby",
    "id": "1691813",
    "h_index": 7,
    "papers": 34
   }
  ],
  "comment": "Under review in IEEE Transactions on Network and Service Management",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.25989v1",
  "pdf_url": "https://arxiv.org/pdf/2607.25989v1",
  "html_url": "https://arxiv.org/html/2607.25989v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.25918",
  "slug": "dc-wam-dynamic-centric-visual-supervision-and-reasoning-for-world-acti",
  "title": "DC-WAM: Dynamic-Centric Visual Supervision and Reasoning for World-Action Models",
  "abstract": "World-Action Models (WAMs) augment robot policies with future visual prediction, but it remains unclear what the visual modality should learn for control. While photorealistic future prediction provides dense supervision, it also incurs substantial computation and can allocate capacity to texture, illumination, and background variations that are only weakly related to action selection. Recent efficient WAM variants suggest that the main benefit of the video branch may not lie in the rendered future itself, but in the control-relevant visual representations induced during training. In this work, we revisit future video prediction from a dynamic-centric perspective and ask whether an existing RGB-based WAM can be redirected from appearance-dominated reconstruction toward interaction-induced visual dynamics without introducing additional modality-specific predictions or online inputs at deployment. We propose DC-WAM, a dynamic-centric WAM framework that redistributes supervision and computation in the RGB video branch. At the supervision level, DC-WAM combines temporal-difference flow matching with trajectory-guided weighting, emphasizing dense temporal changes and localized regions where the gripper, manipulated objects, and contact areas move. At the reasoning level, DynaRoute predicts token-wise dynamic relevance and converts it into an attention bias, guiding the model toward control-relevant future tokens. Experiments in simulation and on real-world manipulation tasks show that DC-WAM consistently improves policy performance, especially under out-of-distribution perturbations in lighting, object appearance, and background texture.",
  "published": "2026-07-28",
  "updated": "2026-07-28",
  "year": "2026",
  "authors": [
   "Haoyuan Ji",
   "Lingxiang Fan",
   "Shang Su",
   "Yinqiao Lu",
   "Mengkai Shi",
   "Jun Gao",
   "Shuo Feng"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes DC-WAM, a dynamic-centric WAM framework that redistributes supervision and computation in the RGB video branch that consistently improves policy performance, especially under out-of-distribution perturbations in lighting, object appearance, and background texture.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Haoyuan Ji",
    "id": "2288273640",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Lingxiang Fan",
    "id": "2397439599",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Shang Su",
    "id": "2454002707",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yinqiao Lu",
    "id": "2454067173",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Mengkai Shi",
    "id": "2454013357",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Jun Gao",
    "id": "2341251728",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Shuo Feng",
    "id": "2376159357",
    "h_index": 1,
    "papers": 4
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.25918v1",
  "pdf_url": "https://arxiv.org/pdf/2607.25918v1",
  "html_url": "https://arxiv.org/html/2607.25918v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.25895",
  "slug": "hifi-umi-learning-deployable-manipulation-policies-from-high-fidelity",
  "title": "HiFi-UMI: Learning Deployable Manipulation Policies from High-Fidelity UMI Data Alone",
  "abstract": "Learning deployable manipulation policies is bottlenecked by the scarcity of data that is both high-fidelity and scalable. Real-robot teleoperation is accurate but costly to scale; robot-free UMI capture scales readily, and current practice uses the resulting data mainly for pre-training, adding a small real-robot \"anchor\" at post-training. We ask whether raising the fidelity of robot-free UMI data, rather than shrinking the real-robot fraction, can remove that anchor. We present HiFi-UMI, a portable UMI data-production system co-designed for trajectory accuracy, inter-gripper relative pose, synchronization, and field of view: head-mounted offline stereo-inertial SLAM, native rather than reconstructed relative pose, a shared microsecond GPIO trigger, and two wide-angle cameras per hand covering ~200 degrees. It reaches 3 mm workspace-local end-effector accuracy without external tracking infrastructure. Using this corpus, we demonstrate zero-robot post-training: a policy post-trained solely on HiFi-UMI demonstrations deploys directly on a real robot and matches in-domain teleoperation across three backbones spanning the vision-language-action and world-action-model families, with success-rate differences of -2.5, +3.1, and -0.6 percentage points on StarVLA-QwenPI, OpenPI-pi_0.5, and LingBot-VA; the strongest policy reaches 85% on a precision insertion task, even though the teleoperation baseline is collected in the evaluation scene and no HiFi-UMI trajectory is. Pre-training on 4,000 hours from the same corpus lowers action error on ten unseen tasks by 41% and, on StarVLA-QwenPI, raises real-robot success by a further 18.1 percentage points. We open-source HiFi-UMI-2K, 2,000 hours of microsecond-synchronized, ultra-wide-FoV demonstrations, each automatically reconstructed and validated through simulation replay, as a large-scale, high-fidelity resource for the robot-learning community.",
  "published": "2026-07-28",
  "updated": "2026-07-28",
  "year": "2026",
  "authors": [
   "Simple AI",
   " :",
   "Yuteng Wei",
   "Jinming Ma",
   "Jiawei Wang",
   "Weitao Zhou",
   "Yushen Zuo",
   "Ke Rui",
   "Minglei Li",
   "Jinhao Zhang",
   "Zhikang Pan",
   "Xiang Wang",
   "Haoran Jia",
   "Huan Du",
   "Zicheng Zeng",
   "Jun Ma",
   "Guiyu Qin",
   "Di Zhang",
   "Xiaofei Li"
  ],
  "author_count": 19,
  "categories": [
   "cs.RO",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "It is asked whether raising the fidelity of robot-free UMI data, rather than shrinking the real-robot fraction, can remove that anchor at post-training, and open-source HiFi-UMI, a portable UMI data-production system co-designed for trajectory accuracy, inter-gripper relative pose, synchronization, and field of view.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Simple AI Yuteng Wei",
    "id": "2454060763",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Jinming Ma",
    "id": "2383577701",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Jiawei Wang",
    "id": "2448016120",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Weitao Zhou",
    "id": "2148943699",
    "h_index": 7,
    "papers": 29
   },
   {
    "name": "Yushen Zuo",
    "id": "2331174834",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Ke Rui",
    "id": "2448130262",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Minglei Li",
    "id": "2448263547",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Jinhao Zhang",
    "id": "2348679757",
    "h_index": 1,
    "papers": 10
   },
   {
    "name": "Zhi Pan",
    "id": "2449470231",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Xiang Wang",
    "id": "2457308569",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Haoran Jia",
    "id": "2447074517",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Hua Du",
    "id": "2065005586",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Zicheng Zeng",
    "id": "2406449860",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Jun Ma",
    "id": "2452250522",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Guiyu Qin",
    "id": "2453997812",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Di Zhang",
    "id": "2381837030",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Xiaofei Li",
    "id": "2454201183",
    "h_index": 0,
    "papers": 2
   }
  ],
  "comment": "33 pages, 15 figures, 4 tables. Project page: https://cloud.simpleai.tech/simple-world-lab/hifi-umi/ Dataset: https://huggingface.co/datasets/simple-world-lab/HiFi-UMI-2K",
  "topics": [
   "vla",
   "spatial-3d",
   "foundation-pretraining",
   "data-teleop",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.25895v1",
  "pdf_url": "https://arxiv.org/pdf/2607.25895v1",
  "html_url": "https://arxiv.org/html/2607.25895v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.25798",
  "slug": "transformer-transformer-a-unified-model-for-motion-conditioned-robot-c",
  "title": "Transformer Transformer: A Unified Model for Motion-Conditioned Robot Co-design",
  "abstract": "An often overlooked factor of robot manipulation performance is the embodiment of the robot itself. Motivated by this problem, we study motion-conditioned robot co-design, where the goal is to generate complete robot designs that track target end-effector trajectories (from human demonstrations) while optimizing user-defined rewards. We introduce Transformer Transformer, a diffusion transformer trained on RoboTokens, a unified tokenization of robot embodiments, states, and actions. The same architecture can be used across embodiment spaces (e.g., wheeled bimanual, quadrupeds, humanoids) and use cases (embodiment generation, cross embodiment controller). Rather than overfitting to one reward function, Transformer Transformer is a dynamics model, whose reward-agnostic state and action predictions can be converted into reward-specific value predictions. These value predictions are used to steer embodiment diffusion towards high value robot designs, through a procedure we call Dynamics Self-Guidance. Experiments across multiple design spaces show zero-shot optimization of unseen rewards and trajectories, improving performance and runtime over the evolutionary baseline. Finally, we fabricated an optimized ALOHA design, which reduced tracking error by over 70% compared to the original design.",
  "published": "2026-07-28",
  "updated": "2026-07-28",
  "year": "2026",
  "authors": [
   "Huy Ha",
   "C. Karen Liu",
   "Shuran Song"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work fabricated an optimized ALOHA design, which reduced tracking error by over 70% compared to the original design, and introduced Transformer Transformer, a diffusion transformer trained on RoboTokens, a unified tokenization of robot embodiments, states, and actions.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Huy Ha",
    "id": "2291134164",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Karen Liu",
    "id": "2286671296",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Shuran Song",
    "id": "2289085682",
    "h_index": 7,
    "papers": 13
   }
  ],
  "comment": "26 pages, 12 figures, 20 tables. Project page: https://transformer-transformer.github.io",
  "topics": [
   "humanoids",
   "egocentric-data",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.25798v1",
  "pdf_url": "https://arxiv.org/pdf/2607.25798v1",
  "html_url": "https://arxiv.org/html/2607.25798v1",
  "code_url": "https://transformer-transformer.github.io",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2607.25781",
  "slug": "tripody-an-overconstrained-3-spr-like-parallel-robot-for-high-reach-co",
  "title": "Tripody: An Overconstrained 3-SPR-like Parallel Robot for High-Reach Construction Tasks",
  "abstract": "Many ceiling construction tasks still rely on heavy serial manipulators that are difficult to deploy in cluttered interiors, motivating lightweight, field-ready alternatives that reach ceiling height while maintaining millimeter-level accuracy and the stiffness demanded by overhead tool loads. We introduce Tripody, a wheeled 3-DoF parallel robot for high-reach tasks that replaces the base spherical joints of a classical 3-SPR (3 legs; S: base spherical joint; P: actuated prismatic joint; R: end-effector revolute joint) morphology with universal joints, intentionally overconstraining the mechanism; small, distributed elastic deflections absorb the resulting incompatibilities, preserving predominantly translational motion. The 33kg system extends from 1.7m to 3.4m in height, supports a continuous 32kg payload, and offers a modular end-effector interface for ceiling operations. We detail the mechanical design - including custom linear actuators and a kinematic-compatibility analysis - and a control stack for accurate positioning that combines SE(3) state estimation, forward kinematics, and task-space control. In experiments, Tripody exhibits similar in-plane stiffness to a spherical-base variant but substantially higher torsional stiffness - an increase of 67% at 1.7m, 196% at 2.6m, and 454% at 3.4m - while maintaining negligible cross-axis coupling. Closed-loop positioning with a total station converges below 0.6mm across the entire workspace; pure model extrapolation achieves a 95th-percentile error of 2.7mm (max 3.6mm). Finally, we demonstrate task-level ceiling-drilling feasibility in an open-loop study by drilling a 15-hole pattern with 4.5mm maximum relative hole-position error after rigid alignment. These results support overconstrained, compliance-absorbing 3-SPR-like architectures as a practical path to lightweight, high- reach, millimeter-accurate construction robots.",
  "published": "2026-07-28",
  "updated": "2026-07-28",
  "year": "2026",
  "authors": [
   "Julien Kindle",
   "Jakub Raczy",
   "Riccardo Balbi",
   "Andrea Alessandretti",
   "Cesar Cadena",
   "Marco Hutter"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "RA-L",
  "venue_source": "semantic-scholar",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "10.1109/LRA.2026.3713724",
  "oa_pdf": "https://arxiv.org/pdf/2607.25781",
  "s2_authors": [
   {
    "name": "Julien Kindle",
    "id": "1522048638",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "J. R\u0105czy",
    "id": "87088103",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Riccardo Balbi",
    "id": "2450360604",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "A. Alessandretti",
    "id": "1966842",
    "h_index": 11,
    "papers": 34
   },
   {
    "name": "C\u00e9sar Cadena",
    "id": "2254273659",
    "h_index": 9,
    "papers": 24
   },
   {
    "name": "Marco Hutter",
    "id": "2273650854",
    "h_index": 4,
    "papers": 13
   }
  ],
  "comment": "8 pages, 11 figures, Accepted in July 2026 at IEEE Robotics and Automation Letters (RA-L)",
  "topics": [
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.25781v1",
  "pdf_url": "https://arxiv.org/pdf/2607.25781v1",
  "html_url": "https://arxiv.org/html/2607.25781v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2607.25731",
  "slug": "tri-manual-visuomotor-imitation-learning-of-robot-policies",
  "title": "Tri-Manual Visuomotor Imitation Learning of Robot Policies",
  "abstract": "Bimanual teleoperation provides an effective way to collect robot demonstrations, but it assumes that the operator and robot have matching numbers of simultaneous control channels. This assumption breaks for tri-manual systems: the robot can coordinate three arms concurrently, whereas a single operator can continuously control only two. Pairwise mode switching may therefore record otherwise independent motions sequentially, causing behaviour cloning to reproduce delays imposed by the interface rather than required by the task. We present TriManPolicy, a tri-manual imitation learning system that allows one operator to demonstrate behaviours for three arms. Its central component is Dependency-Aware Tri-Arm Scheduling (DATS). The key idea is to preserve the demonstrated arm motions while reconsidering when they occur. DATS retimes demonstrations offline by preserving local sensorimotor segments of fixed duration and repositioning them according to constraints on task order and arm usage that are reviewed by a human. The resulting data train a single synchronous policy for all three arms, while deployment requires neither the dependency graph nor the scheduler. Across six challenging tasks performed in the real world, policies trained on demonstrations retimed by DATS exhibit more efficient coordination while maintaining comparable observed task success. Offline analysis further shows that DATS changes the supervision across arms rather than merely removing idle periods. Project videos and additional material are available at https://aus.bot/trimanpolicy/.",
  "published": "2026-07-28",
  "updated": "2026-08-11",
  "year": "2026",
  "authors": [
   "James Zhao",
   "Mingyuan Ba",
   "Weiming Zhi"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "TriManPolicy is presented, a tri-manual imitation learning system that allows one operator to demonstrate behaviours for three arms while reconsidering when they occur, and policies trained on demonstrations retimed by DATS exhibit more efficient coordination while maintaining comparable observed task success.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "James Zhao",
    "id": "2454064283",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Mingyuan Ba",
    "id": "2453997493",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Weiming Zhi",
    "id": "32184659",
    "h_index": 8,
    "papers": 28
   }
  ],
  "comment": "9 pages, 9 figures. Project page: https://aus.bot/research/trimanpolicy/ . Equal contribution by James Zhao and Mingyuan Ba. Added the project website and clarified the qualitative figure captions",
  "topics": [
   "imitation-diffusion",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.25731v4",
  "pdf_url": "https://arxiv.org/pdf/2607.25731v4",
  "html_url": "https://arxiv.org/html/2607.25731v4",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.25728",
  "slug": "shared-voxel-map-based-cooperative-indoor-uav-guidance-with-a-multi-ag",
  "title": "Shared Voxel-Map-Based Cooperative Indoor UAV Guidance with a Multi-Agent Soft Actor-Critic Controller",
  "abstract": "This paper presents a cooperative indoor UAV guidance framework that combines a shared voxel-map world model with a multi-agent Soft Actor-Critic (MASAC) controller. Multiple drones fuse 360 LiDAR observations into a common world-frame occupancy map, which is converted into a compact bird's-eye-view (BEV) representation and provided to each agent as an ego-aligned local crop. This integrate-in-world, act-in- ego design enables consistent multi-UAV spatial fusion whilst retaining decentralised continuous control. The policy combines BEV map features, near-field obstacle observations, and compact goal and peer-state information within a centralised-training, decentralised-execution framework. In simulation, the learned controller achieves a 90.3% success rate in corridor navigation, outperforming Astar planning, an artificial potential field controller, and a prior guidance method. To address residual sim-to-real mismatch, the simulation-trained policy is further adapted using offline imitation fine-tuning from real-world data. Real-world experiments in GNSS-denied indoor environments demonstrate stable two-UAV cooperative operation across increasingly chal- lenging obstacle layouts. The results show that shared voxel-map representations provide an effective and scalable spatial substrate for learned cooperative indoor UAV guidance.",
  "published": "2026-07-28",
  "updated": "2026-07-28",
  "year": "2026",
  "authors": [
   "Thomas Hickling",
   "Dylan Wynne",
   "Yu Su",
   "Nabil Aouf"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The results show that shared voxel-map representations provide an effective and scalable spatial substrate for learned cooperative indoor UAV guidance and are adapted using offline imitation fine-tuning from real-world data.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Thomas Hickling",
    "id": "2347531687",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Dylan Wynne",
    "id": "2453997760",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yuhuang Su",
    "id": "2435690165",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Nabil Aouf",
    "id": "2266772237",
    "h_index": 5,
    "papers": 24
   }
  ],
  "comment": "11 pages, 17 figures",
  "topics": [
   "world-models",
   "sim2real",
   "rl-control",
   "spatial-3d",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.25728v1",
  "pdf_url": "https://arxiv.org/pdf/2607.25728v1",
  "html_url": "https://arxiv.org/html/2607.25728v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.25593",
  "slug": "when-does-legacy-data-start-to-help-emergent-transfer-in-cross-configu",
  "title": "When Does Legacy Data Start to Help? Emergent Transfer in Cross-Configuration Robot Learning",
  "abstract": "Robotic hardware evolves over time, but demonstration data is often tied to a specific sensor and actuator configuration. This raises a practical and underexplored question: when does legacy data begin to benefit an upgraded robot? We study this question on a wheeled humanoid platform across two hardware generations, where both the camera and gripper are changed while the overall morphology remains fixed. Contrary to the common assumption that more cross-configuration data is always helpful, we observe a grokking-like transition: legacy data remains ineffective until the upgraded configuration acquires a minimum level of task competence, after which co-training gains rise sharply before diminishing near saturation. We hypothesize that this task-dependent transition is governed by a transfer threshold and characterize the resulting three-phase pattern. Across real-robot manipulation tasks, we observe all three phases: no measurable benefit at low competence ($10.0\\% \\rightarrow 10.0\\%$), a sharp gain after crossing the threshold ($23.3\\% \\rightarrow 86.7\\%$ on flower insertion), and diminishing returns at high competence ($85.0\\% \\rightarrow 93.3\\%$ on pen insertion). We provide a theoretical account based on gradient alignment and residual policy uncertainty, and derive a phase-aware rule for deciding when to collect more new-hardware data and when to reuse legacy demonstrations. We further validate this three-phase pattern on a mobile dual-arm watering task, with results consistent with our predictions.",
  "published": "2026-07-28",
  "updated": "2026-07-28",
  "year": "2026",
  "authors": [
   "Tao Wang",
   "Hudson Hou",
   "Yingdong Hu",
   "Yufeng Liu",
   "Qinghai Li",
   "Yingjie Jiang",
   "Yingzhi Wang",
   "Cheng Ma",
   "Richard Wang",
   "Yang Gao"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "A theoretical account based on gradient alignment and residual policy uncertainty is provided, and a phase-aware rule for deciding when to collect more new-hardware data and when to reuse legacy demonstrations is derived.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Tao Wang",
    "id": "2452307667",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Hudson Hou",
    "id": "2454000423",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yingdong Hu",
    "id": "2149297811",
    "h_index": 13,
    "papers": 30
   },
   {
    "name": "Yufeng Liu",
    "id": "2386638329",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Qinghai Li",
    "id": "2449152552",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yingjie Jiang",
    "id": "2453582075",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yingzhi Wang",
    "id": "2454071591",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Cheng Ma",
    "id": "2454060231",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Richard Wang",
    "id": "2454231068",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yang Gao",
    "id": "2382886824",
    "h_index": 2,
    "papers": 8
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.25593v1",
  "pdf_url": "https://arxiv.org/pdf/2607.25593v1",
  "html_url": "https://arxiv.org/html/2607.25593v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2607.25541",
  "slug": "p3-probabilistic-policy-propagation-for-stable-vae-based-robot-learnin",
  "title": "P3: Probabilistic Policy Propagation for Stable VAE-Based Robot Learning",
  "abstract": "Variational Autoencoders are widely used to encode high-dimensional and noisy observations in robotics. However, their stochastic latent creates a mismatch with Proximal Policy Optimization (PPO): an effective policy marginalizes over the latent distribution, whereas former implementations estimate its probability ratio and KL divergence using only one latent sample. We identify a fundamental but overlooked theoretical cause: naive single-sample approximations in stochastic latent space induce significant variance and bias in the surrogate loss. To address this, we introduce P^3 (Probabilistic Policy Propagation), a distribution-aware optimization framework for VAE-based policies. $P^3$ couples moment-based probabilistic method for stable and efficient learning with sampling-based calibration for robust policy behavior under latent uncertainty. In our experiments, P^3 boosts data efficiency from 64.6% to >96%, reduces convergence steps by >20%. Furthermore, P^3 is evaluated on challenging humanoid parkour tasks and shows an effective foundation for VAE-based PPO. Code is available at https://github.com/ylyem9x/P3_Open.",
  "published": "2026-07-28",
  "updated": "2026-07-28",
  "year": "2026",
  "authors": [
   "Liyun Yan",
   "Jianming Ma",
   "Yang Zhang",
   "Shengcheng Fu",
   "Zhanxiang Cao",
   "Keqi Zhu",
   "Yizhi Chen",
   "Yue Gao"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work introduces P^3 (Probabilistic Policy Propagation), a distribution-aware optimization framework for VAE-based policies that couples moment-based probabilistic method for stable and efficient learning with sampling-based calibration for robust policy behavior under latent uncertainty.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Liyun Yan",
    "id": "2441307187",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Jianming Ma",
    "id": "2328977598",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Yang Zhang",
    "id": "2320900342",
    "h_index": 3,
    "papers": 16
   },
   {
    "name": "Shengcheng Fu",
    "id": "2409890795",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Zhanxiang Cao",
    "id": "2357239038",
    "h_index": 3,
    "papers": 18
   },
   {
    "name": "Keqi Zhu",
    "id": "2292175623",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Yizhi Chen",
    "id": "2368555401",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Yue Gao",
    "id": "2257030263",
    "h_index": 4,
    "papers": 20
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.25541v1",
  "pdf_url": "https://arxiv.org/pdf/2607.25541v1",
  "html_url": "https://arxiv.org/html/2607.25541v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.25516",
  "slug": "a-causality-aware-infer-diagnose-refine-framework-for-test-time-modali",
  "title": "A Causality-aware Infer-diagnose-refine Framework for Test-time Modality Adaptation in VLA Models",
  "abstract": "Vision-language-action (VLA) models predict sequential actions to execute tasks specified by language instructions, conditioned on visual observations and proprioceptive states. However, how to fuse modalities in VLA models remains an open problem, since robot manipulation involves dynamic phases, such as long-distance movements and close-range interactions, in which the importance of visual observations may vary over time. In this paper, we propose an infer-diagnose-refine (IDR) framework, a model-agnostic framework that can be integrated with diverse VLA architectures for refining action predictions at test time. IDR first infers actions under factual and counterfactual scenarios of visual observations, and then diagnoses the causal effects of visual observations as the estimated dynamic importance, which is finally used to refine the action predictions in a training-free manner. We further design a causality-aware action refiner to realize the IDR framework, including zero-padding interventions for inferring counterfactual actions, norm-based quantification for diagnosing causal effects, and gated residual fusion for refining actions. Extensive experiments on both simulation benchmarks and real-world tasks show improvements in overall performance across multiple VLA backbones, demonstrating the efficacy of dynamically adjusting visual importance at test time.",
  "published": "2026-07-28",
  "updated": "2026-07-28",
  "year": "2026",
  "authors": [
   "Haoyu Zhang",
   "Yuwei Wu",
   "Jin Chen",
   "Gao Zhi",
   "Zhenxin Diao",
   "Mingyang Gao",
   "Kun Wu",
   "Yongchun Liu",
   "Fan Li"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper proposes an infer-diagnose-refine (IDR) framework, a model-agnostic framework that can be integrated with diverse VLA architectures for refining action predictions at test time, and designs a causality-aware action refiner to realize the IDR framework.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Haoyu Zhang",
    "id": "2276656133",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Yuwei Wu",
    "id": "150352923",
    "h_index": 20,
    "papers": 86
   },
   {
    "name": "Jin Chen",
    "id": "2373548997",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Gao Zhi",
    "id": "2453964961",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Zhenxin Diao",
    "id": "2438735611",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Mingyang Gao",
    "id": "2361657923",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Kun Wu",
    "id": "2112562410",
    "h_index": 12,
    "papers": 36
   },
   {
    "name": "Yongchun Liu",
    "id": "2451289348",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Fangqing Li",
    "id": "2444683049",
    "h_index": 0,
    "papers": 2
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "sim2real"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.25516v1",
  "pdf_url": "https://arxiv.org/pdf/2607.25516v1",
  "html_url": "https://arxiv.org/html/2607.25516v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.25397",
  "slug": "decompose-and-reorganize-planning-with-primitives-and-visuomotor-polic",
  "title": "Decompose and Reorganize: Planning with Primitives and Visuomotor Policies Learned from Demonstrations",
  "abstract": "Successfully automating dexterous, long-horizon robotic manipulation requires frameworks capable of both high-level reasoning and fine-grained execution. Traditional task and motion planning (TAMP), while excellent at symbolic planning, is often brittle in contact-rich operations. Simultaneously, imitation learning (IL), while effective in manipulation tasks with visual feedback, is limited by its low capability in spatial generalization and multi-stage operation. To reconcile their complementary strengths and limitations, we propose DR-LfD (Decomposed and Reorganized Skills Learned from Demonstrations), a framework that seamlessly integrates visuomotor policies into a TAMP-gated decision-making system. Based on contact relationships, DR-LfD decomposes human demonstrations into atomic skills, which are reproduced as visuomotor policies or object-centric primitives. The initiation, termination, and constraints of the visuomotor policies are carefully modeled and implemented in a TAMP-compatible form, enabling reorganization of skills learned from different sources. DR-LfD transforms the learning problem from one requiring exponential demonstration data over possible skill sequences to one whose demonstration burden scales with the number of distinct skill types, with limited data for each skill. Through comprehensive real-world and simulation benchmarking across diverse scenarios, we demonstrate the strong performance of DR-LfD on tasks involving multiple steps, unseen setups, and physical constraints. Project website: https://dr-lfd.github.io/DR-LfD-website.",
  "published": "2026-07-28",
  "updated": "2026-07-28",
  "year": "2026",
  "authors": [
   "Yizhou Chen",
   "Hang Xu",
   "Dongjie Yu",
   "Yupu Lu",
   "Tengye Xu",
   "Zeqing Zhang",
   "Wei Zhang",
   "Yi Ren",
   "Ben M. Chen",
   "Jia Pan"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The proposed DR-LfD (Decomposed and Reorganized Skills Learned from Demonstrations) framework is a framework that seamlessly integrates visuomotor policies into a TAMP-gated decision-making system, enabling reorganization of skills learned from different sources.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yizhou Chen",
    "id": "2306829484",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Hang Xu",
    "id": "2306863533",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Dongjie Yu",
    "id": "2307759380",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Yupu Lu",
    "id": "1897570842",
    "h_index": 6,
    "papers": 18
   },
   {
    "name": "Tengye Xu",
    "id": "2320264772",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Zeqing Zhang",
    "id": "2288259321",
    "h_index": 4,
    "papers": 19
   },
   {
    "name": "Wei Zhang",
    "id": "2274892274",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Yi Ren",
    "id": "2307995547",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Ben M. Chen",
    "id": "2269691404",
    "h_index": 7,
    "papers": 41
   },
   {
    "name": "Jia Pan",
    "id": "2307900103",
    "h_index": 3,
    "papers": 11
   }
  ],
  "comment": "21 pages, 12 figures",
  "topics": [
   "dexterous-manipulation",
   "egocentric-data",
   "tactile",
   "sim2real",
   "imitation-diffusion"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.25397v1",
  "pdf_url": "https://arxiv.org/pdf/2607.25397v1",
  "html_url": "https://arxiv.org/html/2607.25397v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.25337",
  "slug": "temporal-distance-jepa-plan-aware-representation-learning-for-latent-w",
  "title": "Temporal-Distance JEPA: Plan-Aware Representation Learning for Latent World Model Predictive Control",
  "abstract": "Joint-Embedding Predictive Architectures (JEPAs) learn world models by predicting in representation space rather than reconstructing pixels, making them a natural backbone for latent model predictive control from offline demonstration logs. JEPA-style training optimizes short-horizon latent prediction, whereas planning requires a multi-step ranking of imagined futures by goal progress. Prior JEPA planners often inherit that ranking from embedding geometry, typically latent Euclidean distance, which arises as a byproduct of representation learning rather than as a progress cost mined from the logs. We propose Temporal-Distance-JEPA, which retains the LeWM encoder--predictor backbone and mines a directed temporal cost from reward-free trajectories: same-trajectory step order supplies positive targets, cross-trajectory pairs act as heuristic negatives, and a rollout-consistency term matches the planner horizon. The mined supervision serves two roles: as the deployed planning cost when progress is topological, and as a representation signal that improves Euclidean planning when contact geometry dominates. Under locked evaluation, deploying the mined cost raises Two-Room success to 100.0% versus LeWM's 97.4%, while shared Euclidean planning on the same temporally trained checkpoint raises OGB-Cube by 14.2 points over LeWM and improves Push-T. Against LeWM and the concurrent RC-aux baseline under locked evaluation, Temporal-Distance-JEPA matches or exceeds both methods on every environment. Ablations show that the directed head, cross-trajectory negatives, and rollout consistency each contribute. Temporal-Distance-JEPA narrows the train--plan gap for JEPA world-model planners by discovering temporal progress structure in offline logs and co-designing cost form with plan-time deployment. Code is available at https://github.com/HKBU-KnowComp/Temporal-Distance-JEPA.",
  "published": "2026-07-28",
  "updated": "2026-07-29",
  "year": "2026",
  "authors": [
   "Jiaxin Bai",
   "Jiaxuan Xiong"
  ],
  "author_count": 2,
  "categories": [
   "cs.CL",
   "cs.RO"
  ],
  "primary_category": "cs.CL",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Temporal-Distance-JEPA narrows the train--plan gap for JEPA world-model planners by discovering temporal progress structure in offline logs and co-designing cost form with plan-time deployment.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jiaxin Bai",
    "id": "145677395",
    "h_index": 14,
    "papers": 60
   },
   {
    "name": "Jia\u2013Jie Xiong",
    "id": "2282235882",
    "h_index": 2,
    "papers": 7
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "rl-control",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.25337v2",
  "pdf_url": "https://arxiv.org/pdf/2607.25337v2",
  "html_url": "https://arxiv.org/html/2607.25337v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.25236",
  "slug": "visualpatchworld-code-world-models-as-latent-structured-representation",
  "title": "VisualPatchWorld: Code World Models as Latent Structured Representations for Planning",
  "abstract": "Different research lines use the term world model in different ways, yet they share a common aim: to capture how the world evolves under action in a form that supports perception, simulation, and planning. Two prominent realizations are neural predictors that learn dynamics in continuous vector spaces, and hand-built physics engines that expose explicit state and physical laws. Neural predictors scale from data but leave the form of the dynamics implicit; physics engines are inspectable and editable but difficult to construct at scale. We introduce VisualPatchWorld (VPW), which represents world dynamics as code. VPW first selects a qualitative dynamical form with short active probes, then fits that form's free parameters from recorded state-action traces by minimizing multi-step prediction error. The resulting programs can be rolled forward like a simulator, inspected in source form, and used inside model-predictive control; image-derived scene graphs can supply the live state at replan time. Across comparisons with prior code-based world models, VPW attains 69.0% mean planning success and exceeds the strongest code baseline by 23.5 points. The largest gains arise when choosing the correct qualitative dynamics is essential. Under the same planner, the induced models approach ground-truth engine success on navigation and grasp-rich control; a residual gap remains for contact-rich pushing, and checking a shortlist of promising plans in the engine closes most of that gap. These results establish a practical route toward automatically constructed code world models that are useful for planning. Code is available at https://github.com/HKBU-KnowComp/VisualPatchWorld/.",
  "published": "2026-07-28",
  "updated": "2026-07-28",
  "year": "2026",
  "authors": [
   "Jiaxin Bai",
   "Jiaxuan Xiong"
  ],
  "author_count": 2,
  "categories": [
   "cs.CL",
   "cs.RO"
  ],
  "primary_category": "cs.CL",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "VisualPatchWorld is introduced, which represents world dynamics as code and first selects a qualitative dynamical form with short active probes, then fits that form's free parameters from recorded state-action traces by minimizing multi-step prediction error.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jiaxin Bai",
    "id": "145677395",
    "h_index": 14,
    "papers": 60
   },
   {
    "name": "Jia\u2013Jie Xiong",
    "id": "2282235882",
    "h_index": 2,
    "papers": 7
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "dexterous-manipulation",
   "tactile",
   "sim2real",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.25236v1",
  "pdf_url": "https://arxiv.org/pdf/2607.25236v1",
  "html_url": "https://arxiv.org/html/2607.25236v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.25219",
  "slug": "song-a-photorealistic-3d-gaussian-simulation-platform-for-benchmarking",
  "title": "SONG: A Photorealistic 3D Gaussian Simulation Platform for Benchmarking Social Navigation",
  "abstract": "Social navigation has progressed from simplified 2D environments toward a more general vision-based setting, in which a robot needs to achieve socially compliant behavior purely from onboard visual observations. Yet supporting simulation platforms have not kept pace: existing options either lack visual observations, lack moving human avatars, or fall short of real-world fidelity in appearance and pedestrian behavior, offering limited support for advancing vision-based social navigation. We introduce SONG, a SOcial Navigation platform powered by 3D Gaussian splatting (3DGS). It leverages 3DGS for both scene and avatar representations, drives pedestrians using semantically grounded trajectories generated by a large language model, and synthesizes their full-body motion with a trajectory-conditioned generator to produce continuous, natural movement. On top of the platform, we curate SONG-Bench, a set of evaluation episodes stratified by difficulty, and propose a multi-dimensional metric suite covering effectiveness, safety, and social compliance. A systematic evaluation of representative navigation baselines reveals three findings: (a) vision-based social navigation is far from solved; (b) a critical safety deficit precedes social etiquette; (c) real-world data matters more than model scale. Crucially, we demonstrate that fine-tuning on our curated data effectively improves the success rate in real-world environments. We hope our platform provides a faithful and rigorous testbed for the next generation of vision-based social navigation research.",
  "published": "2026-07-28",
  "updated": "2026-07-28",
  "year": "2026",
  "authors": [
   "Weiqi Huang",
   "Dianyi Yang",
   "Jiaxin Li",
   "Shuangyi Dong",
   "Hao Xu",
   "Zan Wang",
   "Wei Liang"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "SONG is introduced, a SOcial Navigation platform powered by 3D Gaussian splatting (3DGS) that drives pedestrians using semantically grounded trajectories generated by a large language model, and synthesizes their full-body motion with a trajectory-conditioned generator to produce continuous, natural movement.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Weiqi Huang",
    "id": "2337068869",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Dianyi Yang",
    "id": "2311192060",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Jiaxin Li",
    "id": "2337070017",
    "h_index": 2,
    "papers": 11
   },
   {
    "name": "Shuangyi Dong",
    "id": "2401980360",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Haoyang Xu",
    "id": "2451021004",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Zan Wang",
    "id": "2337028259",
    "h_index": 2,
    "papers": 12
   },
   {
    "name": "Wei Liang",
    "id": "2150326083",
    "h_index": 6,
    "papers": 11
   }
  ],
  "comment": "",
  "topics": [
   "spatial-3d",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.25219v1",
  "pdf_url": "https://arxiv.org/pdf/2607.25219v1",
  "html_url": "https://arxiv.org/html/2607.25219v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.25215",
  "slug": "leveraging-semantic-maps-for-city-scale-cross-view-localization",
  "title": "Leveraging Semantic Maps for City-Scale Cross-View Localization",
  "abstract": "We want robots to localize in previously untraversed environments against commonly available prior data. Rich semantic data available from OpenStreetMap can be useful in this task. However, existing methods either ignore this semantic information, directly matching panoramas and overhead imagery, or dramatically compress the semantic information, working with a small set of fixed classes. To leverage this rich semantic information, two challenges need to be overcome. First, useful semantic information needs to be extracted from the robot's egocentric observations. Second, the observed information must be quickly associated with the large prior semantic map (e.g., up to 628 km^2). We show that VLMs are effective at both extracting relevant landmarks from panoramas, and identifying feasible correspondences between these landmarks and prior overhead landmarks. However, using VLMs to propose all correspondences scales poorly as the number of mapped landmarks increases. Instead, we propose distilling a lightweight matcher from a VLM which computes correspondences for all entities in a map. We use this output to form an observation likelihood which is fused over time with a Bayes filter to create a time series of pose estimates. To support further investigation into generalizable cross-view methods that leverage semantic information, we release a dataset of extracted semantics and evaluation trajectories spanning eleven environments, including panoramas we collected in a snowstorm and at night in Boston. We demonstrate our method, trained on a single city's fair-weather data, generalizes across location, lighting, weather, and other challenges. Code and datasets are available at https://efahnestock.github.io/loci/.",
  "published": "2026-07-28",
  "updated": "2026-07-28",
  "year": "2026",
  "authors": [
   "Ethan Fahnestock",
   "Erick Fuentes",
   "Philip R Osteen",
   "Nicholas Roy"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work shows that VLMs are effective at both extracting relevant landmarks from panoramas, and identifying feasible correspondences between these landmarks and prior overhead landmarks, and proposes distilling a lightweight matcher from a VLM which computes correspondences for all entities in a map.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "E. Fahnestock",
    "id": "151431506",
    "h_index": 4,
    "papers": 16
   },
   {
    "name": "Erick Fuentes",
    "id": "2342982758",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Philip R. Osteen",
    "id": "2305360241",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Nicholas Roy",
    "id": "2342984208",
    "h_index": 2,
    "papers": 6
   }
  ],
  "comment": "Equal contribution by Ethan Fahnestock and Erick Fuentes. 13 pages, 7 figures, and 5 tables",
  "topics": [
   "egocentric-data"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.25215v1",
  "pdf_url": "https://arxiv.org/pdf/2607.25215v1",
  "html_url": "https://arxiv.org/html/2607.25215v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.24959",
  "slug": "amortising-trajectory-optimisation-for-residual-mpc-via-implicit-conta",
  "title": "Amortising Trajectory Optimisation for Residual MPC via Implicit Contact Differentiation",
  "abstract": "Differentiable simulation can accelerate contact-rich trajectory optimisation by exposing local sensitivities of task outcomes to controls. Existing approaches either use finite differences, which are expensive and step-size sensitive; differentiate iterative contact solvers by unrolling automatic differentiation (AD), which stores a growing computation trace; or require intricate, solver-specific KKT sensitivity derivations. We introduce an AD-assisted implicit derivative for regularised smooth contacts and apply it to Mujoco MJX, based on the Implicit Function Theorem (IFT). The method differentiates the stationarity residual at the tolerance-converged solution, avoiding both solver unrolling and hand-assembled KKT systems. IFT keeps compiled temporary memory nearly constant with solver effort, changing by less than 4$\\%$ from one to ten iterations versus 10.6$\\times$ growth for unrolled AD. IFT memory grows slower with active contacts and model dimension, using 20$\\times$ less memory at 256 contacts and 6$\\times$ less at 16 contacts and 96 DoF. We further introduce optimiser distillation for residual MPC, amortising batched full-horizon iLQR into a policy that guides short-horizon residual iLQR. Across Finger, Franka, and Unitree, this raises six-step success by 28-98 percentage points over standard iLQR.",
  "published": "2026-07-27",
  "updated": "2026-07-27",
  "year": "2026",
  "authors": [
   "Daniel Layeghi",
   "Thomas Corb\u00e8res",
   "Calum Arnott",
   "Aditya Kamireddypalli",
   "Hashim Al-Obaidi",
   "Steve Tonneau",
   "Michael Mistry"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "cs.DC",
   "cs.LG",
   "eess.SY",
   "math.OC"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work introduces an AD-assisted implicit derivative for regularised smooth contacts and applies it to Mujoco MJX, based on the Implicit Function Theorem, and introduces optimiser distillation for residual MPC, amortising batched full-horizon iLQR into a policy that guides short-horizon residual iLQR.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Daniel Layeghi",
    "id": "2127881913",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Thomas Corb\u00e9res",
    "id": "2103268170",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "C. Arnott",
    "id": "2453962488",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Aditya Kamireddypalli",
    "id": "2351607537",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "H. Al-Obaidi",
    "id": "2129828975",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "S. Tonneau",
    "id": "3084582",
    "h_index": 19,
    "papers": 58
   },
   {
    "name": "M. Mistry",
    "id": "2268562468",
    "h_index": 2,
    "papers": 10
   }
  ],
  "comment": "Code available from https://github.com/calumarnott/mujoco",
  "topics": [
   "tactile"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2607.24959v1",
  "pdf_url": "https://arxiv.org/pdf/2607.24959v1",
  "html_url": "https://arxiv.org/html/2607.24959v1",
  "code_url": "https://github.com/calumarnott/mujoco",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2607.24744",
  "slug": "data-pyramid-for-embodied-manipulation-a-survey",
  "title": "Data Pyramid for Embodied Manipulation: A Survey",
  "abstract": "Multimodal foundation models learned to see and to speak by consuming the whole internet. Embodied agents admit no such shortcut, since they require data that couple observations with physical states and actions. These signals can be provided, to varying degrees, by multiple data sources. In this work, we organize the embodied data ecosystem as a \"pyramid\" spanning five complementary sources: real-robot data, UMI-style data, egocentric and exocentric data, simulation data, and general vision-language data. We organize the pyramid around the tension between scalability and robot alignment, and further characterize each source in terms of data quality, diversity, reusability, and physical fidelity. We then analyze recent embodied foundation models through the lens of their data recipes, examining how different sources are selected, aligned, and mixed during pretraining. For embodied brain models, vision-language-action models, and world-action models alike, we relate data composition to capabilities in perception, reasoning, planning, action generation, and world prediction. We close by discussing six open challenges: building large-scale tactile datasets, collecting failure and recovery data, developing scalable data-collection pipelines, aligning actions across embodiments, leveraging egocentric data for dexterous manipulation, and designing principled data recipes for robot learning. We hope this work paves the foundation for the design of next-generation embodied systems.",
  "published": "2026-07-27",
  "updated": "2026-08-08",
  "year": "2026",
  "authors": [
   "Yifan Ye",
   "Yankai Fu",
   "Yaoxu Lv",
   "Bohan Hou",
   "Jun Cen",
   "Lingdong Kong",
   "Duo Zheng",
   "Tianxing Chen",
   "Jiaming Liu",
   "Ziang Cao",
   "Yunfan Lou",
   "Wei Chow",
   "Xian Sun",
   "Yingshuo Wang",
   "Kuangzhi Ge",
   "Xiaowei Chi",
   "Xidong Zhang",
   "Zhibo Pang",
   "Yiwu Zhong",
   "Sirui Han",
   "Zhihe Lu",
   "Weihao Yuan",
   "Qifeng Chen",
   "Michael Yu Wang",
   "Yao Mu",
   "Ziwei Liu",
   "Jianfei Yang",
   "Ping Luo",
   "Shanghang Zhang"
  ],
  "author_count": 29,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "This work organizes the embodied data ecosystem as a pyramidspanning five complementary sources: real-robot data, UMI-style data, egocentric and exocentric data, simulation data, and general vision-language data, and further characterize each source in terms of data quality, diversity, reusability, and physical fidelity.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yifan Ye",
    "id": "2384823245",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Yankai Fu",
    "id": "2335992133",
    "h_index": 7,
    "papers": 16
   },
   {
    "name": "Ya-hui Lv",
    "id": "2368710300",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Bohan Hou",
    "id": "2335000929",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Jun Cen",
    "id": "2375739984",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Lingdong Kong",
    "id": "2335574081",
    "h_index": 9,
    "papers": 34
   },
   {
    "name": "Duo Zheng",
    "id": "2118974163",
    "h_index": 10,
    "papers": 17
   },
   {
    "name": "Tianxing Chen",
    "id": "2316455829",
    "h_index": 10,
    "papers": 31
   },
   {
    "name": "Jiaming Liu",
    "id": "2258602418",
    "h_index": 14,
    "papers": 37
   },
   {
    "name": "Ziang Cao",
    "id": "2113998517",
    "h_index": 15,
    "papers": 18
   },
   {
    "name": "Y. Lou",
    "id": "2006049963",
    "h_index": 2,
    "papers": 11
   },
   {
    "name": "Wei Chow",
    "id": "2299941238",
    "h_index": 8,
    "papers": 15
   },
   {
    "name": "Xian Sun",
    "id": "2306077347",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Yingshuo Wang",
    "id": "2322820983",
    "h_index": 1,
    "papers": 17
   },
   {
    "name": "Kuangzhi Ge",
    "id": "2336864814",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Xiaowei Chi",
    "id": "2192825554",
    "h_index": 14,
    "papers": 49
   },
   {
    "name": "Xidong Zhang",
    "id": "2373529061",
    "h_index": 2,
    "papers": 12
   },
   {
    "name": "Zhibo Pang",
    "id": "2312050332",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Yiwu Zhong",
    "id": "2255932423",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Sirui Han",
    "id": "2383213418",
    "h_index": 4,
    "papers": 14
   },
   {
    "name": "Zhihe Lu",
    "id": "2382317247",
    "h_index": 2,
    "papers": 11
   },
   {
    "name": "Weihao Yuan",
    "id": "11349534",
    "h_index": 15,
    "papers": 24
   },
   {
    "name": "Qifeng Chen",
    "id": "2115931029",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Michael Yu Wang",
    "id": "2248999920",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Yao Mu",
    "id": "2348606790",
    "h_index": 4,
    "papers": 18
   },
   {
    "name": "Ziwei Liu",
    "id": "2321147297",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Jianfei Yang",
    "id": "2398616773",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Ping Luo",
    "id": "2262515628",
    "h_index": 14,
    "papers": 21
   },
   {
    "name": "Shanghang Zhang",
    "id": "2268817905",
    "h_index": 10,
    "papers": 27
   }
  ],
  "comment": "Awesome Embodied Data Pyramid; Project Page at https://jasper-aaa.github.io/embodied-data-pyramid/ GitHub Repo at https://github.com/worldbench/awesome-embodied-data-pyramid",
  "topics": [
   "vla",
   "dexterous-manipulation",
   "egocentric-data",
   "tactile",
   "foundation-pretraining",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.24744v2",
  "pdf_url": "https://arxiv.org/pdf/2607.24744v2",
  "html_url": "https://arxiv.org/html/2607.24744v2",
  "code_url": "https://jasper-aaa.github.io/embodied-data-pyramid/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.3
 },
 {
  "id": "2607.24485",
  "slug": "learning-touch-augmented-vision-language-action-models-from-future-vis",
  "title": "\u03c4: Learning Touch-Augmented Vision-Language-Action Models from Future Visual Supervision",
  "abstract": "Incorporating tactile sensing into Vision-Language-Action (VLA) models holds promise for contact-rich manipulation, where visual observations alone often fail to capture critical cues about physical interactions. However, learning informative tactile representation while effectively adapting it to pretrained VLA models remains challenging under limited task-specific data. Existing methods either focus on instantaneous contact states or model temporal interaction dynamics using 6D wrench sequences, leaving high-dimensional tactile signals underexplored. To address these challenges, we present \u03c4, a touch-augmented VLA framework that learns an action-conditioned spatiotemporal tactile representation from future visual supervision inspired by the Joint-Embedding Predictive Architecture (JEPA), and fuses it with vision-language features for action generation. This supervision operates in latent space and is used only during training, adding no deployment overhead. We also introduce TacAura, a dataset of synchronized vision, proprioception, and vision-based tactile signals across four representative contact-rich manipulation tasks. Experiments show that \u03c4 outperforms existing models and generalizes to unseen objects and scenes, delivering improved manipulation performance and robustness. Project Page: https://cocacola-lab.github.io/tau-Page/.",
  "published": "2026-07-27",
  "updated": "2026-08-07",
  "year": "2026",
  "authors": [
   "Ning Cheng",
   "Jinan Xu",
   "Wanlin Li",
   "Yangzhi Chen",
   "Jing Gao",
   "Yiqun Wang",
   "Kelan Peng",
   "Wenjuan Han"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Tau is presented, a touch-augmented VLA framework that learns an action-conditioned spatiotemporal tactile representation from future visual supervision inspired by the Joint-Embedding Predictive Architecture (JEPA), and fuses it with vision-language features for action generation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ning Cheng",
    "id": "2208958032",
    "h_index": 6,
    "papers": 14
   },
   {
    "name": "Jinan Xu",
    "id": "2291982150",
    "h_index": 8,
    "papers": 14
   },
   {
    "name": "Wanlin Li",
    "id": "2357100311",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Yangzhi Chen",
    "id": "2453994527",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Jing Gao",
    "id": "2292122105",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Yiqun Wang",
    "id": "2454072079",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Kelan Peng",
    "id": "2453890518",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Wenjuan Han",
    "id": "2261343071",
    "h_index": 6,
    "papers": 15
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "vla",
   "tactile",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.24485v3",
  "pdf_url": "https://arxiv.org/pdf/2607.24485v3",
  "html_url": "https://arxiv.org/html/2607.24485v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.24481",
  "slug": "armnetbench-v0-1-parallel-real-world-evaluation-of-manipulation-polici",
  "title": "ArmnetBench v0.1: Parallel Real-World Evaluation of Manipulation Policies on a Low-Cost Arm Farm",
  "abstract": "Real-world evaluation is a bottleneck in developing generalist robot manipulation policies. Each rollout requires physical hardware and an operator to set up, reset, and score it. We introduce ArmnetBench v0.1, a benchmark run on a fleet of low-cost SO-101 cells under light on-site supervision. v0.1 validates this arm farm end to end and compares 7 policies across 12 tasks with both single-arm and bimanual configurations. Each policy is trained or fine-tuned on 50 demonstrations per task; the benchmark contains 2,518 policy rollouts and 600 reference demonstrations. All 3,118 episodes carry a three-way label (successful, suboptimal, or failure). Policy rollouts are human-scored, while demonstrations are successful by construction. Beyond evaluation, its quality-labelled trajectories support downstream learning, from reward and predictive world models to policies trained on mixed-quality data. The leaderboard is an initial comparison under this shared budget. We release the 3,118 core episodes in LeRobot v3.0 and RoboMeter formats.",
  "published": "2026-07-27",
  "updated": "2026-07-27",
  "year": "2026",
  "authors": [
   "Praveen Selvaraj",
   "Lorenzo Uttini",
   "Ville Kuosmanen"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work releases the 3,118 core episodes in LeRobot v3.0 and RoboMeter formats, a benchmark run on a fleet of low-cost SO-101 cells under light on-site supervision and validates this arm farm end to end.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Praveen Selvaraj",
    "id": "2220825490",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Lorenzo Uttini",
    "id": "2395685960",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Ville Kuosmanen",
    "id": "2453886596",
    "h_index": 0,
    "papers": 1
   }
  ],
  "comment": "11 pages, 6 tables, 3 figures. data available at https://huggingface.co/collections/armnet/armnetbench-v01",
  "topics": [
   "world-models"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.24481v1",
  "pdf_url": "https://arxiv.org/pdf/2607.24481v1",
  "html_url": "https://arxiv.org/html/2607.24481v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.24409",
  "slug": "accuracy-potential-of-visual-localization-exploiting-high-end-street-l",
  "title": "Accuracy potential of visual localization exploiting high-end street-level imagery",
  "abstract": "Accurate and reliable pose information with respect to a reference frame is increasingly demanded across applications such as autonomous navigation, surveying, robotics, and augmented and mixed reality. Visual localization can serve as a complementary positioning modality to GNSS, whose applicability and accuracy are often limited. Yet, the accuracy potential of visual localization has not been systematically investigated against survey-grade demands. This is mainly due to the lack of publicly available, large-scale outdoor datasets with ground-truth poses in the sub-centimeter range. In this work, we address both gaps. We introduce a scalable visual localization pipeline that employs precisely georeferenced, high-resolution street-level imagery directly as the scene representation. It combines prior-guided reference candidate selection with on-the-fly local Structure-from-Motion reconstruction and PnP-based pose estimation. We further present the FHNW Muttenz dataset, a real-world dataset covering a contiguous 10 km street network mapped in two mobile mapping campaigns approximately 1.5 years apart. It consists of high-resolution reference imagery and query sequences acquired by four different cameras across five representative scenes. All images are precisely co-registered, yielding 6-DoF ground-truth poses in the sub-centimeter range. Using this dataset, we evaluate the accuracy potential of visual localization. Our experiments demonstrate median pose accuracies in the range of 1-5 cm for translation and 0.05-0.1\u00b0 for rotation, reaching as low as 1 cm and 0.03\u00b0 under favorable conditions. These results show that visual localization can complement survey-grade GNSS positioning, paving the way for 3D geospatial data acquisition using consumer devices and fully automated georeferencing approaches. The dataset is publicly available at: https://fhnw-muttenz-vl-dataset.github.io/.",
  "published": "2026-07-27",
  "updated": "2026-07-27",
  "year": "2026",
  "authors": [
   "Jonas Meyer",
   "Stephan Nebiker",
   "Pascal Theiler",
   "Norbert Haala"
  ],
  "author_count": 4,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A scalable visual localization pipeline that combines prior-guided reference candidate selection with on-the-fly local Structure-from-Motion reconstruction and PnP-based pose estimation is introduced, paving the way for 3D geospatial data acquisition using consumer devices and fully automated georeferencing approaches.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jonas Meyer",
    "id": "2110861337",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "S. Nebiker",
    "id": "2745588",
    "h_index": 18,
    "papers": 96
   },
   {
    "name": "P. Theiler",
    "id": "70478979",
    "h_index": 7,
    "papers": 15
   },
   {
    "name": "Norbert Haala",
    "id": "2257002242",
    "h_index": 3,
    "papers": 13
   }
  ],
  "comment": "26 pages, 6 figures",
  "topics": [
   "spatial-3d",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.24409v1",
  "pdf_url": "https://arxiv.org/pdf/2607.24409v1",
  "html_url": "https://arxiv.org/html/2607.24409v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.24320",
  "slug": "continual-rl-for-generalization-in-autonomous-racing-on-the-roboracer",
  "title": "Continual-RL for Generalization in Autonomous Racing on the RoboRacer Platform",
  "abstract": "A key challenge in modern robotics is to adapt to changing environments, a challenge that is exacerbated when simulations cannot encompass every possible real-world configuration, and therefore Reinforcement Learning (RL) in the physical world becomes necessary. Continual Reinforcement Learning provides the tools to address this challenge; however, both the frameworks and the methods remain underexplored. Autonomous Racing and in particular the RoboRacer competition provide a testing ground for such methods, as learning to drive on a new track-floor combination with the least amount of new experience naturally frames a continual learning problem. This work tries to address this gap by proposing a continual RL framework based on Continual Backpropagation that is able, with only real-world data, to train a generalistic policy on a set of tracks and then fine- tune it within 15 minutes to outperform classical controllers. Furthermore, a comparison method based on offline RL is proposed, and a simulation analysis of the plasticity properties of the methods is conducted.",
  "published": "2026-07-27",
  "updated": "2026-07-27",
  "year": "2026",
  "authors": [
   "Joel Siegert",
   "Edoardo Ghignone",
   "Michele Magno"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work tries to address the gap in Continual Reinforcement Learning by proposing a continual RL framework based on Continual Backpropagation that is able, with only real-world data, to train a generalistic policy on a set of tracks and then fine- tune it within 15 minutes to outperform classical controllers.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Joel Siegert",
    "id": "2453887336",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Edoardo Ghignone",
    "id": "116272961",
    "h_index": 7,
    "papers": 21
   },
   {
    "name": "Michele Magno",
    "id": "2243234391",
    "h_index": 8,
    "papers": 27
   }
  ],
  "comment": "8 pages, conference",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.24320v1",
  "pdf_url": "https://arxiv.org/pdf/2607.24320v1",
  "html_url": "https://arxiv.org/html/2607.24320v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.24296",
  "slug": "pac-dp-pac-bayesian-diffusion-policy-learning",
  "title": "PAC-DP: PAC-Bayesian Diffusion Policy Learning",
  "abstract": "Diffusion Policies (DPs) are able to perform complex manipulation tasks. However, DPs are typically trained by minimizing a denoising objective, which provides limited control over generalization in the finite-data regimes common in robotics. In this letter, we propose PAC-DP, an approach that increases the performance of DPs in robotic manipulation tasks. By modeling the DP as a Bayesian neural network, and defining a PAC-Bayes generalization bound, we derive a novel training objective that augments the standard denoising loss with a Kullback-Leibler divergence regularizer between the posterior and prior parameter distributions. From the theoretical perspective, our approach provides a principled approach to regularize the training of DPs without significantly increasing the training time. From the practical point of view, experimental results demonstrate improved denoising performance, lower variational negative log-likelihood, and higher success rates across multiple robotic manipulation benchmarks. Crucially, the largest improvements are observed in low-data training regimes and complex tasks, establishing PAC-DP as a theoretically grounded framework for robot policy learning.",
  "published": "2026-07-27",
  "updated": "2026-07-27",
  "year": "2026",
  "authors": [
   "Mohammad Hasan Yeganegi",
   "Dian Yu",
   "Andrea Del Prete",
   "Majid Khadiv",
   "Matteo Saveriano"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "PAC-DP is proposed, an approach that increases the performance of DPs in robotic manipulation tasks by modeling the DP as a Bayesian neural network, and defining a PAC-Bayes generalization bound that augments the standard denoising loss with a Kullback-Leibler divergence regularizer between the posterior and prior parameter distributions.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Mohammad Hasan Yeganegi",
    "id": "147636109",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Dian Yu",
    "id": "2355234295",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Andrea Del Prete",
    "id": "2239201923",
    "h_index": 5,
    "papers": 22
   },
   {
    "name": "M. Khadiv",
    "id": "8134198",
    "h_index": 19,
    "papers": 89
   },
   {
    "name": "Matteo Saveriano",
    "id": "2296186061",
    "h_index": 6,
    "papers": 23
   }
  ],
  "comment": "",
  "topics": [
   "imitation-diffusion"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.24296v1",
  "pdf_url": "https://arxiv.org/pdf/2607.24296v1",
  "html_url": "https://arxiv.org/html/2607.24296v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.24267",
  "slug": "feelworld-visuo-tactile-world-model-for-hierarchical-contact-predictio",
  "title": "FeelWorld: Visuo-Tactile World Model for Hierarchical Contact Prediction and Planning",
  "abstract": "Humans plan physical interactions by imagining the possible outcomes of candidate actions. However, existing visual world models primarily capture appearance dynamics while overlooking the tactile states that govern contact-rich interactions, potentially producing imagined futures that appear visually plausible but violate physical dynamics. We introduce FeelWorld, a hierarchical visuo-tactile world model that jointly predicts future visual latents and three tactile states. FeelWorld organizes these states hierarchically as contact state, a 3D tactile latent that encodes force-related information, and slip state. These states are jointly predicted by a shared latent dynamics model with explicit supervision. To prevent irrelevant tactile signals during free-space motion from degrading visual prediction, we introduce a contact-gated asymmetric attention mechanism that maintains a visual-only prediction pathway before contact and enables joint visuo-tactile dynamics prediction during contact. The model is further trained with autoregressive rollouts and context noise injection to improve robustness to compounding errors. The predicted contact and slip states also support contact-aware CEM planning. Experiments on chip grasping, fruit grasping, and USB insertion show that FeelWorld reduces 10-step LPIPS from 0.084 to 0.058 and maintains an LPIPS that is 61% lower than that of the visual baseline after an 80-step autoregressive rollout. FeelWorld also achieves an average zero-shot planning success rate of 81.7%, providing an effective approach for incorporating tactile sensing into world models.",
  "published": "2026-07-27",
  "updated": "2026-07-27",
  "year": "2026",
  "authors": [
   "Wenxuan Ma",
   "Chaofan Zhang",
   "Chao Xue",
   "Yinghao Cai",
   "Guocai Yao",
   "Shaowei Cui",
   "Shuo Wang"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "To prevent irrelevant tactile signals during free-space motion from degrading visual prediction, a contact-gated asymmetric attention mechanism is introduced that maintains a visual-only prediction pathway before contact and enables joint visuo-tactile dynamics prediction during contact.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Wenxuan Ma",
    "id": "2276183477",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Chaofan Zhang",
    "id": "2256775583",
    "h_index": 7,
    "papers": 19
   },
   {
    "name": "Chao Xue",
    "id": "2454126195",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yinghao Cai",
    "id": "2230979",
    "h_index": 19,
    "papers": 89
   },
   {
    "name": "Guocai Yao",
    "id": "2376597393",
    "h_index": 6,
    "papers": 23
   },
   {
    "name": "Shaowei Cui",
    "id": "1853836031",
    "h_index": 17,
    "papers": 57
   },
   {
    "name": "Shuo Wang",
    "id": "2360886366",
    "h_index": 3,
    "papers": 13
   }
  ],
  "comment": "9 pages, 7 figures",
  "topics": [
   "world-models",
   "dexterous-manipulation",
   "tactile",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.24267v1",
  "pdf_url": "https://arxiv.org/pdf/2607.24267v1",
  "html_url": "https://arxiv.org/html/2607.24267v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.24233",
  "slug": "quality-adaptive-multi-uav-3d-reconstruction-with-sparse-workload-redi",
  "title": "Quality-Adaptive Multi-UAV 3D Reconstruction with Sparse Workload Redistribution",
  "abstract": "3D reconstruction of unknown environments is a key application in robotics but is severely limited by the computational and energy capabilities of current aerial platforms. Deploying multiple UAVs and providing efficient and scalable path planning strategies are common approaches, but effective online coordination among UAVs remains a significant challenge. To address this problem, we propose a quality-adaptive decentralized decision-making strategy to build a 3D map with user-defined degrees of fidelity. The approach integrates a quality-oriented criterion based on TSDF confidence into view generation and information gain estimation to produce viewpoints consistent with the desired fidelity target. Additionally, we employ two levels of coordination: a penalty factor in the viewpoint evaluation to encourage local dispersion among the UAVs and a global imbalance correction mechanism. The latter, based on regularized clustering and optimal task assignment, is only triggered when an unbalanced configuration relative to high-information regions is detected. Simulation results demonstrate that the proposed method improves path efficiency compared to state-of-the-art multi-UAV exploration approaches, while also achieving higher-fidelity reconstructions in terms of coverage and accuracy. We make our code publicly available to the community.",
  "published": "2026-07-27",
  "updated": "2026-07-27",
  "year": "2026",
  "authors": [
   "Benjamin Sportich",
   "Kenza Boubakri",
   "Olivier Simonin",
   "Alessandro Renzaglia"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A quality-adaptive decentralized decision-making strategy to build a 3D map with user-defined degrees of fidelity that improves path efficiency compared to state-of-the-art multi-UAV exploration approaches, while also achieving higher-fidelity reconstructions in terms of coverage and accuracy.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Benjamin Sportich",
    "id": "2005272928",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Kenza Boubakri",
    "id": "2394164943",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Olivier Simonin",
    "id": "2261646086",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Alessandro Renzaglia",
    "id": "2382368822",
    "h_index": 1,
    "papers": 4
   }
  ],
  "comment": "",
  "topics": [
   "spatial-3d",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.24233v1",
  "pdf_url": "https://arxiv.org/pdf/2607.24233v1",
  "html_url": "https://arxiv.org/html/2607.24233v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.24207",
  "slug": "floaff-kitchen-bridging-navigation-and-manipulation-via-canonical-and",
  "title": "FloAff-Kitchen: Bridging Navigation and Manipulation via Canonical and Progressive Floor Affordance Learning",
  "abstract": "Mobile manipulation requires robots to identify Floor Affordance (FloAff) that maximizes downstream manipulation success rather than merely ensuring navigation feasibility. FloAff prediction is a target-conditioned local spatial reasoning problem, yet existing methods suffer from representation ambiguity caused by irrelevant spatial context and arbitrary object orientations, while entangling shared and task-specific knowledge across heterogeneous manipulation skills. To address these challenges, we propose a unified framework for FloAff prediction from egocentric multimodal perception, consisting of canonical representation learning and progressive affordance prior learning. Specifically, we introduce a Canonical Floor Affordance Representation (CFAR), which learns canonical interaction geometry by preserving affordance-relevant local structure while eliminating nuisance spatial variations unrelated to robot base placement. We further propose Progressive Floor Affordance Learning (PFAL), which learns transferable FloAff priors from a foundation manipulation task and progressively adapts them to heterogeneous downstream manipulation skills. To facilitate systematic evaluation, we establish the first cross-scene, multi-view FloAff-Kitchen benchmark covering diverse manipulation skills, scene layouts, furniture styles, and viewpoints. Extensive experiments on three benchmark settings demonstrate that our method consistently outperforms strong baselines, while ablation studies validate the contribution of each proposed component. Project page: https://csu-hero-lab.github.io/FloAff-Kitchen_Web/",
  "published": "2026-07-27",
  "updated": "2026-07-27",
  "year": "2026",
  "authors": [
   "Ping Zhong",
   "Manling Teng",
   "Tao Wu",
   "Bolei Chen",
   "Jiazhi Xia",
   "Jianxin Wang"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A unified framework for FloAff prediction from egocentric multimodal perception is proposed, consisting of canonical representation learning and progressive affordance prior learning, and the first cross-scene, multi-view FloAff-Kitchen benchmark is established.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ping Zhong",
    "id": "2242594347",
    "h_index": 7,
    "papers": 29
   },
   {
    "name": "Manling Teng",
    "id": "2453887874",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Tao Wu",
    "id": "2379983871",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Bolei Chen",
    "id": "2152692482",
    "h_index": 8,
    "papers": 36
   },
   {
    "name": "Jiazhi Xia",
    "id": "2407150825",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Jianxin Wang",
    "id": "2188020750",
    "h_index": 5,
    "papers": 15
   }
  ],
  "comment": "",
  "topics": [
   "egocentric-data",
   "spatial-3d",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.24207v1",
  "pdf_url": "https://arxiv.org/pdf/2607.24207v1",
  "html_url": "https://arxiv.org/html/2607.24207v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.24190",
  "slug": "not-forgotten-implementation-and-evaluation-of-a-personalized-episodic",
  "title": "Not Forgotten: Implementation and Evaluation of a Personalized Episodic Memory for the Humanoid Robot Head Kim",
  "abstract": "Social robots that rely on large language models for conversation are unable to retain information across sessions. This absence of memory violates social expectations, potentially preventing the formation of persistent relationships. This paper presents a lightweight episodic memory module that integrates vector-based semantic retrieval with an LLM-controlled dialog system, deployed on the humanoid robot head Kim. The module employs a hybrid scoring function combining cosine similarity with a memory strength metric to retrieve contextually relevant past interactions and inject them into the generation prompt. The system was evaluated in a within-subjects video-based online study (N = 43) using the Human-Robot Interaction Evaluation Scale (HRIES). Results show that episodic memory significantly increased perceived sociability (d = 0.60, p < .001), with the strongest effects on perceived trustworthiness (d = 0.62) and warmth (d = 0.56). Perceived disturbance remained unchanged (d = 0.00), indicating that the implemented approach to personalized recall did not trigger privacy-related discomfort or uncanny valley effects. These findings suggest that episodic memory serves as a social lubricant in embodied Human-Robot Interaction, enhancing relational quality without eliciting negative affective responses.",
  "published": "2026-07-27",
  "updated": "2026-07-27",
  "year": "2026",
  "authors": [
   "Steve Aschenbrenner",
   "Marcel Heisler",
   "Thomas Sievers",
   "Christian Becker-Asano"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.HC"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A lightweight episodic memory module that integrates vector-based semantic retrieval with an LLM-controlled dialog system, deployed on the humanoid robot head Kim, suggests that episodic memory serves as a social lubricant in embodied Human-Robot Interaction, enhancing relational quality without eliciting negative affective responses.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Steven Aschenbrenner",
    "id": "108539395",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Marcel Heisler",
    "id": "2217471473",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Thomas Sievers",
    "id": "2272524905",
    "h_index": 4,
    "papers": 14
   },
   {
    "name": "C. Becker-Asano",
    "id": "2266560677",
    "h_index": 4,
    "papers": 16
   }
  ],
  "comment": "Acceoted at the 35th IEEE International Conference on Robot and Human Interactive Communication (RO-MAN 2026) Kitakyushu, Japan",
  "topics": [
   "humanoids",
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.24190v1",
  "pdf_url": "https://arxiv.org/pdf/2607.24190v1",
  "html_url": "https://arxiv.org/html/2607.24190v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.24159",
  "slug": "deva-decoupled-video-action-model-with-physical-guidance-for-robot-pol",
  "title": "DeVA: Decoupled Video-Action Model with physical guidance for robot policy learning",
  "abstract": "Generalizable robot manipulation requires policies that can anticipate how visual scenes evolve while executing language instructions. While recent Vision-Language-Action models benefit from large-scale pretraining, their predominantly static pretraining objectives provide limited supervision for physical dynamics and temporal causality, leaving control-relevant knowledge to be learned from downstream robot demonstrations. Video generative models offer a promising foundation by encoding rich spatiotemporal priors through future predictions. However, existing Video-Action Models either couple video and action prediction in a shared backbone, making policy adaptation harder to optimize, or under-utilize video information when guiding the action branch. In this work, we introduce DeVA, a Decoupled Video-Action model with specialized video and action experts, multi-level feature transfer, and physically salient guidance. DeVA transfers representations from multiple video layers to the action expert, enabling rich information exchange while making policy learning more tractable. It further supervises intermediate video features and the action stream with physically salient guidance (affordance/depth). Experiments on both simulation benchmarks and real-world deployment demonstrate strong performance with limited data, faster convergence than a unified architecture, and clear performance gains from physical guidance.",
  "published": "2026-07-27",
  "updated": "2026-07-27",
  "year": "2026",
  "authors": [
   "Mengqi Zhang",
   "Sahil Khose",
   "Simar Kareer",
   "Yuchen Song",
   "Unnat Jain",
   "Judy Hoffman"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "DeVA is introduced, a Decoupled Video-Action model with specialized video and action experts, multi-level feature transfer, and physically salient guidance, enabling rich information exchange while making policy learning more tractable.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Meng Zhang",
    "id": "2447863906",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Sahil Khose",
    "id": "2077429338",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Simar Kareer",
    "id": "2188833033",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Yuchen Song",
    "id": "2450407831",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Unnat Jain",
    "id": "2387697720",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Judy Hoffman",
    "id": "2328413304",
    "h_index": 4,
    "papers": 7
   }
  ],
  "comment": "Project page with videos, code, and checkpoints: https://deva-model.github.io/",
  "topics": [
   "vla",
   "sim2real",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.24159v1",
  "pdf_url": "https://arxiv.org/pdf/2607.24159v1",
  "html_url": "https://arxiv.org/html/2607.24159v1",
  "code_url": "https://deva-model.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2607.24083",
  "slug": "learning-reusable-hybrid-motion-priors-for-humanoid-locomotion-from-mo",
  "title": "Learning Reusable Hybrid Motion Priors for Humanoid Locomotion from Motion Imitation",
  "abstract": "Reinforcement learning can produce robust humanoid controllers, but each new task is typically trained as a separate policy with its own reward design and training process. Motion imitation provides an alternative source of motor competence by training policies to track retargeted human motions, yet the resulting controllers remain reference trackers and are not directly usable as task policies. We propose a three-stage pipeline that turns motion-imitation skills into a reusable hybrid motion prior (HMP) for humanoid locomotion. First, an expert policy is trained to imitate retargeted human motion-capture clips. Second, the expert is distilled into a frozen architecture composed of a proprioceptive encoder, a residual vector-quantized (RVQ) codebook, and an action decoder. Third, task-level policies are trained to solve locomotion tasks by selecting discrete codebook entries while the HMP remains frozen. We evaluate the method on velocity tracking, point-goal navigation, and fall-recovery velocity tracking in simulation, and deploy the velocity-tracking policy on a real Unitree G1 robot. The distillation process preserves the tracking behavior of the expert, while the resulting HMP can be reused without retraining as the action interface for different downstream locomotion policies. The learned HMP reveals an interpretable codebook structure in which the number of active RVQ stages modulates the available gait patterns. We further show that training the codebook with the rotation trick improves latent organization and reduces downstream falls compared with a standard straight-through estimator.",
  "published": "2026-07-27",
  "updated": "2026-07-27",
  "year": "2026",
  "authors": [
   "Valerio Belli",
   "Valerio Modugno",
   "Enrico Mingo Hoffman",
   "Fabio Amadio"
  ],
  "author_count": 4,
  "categories": [
   "cs.LG",
   "cs.RO"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A three-stage pipeline that turns motion-imitation skills into a reusable hybrid motion prior (HMP) for humanoid locomotion and shows that training the codebook with the rotation trick improves latent organization and reduces downstream falls compared with a standard straight-through estimator.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Valerio Belli",
    "id": "2335668605",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Valerio Modugno",
    "id": "48302763",
    "h_index": 15,
    "papers": 50
   },
   {
    "name": "Enrico Mingo Hoffman",
    "id": "2336951546",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Fabio Amadio",
    "id": "2336951383",
    "h_index": 7,
    "papers": 22
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "rl-control",
   "navigation"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2607.24083v1",
  "pdf_url": "https://arxiv.org/pdf/2607.24083v1",
  "html_url": "https://arxiv.org/html/2607.24083v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2607.24079",
  "slug": "effective-parameters-real-behavior-renormalization-for-robotics-from-i",
  "title": "Effective Parameters, Real Behavior: Renormalization for Robotics -- From Infinite Electron Mass to Sim-to-Real Gap",
  "abstract": "Bridging the sim-to-real gap is a central problem in robotics, and the prevailing approach is to build increasingly accurate simulators. Here, we propose another approach based on renormalization: using effective, resolution-dependent parameters to absorb details omitted by the simulator and reproduce real behavior. These parameters may differ from measured physical values because they compensate for what the simulator leaves out. We demonstrate this mechanism analytically for proportional--derivative (PD) control at finite simulation frequency, where proportional feedback changes the effective derivative gain and derivative feedback changes the effective inertia. We then interpret dynamic rope manipulation and underwater swimming through the same perspective. Finally, we present a practical procedure for choosing observables, identifying omitted physics, and determining effective parameters. Renormalization offers robotics a complementary path across the sim-to-real gap: effective parameters, real behavior.",
  "published": "2026-07-27",
  "updated": "2026-07-27",
  "year": "2026",
  "authors": [
   "Youran Sun",
   "Jiaxuan Guo",
   "Xingyu Ren",
   "Chugang Yi",
   "Haizhao Yang"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "hep-th"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Youran Sun",
    "id": "2343218681",
    "h_index": 1,
    "papers": 11
   },
   {
    "name": "Jiaxuan Guo",
    "id": "2374348773",
    "h_index": 10,
    "papers": 56
   },
   {
    "name": "Xingyu Ren",
    "id": "2310403276",
    "h_index": 3,
    "papers": 26
   },
   {
    "name": "Chugang Yi",
    "id": "2324054562",
    "h_index": 2,
    "papers": 11
   },
   {
    "name": "Haizhao Yang",
    "id": "2404064460",
    "h_index": 1,
    "papers": 9
   }
  ],
  "comment": "",
  "topics": [
   "sim2real"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.24079v1",
  "pdf_url": "https://arxiv.org/pdf/2607.24079v1",
  "html_url": "https://arxiv.org/html/2607.24079v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.24029",
  "slug": "moving-horizon-estimation-and-nonlinear-model-predictive-control-of-ca",
  "title": "Moving-Horizon Estimation and Nonlinear Model Predictive Control of Cable-Driven Soft Manipulators",
  "abstract": "Precise control of soft manipulators remains challenging due to the difficulty of developing accurate yet computationally tractable models for model-based estimation and control. Reduced Cosserat-rod models provide a physics-based and control-oriented description of soft-robot dynamics, offering an explicit alternative to purely data-driven input-output representations. In this paper, we propose a moving-horizon estimation (MHE) and nonlinear model predictive control (NMPC) framework for cable-driven soft manipulators based on reduced Cosserat dynamics. A smooth cable-length-driven modeling formulation is developed by approximating the complementarity relationship between cable tension and cable slackness, enabling cable-length control without direct tension sensing. Based on this formulation, an MHE method is introduced to estimate the reduced state and reconstruct the manipulator configuration from end-effector pose measurements and cable-length information. An NMPC controller is then formulated to achieve task-space control under cable-length and cable-rate constraints. The proposed framework is validated through numerical simulations and experiments. Simulation results demonstrate the effectiveness of the estimator and controller for pose and strain-related regulation on a multi-cable soft manipulator. Experimental results on a four-cable prototype further show that the proposed MHE-NMPC scheme can be implemented in real time and enables accurate end-effector position tracking through cable-length control.",
  "published": "2026-07-27",
  "updated": "2026-07-27",
  "year": "2026",
  "authors": [
   "Lingxiao Xun",
   "Haihong Li",
   "Gang Zheng"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A moving-horizon estimation (MHE) and nonlinear model predictive control (NMPC) framework for cable-driven soft manipulators based on reduced Cosserat dynamics is proposed and results demonstrate the effectiveness of the estimator and controller for pose and strain-related regulation on a multi-cable soft manipulator.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Lingxiao Xun",
    "id": "2168878137",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Haihong Li",
    "id": "2144398521",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Gang Zheng",
    "id": "2365282988",
    "h_index": 1,
    "papers": 6
   }
  ],
  "comment": "",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.24029v1",
  "pdf_url": "https://arxiv.org/pdf/2607.24029v1",
  "html_url": "https://arxiv.org/html/2607.24029v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.23969",
  "slug": "leapbot-wa-world-anchor-action-models-via-predictive-latent-alignments",
  "title": "LeapBot-WA: World-Anchor Action Models via Predictive Latent Alignments",
  "abstract": "World Action Models (WAMs) have emerged as a powerful paradigm for embodied intelligence, yet the prevailing reliance on pixel-level video generation creates a fundamental bottleneck. Forcing models to reconstruct task-irrelevant visual details dissipates representational capacity and renders policies vulnerable to visual distractors. In this paper, we propose LeapBot-WA, which establishes a novel Predictive-Latent paradigm for WAMs by operationalizing the Joint-Embedding Predictive Architecture (JEPA) as a World-Anchor. Departing from the traditional reliance on visual synthesis, LeapBot-WA shifts the core of world modeling to Predictive Semantic Alignment, extracting abstract physical dynamics directly within a latent foundation space. To bridge the modality gap between non-Gaussian predictive features and diffusion priors, we introduce the Isotropic Semantic Autoencoder (ISAE), which reshapes the anchor's latent space into a diffusion-friendly manifold to prevent off-manifold drift. Furthermore, we design an Asymmetric Mixture-of-Transformers (MoT) architecture. During training, an Anchor Diffusion Transformer acts as a privileged dynamics expert to guide the Action Diffusion Transformer; at inference, this heavy dynamics branch is pruned, enabling zero-overhead execution. LeapBot-WA achieves state-of-the-art performance among predictive models on LIBERO and matches top-tier generative WAMs on RoboTwin 2.0 without requiring large-scale trajectory pre-training. It further demonstrates superior zero-shot robustness to unseen environments and successful real-world transfer, establishing a highly efficient and robust latent-centric paradigm for scalable robotic control. Code: https://github.com/LeapWM/leapbot-wa.",
  "published": "2026-07-27",
  "updated": "2026-07-30",
  "year": "2026",
  "authors": [
   "Pei Liu",
   "Nan Zheng",
   "Lang Zhang",
   "Daojie Peng",
   "Yanan Zhang",
   "Feilong Kong",
   "Mingyue Feng",
   "Jiachao Liu",
   "Yaonong Wang",
   "Qifeng Chen",
   "Jun Ma"
  ],
  "author_count": 11,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "LeapBot-WA establishes a novel Predictive-Latent paradigm for WAMs by operationalizing the Joint-Embedding Predictive Architecture (JEPA) as a World-Anchor and introduces the Isotropic Semantic Autoencoder (ISAE), which reshapes the anchor's latent space into a diffusion-friendly manifold to prevent off-manifold drift.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Pei Liu",
    "id": "2359687414",
    "h_index": 3,
    "papers": 19
   },
   {
    "name": "Nan Zheng",
    "id": "2453921996",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Lang Zhang",
    "id": "2374280219",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Daojie Peng",
    "id": "2372268470",
    "h_index": 2,
    "papers": 11
   },
   {
    "name": "Yanan Zhang",
    "id": "2453836043",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Feilong Kong",
    "id": "2453891580",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Mingyue Feng",
    "id": "2310397283",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Jiachao Liu",
    "id": "2376371031",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Yaonong Wang",
    "id": "2378861215",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Qifeng Chen",
    "id": "2317122186",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Jun Ma",
    "id": "2374084860",
    "h_index": 3,
    "papers": 10
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "foundation-pretraining",
   "video-generation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.23969v2",
  "pdf_url": "https://arxiv.org/pdf/2607.23969v2",
  "html_url": "https://arxiv.org/html/2607.23969v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.23909",
  "slug": "worlddit-a-unified-diffusion-architecture-for-world-and-action-modelin",
  "title": "WorldDiT: A Unified Diffusion Architecture for World and Action Modeling",
  "abstract": "Many recent robot policies pursue stronger control by using large pretrained vision-language models (VLMs) as the action backbone. We introduce WorldDiT, a unified diffusion transformer architecture that couples action generation with visual world modeling and achieves strong performance without a large pretrained VLM action backbone. During training, a single diffusion transformer generates continuous action chunks and predicts normalized RGB patch targets from future camera frames. Across four LIBERO simulation suites, WorldDiT lies on the reported Pareto frontier for total model parameters and mean success among methods reporting all four suites. These results provide a strong sub-billion-parameter baseline for future scaling studies.",
  "published": "2026-07-27",
  "updated": "2026-07-31",
  "year": "2026",
  "authors": [
   "Sen Wang",
   "R. Gnana Praveen",
   "Bidhan Roy",
   "Marcos Villagra"
  ],
  "author_count": 4,
  "categories": [
   "cs.LG",
   "cs.RO"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "WorldDiT, a unified diffusion transformer architecture that couples action generation with visual world modeling and achieves strong performance without a large pretrained VLM action backbone, lies on the reported Pareto frontier for total model parameters and mean success among methods reporting all four suites.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Sen Wang",
    "id": "2448250408",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "R. G. Praveen",
    "id": "2072830658",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Bidhan Roy",
    "id": "2384135588",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "M. Villagra",
    "id": "2342275351",
    "h_index": 2,
    "papers": 8
   }
  ],
  "comment": "9 pages, 4 figures",
  "topics": [
   "world-models",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.23909v2",
  "pdf_url": "https://arxiv.org/pdf/2607.23909v2",
  "html_url": "https://arxiv.org/html/2607.23909v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.23899",
  "slug": "embodied-gpt-5-1-evidence-of-a-world-model",
  "title": "Embodied GPT-5.1: Evidence of a World Model?",
  "abstract": "This exploratory study examines whether a large multimodal language model, GPT-5.1, can serve as the high-level controller of a physical mobile robot despite having no prior embodiment, no training in simulated environments, and no exposure to sensorimotor experience. Using only low-resolution first-person images and a discrete action set, the model was tasked with navigation and object-directed behaviors such as locating and contacting a target toy. Across multiple trials, GPT-5.1 demonstrated emergent capabilities that suggest elements of spatial reasoning and physical understanding. These included maintaining short-term memory of object locations after they left the camera frame, inferring the physical consequences of its own movements, and executing coherent action sequences such as colliding with an object and reversing to visually verify the outcome. At the same time, the model displayed inefficiencies and perceptual limitations, including imprecise alignment strategies and occasional misidentification of distant distractors. Overall, the results indicate that GPT-5.1 exhibits signs of world-model-like behavior in an embodied setting, despite the absence of any embodiment-related training, a finding that challenges long-standing views in cognitive science and robotics which hold that a physical body is a necessary prerequisite for developing such forms of intelligence. The findings motivate deeper investigation into the emergence, limits, and robustness of physical understanding in large language models.",
  "published": "2026-07-27",
  "updated": "2026-07-27",
  "year": "2026",
  "authors": [
   "Roberto Spinelli",
   "Thiago C. Martins"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The results indicate that GPT-5.1 exhibits signs of world-model-like behavior in an embodied setting, despite the absence of any embodiment-related training, a finding that challenges long-standing views in cognitive science and robotics which hold that a physical body is a necessary prerequisite for developing such forms of intelligence.",
  "doi": "10.1109/CROS69211.2026.11565684",
  "oa_pdf": "https://arxiv.org/pdf/2607.23899",
  "s2_authors": [
   {
    "name": "Robert J. Spinelli",
    "id": "15002128",
    "h_index": 4,
    "papers": 24
   },
   {
    "name": "T. Martins",
    "id": "2370786579",
    "h_index": 0,
    "papers": 7
   }
  ],
  "comment": "6 pages, 16 figures. Published in the 2026 Brazilian Conference on Robotics (CROS)",
  "topics": [
   "world-models",
   "egocentric-data",
   "spatial-3d",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.23899v1",
  "pdf_url": "https://arxiv.org/pdf/2607.23899v1",
  "html_url": "https://arxiv.org/html/2607.23899v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.24860",
  "slug": "egocentric-station-holding-of-robotic-fish-in-unknown-turbulent-backgr",
  "title": "Egocentric Station Holding of Robotic Fish in Unknown Turbulent Background Flow",
  "abstract": "Approaching a target position and holding station in flowing water is a fundamental and critical capability for robotic fish operating in natural aquatic environments. Despite decades of advances in enhancing swimming efficiency and maneuverability, this capability remains underdeveloped, largely owing to the insufficiently characterized, highly nonlinear fluid-structure interactions inherent to freely swimming robotic fish in flows. To bridge this gap, we propose the SWiFT framework, a Swimming With Flow Toolbox that enables the efficient exploration of an egocentric station-holding policy for a body and/or caudal fin (BCF) robotic fish in unknown and turbulent background flows via reinforcement learning (RL). Our SWiFT integrates a free-swimming flow-tank experimental setup with a highly efficient, physically consistent computational fluid dynamics (CFD)-based simulator and a systematic sim-to-real transfer pipeline. The resulting policy achieves substantial improvements over state-of-the-art methods across all metrics, most notably root-mean-square error (RMSE) of distance. Furthermore, we validated that egocentric feedback alone, without any explicit flow sensing, enables station-holding in unknown turbulent flows, closely mirroring the biological phenomenon of rheotaxis. Accordingly, the success of this egocentric station-holding policy not only advances robotic fish control toward real-world deployment, but also highlights SWiFT's promise as a foundation for tackling complex swimming tasks for underwater robots.",
  "published": "2026-07-26",
  "updated": "2026-07-26",
  "year": "2026",
  "authors": [
   "Xiaozhu Lin",
   "Xu Huang",
   "Hongru Dai",
   "Xiaopei Liu",
   "Junzhi Yu",
   "Yang Wang"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xiaozhu Lin",
    "id": "2274176767",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Xuejiao Huang",
    "id": "2451430596",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Hongru Dai",
    "id": "2379570816",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Xiaopei Liu",
    "id": "2267937833",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Junzhi Yu",
    "id": "2153201680",
    "h_index": 19,
    "papers": 164
   },
   {
    "name": "Yang Wang",
    "id": "2274044299",
    "h_index": 3,
    "papers": 9
   }
  ],
  "comment": "20 pages, 18 figures",
  "topics": [
   "egocentric-data",
   "sim2real",
   "rl-control",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.24860v1",
  "pdf_url": "https://arxiv.org/pdf/2607.24860v1",
  "html_url": "https://arxiv.org/html/2607.24860v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.23783",
  "slug": "n-0-twam-scaling-tactile-native-world-action-model-for-contact-rich-ma",
  "title": "$N_0$-TWAM: Scaling Tactile-Native World-Action Model for Contact-Rich Manipulation",
  "abstract": "We present $N_0$-TWAM, a tactile-native world-action model for contact-rich manipulation that predicts both future vision and future contact. To our knowledge, it is the first tactile world-action model trained at large scale, and it shows strong capability on contact-rich tasks. We pre-train $N_0$-TWAM at large scale with visuo-tactile joint training over tactile-rich demonstrations spanning six embodiments and 450 tasks. We use NeoForce, a unified force-based tactile representation, to form a physically grounded contact signal that conditions action generation. To improve long-horizon and multi-stage manipulation, we introduce tactile contact events for task staging and advance through them during execution. For real-time efficiency, we adopt an asymmetric Mixture-of-Transformers architecture that pairs a full-width expert for video prediction with slim experts for downstream action and tactile prediction. Evaluations on both real and simulated benchmarks justify the capabilities of $N_0$-TWAM across a range of contact-rich tasks, and demonstrate the benefit of data scaling for precise tactile and action prediction. In summary, $N_0$-TWAM endows a world-action model with predictive capabilities to foresee vision, touch and action, building a solid foundation for fine-grained manipulation on open contact-rich tasks. The codebase and model checkpoints will be made publicly available to foster further research and development in tactile-enabled robotic manipulation.",
  "published": "2026-07-26",
  "updated": "2026-07-26",
  "year": "2026",
  "authors": [
   " NeoteAI Team",
   " Fudan TEAI Team"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A tactile-native world-action model for contact-rich manipulation that predicts both future vision and future contact and endows a world-action model with predictive capabilities to foresee vision, touch and action, building a solid foundation for fine-grained manipulation on open contact-rich tasks.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "NeoteAI Team",
    "id": "2453856308",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Fudan Teai Team",
    "id": "2453854369",
    "h_index": 0,
    "papers": 2
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.23783v1",
  "pdf_url": "https://arxiv.org/pdf/2607.23783v1",
  "html_url": "https://arxiv.org/html/2607.23783v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.23782",
  "slug": "n-0-vtla-scaling-vision-tactile-language-action-model-with-latent-tact",
  "title": "$N_0$-VTLA: Scaling Vision-Tactile-Language-Action Model with Latent Tactile Tokens",
  "abstract": "We present $N_0$-VTLA, a vision-tactile-language-action (VTLA) foundation model capable of (1) fine-grained contact-rich manipulation with tactile perception and tactile-feedback control, and (2) offline policy improvement from stored deployment data. Building on current vision-based backbones, we propose a training recipe for tactile integration consisting of visuo-tactile pre-training, staged tactile-pathway integration, and advantage-conditioned offline policy improvement. During pre-training, the policy learns broad contact priors from NeoData, our large-scale visuo-tactile robot dataset; to our knowledge, $N_0$-VTLA is the first VTLA model pretrained on tactile data at scale. During post-training, we augment the policy with a predictive tactile pathway that distills the contact patterns learned at scale into the fine motion adjustments required by downstream tactile-centric manipulation. For offline policy improvement, we introduce ALTER, an advantage-conditioned offline reinforcement learning method that converts relative progress and trajectory-event comparisons into binary advantage labels for policy training on a fixed deployment corpus, further improving task-specific learning on contact-rich skills such as deformable object manipulation. Across contact-rich benchmarks, $N_0$-VTLA outperforms strong baselines by wide margins: it wins all nine real-robot NeoReal tasks and reaches 63.8% mean success on a twenty-task simulation suite, against 44.0% for the strongest baseline. $N_0$-VTLA policies trained with ALTER reach 75-95% success on three long-horizon real-robot tasks. These results lay a foundation for versatile tactile-driven manipulation policies.",
  "published": "2026-07-26",
  "updated": "2026-07-26",
  "year": "2026",
  "authors": [
   " NeoteAI Team",
   " Fudan TEAI Team"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 2,
  "influential_citations": 0,
  "tldr": "AlTER, an advantage-conditioned offline reinforcement learning method that converts relative progress and trajectory-event comparisons into binary advantage labels for policy training on a fixed deployment corpus, further improving task-specific learning on contact-rich skills such as deformable object manipulation is introduced.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "NeoteAI Team",
    "id": "2453856308",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Fudan Teai Team",
    "id": "2453854369",
    "h_index": 0,
    "papers": 2
   }
  ],
  "comment": "",
  "topics": [
   "tactile",
   "rl-control",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.23782v1",
  "pdf_url": "https://arxiv.org/pdf/2607.23782v1",
  "html_url": "https://arxiv.org/html/2607.23782v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.48
 },
 {
  "id": "2607.23704",
  "slug": "labrobfail-a-benchmark-for-robotic-failure-analysis-in-chemical-self-d",
  "title": "LabRobFail: A Benchmark for Robotic Failure Analysis in Chemical Self-driving Laboratory",
  "abstract": "The deployment of embodied agents in self-driving laboratories could accelerate scientific discovery, yet their reliability is constrained by the irreversible and safety-critical nature of chemical experiments. Progress is further hindered by scarce failure data and the lack of fine-grained evaluation protocols. To address these challenges, we introduce LabRobFail, a failure-centric framework for learning and evaluating robotic failure analysis in chemical laboratories. LabRobFail-Sim injects controllable failures at the control, physics, and semantic levels, enabling the construction of LabRobFail-Data, which contains over 20,000 trajectories across 70+ task scenarios, five failure categories, and 11 fine-grained failure types. LabRobFail-Bench evaluates six capabilities spanning task understanding, failure detection, temporal localization, severity assessment, failure classification, and actionable correction. We further develop LabRobFail-VLM, a domain-specialized vision-language model that generates structured failure diagnoses and recovery instructions. On seen environments, it achieves 90.83% failure-detection accuracy and 77.21% temporal-localization accuracy, substantially outperforming general-purpose VLMs. When integrated as a real-time supervisor, it improves downstream task success rates by 4-16 percentage points, demonstrating the value of fine-grained failure understanding for closed-loop recovery and reliable laboratory autonomy. Our code and data are available at https://github.com/Su-ISE-2001/SciRobo",
  "published": "2026-07-26",
  "updated": "2026-07-29",
  "year": "2026",
  "authors": [
   "Haobo Wang",
   "Baoli Sun",
   "Anqi Zou",
   "Dongsheng Huang",
   "Zelin Lv",
   "Ning Wang",
   "Rui Li",
   "Dongzhan Zhou",
   "Weiyu Guo",
   "Zhihui Wang",
   "Wanli Ouyang"
  ],
  "author_count": 11,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "LabRobFail, a failure-centric framework for learning and evaluating robotic failure analysis in chemical laboratories, and LabRobFail-VLM, a domain-specialized vision-language model that generates structured failure diagnoses and recovery instructions, demonstrate the value of fine-grained failure understanding for closed-loop recovery and reliable laboratory autonomy.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Haobo Wang",
    "id": "2359165325",
    "h_index": 1,
    "papers": 9
   },
   {
    "name": "Baoli Sun",
    "id": "4693885",
    "h_index": 10,
    "papers": 29
   },
   {
    "name": "Anqi Zou",
    "id": "2282282835",
    "h_index": 2,
    "papers": 13
   },
   {
    "name": "Dongsheng Huang",
    "id": "2453843323",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Zelin Lv",
    "id": "2453905500",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Ning Wang",
    "id": "2453026178",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Rui Li",
    "id": "2363818030",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Dongzhan Zhou",
    "id": "2359612649",
    "h_index": 6,
    "papers": 29
   },
   {
    "name": "Weiyu Guo",
    "id": "2348877142",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Zhihui Wang",
    "id": "2312338773",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Wanli Ouyang",
    "id": "2383976949",
    "h_index": 2,
    "papers": 7
   }
  ],
  "comment": "Under review. Haobo Wang and Baoli Sun contributed equally. Code and data: https://github.com/Su-ISE-2001/SciRobo",
  "topics": [
   "sim2real",
   "video-generation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.23704v2",
  "pdf_url": "https://arxiv.org/pdf/2607.23704v2",
  "html_url": "https://arxiv.org/html/2607.23704v2",
  "code_url": "https://github.com/Su-ISE-2001/SciRobo",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2607.23602",
  "slug": "action-from-adjacent-set-in-physical-space-outperforms-the-best-predic",
  "title": "Action from Adjacent Set in Physical Space Outperforms the Best Prediction in World Models",
  "abstract": "Controllers based on sampling and latent world models assign a predicted terminal cost to each candidate action sequence, choose the minimum, execute its first action block, and replan. This rule can fail even when the terminal cost perfectly and accurately reflects the true task objective in the physical world. Residual prediction error can give an infeasible sequence an anomalously low cost, and a larger proposal pool gives such errors more chances to outrank feasible alternatives. We call this conditional failure proposal overgeneration. In Cube candidate execution audits, increasing the total proposal budget from 72 to 288 reduces the feasibility of selection by minimum latent cost from .375 to .062 for position targets and from .344 to .031 for targets defined by position and yaw, although every larger pool contains a feasible sequence. We introduce Adjacent Set Action Reconstruction (ASAR). Among proposals with low cost, ASAR measures density from standardized early action prefixes and reconstructs a full sequence from an adjacent set with a light anchor from the sequence with minimum cost. On a Carry and Release evaluation set of 75 queries, Kernel ASAR improves event completion success over matching selection by 28.0, 24.0, and 18.7 percentage points under latent cost and by 18.7, 20.0, and 17.3 points under a trajectory reachability cost at 72, 144, and 288 proposals. Analysis of finite proposal pools characterizes selection risk from the lower tail, separation by a related radius support statistic, and sequence containment under an explicit local feasibility condition.",
  "published": "2026-07-26",
  "updated": "2026-07-26",
  "year": "2026",
  "authors": [
   "Liangyu Li",
   "Qingwen Liu",
   "Mingqing Liu"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work introduces Adjacent Set Action Reconstruction (ASAR), which measures density from standardized early action prefixes and reconstructs a full sequence from an adjacent set with a light anchor from the sequence with minimum cost.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Lian Li",
    "id": "2393056808",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Qingwen Liu",
    "id": "2112301006",
    "h_index": 16,
    "papers": 53
   },
   {
    "name": "Mingqing Liu",
    "id": "2108366158",
    "h_index": 14,
    "papers": 75
   }
  ],
  "comment": "26 pages, 7 figures. Includes supplementary material",
  "topics": [
   "world-models",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.23602v1",
  "pdf_url": "https://arxiv.org/pdf/2607.23602v1",
  "html_url": "https://arxiv.org/html/2607.23602v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.23565",
  "slug": "anticipatory-risk-guided-reinforcement-learning-for-safe-flight-throug",
  "title": "Anticipatory Risk-Guided Reinforcement Learning for Safe Flight Through Dynamic Clutter",
  "abstract": "Safe quadrotor navigation in cluttered and dynamic environments depends not only on instantaneous geometric perception, but more critically on anticipating collision risks induced by relative motion. Conventional modular pipelines frequently suffer from perception latency, while end-to-end learning methods relying on implicit scalar rewards often struggle to extract reliable spatio-temporal features without physics-grounded supervision. To address this, we propose an anticipatory risk-guided reinforcement learning framework. Leveraging privileged simulator states, we construct a directionally aligned future collision risk map based on the Closest Point of Approach (CPA). Through an asymmetric actor-critic architecture, the network is trained to self-predict this structured risk, which explicitly guides the visual policy during deployment. A lightweight spatio-temporal encoder extracts motion cues directly from onboard depth sequences, bypassing explicit object tracking or optical flow estimation. Extensive simulated and real-world experiments demonstrate that our method effectively improves safety margins and flight efficiency in dense dynamic clutters compared to existing baselines. Furthermore, the learned policy achieves robust zero-shot Sim-to-Real transfer on a physical quadrotor, relying purely on abstracted spatio-temporal depth sequences and its self-predicted risk priors, validating the effectiveness of our approach and its robust generalization from simulation to reality.",
  "published": "2026-07-26",
  "updated": "2026-07-26",
  "year": "2026",
  "authors": [
   "Yuchao Mei",
   "Guohao Zhang",
   "Luxia Ai",
   "Haopeng Chen",
   "Wenbing Tao"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "An anticipatory risk-guided reinforcement learning framework that achieves robust zero-shot Sim-to-Real transfer on a physical quadrotor, relying purely on abstracted spatio-temporal depth sequences and its self-predicted risk priors, validating the effectiveness of the approach and its robust generalization from simulation to reality.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yuchao Mei",
    "id": "2453847203",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Guohao Zhang",
    "id": "2453861780",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Luxia Ai",
    "id": "2453857631",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Haopeng Chen",
    "id": "2376392566",
    "h_index": 1,
    "papers": 12
   },
   {
    "name": "Wenbing Tao",
    "id": "2269141370",
    "h_index": 7,
    "papers": 16
   }
  ],
  "comment": "8 pages, 7 figures. Accepted to the 2026 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS 2026)",
  "topics": [
   "sim2real",
   "rl-control",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.23565v1",
  "pdf_url": "https://arxiv.org/pdf/2607.23565v1",
  "html_url": "https://arxiv.org/html/2607.23565v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2607.23473",
  "slug": "prism-polynomial-representations-for-interaction-structured-motor-cont",
  "title": "PRISM: Polynomial Representations for Interaction-Structured Motor Control",
  "abstract": "Robot policies are typically MLPs mapping observations to actions. Yet robot observations are physical variables, and many action-relevant cues arise not from individual variables but from their interactions; power, inertial effects, contact, slip, and compliance depend on products among observable signals. We introduce PRISM, a policy representation that makes polynomial interactions among observable physical variables explicit, learnable, and compact. Rather than listing all polynomial terms, PRISM uses a factorized polynomial module to expose higher-order interaction features efficiently. In reinforcement learning, it keeps the standard MLP backbone but applies a gradually activated element-wise polynomial function after it. In imitation learning, it replaces linear proprioceptive conditioning in Diffusion Policy with a polynomial layer trained end-to-end. Across humanoid locomotion and contact-rich manipulation, PRISM improves performance over standard MLP policies and larger MLPs with matched capacity, showing that interaction structure cannot be replaced by capacity alone. It also yields sensorless compliant behavior without force, wrench, tactile input, contact labels, or admittance control. These results suggest that polynomial representations should become a standard architectural choice for embodied motor control. The project page is available at https://lsh3163.github.io/prism/",
  "published": "2026-07-26",
  "updated": "2026-07-26",
  "year": "2026",
  "authors": [
   "Seung Hyun Lee",
   "Stella X. Yu"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "PRISM is introduced, a policy representation that makes polynomial interactions among observable physical variables explicit, learnable, and compact, and suggests that polynomial representations should become a standard architectural choice for embodied motor control.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Seung Hyun Lee",
    "id": "2392715699",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Stella X. Yu",
    "id": "2391343360",
    "h_index": 1,
    "papers": 8
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "tactile",
   "imitation-diffusion",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.23473v1",
  "pdf_url": "https://arxiv.org/pdf/2607.23473v1",
  "html_url": "https://arxiv.org/html/2607.23473v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.23384",
  "slug": "semantic-semi-incremental-data-association-free-object-slam",
  "title": "Semantic Semi-Incremental Data-Association-Free Object SLAM",
  "abstract": "Data association between landmark measurements and landmark variables has long been a central challenge in SLAM, as estimation accuracy depends critically on associating measurements with the correct landmark variables. Recent advances in deep learning have created new opportunities for the problem; data association can now leverage not only positional measurements but also semantic information about object landmarks, such as class labels from neural object detectors and feature vectors from visual foundation models. In this paper, we present a generalized data-association-free SLAM framework that jointly estimates data associations, robot poses, landmark positions, and landmark semantics from odometry, and positional and semantic measurements of landmarks. The proposed framework (i) creates a synergy between data association and landmark semantics estimation; (ii) adopts a semi-incremental estimation scheme for improved accuracy and computational efficiency; and (iii) provides a principled justification, guidelines, and heuristics for landmark-number estimation, improving the interpretability and practical usability of the framework. The proposed framework and algorithms are evaluated on synthetic and real-world datasets with two types of semantic information, class labels and real-valued feature vectors, and demonstrate superior performance compared to strong baselines.",
  "published": "2026-07-25",
  "updated": "2026-07-25",
  "year": "2026",
  "authors": [
   "Yihao Zhang",
   "Jungseok Hong",
   "John J. Leonard"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A generalized data-association-free SLAM framework that jointly estimates data associations, robot poses, landmark positions, and landmark semantics from odometry, and positional and semantic measurements of landmarks and adopts a semi-incremental estimation scheme.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yihao Zhang",
    "id": "1971137",
    "h_index": 8,
    "papers": 23
   },
   {
    "name": "Jungseok Hong",
    "id": "2292217130",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "John J. Leonard",
    "id": "2328976777",
    "h_index": 3,
    "papers": 5
   }
  ],
  "comment": "",
  "topics": [
   "spatial-3d",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.23384v1",
  "pdf_url": "https://arxiv.org/pdf/2607.23384v1",
  "html_url": "https://arxiv.org/html/2607.23384v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.23108",
  "slug": "the-curse-of-precision-a-data-scaling-law-for-high-precision-robotic-m",
  "title": "The Curse of Precision: A Data Scaling Law for High-Precision Robotic Manipulation",
  "abstract": "While scaling laws for imitation learning have primarily focused on generalization in open-world settings, the relationship between data and precision in closed-world tasks like robotic assembly remains largely unexplored. This paper systematically investigates this relationship and introduces a novel scaling law. We find that to achieve a fixed success rate, the required number of demonstrations $N$ grows super-exponentially as the target precision $P$ approaches a limit $c$. This relationship is accurately captured by the model $\\log(N) \\propto 1/(P-c)$. Crucially, we reveal that the limit precision $c$ is not a static physical constant of the task but an emergent property of the entire agent system, including its sensors and expert policy. Through experiments on canonical manipulation tasks, we validate this law and demonstrate that improving system components, such as adding a wrist camera or using a more effective expert, measurably lowers $c$, thus expanding the system's achievable precision. Our work provides a new theoretical framework for precision in robotics and a quantitative metric to evaluate system capabilities. Furthermore, these findings provide a practical methodology for guiding the development and debugging of high-precision manipulation systems.",
  "published": "2026-07-25",
  "updated": "2026-07-25",
  "year": "2026",
  "authors": [
   "Cuijie Xu",
   "Yuanfan Xu",
   "Min Xue",
   "Jianjie Lin",
   "Jian Wang",
   "Xudong Zhang",
   "Yu Wang",
   "Jincheng Yu"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA 2026",
  "venue_source": "arxiv-comment",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "This paper systematically investigates the relationship between data and precision in closed-world tasks like robotic assembly and introduces a novel scaling law that provides a new theoretical framework for precision in robotics and a quantitative metric to evaluate system capabilities.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Cuijie Xu",
    "id": "2373582405",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Yuanfan Xu",
    "id": "2110355949",
    "h_index": 6,
    "papers": 21
   },
   {
    "name": "Min Xue",
    "id": "2187523961",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Jianjie Lin",
    "id": "30903208",
    "h_index": 10,
    "papers": 26
   },
   {
    "name": "Jian Wang",
    "id": "2152767581",
    "h_index": 19,
    "papers": 86
   },
   {
    "name": "Xudong Zhang",
    "id": "2124923863",
    "h_index": 13,
    "papers": 46
   },
   {
    "name": "Yu Wang",
    "id": "2153603995",
    "h_index": 7,
    "papers": 26
   },
   {
    "name": "Jincheng Yu",
    "id": "2381125782",
    "h_index": 2,
    "papers": 10
   }
  ],
  "comment": "8 pages, 4 figures. Accepted to the 2026 IEEE International Conference on Robotics and Automation (ICRA 2026)",
  "topics": [
   "imitation-diffusion",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.23108v1",
  "pdf_url": "https://arxiv.org/pdf/2607.23108v1",
  "html_url": "https://arxiv.org/html/2607.23108v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.8
 },
 {
  "id": "2607.22999",
  "slug": "wcm-world-cognition-model-for-generalizable-human-robot-interaction",
  "title": "WCM: World-Cognition Model for Generalizable Human-Robot Interaction",
  "abstract": "Language agents can now interact fluently with users in software, but robots still struggle to bring comparable interaction to physical tasks. Current robot-control paradigms, including vision-language-action policies and world-model-based planners, are mainly optimized for instruction execution, leaving users with little visibility into why an action is chosen and few mechanisms to redirect, correct, or teach the robot through interaction. To solve this problem, we present the World-Cognition Model (WCM), a human-centered embodied agent built on the SLAK architecture (Sensing, Logic, Action, and Knowledge) and an asynchronous runtime. SLAK separates perception, reasoning, control, and memory, while the runtime allows reasoning, dialogue, and execution to proceed concurrently. WCM further introduces a human-in-the-loop teaching mode that enables users to interactively teach the robot difficult or long-horizon tasks. Teaching episodes and autonomous task rollouts are refined into chain-of-thought supervision to continually improve the model. WCM achieves a 73.8% average success rate across nine real-world human-robot interaction tasks, including tasks held out from CoT fine-tuning and a long-horizon task learned through teaching.",
  "published": "2026-07-25",
  "updated": "2026-08-03",
  "year": "2026",
  "authors": [
   "Yuzhen Chen",
   "KC Zhou"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.HC",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The World-Cognition Model is presented, a human-centered embodied agent built on the SLAK architecture (Sensing, Logic, Action, and Knowledge) and an asynchronous runtime and introduces a human-in-the-loop teaching mode that enables users to interactively teach the robot difficult or long-horizon tasks.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yuzhen Chen",
    "id": "2283413600",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "K. Zhou",
    "id": "2352012815",
    "h_index": 1,
    "papers": 2
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "vla",
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.22999v2",
  "pdf_url": "https://arxiv.org/pdf/2607.22999v2",
  "html_url": "https://arxiv.org/html/2607.22999v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.22997",
  "slug": "real2sim2real-for-vision-language-action-manipulation-an-amd-rocm-base",
  "title": "Real2Sim2Real for Vision-Language-Action Manipulation: An AMD ROCm-Based Pipeline",
  "abstract": "Physical AI -- the integration of large vision-language-action (VLA) models with embodied agents that act in the real world -- has emerged as the next major frontier for AI, echoed by industry leaders such as Jensen Huang (``the next big thing is Physical AI, AI with a body,'' GTC Paris, June 2025) and Dr. Lisa Su (`we're entering the world of Physical AI ... this is where AI enters the real world,' CES 2026). This paper presents an end-to-end, fully AMD-accelerated technology stack for embodied manipulation, spanning data-center training silicon, Radeon PRO simulation/rendering GPUs, and Ryzen AI edge compute, unified by the open ROCm software stack. We demonstrate that training and deploying VLA-based manipulation policies does not require a CUDA-locked ecosystem. Four progressive demonstrations are presented: (1) a Sim-to-Real manipulation pipeline trained with SmolVLA and deployed on a physical Franka arm; (2) a semantic, language-grounded object-selection task (`one-of-three'); (3) a Real2Sim synthetic-data generation pipeline that fuses 3D Gaussian Splatting (3DGS) reconstructions of real scenes with the Genesis physics engine; and (4) large-scale reinforcement learning for quadruped and humanoid locomotion benchmarked across multiple hardware platforms. All pipelines run natively on ROCm + PyTorch on RDNA4 (Radeon AI PRO R9700) and RDNA3.5 (Radeon PRO W7900) hardware and are reproducible on the free Radeon Cloud Platform.",
  "published": "2026-07-25",
  "updated": "2026-07-25",
  "year": "2026",
  "authors": [
   "Qing Yang",
   "Xun Wang",
   "Ziguan Wang",
   "Zhenjiang Li",
   "Hongqiang Wang",
   "Dongdong Weng"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.GR",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "An end-to-end, fully AMD-accelerated technology stack for embodied manipulation, spanning data-center training silicon, Radeon PRO simulation/rendering GPUs, and Ryzen AI edge compute, unified by the open ROCm software stack is presented.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Qingwen Yang",
    "id": "2445571701",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Xunyue Wang",
    "id": "2447889274",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Ziguan Wang",
    "id": "2453958297",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Zhenjiang Li",
    "id": "2453951630",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Hongqiang Wang",
    "id": "2455562415",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Dongdong Weng",
    "id": "2270360267",
    "h_index": 5,
    "papers": 33
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "humanoids",
   "sim2real",
   "rl-control",
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.22997v1",
  "pdf_url": "https://arxiv.org/pdf/2607.22997v1",
  "html_url": "https://arxiv.org/html/2607.22997v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.22964",
  "slug": "pose-aware-modeling-to-mitigate-pose-related-artifacts-in-tactile-glov",
  "title": "Pose-Aware Modeling to Mitigate Pose-Related Artifacts in Tactile Gloves",
  "abstract": "Tactile gloves digitize contact and force during hand-object interactions, enabling robotics applications in dexterous manipulation, teleoperation, and learning from demonstration. To preserve hand dexterity and capture the nuances of natural interactions, these gloves and the integrated tactile sensors are designed to be soft, flexible, and comfortable. However, such flexible sensors are sensitive not only to contact forces but also unavoidably to hand pose changes, resulting in pose-related artifacts (PRAs). PRAs are especially problematic in the low-force range, resulting in misdetections or late-onset detections of contact, which raises the minimum detectable force (MDF) of the glove. In this work, we characterize the PRAs in relation to pose and force. Building on these insights, we introduce a glove-agnostic algorithmic framework that leverages hand pose information, which is increasingly available, to mitigate PRAs without glove modifications. Our pose-aware force estimation model augments tactile-to-force pipelines with a residual prediction branch that explicitly accounts for pose-induced sensor deformations. We validate our approach across 3 glove designs and 15 users, reducing MDF by 10.4%, 12.2%, and 18.3%, with consistent improvements across all evaluated metrics. This method provides a practical path to improving the usability of tactile gloves in data collection and diverse robotic applications.",
  "published": "2026-07-25",
  "updated": "2026-07-25",
  "year": "2026",
  "authors": [
   "Tianhong Catherine Yu",
   "Ziyi Kou",
   "Mia Huang",
   "Taylor Niehues",
   "Yiyue Luo",
   "Li Guan",
   "Dingtian Zhang"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "cs.HC"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A glove-agnostic algorithmic framework that leverages hand pose information, which is increasingly available, to mitigate PRAs without glove modifications is introduced, and a pose-aware force estimation model augments tactile-to-force pipelines with a residual prediction branch that explicitly accounts for pose-induced sensor deformations.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Tianhong Catherine Yu",
    "id": "2282248672",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Ziyi Kou",
    "id": "2399163710",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Mianbo Huang",
    "id": "2196819171",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Taylor D. Niehues",
    "id": "21314068",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Yiyue Luo",
    "id": "2364578948",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Li Guan",
    "id": "2287035186",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Dingtian Zhang",
    "id": "2454073320",
    "h_index": 0,
    "papers": 1
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.22964v1",
  "pdf_url": "https://arxiv.org/pdf/2607.22964v1",
  "html_url": "https://arxiv.org/html/2607.22964v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.22858",
  "slug": "a-replay-constrained-simulation-framework-for-personalization-of-power",
  "title": "A Replay-Constrained Simulation Framework for Personalization of Powered Knee--Ankle Prosthesis Controllers",
  "abstract": "Personalization of impedance controllers for powered prosthetic legs is critical to accommodating individual gait biomechanics but remains challenging. Existing methods rely on time-intensive human-in-the-loop exploration and/or constrain optimization to low-dimensional, single-joint parameter subspaces. Sim-to-real transfer has enabled high-dimensional locomotion control for legged robots, but in assistive device control the human partner remains un-modelable. We present a replay-constrained simulation framework: a MuJoCo-based simulator reproduces prosthetic knee-ankle dynamics while replaying recorded hip kinematics and feedback-based ground reaction forces from individual walking data, bypassing the need to model complex human neuromuscular control mechanisms. We demonstrate the framework with a deep reinforcement learning policy that personalizes phase-dependent stiffness, damping, and equilibrium angle at both joints simultaneously, maximizing a biomimicry-based reward computed solely from onboard prosthesis measurements. Experiments with three participants with transfemoral amputation during level-ground walking at 0.8~m/s demonstrate strong simulation-to-hardware predictive validity (Pearson $r=0.96$--$0.997$). The best-performing policy on hardware was consistently predicted within the top five simulation policies for all participants. The learned controllers improved overall biomimicry rewards by 42--59\\% relative to the unpersonalized baseline. The framework supports scalable high-dimensional personalization of powered prosthetic legs and is amenable to extension to higher-dimensional controller parameterizations such as neural-network controllers.",
  "published": "2026-07-24",
  "updated": "2026-07-24",
  "year": "2026",
  "authors": [
   "Duong Le",
   "Ryan Posh",
   "Shihao Cheng",
   "Maani Ghaffari",
   "Robert D. Gregg"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A replay-constrained simulation framework that supports scalable high-dimensional personalization of powered prosthetic legs and is amenable to extension to higher-dimensional controller parameterizations such as neural-network controllers is presented.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Duong Le",
    "id": "2351147918",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Ryan R. Posh",
    "id": "2101475224",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Shihao Cheng",
    "id": "2353685843",
    "h_index": 8,
    "papers": 13
   },
   {
    "name": "Maani Ghaffari",
    "id": "1389560593",
    "h_index": 17,
    "papers": 108
   },
   {
    "name": "Robert D. Gregg",
    "id": "2249111585",
    "h_index": 8,
    "papers": 33
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "sim2real",
   "rl-control",
   "navigation",
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.22858v1",
  "pdf_url": "https://arxiv.org/pdf/2607.22858v1",
  "html_url": "https://arxiv.org/html/2607.22858v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.22789",
  "slug": "learning-based-hierarchical-tracheal-anatomy-understanding-from-sparse",
  "title": "Learning-based Hierarchical Tracheal Anatomy Understanding from Sparse Surgical Demonstration Annotations for Ultrasound Robots",
  "abstract": "Tracheostomy requires precise localization of the tracheal incision site; however, conventional manual palpation is subjective and often unreliable, while ultrasound utility remains operator-dependent. This work presents a learning-based framework for hierarchical tracheal anatomy understanding, designed specifically for ultrasound-guided robotic systems. We propose a two-stage perception pipeline integrating a YOLOv8n localization backbone with a sparse, prompt-optimized SAM2 decoder to achieve high-fidelity segmentation from sparse surgical annotations. Our hybrid training strategy, bridging curated laboratory data with unconstrained sequences, ensures clinical robustness. Experimental benchmarks demonstrate that this decoupled architecture effectively balances generalization, precision, and efficiency. The YOLOv8n and SAM2 framework achieves a consistent Mean Dice Similarity Coefficient (DSC) of 0.777 across both controlled and generalized domains. This significantly outperforms U-Net baselines, which often suffer from anatomical fragmentation and performance degradation (Generalization DSC $\\le$ 0.494). By constraining mask decoding to targeted, sparse regions of interest, our model achieves a throughput of 6.92 FPS, which is vital for closed-loop robotic teleoperation. This study confirms that a robust hierarchical understanding of tracheal anatomy can be derived by coupling lightweight localization with foundation-scale visual models. Our framework establishes a scalable foundation for standardized, autonomous surgical assistance, effectively navigating the variability of real-world ultrasound to enhance the safety and precision of robotic-assisted tracheostomy.",
  "published": "2026-07-24",
  "updated": "2026-07-24",
  "year": "2026",
  "authors": [
   "Hiu Ching Cheung",
   "Wenchao Yue",
   "Zhengran Han",
   "Mingcong Chen",
   "Guanglin Cao",
   "Hongbin Liu",
   "Hongliang Ren"
  ],
  "author_count": 7,
  "categories": [
   "eess.IV",
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "eess.IV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This study confirms that a robust hierarchical understanding of tracheal anatomy can be derived by coupling lightweight localization with foundation-scale visual models, and establishes a scalable foundation for standardized, autonomous surgical assistance.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hiu Ching Cheung",
    "id": "2143023584",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Wenchao Yue",
    "id": "2120213158",
    "h_index": 6,
    "papers": 26
   },
   {
    "name": "Zheng Han",
    "id": "2448002453",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Mingcong Chen",
    "id": "2221136265",
    "h_index": 6,
    "papers": 20
   },
   {
    "name": "Guanglin Cao",
    "id": "2238600925",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Hongbin Liu",
    "id": "2313870387",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Hongliang Ren",
    "id": "2408853435",
    "h_index": 1,
    "papers": 5
   }
  ],
  "comment": "17 pages, 8 figures, accepted at the 2026 International Conference on Cyborg and Bionic Systems (2026ICCBS)",
  "topics": [
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.22789v1",
  "pdf_url": "https://arxiv.org/pdf/2607.22789v1",
  "html_url": "https://arxiv.org/html/2607.22789v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.22535",
  "slug": "robot-factored-world-models-via-robot-rendering",
  "title": "Robot-Factored World Models via Robot Rendering",
  "abstract": "Action-conditioned video world models predict future observations from an initial observation and an action signal. In robotics, actions influence future observations through two distinct processes: they are first realized into robot motion by the robot body and controller, and the scene then responds through contact and object motion. Conditioning directly on action commands asks the world model to learn the realization process itself, while conditioning on logged future states leaks the interaction outcomes it is meant to predict. We propose robot-factored world models, which move two robot-specific factors outside the world model. First, action realization: each command is rolled through the robot's own controller and kinematics into a deployment-available nominal trajectory, a middle signal that avoids both action-realization learning and future-state leakage. Second, robot rendering: this nominal trajectory is rendered through the robot URDF, factoring the robot's geometry, kinematics, and appearance out of the model and into explicit rendered robot geometry. To resolve depth ambiguity, we pair end-effector depth with scene depth, giving geometric cues for contact and occlusion beyond image-plane overlap. Together, camera-aware static RGB/depth context and rendered robot geometry form a shared visual world-model interface that stays consistent across viewpoints and robot embodiments, so the model sees the action only as visible robot geometry and learns how objects respond to it. Our experiments show that the rendered interface outperforms vector-conditioned baselines and generalizes to unseen robot embodiments at inference. We further demonstrate that our model generates robot manipulation videos from human demonstrations by retargeting and rendering the hand motion as robot geometry.",
  "published": "2026-07-24",
  "updated": "2026-07-24",
  "year": "2026",
  "authors": [
   "Byungjun Kim",
   "Taeksoo Kim",
   "Hyunsoo Cha",
   "Hanbyul Joo"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 2,
  "influential_citations": 0,
  "tldr": "This work proposes robot-factored world models, which move two robot-specific factors outside the world model, and shows that the rendered interface outperforms vector-conditioned baselines and generalizes to unseen robot embodiments at inference.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Byungjun Kim",
    "id": "2211214359",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Taeksoo Kim",
    "id": "2280709385",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Hyunsoo Cha",
    "id": "2284593116",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Hanbyul Joo",
    "id": "2277246764",
    "h_index": 8,
    "papers": 27
   }
  ],
  "comment": "Project Page: https://bjkim95.github.io/rofacto/",
  "topics": [
   "world-models",
   "egocentric-data"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.22535v1",
  "pdf_url": "https://arxiv.org/pdf/2607.22535v1",
  "html_url": "https://arxiv.org/html/2607.22535v1",
  "code_url": "https://bjkim95.github.io/rofacto/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.48
 },
 {
  "id": "2607.22534",
  "slug": "sm4rt-learning-structured-motion-geometry-for-4d-reconstruction",
  "title": "SM4RT: Learning Structured Motion Geometry for 4D Reconstruction",
  "abstract": "Geometry Foundation Models (GFMs) have substantially advanced monocular 3D reconstruction, yet extending this capability to 4D dynamic understanding remains a fundamental challenge. Most existing motion perception methods (e.g., sparse tracking, dense point-wise flow) treat motion as independent point-wise displacements, ignoring the structured nature of physical motion. However, real-world objects usually obey rigid-body kinematics, and points thus usually move collectively, not in isolation. Motion itself possesses geometric structure: physical objects undergo a set of rigid-body transformations governed by SE(3), rather than unstructured point-wise displacements. Building on this insight, we propose SM4RT, a Structured Motion 4D Reconstruction Transformer for end-to-end 3D reconstruction and structured motion perception. SM4RT introduces Structure-of-Motion to represent scene dynamics, where scene motion is decomposed into a compact set of motion bases, each represented as a temporal sequence of 6D twists in SE(3). Dense scene motion is then recovered by sparse, time-shared per-pixel assignment weights over these bases, ensuring points on the same object share a common rigid-body motion trajectory. SM4RT introduces a parallel motion geometry encoder and decoder that jointly infer 3D geometry, world-coordinate motion, and scene kinematic structure in a single forward pass from monocular RGB video. SM4RT achieves strong motion reconstruction performance while preserving the geometric structure of scene motion.",
  "published": "2026-07-24",
  "updated": "2026-07-24",
  "year": "2026",
  "authors": [
   "Shing Ho J. Lin",
   "Wenzhao Zheng",
   "Dong Zhuo",
   "Yuqi Wu",
   "Jie Zhou",
   "Jiwen Lu"
  ],
  "author_count": 6,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "SM4RT introduces a parallel motion geometry encoder and decoder that jointly infer 3D geometry, world-coordinate motion, and scene kinematic structure in a single forward pass from monocular RGB video, and achieves strong motion reconstruction performance while preserving the geometric structure of scene motion.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Shing Ho J. Lin",
    "id": "2453684813",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Wenzhao Zheng",
    "id": "72315096",
    "h_index": 25,
    "papers": 92
   },
   {
    "name": "Dong Zhuo",
    "id": "2380965458",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Yuqi Wu",
    "id": "2333970381",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Jie Zhou",
    "id": "2267506266",
    "h_index": 14,
    "papers": 48
   },
   {
    "name": "Jiwen Lu",
    "id": "2268428068",
    "h_index": 17,
    "papers": 73
   }
  ],
  "comment": "Code is available at: https://github.com/wzzheng/SM4RT",
  "topics": [
   "spatial-3d",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.22534v1",
  "pdf_url": "https://arxiv.org/pdf/2607.22534v1",
  "html_url": "https://arxiv.org/html/2607.22534v1",
  "code_url": "https://github.com/wzzheng/SM4RT",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2607.22530",
  "slug": "vitacworld-scaling-visuo-tactile-world-models-for-contact-rich-robot-m",
  "title": "ViTacWorld: Scaling Visuo-Tactile World Models for Contact-Rich Robot Manipulation",
  "abstract": "Contact-rich robot manipulation requires physical interaction cues that are often invisible to cameras, making tactile sensing essential for robust control. However, scaling visuo-tactile robot learning remains difficult because real tactile interaction data are expensive to collect, hardware-dependent, and limited in task and scene diversity. We present ViTacWorld, an action-conditioned visuo-tactile world model for scalable contact-rich robot manipulation. ViTacWorld leverages public real tactile datasets and a constructed simulation environment to scale visuo-tactile-action data, exploiting the fact that tactile signals are directly grounded in physical contact and can exhibit a smaller simulation-to-real gap than purely visual observations. The model is first pretrained with large-scale real and simulated visuo-tactile trajectories, and then finetuned with real-world policy rollouts to better match downstream manipulation behaviors. Given robot actions, ViTacWorld predicts temporally aligned visual observations and tactile feedback, enabling visuo-tactile-action rollout generation. To the best of our knowledge, ViTacWorld is the first framework that uses a world model for robot visuo-tactile-action trajectory generation and policy evaluation. It serves two roles: synthesizing rollouts to improve downstream tactile policies, and evaluating policies by predicting action-conditioned visuo-tactile outcomes under controlled action sequences. Experiments on contact-rich manipulation tasks show that ViTacWorld generates physically meaningful rollouts, improves policy performance through scalable data augmentation, and enables action-conditioned policy evaluation. Project page: https://vitacworld.github.io/",
  "published": "2026-07-24",
  "updated": "2026-07-24",
  "year": "2026",
  "authors": [
   "Yunao Huang",
   "Shiyu Sang",
   "Haotao Lu",
   "Suting Ni",
   "Shijie Wu",
   "Ziyang Guo",
   "Ye Shi",
   "Jingya Wang"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 2,
  "influential_citations": 0,
  "tldr": "ViTacWorld is the first framework that uses a world model for robot visuo-tactile-action trajectory generation and policy evaluation, and serves two roles: synthesizing rollouts to improve downstream tactile policies, and evaluating policies by predicting action-conditioned visuo-tactile outcomes under controlled action sequences.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yunao Huang",
    "id": "2333799509",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Shiyu Sang",
    "id": "2453619932",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Haotao Lu",
    "id": "2351826252",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Suting Ni",
    "id": "2338898182",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Shijie Wu",
    "id": "2333604980",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Ziyang Guo",
    "id": "2453812112",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Ye Shi",
    "id": "2192580886",
    "h_index": 17,
    "papers": 50
   },
   {
    "name": "Jingya Wang",
    "id": "2273021500",
    "h_index": 11,
    "papers": 42
   }
  ],
  "comment": "18 pages, 6 figures, 5 tables. Project page: https://vitacworld.github.io/",
  "topics": [
   "world-models",
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.22530v1",
  "pdf_url": "https://arxiv.org/pdf/2607.22530v1",
  "html_url": "https://arxiv.org/html/2607.22530v1",
  "code_url": "https://vitacworld.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.48
 },
 {
  "id": "2607.22483",
  "slug": "plug-play-and-comply-a-modular-framework-for-online-variable-impedance",
  "title": "Plug, Play, and Comply: A Modular Framework for Online Variable Impedance with Arbitrarily Oriented Compliance Axes",
  "abstract": "The paper proposes a robot-agnostic compliant-control framework that extends the ROS control ecosystem with standardized joint and Cartesian command interfaces. It addresses a key limitation of existing control software: no reusable infrastructure for implementing compliant-control algorithms across different manipulators while preserving a common interface to higher-level applications. A plugin-based architecture separates controller infrastructure from control-law implementation. Generic wrappers use existing hardware abstractions to interface with different manipulators, while runtime-loaded plugins implement only the control law. Command interfaces support joint- and Cartesian-space references, stiffness and damping gains, nullspace targets, and feedforward terms, enabling variable impedance and diverse compliant-control formulations. Robot kinematics and dynamics are computed from URDF models using Pinocchio. The architecture facilitates the development of compliant-control strategies and enables the same implementation to be deployed across platforms unchanged. The complete framework, including reference controllers, high-level task interfaces, and example configurations for various manipulators, is open-sourced. The reference Cartesian impedance controller supports task-dependent compliance by rotating translational and rotational stiffness and damping, allowing the principal compliance directions to be updated online according to local task geometry rather than remaining fixed in the robot base or TCP frame. This is particularly important in contact-rich manipulation, where the desired directions of motion, constraints, and compliance directions may vary throughout task execution. Real-robot experiments demonstrate task-dependent compliance in contact-rich manipulation, while simulations show portability across manipulators with distinct kinematic and dynamic characteristics.",
  "published": "2026-07-24",
  "updated": "2026-07-24",
  "year": "2026",
  "authors": [
   "Mihael Simoni\u010d",
   "Xiaocong Li"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A robot-agnostic compliant-control framework that extends the ROS control ecosystem with standardized joint and Cartesian command interfaces that facilitates the development of compliant-control strategies and enables the same implementation to be deployed across platforms unchanged.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Mihael Simoni\u010d",
    "id": "32024567",
    "h_index": 8,
    "papers": 27
   },
   {
    "name": "Xiaocong Li",
    "id": "2378900876",
    "h_index": 1,
    "papers": 6
   }
  ],
  "comment": "8 pages, 7 figures, 2 tables. Paper page: https://smihael.github.io/plug-play-comply/",
  "topics": [
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.22483v1",
  "pdf_url": "https://arxiv.org/pdf/2607.22483v1",
  "html_url": "https://arxiv.org/html/2607.22483v1",
  "code_url": "https://smihael.github.io/plug-play-comply/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2607.22434",
  "slug": "robot-learning-to-communicate-through-projected-visual-abstractions",
  "title": "Robot Learning to Communicate through Projected Visual Abstractions",
  "abstract": "Humans routinely communicate through abstractions of their bodies, including shadows, silhouettes, and reflections. Yet robots remain largely confined to expressing themselves through their physical morphology. Enabling robots to communicate through such projected visual abstractions requires reasoning not only about bodily motion but also about how that motion is transformed into an external representation perceived by an observer. Among these abstractions, shadows provide a particularly compelling example because they emerge directly from the robot's embodiment while remaining visually distinct from the body itself. Here, we present a robotic system capable of dynamic shadow expression using a 21-degree-of-freedom dexterous hand with compliant soft skin and a learned shadow self-model. The soft-skinned embodiment reduces light leakage to produce visually continuous silhouettes, while the differentiable self-model learns the mapping between hand configurations and projected shadow appearance through task-agnostic self-exploration. Given a target shadow image or video, the robot optimizes its hand configurations through gradient-based search over 1 the learned self-model and refines the solution through collision-aware simulation to obtain physically feasible motions. For dynamic shadow performance, we further introduce expressive-region objectives, temporal smoothness regularization, and keyframe-based optimization to preserve visually important motion cues while reducing optimization complexity. We demonstrate robotic shadow expression across sign-language gestures, hand-shadow puppetry, and animal motion imitation in both simulation and physical experiments. These results establish a framework for enabling robots to manipulate projected visual abstractions of themselves for communication and visual storytelling.",
  "published": "2026-07-24",
  "updated": "2026-07-24",
  "year": "2026",
  "authors": [
   "Danyang Yan",
   "Boyuan Wang",
   "Jiaxun Liu",
   "Boyuan Chen"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work presents a robotic system capable of dynamic shadow expression using a 21-degree-of-freedom dexterous hand with compliant soft skin and a learned shadow self-model, and establishes a framework for enabling robots to manipulate projected visual abstractions of themselves for communication and visual storytelling.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Dan Yan",
    "id": "2447537890",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Boyuan Wang",
    "id": "2295591878",
    "h_index": 11,
    "papers": 23
   },
   {
    "name": "Jiaxun Liu",
    "id": "2308349611",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Boyuan Chen",
    "id": "2308273045",
    "h_index": 3,
    "papers": 6
   }
  ],
  "comment": "Our project website is at:https://generalroboticslab.com/shadow",
  "topics": [
   "dexterous-manipulation",
   "navigation",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.22434v1",
  "pdf_url": "https://arxiv.org/pdf/2607.22434v1",
  "html_url": "https://arxiv.org/html/2607.22434v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.22320",
  "slug": "a-monolithic-hand-with-asymmetric-origami-bending-and-dual-chamber-act",
  "title": "A Monolithic Hand with Asymmetric Origami Bending and Dual-chamber Actuators",
  "abstract": "The passive adaptability inherent in soft robotic hands affords them advantages in applications that require safe and compliant interaction. However, existing soft robotic hands often struggle to simultaneously achieve adequate output performance and easy manufacturing due to their complicated structures. In this paper, we introduce the asymmetric origami bending (AOB) pattern for generating bending motion and the asymmetric dual-chamber (ADC) design for obtaining multifunction capability. The AOB single (AOB-S) chamber and AOB dual-chamber (AOB-D) units are designed and constitute the finger and palm actuators of the proposed Origami-inspired SOft Robotic (OSOR) hand. The OSOR hand achieves bio-inspired fingers-palm motions and adequate output performance within a monolithic structure that significantly simplifies the manufacturing process. By defining the asymmetric ratio to characterize the geometric asymmetry of the unit, the analytical models of the AOB and ADC structures are proposed. The Finite Element Analysis tool for the design of AOB actuators is obtained by geometric analysis. The asymmetric origami design grants the integrated manufacturing of the OSOR hand through a Selective Laser Sintering printing process with a single thermoplastic polyurethane material. The model and simulations are validated by experimental results. Experiments show the finger and palm maximum bending motion range of 203\u00b0 and 40\u00b0, respectively, with output forces of 6.3 N and 16 N. The OSOR hand is capable of pinching a piece of tissue, stably grasping water bottles with two fingers, palm-only grasping, and completing the power grasps in the taxonomy of manufacturing grasps. The compactness, performance, and easy manufacturing of the proposed hand benefit the development of the soft robotic hand with new possibilities.",
  "published": "2026-07-24",
  "updated": "2026-07-24",
  "year": "2026",
  "authors": [
   "Nan Huang",
   "Yuming Zhu",
   "Zicong Zhang",
   "Jianhui Liu",
   "Xiaohuang Liu",
   "Dihan Liu",
   "Jiansheng Dai",
   "Sicong Liu"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Nan Huang",
    "id": "2392013657",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yuming Zhu",
    "id": "2109426881",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Zicong Zhang",
    "id": "2144371386",
    "h_index": 14,
    "papers": 41
   },
   {
    "name": "Jianhui Liu",
    "id": "2109452383",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Xiaohuang Liu",
    "id": "2316098854",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Dihan Liu",
    "id": "2291865664",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Jiansheng Dai",
    "id": "2153152661",
    "h_index": 7,
    "papers": 29
   },
   {
    "name": "Sicong Liu",
    "id": "2108637700",
    "h_index": 15,
    "papers": 56
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.22320v1",
  "pdf_url": "https://arxiv.org/pdf/2607.22320v1",
  "html_url": "https://arxiv.org/html/2607.22320v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.22249",
  "slug": "design-and-human-evaluation-of-tactile-withdrawal-reflexes-for-a-skin",
  "title": "Design and Human Evaluation of Tactile Withdrawal Reflexes for a Skin-Covered Robot Arm",
  "abstract": "Nociception is a protective biological mechanism that links harmful stimulation to a reaction. This paper investigates artificial nociception for a robotic arm with whole-body tactile sensing. We present a complete pipeline that maps pressure changes from sensitive skin on a robot manipulator to bio-inspired withdrawal motions. The system first converts skin pressure into a scalar pain gain using a nonlinear continuous model. We compare three reflexes: (i) uniform reflex moves four robot joints by a fixed amount, whereby the withdrawal is approximated by a movement of the arm \"toward the base\", independent of where the robot was touched; (ii) biologically motivated location-dependent joint-space withdrawal derived from human withdrawal reflex characteristics; (iii) Cartesian space withdrawal along the surface normal of the contacted skin pad. All behaviors are integrated in a reflex controller that interrupts the task, executes the withdrawal, and returns to a pre-contact pose. A user study with 15 participants compared the strategies using Godspeed questionnaire subscales, custom perceived-naturalness and safety items, forced-choice comparisons, and qualitative feedback. Interestingly, participants rated more highly the uniform reflex behavior over one or both competitors on the anthropomorphism, animacy, and likeability Godspeed subscales and on the Naturalness and Realism custom scale. When asked to compare the conditions, the uniform reflex was scored best in \"felt safest\", \"most human-like\", and \"most natural\". This suggests that predictability of the robot behavior is key for user acceptance. The Cartesian reflex was judged the most appropriate reaction to touch. The bio-inspired reflex did not lead any evaluated measure. This may be partly attributed to the embodiment gap between the robot arm and human arm and participants having different expectations from a robot manipulator.",
  "published": "2026-07-24",
  "updated": "2026-07-24",
  "year": "2026",
  "authors": [
   "Laura Babayeva",
   "Lukas Rustler",
   "Matej Hoffmann"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "Humanoids 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Laura Babayeva",
    "id": "2453618236",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Lukas Rustler",
    "id": "2105343640",
    "h_index": 7,
    "papers": 17
   },
   {
    "name": "Matej Hoffmann",
    "id": "2308099253",
    "h_index": 3,
    "papers": 8
   }
  ],
  "comment": "Submitted to IEEE Humanoids 2026",
  "topics": [
   "humanoids",
   "tactile",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.22249v1",
  "pdf_url": "https://arxiv.org/pdf/2607.22249v1",
  "html_url": "https://arxiv.org/html/2607.22249v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2607.22226",
  "slug": "offline-vision-language-navigation-with-geometric-goal-localization-fo",
  "title": "Offline Vision-Language Navigation with Geometric Goal Localization for Outdoor Environments",
  "abstract": "Foundation-model-based vision-language navigation (VLN) has advanced autonomous robot navigation by enabling robots to interpret natural-language instructions, identify semantic goals, and follow user-specified behavioral rules. However, existing VLN systems rely heavily on cloud-hosted foundation models for language understanding and semantic grounding, limiting their applicability where network connectivity is unavailable and reliable metric goal localization is required. Although recent small language models (SLMs) enable fully onboard inference, their suitability for navigation instruction decomposition has not been systematically evaluated. This paper makes three contributions toward fully onboard VLN for outdoor environments. First, we present the first systematic benchmark of 17 edge-deployable SLMs against 4 online APIs for robotic navigation instruction decomposition, evaluating accuracy and latency on human-annotated instructions across three computing platforms and providing practical guidance for selecting onboard language models. Second, we propose a lightweight hybrid semantic-geometric goal localization framework that combines open-vocabulary object detection, prompted segmentation, and LiDAR geometry to estimate metric goals, while maintaining visual bearing guidance when reliable geometric observations are unavailable. Third, we integrate these advances into Edge-BehAV, a fully onboard extension of the BehAV architecture that enables cloud-independent behavior-guided navigation. Experimental results show that the best offline SLM matches the instruction decomposition performance of the strongest cloud API while running approximately 9x faster and without network connectivity. The proposed goal localization framework reduces mean goal-distance error from 2.05 m to 0.20 m at lower computational cost, and the complete system succeeds in 31 of 32 closed-loop outdoor trials.",
  "published": "2026-07-24",
  "updated": "2026-07-24",
  "year": "2026",
  "authors": [
   "Ali Salmasi",
   "Xianjia Yu",
   "Tomi Westerlund"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper presents the first systematic benchmark of 17 edge-deployable SLMs against 4 online APIs for robotic navigation instruction decomposition, and proposes a lightweight hybrid semantic-geometric goal localization framework that combines open-vocabulary object detection, prompted segmentation, and LiDAR geometry to estimate metric goals.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ali Salmasi",
    "id": "2389632157",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Xianjia Yu",
    "id": "122158191",
    "h_index": 13,
    "papers": 41
   },
   {
    "name": "Tomi Westerlund",
    "id": "2265679700",
    "h_index": 3,
    "papers": 24
   }
  ],
  "comment": "",
  "topics": [
   "navigation",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.22226v1",
  "pdf_url": "https://arxiv.org/pdf/2607.22226v1",
  "html_url": "https://arxiv.org/html/2607.22226v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.22119",
  "slug": "one-hand-watches-the-other-dynamic-multi-agent-cooperation-for-sample",
  "title": "One Hand Watches The Other: Dynamic Multi-Agent Cooperation for Sample-Efficient Bimanual Manipulation in Dynamic Environments",
  "abstract": "Multi-stream robot manipulation policies achieve unparalleled sample efficiency and generalization by modeling actions relative to environmental reference frames. However, existing approaches typically assume these frames to be strictly exogenous. This causal assumption collapses in dynamic settings, such as when a single robot arm manipulates a moving object or when two arms coordinate, where each arm effectively becomes part of the dynamic environment of the other. We propose DynaMAC, a lightweight, policy-agnostic framework that resolves this causal limitation while preserving the sample efficiency, computational speed, and flexibility of multi-stream policies, DynaMAC treats the opposite arm as a dynamic task parameter, thereby providing a unified formulation for dynamic manipulation and bimanual coordination without requiring an explicit leader-follower relationship. To rigorously evaluate these capabilities, we introduce DynaBench, a novel benchmark for robot manipulation in dynamic environments. Across both dynamic environments and bimanual manipulation tasks, DynaMAC outperforms leading probabilistic and generative baselines by over 35 percentage points while requiring 20 times fewer samples. Crucially, DynaMAC generalizes zero-shot from static demonstrations to dynamic environments, substantially simplifying data collection and establishing an elegant bridge toward human-robot collaboration.",
  "published": "2026-07-24",
  "updated": "2026-07-24",
  "year": "2026",
  "authors": [
   "Jan Ole von Hartz",
   "Abhinav Valada",
   "Joschka Boedecker"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "DynaMAC treats the opposite arm as a dynamic task parameter, thereby providing a unified formulation for dynamic manipulation and bimanual coordination without requiring an explicit leader-follower relationship, and substantially simplifying data collection and establishing an elegant bridge toward human-robot collaboration.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "J. Hartz",
    "id": "101921793",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "A. Valada",
    "id": "2131132945",
    "h_index": 17,
    "papers": 86
   },
   {
    "name": "Joschka Boedecker",
    "id": "2370858554",
    "h_index": 4,
    "papers": 9
   }
  ],
  "comment": "",
  "topics": [
   "sim2real",
   "data-teleop",
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.22119v1",
  "pdf_url": "https://arxiv.org/pdf/2607.22119v1",
  "html_url": "https://arxiv.org/html/2607.22119v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.22030",
  "slug": "impedance-control-of-ship-borne-manipulators-via-optimization-based-ta",
  "title": "Impedance Control of Ship-Borne Manipulators via Optimization-based Task-Space Inverse Dynamics",
  "abstract": "Ship-borne manipulators operating in maritime environments are subject to stochastic wave-induced base motions that introduce kinematic disturbances and dynamic coupling, degrading trajectory tracking accuracy and complicating safe, contact-rich manipulation. This paper proposes a torque-level optimization-based control framework that integrates high-precision trajectory tracking with task-space impedance for ship-borne manipulators. The controller is formulated using task-space inverse dynamics (TSID) and solved via quadratic programming to explicitly compensate for the dynamic coupling introduced by base motion. To enable accurate feedforward compensation, an error-state Kalman filter (ESKF) is developed to estimate the base state by fusing inertial measurements with end-effector pose feedback. The framework is validated in simulation and real-world experiments using a 7-DOF manipulator mounted on a 6-DOF Stewart platform. The proposed method reduces real-world end-effector position tracking error by over 25.7% compared with the best baseline. Furthermore, the controller enables dynamic peg-in-hole insertion with 1~mm clearance under base motion, increasing the success rate while reducing average contact forces by 45%, demonstrating precise and compliant manipulation in contact-rich environments.",
  "published": "2026-07-24",
  "updated": "2026-07-24",
  "year": "2026",
  "authors": [
   "Lingxiao Meng",
   "Bi-Ke Zhu",
   "Xuheng Gao",
   "Zhe Zhang",
   "Jiankun Yang",
   "Jiankun Wang",
   "Haibo Lu",
   "Max Q. -H. Meng"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Lingxiao Meng",
    "id": "2279801893",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Bi-Ke Zhu",
    "id": "2367602393",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Xuheng Gao",
    "id": "2232279556",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Zhe Zhang",
    "id": "2117995148",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Jiankun Yang",
    "id": "2293551323",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Jiankun Wang",
    "id": "51068901",
    "h_index": 25,
    "papers": 126
   },
   {
    "name": "Haibo Lu",
    "id": "2115607652",
    "h_index": 8,
    "papers": 23
   },
   {
    "name": "M. Meng",
    "id": "2054282968",
    "h_index": 15,
    "papers": 48
   }
  ],
  "comment": "",
  "topics": [
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.22030v1",
  "pdf_url": "https://arxiv.org/pdf/2607.22030v1",
  "html_url": "https://arxiv.org/html/2607.22030v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.22014",
  "slug": "zero-shot-mission-level-evaluation-for-aerial-mllm-agents",
  "title": "Zero-Shot Mission-Level Evaluation for Aerial MLLM Agents",
  "abstract": "Multimodal Large Language Models (MLLMs) are emerging as core reasoning modules for embodied agents, yet it remains unclear how well general-purpose models can solve long-horizon embodied tasks from a single high-level instruction. We introduce MissionBench, a benchmark for mission-level evaluation of MLLMs in aerial 3D environments. It comprises 120 missions across five simulated 3D environments and four task families. Agents must autonomously plan, navigate, and report outcomes using only egocentric observations and its action history, without aerial-specific fine-tuning. Across 22 open- and closed-source MLLMs, the strongest model succeeds on fewer than 35% of missions compared to 84.4% human performance, highlighting the difficulty of multi-step embodied tasks. Despite large variations between model families, we observe gains from scaling, indicating that larger general-purpose models possess stronger zero-shot embodied capabilities. Our analysis shows that mission-level competence requires coordinating multiple capabilities beyond spatial perception, including multi-step planning and adaptive reasoning. This motivates closed-loop evaluation and highlights both the promise and risk of scaling-driven improvements for embodied AI.",
  "published": "2026-07-24",
  "updated": "2026-07-24",
  "year": "2026",
  "authors": [
   "Suman Navaratnarajah",
   "Taehyoung Kim",
   "Jona Ruthardt",
   "Ishaan Bhimwal",
   "Ryousuke Yamada",
   "Yannik Blei",
   "Wolfram Burgard",
   "Yuki M Asano"
  ],
  "author_count": 8,
  "categories": [
   "cs.AI",
   "cs.CL",
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.AI",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work introduces MissionBench, a benchmark for mission-level evaluation of MLLMs in aerial 3D environments, and shows that mission-level competence requires coordinating multiple capabilities beyond spatial perception, including multi-step planning and adaptive reasoning.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Suman Navaratnarajah",
    "id": "2223870381",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Taehyoung Kim",
    "id": "2453805233",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Jona Ruthardt",
    "id": "2178997817",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Ishaan Bhimwal",
    "id": "2453617928",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Ryousuke Yamada",
    "id": "2362493939",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Yannik Blei",
    "id": "2183598353",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Wolfram Burgard",
    "id": "2322439428",
    "h_index": 5,
    "papers": 20
   },
   {
    "name": "Yuki Asano",
    "id": "2249758007",
    "h_index": 2,
    "papers": 8
   }
  ],
  "comment": "Preprint",
  "topics": [
   "egocentric-data",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.22014v1",
  "pdf_url": "https://arxiv.org/pdf/2607.22014v1",
  "html_url": "https://arxiv.org/html/2607.22014v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.21964",
  "slug": "acme-a-multi-cultural-multi-embodiment-social-navigation-dataset",
  "title": "ACME: A Multi-Cultural, Multi-Embodiment Social-Navigation Dataset",
  "abstract": "Understanding how robots and humans move in shared spaces is essential for designing effective social robot navigation policies and predicting human behavior. However, existing datasets often lack the diversity needed to capture differences in culture, geography, and human-robot interaction-factors that strongly shape appropriate social behavior. To address this gap, we introduce ACME: A Cross-cultural, Multi-Embodiment dataset for social navigation. A large-scale data collection effort across 8 sites in 5 countries, using 7 robot embodiments, ACME is a large and diverse multi-modal dataset aimed at advancing social navigation research, providing 29.35 hours of onboard robot data and 43.5 hours of overhead pedestrian tracking data. Unlike prior datasets, it focuses on capturing goal-driven social navigation behavior in complex social scenarios with explicit robot-crowd interaction through robot speech. To facilitate learning navigation policies and predicting pedestrian trajectories, ACME provides 3D and 2D scene features, odometry, interaction information, and human-annotated pedestrian trajectory labels. We make ACME easy to use by providing both human-readable data for each sensor modality as well as raw binary data. Our qualitative and quantitative analyses show that our dataset captures more challenging scenarios and a broader distribution of pedestrian behavior than previous datasets.",
  "published": "2026-07-24",
  "updated": "2026-07-24",
  "year": "2026",
  "authors": [
   "Shashank Rao Marpally",
   "Allan Wang",
   "Atharva Ghotavadekar",
   "Renato Alexandre Ribeiro",
   "Nhat Le",
   "Pilar Bachiller-Burgos",
   "Pranav Goyal",
   "Subham Agrawal",
   "Yasuhiro Nitta",
   "Howard Ziyu Han",
   "Daeun Song",
   "Masaki Kuribayashi",
   "Kohei Uehara",
   "Xiyue Wang",
   "Yangzhe Kong",
   "Duc M. Nguyen",
   "Amirreza Payandeh",
   "Gerardo P\u00e9rez-Gonz\u00e1lez",
   "Alejandro Torrej\u00f3n-Harto",
   "Jeeho Ahn",
   "Tisha Jain",
   "Andrew Stratton",
   "Elvin Yang",
   "Jorge de Heuvel",
   "Nico Ostermann-Myrau",
   "Sai Anudeep Sajja",
   "Mithilya Raj",
   "Daisuke Sato",
   "Gaston Rouquette",
   "Nikolas Martelaro",
   "Maki Sugimoto",
   "Hironobu Takagi",
   "Chieko Asakawa",
   "Maren Bennewitz",
   "Aaron Steinfeld",
   "Xuesu Xiao",
   "Christoforos Mavrogiannis",
   "Harold Soh"
  ],
  "author_count": 38,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "ACME is a large and diverse multi-modal dataset aimed at advancing social navigation research, providing 29.35 hours of onboard robot data and 43.5 hours of overhead pedestrian tracking data that captures more challenging scenarios and a broader distribution of pedestrian behavior than previous datasets.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Shashank Rao Marpally",
    "id": "1641679453",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Allan Wang",
    "id": "2301165429",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Atharva Ghotavadekar",
    "id": "2335708723",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Renato Alexandre Ribeiro",
    "id": "2301115327",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Nhat Le",
    "id": "1832319766",
    "h_index": 8,
    "papers": 17
   },
   {
    "name": "P. Bachiller-Burgos",
    "id": "1500527586",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Pranav Goyal",
    "id": "150259972",
    "h_index": 4,
    "papers": 20
   },
   {
    "name": "Subham Agrawal",
    "id": "2248214439",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Yasuhiro Nitta",
    "id": "2197531188",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "H. Han",
    "id": "2260512630",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Daeun Song",
    "id": "2292142152",
    "h_index": 6,
    "papers": 19
   },
   {
    "name": "Masaki Kuribayashi",
    "id": "2089551237",
    "h_index": 7,
    "papers": 20
   },
   {
    "name": "Kohei Uehara",
    "id": "2301157895",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Xiyue Wang",
    "id": "2452156881",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yangzhe Kong",
    "id": "2324918183",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "D. Nguyen",
    "id": "2212685734",
    "h_index": 3,
    "papers": 20
   },
   {
    "name": "Amirreza Payandeh",
    "id": "2267578031",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Gerardo P\u00e9rez-Gonz\u00e1lez",
    "id": "2360091512",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Alejandro Torrej\u00f3n-Harto",
    "id": "2453616276",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Jeeho Ahn",
    "id": "2134912228",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Tisha Jain",
    "id": "2453616914",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Andrew Stratton",
    "id": "2278427712",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Elvin Yang",
    "id": "2221004818",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Jorge de Heuvel",
    "id": "2160540862",
    "h_index": 4,
    "papers": 20
   },
   {
    "name": "Nico Ostermann-Myrau",
    "id": "2351601688",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Sai Anudeep Sajja",
    "id": "2453617112",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Mithilya Raj",
    "id": "2453620423",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Daisuke Sato",
    "id": "2249535095",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "G. Rouquette",
    "id": "16786621",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Nikolas Martelaro",
    "id": "2342993799",
    "h_index": 4,
    "papers": 25
   },
   {
    "name": "Maki Sugimoto",
    "id": "2325403343",
    "h_index": 0,
    "papers": 5
   },
   {
    "name": "Hironobu Takagi",
    "id": "2159987072",
    "h_index": 7,
    "papers": 22
   },
   {
    "name": "Chieko Asakawa",
    "id": "2214763304",
    "h_index": 8,
    "papers": 28
   },
   {
    "name": "Maren Bennewitz",
    "id": "2249760470",
    "h_index": 7,
    "papers": 44
   },
   {
    "name": "Aaron Steinfeld",
    "id": "2266681648",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Xuesu Xiao",
    "id": "2283929715",
    "h_index": 9,
    "papers": 21
   },
   {
    "name": "Christoforos Mavrogiannis",
    "id": "34531831",
    "h_index": 19,
    "papers": 62
   },
   {
    "name": "Harold Soh",
    "id": "2300172476",
    "h_index": 2,
    "papers": 5
   }
  ],
  "comment": "24 Pages, 19 Figures, Submitted to IJRR on June 29th 2026",
  "topics": [
   "navigation",
   "data-teleop",
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.21964v1",
  "pdf_url": "https://arxiv.org/pdf/2607.21964v1",
  "html_url": "https://arxiv.org/html/2607.21964v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.21918",
  "slug": "action-conditioned-world-model-for-goal-plane-probe-guidance-in-roboti",
  "title": "Action-Conditioned World Model for Goal Plane Probe Guidance in Robotic Ultrasound",
  "abstract": "We present an action-conditioned world model framework for goal plane probe guidance in robotic ultrasound, with a focus on neck ultrasound scanning. Autonomous ultrasound tasks often require large numbers of probe-motion trajectories for training, but collecting high-quality demonstrations is labor-intensive and explicit simulators are difficult to build because ultrasound appearance depends on contact, tissue deformation, and view-dependent acoustic artifacts. We address this problem with a two-stage model-based learning pipeline. First, a latent conditional diffusion world model predicts future ultrasound observations from recent context frames, probe motions and temporal offset. Second, a goal-conditioned temporal transformer predicts ordered probe motions and is fine-tuned using rewards from the frozen world model. Experiments on the self-collected dataset show that the world model preserves action-dependent anatomical structure on target-directed scans. In real-world closed loop experiments, the framework achieves success rates of 70.0\\% for carotid guidance and 65.0\\% for thyroid guidance. These results demonstrate the potential of learned ultrasound dynamics for training goal-directed robotic probe navigation.",
  "published": "2026-07-24",
  "updated": "2026-08-01",
  "year": "2026",
  "authors": [
   "Siqi Fan",
   "Mingcong Chen",
   "Ran Liu",
   "Zixuan Yang",
   "Xiaoyu Fu",
   "Xiaoqing Gao",
   "Yunhui Liu",
   "Hongbin Liu"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "An action-conditioned world model framework for goal plane probe guidance in robotic ultrasound, with a focus on neck ultrasound scanning, demonstrates the potential of learned ultrasound dynamics for training goal-directed robotic probe navigation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Siqi Fan",
    "id": "2364116307",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Mingcong Chen",
    "id": "2221136265",
    "h_index": 6,
    "papers": 20
   },
   {
    "name": "Rang Liu",
    "id": "2448647695",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Zixuan Yang",
    "id": "2450179455",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Xiaoyun Fu",
    "id": "2448645062",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Xiaoqing Gao",
    "id": "2453851524",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yunhui Liu",
    "id": "2453820911",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Hongbin Liu",
    "id": "2313870387",
    "h_index": 3,
    "papers": 8
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "sim2real",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.21918v2",
  "pdf_url": "https://arxiv.org/pdf/2607.21918v2",
  "html_url": "https://arxiv.org/html/2607.21918v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.21802",
  "slug": "stars-socially-appropriate-robot-actions-via-a-recommender-system-driv",
  "title": "StARS: Socially Appropriate Robot Actions via a Recommender System-Driven Approach",
  "abstract": "Social appropriateness in human-robot interaction (HRI) is not universal: different people can judge the same robot action differently in the same situation. To capture this inter-subject variability, we reformulate socially appropriate action generation as a preference modelling problem inspired by recommender systems, treating annotators as users, contexts/scenes as items, and appropriateness scores over a set of candidate robot actions as targets. We propose StARS, a novel model-agnostic framework that integrates collaborative filtering with learnable scene representations to generate user-specific appropriateness scores over candidate robot actions. StARS is model-agnostic: it can be integrated with various scene encoders and backbones, enabling personalisation without redesigning the underlying model. We evaluate StARS on two socially aware robotics datasets, MannersDB+ and SocNav1, and analyse robustness under sparse preference feedback. Across datasets and backbones, StARS consistently improves performance and agreement with annotators, supporting personalised action selection aligned with user norms. Our code is publicly available at https://github.com/Cambridge-AFAR/StARS.git.",
  "published": "2026-07-23",
  "updated": "2026-07-23",
  "year": "2026",
  "authors": [
   "Erencem Ozbey",
   "Fethiye Irmak Dogan",
   "Jin Huang",
   "Hatice Gunes"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.IR"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "StARS is a novel model-agnostic framework that integrates collaborative filtering with learnable scene representations to generate user-specific appropriateness scores over candidate robot actions, enabling personalisation without redesigning the underlying model.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Erencem Ozbey",
    "id": "2321030601",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Fethiye Irmak Do\u011fan",
    "id": "9074985",
    "h_index": 9,
    "papers": 42
   },
   {
    "name": "Jin Huang",
    "id": "2261085212",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Hatice Gunes",
    "id": "2351608112",
    "h_index": 2,
    "papers": 14
   }
  ],
  "comment": "IROS 2026",
  "topics": [
   "spatial-3d",
   "hri",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.21802v1",
  "pdf_url": "https://arxiv.org/pdf/2607.21802v1",
  "html_url": "https://arxiv.org/html/2607.21802v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2607.21725",
  "slug": "addressing-the-orchestration-gap-in-generalist-robots-via-physical-age",
  "title": "Addressing the Orchestration Gap in Generalist Robots via Physical Agency",
  "abstract": "General-purpose robots need to reason about their actions, combining perception, world knowledge, planning, success detection, recovery, and low-level control. Today's state-of-the-art models attempt to combine all these capabilities into the learned policy via large-scale pre-training. Instead, we show that these capabilities can be decomposed into a general language-conditioned policy/control agent and a high-level agent manager/orchestrator. Rather than training policies to reason via pre-training, we build a closed-loop physical agent orchestrator that can do high-level planning, decompose the goal into achievable subgoals, command low-level motor commands, track and verify the outcome from low-level observations, and recover from failures. Our Physical Agency orchestrator (Pigey) can control existing vision-language-action (VLA) policies as well as parametrized skills to solve complex reasoning tasks in the real world, without any additional data collection or post-training. We evaluate Pigey extensively across simulation benchmarks and challenging real-world robotic manipulation tasks, and demonstrate significant performance improvements over existing generalist policies. On LIBERO-PRO, Pigey advances the state-of-the-art by over 4x (12.8% -> 53.3%) with no task-specific fine-tuning. On a real robot, Pigey lifts the frozen policy from near-zero to over 90% on reasoning-limited tasks. We call the difference between what frozen motor skills achieve alone and inside the agentic loop the orchestration gap.",
  "published": "2026-07-23",
  "updated": "2026-07-23",
  "year": "2026",
  "authors": [
   "Liane Galanti",
   "Dhruv Shah",
   "Tri Dao"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The Physical Agency orchestrator (Pigey) can control existing vision-language-action policies as well as parametrized skills to solve complex reasoning tasks in the real world, without any additional data collection or post-training.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Liane Galanti",
    "id": "2319409742",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Dhruv Shah",
    "id": "2322628540",
    "h_index": 29,
    "papers": 63
   },
   {
    "name": "Tri Dao",
    "id": "2269146652",
    "h_index": 9,
    "papers": 15
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "sim2real",
   "foundation-pretraining",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.21725v1",
  "pdf_url": "https://arxiv.org/pdf/2607.21725v1",
  "html_url": "https://arxiv.org/html/2607.21725v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.21670",
  "slug": "ordered-action-tokens-for-visuomotor-policy-learning",
  "title": "Ordered Action Tokens for Visuomotor Policy Learning",
  "abstract": "Action tokenization maps continuous robot action chunks to discrete tokens and has become an important interface for modern visuomotor policies. Existing approaches either rely on analytical discretization methods that produce prohibitively long token sequences or learned latent tokenizers that lack structure, limiting their compatibility with downstream policies. In this work, we identify three desiderata for action tokenization - high compression, total decodability, and an ordered token space - and introduce Ordered Action Tokenization (OAT), a learned action tokenizer that satisfies all three. OAT discretizes action chunks into an ordered sequence of tokens using a transformer with registers, finite scalar quantization, and ordering-inducing training mechanisms. By training each token prefix to decode into a valid action chunk, OAT places coarse control information in early tokens and uses later tokens to refine residual detail, yielding an anytime tradeoff between inference cost and action fidelity. We validate OAT in two prevailing uses of action tokens: autoregressive policies that generate tokens for control, and token co-training policies that use token losses to shape the vision-language model context consumed by a flow-based action expert. Across three policy backbones and more than 60 tasks spanning five simulation benchmarks and real-world settings, OAT consistently delivers strong policy performance while offering significantly greater flexibility at inference time.",
  "published": "2026-07-23",
  "updated": "2026-07-23",
  "year": "2026",
  "authors": [
   "Chaoqi Liu",
   "Yue Zhao",
   "Haonan Chen",
   "Xiaoshen Han",
   "Jiawei Gao",
   "Ehsan Adeli",
   "Yilun Du"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work introduces Ordered Action Tokenization (OAT), a learned action tokenizer that satisfies three desiderata for action tokenization - high compression, total decodability, and an ordered token space - and introduces OAT, a learned action tokenizer that satisfies all three.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Chaoqi Liu",
    "id": "2353314038",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Yue Zhao",
    "id": "2116810910",
    "h_index": 20,
    "papers": 27
   },
   {
    "name": "Haonan Chen",
    "id": "2309175413",
    "h_index": 6,
    "papers": 19
   },
   {
    "name": "Xiaoshen Han",
    "id": "2304069708",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Jiawei Gao",
    "id": "2397581385",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Ehsan Adeli",
    "id": "2277245601",
    "h_index": 10,
    "papers": 38
   },
   {
    "name": "Yilun Du",
    "id": "2383299857",
    "h_index": 2,
    "papers": 14
   }
  ],
  "comment": "",
  "topics": [
   "sim2real",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.21670v1",
  "pdf_url": "https://arxiv.org/pdf/2607.21670v1",
  "html_url": "https://arxiv.org/html/2607.21670v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.21588",
  "slug": "axis-a-growable-community-driven-data-engine-for-scalable-robot-manipu",
  "title": "AXIS: A Growable Community-Driven Data Engine for Scalable Robot Manipulation",
  "abstract": "Learning effective robot manipulation policies requires diverse, high-quality demonstrations, yet existing data pipelines are often difficult to scale because they rely on specialized hardware, centralized operators, or fixed task suites. We present AXIS, a growable community-driven data engine and benchmark for scalable robot learning, which enables browser-based teleoperation for large-scale demonstration collection, automatically generates and validates new manipulation tasks, and transforms community-collected demonstrations into training-ready data through automated success checking, quality filtering, trajectory smoothing, and visual and physics-based augmentation. The AXIS dataset currently contains 207 diverse tasks and 50K+ trajectories. Meanwhile, AXIS organizes data into task snapshots and evaluates policies with a systematic held-out protocol. We compare vision-language-action (VLA) policies under a unified AXIS evaluation suite and analyze scaling behavior across different data volumes. Continual pretraining on AXIS substantially improves the overall success rate of $\u03c0_{0.5}$ by 5.8%, outperforms the model pretrained on RoboCasa365 by 37.3%, and exhibits consistent scaling with increasing data volume, with the largest gains observed under layout, sensor-noise, and camera perturbations.",
  "published": "2026-07-23",
  "updated": "2026-07-23",
  "year": "2026",
  "authors": [
   "Mengfei Zhao",
   "Dihong Huang",
   "Yikai Tang",
   "Peihao Li",
   "Mingxuan Yan",
   "Ruiqi Zhuang",
   "Yanjia Huang",
   "Jie Wang",
   "Hai Zhai",
   "Tony Zhou",
   "Rui Zhang",
   "Zhexi Luo",
   "Yuchen Huang",
   "Jianfei Yang",
   "Jiachen Li"
  ],
  "author_count": 15,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "AXIS, a growable community-driven data engine and benchmark for scalable robot learning, is presented, which enables browser-based teleoperation for large-scale demonstration collection, automatically generates and validates new manipulation tasks, and transforms community-collected demonstrations into training-ready data through automated success checking, quality filtering, trajectory smoothing, and visual and physics-based augmentation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Mengfei Zhao",
    "id": "2294682220",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Dihong Huang",
    "id": "2452961019",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yikai Tang",
    "id": "2299549083",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Peihao Li",
    "id": "2358095761",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Mingxuan Yan",
    "id": "2354046119",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Ruiqi Zhuang",
    "id": "2452642370",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yanjia Huang",
    "id": "2351000265",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Jie Wang",
    "id": "2353064089",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Haining Zhai",
    "id": "2450353949",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Tony Zhou",
    "id": "2452680602",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Rui Zhang",
    "id": "2449181644",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Zhexi Luo",
    "id": "2408382791",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Yuchen Huang",
    "id": "2364088921",
    "h_index": 5,
    "papers": 17
   },
   {
    "name": "Jianfei Yang",
    "id": "2404007795",
    "h_index": 2,
    "papers": 26
   },
   {
    "name": "Jiachen Li",
    "id": "2343596356",
    "h_index": 4,
    "papers": 10
   }
  ],
  "comment": "Project Website: https://axisaiorg.github.io/AXIS-V1/",
  "topics": [
   "vla",
   "foundation-pretraining",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.21588v1",
  "pdf_url": "https://arxiv.org/pdf/2607.21588v1",
  "html_url": "https://arxiv.org/html/2607.21588v1",
  "code_url": "https://axisaiorg.github.io/AXIS-V1/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2607.21582",
  "slug": "scale-up-strategically-learning-compositional-generalization-via-bias",
  "title": "Scale Up Strategically: Learning Compositional Generalization via Bias-Aware Evaluation and Data Collection for Robotic Manipulation",
  "abstract": "Compositional generalization is essential for robot to follow diverse instructions. However, pretrained policies are known to take shortcuts, deferring to salient cues rather than grounding language. We introduce a diagnostic framework that localizes this failure to individual \\textit{instruction factors}, \\textit{e.g.,} reusable semantic components such as color, verb, object, size, and spatial attribute. Our framework formalizes instruction factor bias, the tendency of fine-tuned policies to over-rely on dominant factors as shortcuts, and quantifies it through two metrics: Factor Dominance Rate (FDR), capturing pairwise bias between factors, and Factor Dominance Hierarchy (FDH), aggregating these into a global ranking. Evaluation on six foundation policies reveals broadly consistent ordering, \\textit{i.e.}, color $\\geq$ object $\\geq$ spatial $\\geq$ verb $\\geq$ size, with color dominant, and verb and size most under-grounded. We further show the diagnosis is actionable: a bias-aware data collection strategy that reallocates a fixed budget toward under-grounded factors outperforms baselines in simulation and on a real robot using half the demonstrations, thereby enabling more sample-efficient and generalizable policy learning.",
  "published": "2026-07-23",
  "updated": "2026-07-23",
  "year": "2026",
  "authors": [
   "Yu Qi",
   "Zhang Ye",
   "Xinyi Xu",
   "Yuxuan Lu",
   "Amitoj Sandhu",
   "Boce Hu",
   "Haojie Huang",
   "Jonathan Tremblay",
   "Lawson L. S. Wong"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This framework formalizes instruction factor bias, the tendency of fine-tuned policies to over-rely on dominant factors as shortcuts, and quantifies it through two metrics: Factor Dominance Rate (FDR), capturing pairwise bias between factors, and Factor Dominance Hierarchy (FDH), aggregating these into a global ranking.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yu Qi",
    "id": "2311499051",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Zhangchen Ye",
    "id": "2402503135",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Xinyi Xu",
    "id": "2392974908",
    "h_index": 4,
    "papers": 14
   },
   {
    "name": "Yuxuan Lu",
    "id": "2155710822",
    "h_index": 14,
    "papers": 42
   },
   {
    "name": "A. Sandhu",
    "id": "2452636904",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Boce Hu",
    "id": "2312110011",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Haojie Huang",
    "id": "2332517376",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Jonathan Tremblay",
    "id": "2294175839",
    "h_index": 6,
    "papers": 21
   },
   {
    "name": "Lawson L. S. Wong",
    "id": "2263539652",
    "h_index": 6,
    "papers": 13
   }
  ],
  "comment": "",
  "topics": [
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.21582v1",
  "pdf_url": "https://arxiv.org/pdf/2607.21582v1",
  "html_url": "https://arxiv.org/html/2607.21582v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.21571",
  "slug": "beyond-episodic-evaluation-memory-architectural-bottlenecks-in-sequent",
  "title": "Beyond Episodic Evaluation: Memory Architectural Bottlenecks in Sequential Embodied Question Answering",
  "abstract": "Embodied question answering (EQA) is traditionally evaluated under an episodic formulation, where agents solve each task independently and reset internal state between episodes. However, real-world robots operate continuously and must accumulate, retain, and selectively reuse information acquired from prior interactions. Despite this practical requirement, the architectural mechanisms needed to support sequential memory in EQA remain underexplored. In this work, we investigate how different memory architectures behave when EQA agents are evaluated sequentially, with multiple questions answered in the same scene while memory is carried forward across queries. We find that simply preserving existing memory is often insufficient. Agents that retain only traversability information, such as 2D occupancy maps, remember where the robot has explored but not the visual-semantic evidence needed for later questions. Agents trained on short-horizon episodic data face a different challenge: when exposed to continuous, multi-query histories, their inherited context suffers from severe temporal mismatch, rather than forming a reusable scene representation. To overcome this architectural bottleneck, we highlight the necessity of structured, spatially grounded memory: architectures that map persistent visual observations onto metric 3D geometry preserve visual-semantic evidence in a coherent scene representation. Extensive experiments in simulated environments reveal that this form of memory breaks the accuracy-efficiency tradeoff in sequential settings, simultaneously achieving higher answer accuracy and lower navigation costs. We further validate these findings on a real-world mobile robot, demonstrating that spatially grounded visual memory is critical for enabling continuous, intelligent operation in physical environments.",
  "published": "2026-07-23",
  "updated": "2026-07-23",
  "year": "2026",
  "authors": [
   "Zikui Cai",
   "Kaushal Janga",
   "Tan Dat Dao",
   "Seungjae Lee",
   "Shivin Dass",
   "Mingyo Seo",
   "Kaiyu Yue",
   "Mintong Kang",
   "Nandhu Pillai",
   "Monte Hoover",
   "Aadi Palnitkar",
   "Ruchit Rawal",
   "Ruijie Zheng",
   "Bo Li",
   "Yuke Zhu",
   "Roberto Mart\u00edn-Mart\u00edn",
   "Tom Goldstein",
   "Furong Huang"
  ],
  "author_count": 18,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work investigates how different memory architectures behave when EQA agents are evaluated sequentially, with multiple questions answered in the same scene while memory is carried forward across queries, and reveals that structured, spatially grounded memory breaks the accuracy-efficiency tradeoff in sequential settings.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zikui Cai",
    "id": "2346643861",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Kaushal Janga",
    "id": "2452654592",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "T. Dao",
    "id": "113752892",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Seungjae Lee",
    "id": "2347573653",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Shivin Dass",
    "id": "2193057311",
    "h_index": 9,
    "papers": 15
   },
   {
    "name": "Mingyo Seo",
    "id": "23190833",
    "h_index": 7,
    "papers": 12
   },
   {
    "name": "Kaiyu Yue",
    "id": "39826117",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Mintong Kang",
    "id": "2153110066",
    "h_index": 13,
    "papers": 32
   },
   {
    "name": "N. Pillai",
    "id": "2449905826",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Monte Hoover",
    "id": "2241642727",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Aadi Palnitkar",
    "id": "2230380160",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Ruchit Rawal",
    "id": "1658305348",
    "h_index": 6,
    "papers": 18
   },
   {
    "name": "Ruijie Zheng",
    "id": "2345931905",
    "h_index": 9,
    "papers": 15
   },
   {
    "name": "Bo Li",
    "id": "2449408485",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yuke Zhu",
    "id": "2253507326",
    "h_index": 16,
    "papers": 20
   },
   {
    "name": "Roberto Mart\u00edn-Mart\u00edn",
    "id": "2380440732",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Tom Goldstein",
    "id": "2279757591",
    "h_index": 7,
    "papers": 14
   },
   {
    "name": "Furong Huang",
    "id": "2347721825",
    "h_index": 3,
    "papers": 7
   }
  ],
  "comment": "Accepted to IROS 2026",
  "topics": [
   "spatial-3d",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.21571v1",
  "pdf_url": "https://arxiv.org/pdf/2607.21571v1",
  "html_url": "https://arxiv.org/html/2607.21571v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2607.21522",
  "slug": "gs-agent-creating-4d-physical-worlds-with-generative-simulation",
  "title": "GS-Agent: Creating 4D Physical Worlds With Generative Simulation",
  "abstract": "Creating dynamic and physically realistic 4D worlds from natural language descriptions is both fascinating and challenging. Traditional computer graphics methods rely on manual creation, requiring extensive human effort to fine-tune materials, motions, and visual fidelity. Recent advances in generative foundation models have sparked interest in learning to generate such 4D worlds from large-scale data; however, existing methods still struggle to ensure physical plausibility and controllability. In this work, we take a different path by leveraging foundation models to construct an agentic system that emulates how humans traditionally create 4D worlds, yet automates the entire process. We present GS-Agent, an end-to-end multi-agent framework that integrates physics engines in the loop to generate realistic, dynamic, and controllable 4D physical worlds from natural language. Inspired by how humans build 4D worlds, GS-Agent decomposes the task into entity management, covering 3D asset curation, material tuning, placement, and motion control, and rendering configuration, including camera and lighting manipulation. Multiple agents with distinct expertise interact with the physics engine via code, seek multimodal feedback, and collaborate to iteratively construct 4D worlds that align with the given descriptions. Experimental results show that GS-Agent effectively converts natural language into diverse and physically plausible 4D worlds exhibiting rich interactions among liquids, deformable objects, and rigid bodies, while achieving cinematic camera and lighting control. We envision GS-Agent as a foundation for a new paradigm in 4D world generation, empowering creative content creation and physical AI. Project page at https://umass-embodied-agi.github.io/gs-agent/",
  "published": "2026-07-23",
  "updated": "2026-07-23",
  "year": "2026",
  "authors": [
   "Hongxin Zhang",
   "Chunru Lin",
   "Junyan Li",
   "Zhou Xian",
   "Tsun-Hsuan Wang",
   "Chuang Gan"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CL",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "GS-Agent is presented, an end-to-end multi-agent framework that integrates physics engines in the loop to generate realistic, dynamic, and controllable 4D physical worlds from natural language, envisioned as a foundation for a new paradigm in 4D world generation, empowering creative content creation and physical AI.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hongxin Zhang",
    "id": "2332592758",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Chun-Tse Lin",
    "id": "2146246212",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Junyan Li",
    "id": "2265618868",
    "h_index": 10,
    "papers": 14
   },
   {
    "name": "Zhou Xian",
    "id": "2291070250",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Tsun-Hsuan Wang",
    "id": "2264982142",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Chuang Gan",
    "id": "2291068301",
    "h_index": 6,
    "papers": 11
   }
  ],
  "comment": "",
  "topics": [
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.21522v1",
  "pdf_url": "https://arxiv.org/pdf/2607.21522v1",
  "html_url": "https://arxiv.org/html/2607.21522v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.21416",
  "slug": "glam-slam-real-time-gaussian-large-scale-mapping-via-flow-densificatio",
  "title": "GLAM-SLAM: Real-time Gaussian Large-scale Mapping via Flow Densification and Spatial Decomposition",
  "abstract": "Existing Gaussian-splatting-based monocular Simultaneous Localization and Mapping (SLAM) systems are either tailored to short sequences, are not real-time, or suffer from prohibitive GPU memory requirements, limiting their applicability in realistic, long-horizon scenarios. To address this, we present GLAM-SLAM, a real-time, decoupled Gaussian-splatting SLAM system designed for large-scale outdoor scenes. We ensure lightweight tracking using a robust, feature-based SLAM frontend, while for mapping, we adopt a structured, sparse anchor grid representation that ensures scalable operation and maintains scene coherence across long-term sequences. To satisfy the dense initialization requirements of 3D Gaussian Splatting (3DGS), we introduce a geometry-based flow-densification anchoring strategy using epipolar constraints. Furthermore, by treating mapping as a multi-scene problem, we propose a scene-partitioning strategy that introduces a strong spatial inductive bias via MLP initializations to generate localized Gaussians. We evaluate our system on the challenging, long-sequence KITTI Odometry, Oxford RobotCar, and M'alaga datasets. Extensive ablations and comparisons demonstrate a 15% improvement in reconstruction quality over the second-best performer, while maintaining real-time performance and the ability to scale to longer sequences. Code is publicly available for the benefit of the community.",
  "published": "2026-07-23",
  "updated": "2026-07-23",
  "year": "2026",
  "authors": [
   "Panagiotis Mermigkas",
   "Argyris Manetas",
   "Petros Maragos"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "GLAM-SLAM is presented, a real-time, decoupled Gaussian-splatting SLAM system designed for large-scale outdoor scenes, and proposes a scene-partitioning strategy that introduces a strong spatial inductive bias via MLP initializations to generate localized Gaussians.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Panagiotis Mermigkas",
    "id": "2176425296",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Argyris Manetas",
    "id": "2337119885",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Petros Maragos",
    "id": "2292179285",
    "h_index": 5,
    "papers": 40
   }
  ],
  "comment": "Accepted to IROS 2026. Project page: https://glamslam.github.io/ Code: https://github.com/pmermigkas/GLAM-SLAM/",
  "topics": [
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.21416v1",
  "pdf_url": "https://arxiv.org/pdf/2607.21416v1",
  "html_url": "https://arxiv.org/html/2607.21416v1",
  "code_url": "https://glamslam.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.5
 },
 {
  "id": "2607.21341",
  "slug": "grasp-handover-rotate-bimanual-object-reorientation-via-compositional",
  "title": "Grasp, Handover, Rotate: Bimanual Object Reorientation via Compositional Diffusion and Energy-Based Optimization",
  "abstract": "Bimanual object reorientation - picking an object, handing it over between two arms, and placing it in a desired target pose - is valuable when direct placement from the initial grasp is infeasible due to collisions, kinematic constraints, or poor final orientation. However, achieving this under multiple competing objectives remains challenging. We introduce BiCompoDiff, a compositional diffusion and energy-based framework that jointly optimizes grasp selection, handover, regrasp, and motion planning under multiple constraints. By combining a pretrained grasp diffusion model with bimanual planning energy-based models (EBMs), our method injects gradient guidance during reverse diffusion to enforce collision avoidance, trajectory smoothness (via differentiable inverse kinematics), handover feasibility, and regrasp safety. Annealed MCMC sampling further refines grasp poses over the composite energy landscape. Experiments across diverse simulated household reorientation tasks demonstrate that BiCompoDiff achieves over 20% higher success rates and up to 37% smoother trajectories (measured by joint displacement) compared to strong sampling-based baselines. Real-world validation confirms effective sim-to-real transfer and robust performance on challenging scenes.",
  "published": "2026-07-23",
  "updated": "2026-07-23",
  "year": "2026",
  "authors": [
   "Wun Lam Yeung",
   "Wenjun Liu",
   "Yui Cheung Yu",
   "Zhengyan Lambo Qin",
   "Qijin She",
   "Heng Li",
   "Ziqi Wang",
   "Ping Tan"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "BiCompoDiff is introduced, a compositional diffusion and energy-based framework that jointly optimizes grasp selection, handover, regrasp, and motion planning under multiple constraints that achieves effective sim-to-real transfer and robust performance on challenging scenes.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Wun Lam Yeung",
    "id": "2452636872",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Wenjun Liu",
    "id": "2452960125",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yui Cheung Yu",
    "id": "2453010518",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Zheng Qin",
    "id": "2450995379",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Qijin She",
    "id": "1768846972",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Heng Li",
    "id": "2341961642",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Ziqi Wang",
    "id": "2108458533",
    "h_index": 8,
    "papers": 17
   },
   {
    "name": "Ping Tan",
    "id": "2334748460",
    "h_index": 1,
    "papers": 5
   }
  ],
  "comment": "IROS 2026",
  "topics": [
   "dexterous-manipulation",
   "sim2real",
   "video-generation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.21341v1",
  "pdf_url": "https://arxiv.org/pdf/2607.21341v1",
  "html_url": "https://arxiv.org/html/2607.21341v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2607.21227",
  "slug": "forge-plus-force-budgeted-recovery-for-contact-rich-assembly-with-a-fr",
  "title": "FORGE-plus: Force-Budgeted Recovery for Contact-Rich Assembly with a Frozen LLM Supervisor",
  "abstract": "Force-conditioned reinforcement learning (RL) enables tight-clearance assembly under a commanded force ceiling, but practical deployment requires determining an appropriate force limit for each object and recovering from insertion failures without exceeding it. We present a two-layer framework in which a frozen, text-only large language model (LLM) assigns a per-object force ceiling before execution and selects recovery maneuvers from a fixed action menu using compact textual force signatures. The LLM never controls force directly: a low-level controller enforces the force ceiling, the recovery policy cannot increase it, and the hidden breaking-force threshold is known only to the evaluator. We evaluate the framework on fragile bottle placement and 0.4 mm diametral-clearance gear insertion using two grippers (Robotiq 2F-140 and Franka Panda hand). A single policy passes 256/256 evaluation episodes on both fragile and robust objects without breakage, correctly predicts release timing, and completes a full table-pick-and-insert pipeline with a mean peak force of 5.4 N. Under injected in-grip slip, the force-signature recovery strategy resolves 40% and 64% of failures on the two grippers, whereas a press-harder baseline is either ineffective or causes frequent breakage. We also report negative results, including the failure of PPO to solve the task under strict force constraints and unsuccessful learned release strategies. All experiments are conducted in rigid-body simulation with hidden force-threshold breakage; no sim-to-real claim is made.",
  "published": "2026-07-23",
  "updated": "2026-07-23",
  "year": "2026",
  "authors": [
   "Kyupaeck Jeff Rah",
   "Midum Oh"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A two-layer framework in which a frozen, text-only large language model (LLM) assigns a per-object force ceiling before execution and selects recovery maneuvers from a fixed action menu using compact textual force signatures is presented.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Kyupaeck Jeff Rah",
    "id": "2452537382",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Midum Oh",
    "id": "2296828371",
    "h_index": 1,
    "papers": 3
   }
  ],
  "comment": "",
  "topics": [
   "tactile",
   "sim2real",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.21227v1",
  "pdf_url": "https://arxiv.org/pdf/2607.21227v1",
  "html_url": "https://arxiv.org/html/2607.21227v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.21071",
  "slug": "transbiolab-a-real-world-multi-view-dataset-of-cluttered-transparent-b",
  "title": "TransBiolab: A Real-World Multi-View Dataset of Cluttered Transparent Biomedical Objects",
  "abstract": "Autonomous biomedical laboratories increasingly rely on visual perception to recognize, localize, and manipulate transparent plasticware, yet high-quality real-world datasets for this setting remain limited. The scarcity of domain-relevant data is particularly restrictive in cluttered multi-object scenes, where mutual occlusion and view-dependent appearance changes remain challenging even for contemporary visual foundation models. Existing transparent-object datasets have advanced segmentation, depth, and pose estimation, but they usually do not evaluate the combined setting of multi-object clutter, occlusion, and calibrated multi-view capture that characterizes real laboratory manipulation scenes. To address this gap, we present TrainsBiolab, a real-world RGB-D dataset of cluttered transparent biomedical objects captured as calibrated multi-view sequences. TrainsBiolab contains 161,315 frames from 98 scenes and 1.03M instance annotations over 15 laboratory object types, including 6D poses, full and visible masks, depth, and per-frame camera calibration. The dataset is organized along three axes that reflect operational difficulty: object category, the total number of objects in a frame, and camera viewpoint. We further define dataset-centric benchmarks for segmentation, depth estimation and completion, and 6D pose estimation, and report a system-level robot manipulation evaluation enabled by the released annotations and calibrations. By focusing on repeated transparent instances, clutter, and multi-view laboratory capture, TrainsBiolab provides a resource for segmentation, depth estimation, 6D pose estimation, and multi-view reasoning in autonomous laboratory manipulation. Project page: https://dualtransparency.github.io/TransBiolab/.",
  "published": "2026-07-23",
  "updated": "2026-07-23",
  "year": "2026",
  "authors": [
   "Ke Ma",
   "Yifei Wang",
   "Meng Wang",
   "Tian Xia"
  ],
  "author_count": 4,
  "categories": [
   "cs.CV",
   "cs.MM",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "A real-world RGB-D dataset of cluttered transparent biomedical objects captured as calibrated multi-view sequences focusing on repeated transparent instances, clutter, and multi-view laboratory capture, TrainsBiolab provides a resource for segmentation, depth estimation, 6D pose estimation, and multi-view reasoning in autonomous laboratory manipulation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ke Ma",
    "id": "2075322416",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Yifei Wang",
    "id": "2345821544",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Meng Wang",
    "id": "2327563551",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Tian Xia",
    "id": "2282255972",
    "h_index": 2,
    "papers": 4
   }
  ],
  "comment": "9 pages, 10 figures, accepted by ACM Multimedia 2026",
  "topics": [
   "spatial-3d",
   "foundation-pretraining",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.21071v1",
  "pdf_url": "https://arxiv.org/pdf/2607.21071v1",
  "html_url": "https://arxiv.org/html/2607.21071v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2607.21049",
  "slug": "guidedattention-interpretable-and-correctable-visual-attention-for-ood",
  "title": "GuidedAttention: Interpretable and Correctable Visual Attention for OOD-Robust Robot Manipulation via Imitation Learning",
  "abstract": "End-to-end visuomotor policies provide little opportunity for humans to understand or correct the policy's visual attention. We propose GuidedAttention, a visuomotor imitation learning framework that introduces interpretable and correctable visual attention as an explicit intermediate representation. Task-relevant attention keypoints are predicted from camera images and condition a diffusion-based action policy. Users can inspect and optionally correct selected keypoints once at rollout initialization, after which the corrected attention is automatically propagated throughout execution by a tracking module. Experiments in simulation and the real world demonstrate that GuidedAttention consistently improves robot manipulation performance, particularly under positional and appearance out-of-distribution (OOD) conditions. https://mmurooka.github.io/guided-attention-project-page",
  "published": "2026-07-23",
  "updated": "2026-07-24",
  "year": "2026",
  "authors": [
   "Masaki Murooka",
   "Ryoichi Nakajo",
   "Keisuke Shirai",
   "Tomohiro Motoda",
   "Hanbit Oh",
   "Ryo Hanai",
   "Yukiyasu Domae"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "GuidedAttention, a visuomotor imitation learning framework that introduces interpretable and correctable visual attention as an explicit intermediate representation, consistently improves robot manipulation performance, particularly under positional and appearance out-of-distribution (OOD) conditions.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Masaki Murooka",
    "id": "2350756597",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Ryoichi Nakajo",
    "id": "3411577",
    "h_index": 3,
    "papers": 19
   },
   {
    "name": "Keisuke Shirai",
    "id": "2355353380",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Tomohiro Motoda",
    "id": "2328411976",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Hanbit Oh",
    "id": "2367643082",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Ryo Hanai",
    "id": "2243398622",
    "h_index": 3,
    "papers": 17
   },
   {
    "name": "Y. Domae",
    "id": "2512607",
    "h_index": 13,
    "papers": 128
   }
  ],
  "comment": "Project page added",
  "topics": [
   "imitation-diffusion",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.21049v2",
  "pdf_url": "https://arxiv.org/pdf/2607.21049v2",
  "html_url": "https://arxiv.org/html/2607.21049v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.21661",
  "slug": "grace-gradient-free-robot-action-generation-via-combined-diffusion-mpp",
  "title": "GRACE: Gradient-Free Robot Action Generation via Combined Diffusion-MPPI Posterior Mean Estimation",
  "abstract": "Diffusion policies generate multimodal robot action sequences from demonstrations, but steering them toward deployment-time constraints typically relies on differentiable guidance costs. This excludes many practical safety constraints, such as binary collision checks, joint limits, and black-box rollout costs that are nondifferentiable. We propose Gradient-free Robot Action generation via Combined diffusion-MPPI posterior mean Estimation (GRACE), which guides a pretrained diffusion policy with Model Predictive Path Integral (MPPI) control using only forward cost evaluations. Building on the common score-ascent structure of diffusion and MPPI, GRACE constructs a cost-conditioned guidance posterior at each reverse step and estimates its mean with a single MPPI update centered at the diffusion reverse mean. For differentiable costs, GRACE recovers conventional gradient guidance under a first-order, matched-covariance approximation. GRACE attains higher success rates than diffusion-based and sampling-based baselines in simulation. On a real 7-DoF manipulator, GRACE avoids a deployment-time obstacle that the unguided prior collides with in every trial. Code and experiment videos are available at https://anonymous.4open.science/w/grace-70BB/.",
  "published": "2026-07-22",
  "updated": "2026-07-22",
  "year": "2026",
  "authors": [
   "Leesai Park",
   "Jiho HOng",
   "Sanghyun Kim"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Gradient-free Robot Action generation via Combined diffusion-MPPI posterior mean Estimation (GRACE), which guides a pretrained diffusion policy with Model Predictive Path Integral (MPPI) control using only forward cost evaluations.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Leesai Park",
    "id": "2314263170",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "J. Hong",
    "id": "2453859988",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Sanghyun Kim",
    "id": "2314327043",
    "h_index": 2,
    "papers": 7
   }
  ],
  "comment": "",
  "topics": [
   "imitation-diffusion",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.21661v1",
  "pdf_url": "https://arxiv.org/pdf/2607.21661v1",
  "html_url": "https://arxiv.org/html/2607.21661v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.21648",
  "slug": "learning-diverse-humanoid-tasks-via-synthetic-video-scenarios-without",
  "title": "Learning Diverse Humanoid Tasks via Synthetic Video Scenarios without Real World Data",
  "abstract": "The human-like morphology of humanoid robots grants them exceptional potential for agile and versatile motor capabilities, but it also introduces significant challenges in acquiring complex skills. Traditional Learning-from-Demonstrations methods are often constrained by the high cost of collecting real-world data, the difficulty of capturing motion-specific behaviors, and the limited diversity of demonstrations across individuals. Moreover, even for the same task, humans may execute the motion in multiple distinct ways. In this paper, we propose a new framework that leverages the power of Generative AI to convert textual prompts into realistic and diverse sequences of human body movements, enabling the robot to observe multiple variations of how a single task can be performed. These synthetic demonstrations are then used as a training resource, allowing the robot to learn a broad range of task-execution styles without requiring direct human intervention. We evaluate the proposed method across four simulation scenarios. Experimental results show that the robot not only completes the tasks successfully but also demonstrates strong adaptability to complex variations in motion.",
  "published": "2026-07-22",
  "updated": "2026-07-22",
  "year": "2026",
  "authors": [
   "Yun-Hao Tsai",
   "Cong-Thanh Vu",
   "Yen-Chen Liu"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper proposes a new framework that leverages the power of Generative AI to convert textual prompts into realistic and diverse sequences of human body movements, enabling the robot to observe multiple variations of how a single task can be performed.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yun-Hao Tsai",
    "id": "2449504929",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Cong-Thanh Vu",
    "id": "2313838340",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Yen-Chen Liu",
    "id": "2266552616",
    "h_index": 3,
    "papers": 19
   }
  ],
  "comment": "Accepted to the 2026 IEEE/ASME International Conference on Advanced Intelligent Mechatronics (AIM)",
  "topics": [
   "humanoids",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.21648v1",
  "pdf_url": "https://arxiv.org/pdf/2607.21648v1",
  "html_url": "https://arxiv.org/html/2607.21648v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.20785",
  "slug": "robostral-navigate",
  "title": "Robostral Navigate",
  "abstract": "Deploying navigation systems at scale requires a recipe that minimizes sensor assumptions, generalizes across robot embodiments, and trains efficiently. Yet, today's best systems depend on depth sensors, multi-camera rigs, or pre-built maps, limiting the hardware they support and increasing deployment cost. We introduce Robostral Navigate, an 8B vision-language model built around this scalability objective. The model consumes only a stream of monocular RGB images - the most ubiquitous sensor across robotic platforms and predicts waypoints by pointing to the next target location in the current camera view. Operating purely in image space, rather than robot-specific coordinates, makes the policy naturally robust to changes in camera intrinsics and scene scale, enabling deployment across wheeled, legged, and aerial robots without recalibration. We generate 2.4 million trajectories across 350k simulated scenes to reduce the reliance on real-world data collection and scale easily. We further introduce a prefix-caching training recipe that packs entire episodes into single training sequences, reducing training tokens by 22x and cutting training time from months to days. A tree-based attention mask prevents conditioning on previous ground-truth actions, encouraging visually grounded action prediction, and reinforcement learning is used to further improve exploration and recovery capabilities. On the Room-to-Room and Room-Across-Room in Continuous Environments (R2R-CE and RxR-CE) benchmarks, Robostral Navigate sets a new state of the art. On R2R-CE, it achieves a 77.4% success rate, surpassing the best monocular method by 10.5 points and the strongest depth- or multi-camera system by 5.3 points despite using only a single RGB camera. On RxR-CE, it reaches 75.1% success rate, outperforming all monocular baselines.",
  "published": "2026-07-22",
  "updated": "2026-07-31",
  "year": "2026",
  "authors": [
   "Abdelaziz Bounhar",
   "Abhijeet Somani",
   "Aditi Kabra",
   "Adrian Valente",
   "Adrien Petralia",
   "Adrien Sade",
   "Alan Jeffares",
   "Albert Jiang",
   "Aleksandr Timashov",
   "Alexandre Cahill",
   "Alexandre Gavaudan",
   "Alexandre Laval",
   "Alexandre Sablayrolles",
   "Amelie Heliou",
   "Amos You",
   "Andre Jonasson",
   "Andrew Bai",
   "Andrew Ehrenberg",
   "Andrew Zhao",
   "Angele Lenglemetz",
   "Anmol Agarwal",
   "Antonia Calvi",
   "Arata Suzuki",
   "Arjun Majumdar",
   "Arthur Fournier",
   "Artjom Joosen",
   "Avinash Sooriyarachchi",
   "Aylin Guliz Akkus",
   "Aysenur Karaduman",
   "Baptiste Bout",
   "Baptiste Roziere",
   "Baudouin De Monicault",
   "Benjamin Holzschuh",
   "Benjamin Lefaudeux",
   "Benjamin Tibi",
   "Bernhard Stadlbauer",
   "Blazej Osinski",
   "Camille Le Scao",
   "Chaoran Yu",
   "Charlotte Cronjager",
   "Chen-Yo Sun",
   "Chris Bamford",
   "Christian Wallenwein",
   "Christophe Renaudin",
   "Clemence Lanfranchi",
   "Corentin Barreau",
   "Corentin Sautier",
   "Cristiana-Diana Diaconu",
   "Cyprien Courtot",
   "Daniel Marczak",
   "Darius Dabert",
   "Diego de Las Casas",
   "Dominik Nuss",
   "Dylan Rubini",
   "Dzmitry Soupel",
   "Elizaveta Demyanenko",
   "Elliot Chane-Sane",
   "Emilien Fugier",
   "Emmanuel Gottlob",
   "Erik Aas",
   "Etienne Goffinet",
   "Etienne Millon",
   "Eujeong Choi",
   "Fabian Paischer",
   "Fabian Schlager",
   "Faruk Ahmed",
   "Federico Baldassarre",
   "Filip Szatkowski",
   "Florian Wiesner",
   "Gabrielle Berrada",
   "Gaetan Ecrepont",
   "Gaetan Lepage",
   "Gaspard Blanchet",
   "Gaspard Donada-Vidal",
   "Gauthier Delerce",
   "Gauthier Guinet",
   "Genevieve Hayes",
   "Georgii Novikov",
   "Giada Pistilli",
   "Gianluca Galletti",
   "Guillaume Breton",
   "Guillaume Kunsch",
   "Guillaume Lample",
   "Guillaume Martin",
   "Guillaume Raille",
   "Gunjan Dhanuka",
   "Gunshi Gupta",
   "Han Zhou",
   "Harshil Shah",
   "Hasan Furkan Vural",
   "Hedi Hadiji",
   "Hope McGovern",
   "Hugo Cisneros",
   "Hugo Thimonier",
   "Indraneel Mukherjee",
   "Ivan Cuevas Salazar",
   "Jacques Sun",
   "Jan Ludziejewski",
   "Jason Rute",
   "Jean Quentin",
   "Jean-Hadrien Chabran",
   "Jean-Malo Delignon",
   "Jie Zhang",
   "Joachim Studnia",
   "Joep Barmentlo",
   "Johannes Brandstetter",
   "John Harvill",
   "Jonas Amar",
   "Jonas Schweizer",
   "Josephine Delas",
   "Josselin Somerville",
   "Julien Denize",
   "Julien Tauran",
   "Kartik Khandelwal",
   "Khyathi Raghavi Chandu",
   "Kilian Tep",
   "Kush Jain",
   "Larissa Laich",
   "Laura Calem",
   "Laurence Aitchison",
   "Laurent Callot",
   "Laurent Fainsin",
   "Leo Cotteleer",
   "Leonard Blier",
   "Lingxiao Zhao",
   "Louis Martin",
   "Louis Serrano",
   "Lucile Saulnier",
   "Ludovic Ho Fuh",
   "Luis Montero",
   "Maarten Buyl",
   "Manon Chossegros",
   "Marcin Mozejko",
   "Margaret Jennings",
   "Markus Hennerbichler",
   "Martin Alexandre",
   "Mathieu Poiree",
   "Mathieu Schmitt",
   "Mathilde Guillaumin",
   "Matthieu Andre",
   "Matthieu Dinot",
   "Matthieu Futeral",
   "Maurits Bleeker",
   "Mauro Comi",
   "Max Mynter",
   "Maxim Berman",
   "Maxime Darrin",
   "Maxime Louis",
   "Maximilian Augustin",
   "Maximilian Muller",
   "Melina Jingting Laimon",
   "Mert Unsal",
   "Mia Chiquier",
   "Michael Pilcer",
   "Michal Pietruszka",
   "Michal Zajac",
   "Mikhail Biriuchinskii",
   "Minh-Quang Pham",
   "Minwoo Kang",
   "Morgane Riviere",
   "Namit Katariya",
   "Nathan Grinsztajn",
   "Nathan Simpson",
   "Neeraj Aggarwal",
   "Neha Gupta",
   "Ola Mysiak",
   "Oliver Leicht",
   "Olivier Bousquet",
   "Olivier Duchenne",
   "Parag Jain",
   "Patricia Wang",
   "Patrick Blies",
   "Patrick von Platen",
   "Paul Jacob",
   "Paul Wambergue",
   "Paula Kurylowicz",
   "Pavan Kumar Reddy",
   "Pavel Kuksa",
   "Philippe Pinel",
   "Philomene Chagniot",
   "Pierre Stock",
   "Pierre-Andre Savalle",
   "Piotr Milos",
   "Prateek Gupta",
   "Pravesh Agrawal",
   "Quentin Desreumaux",
   "Quentin Torroba",
   "Quercus Hernandez",
   "Ram Ramrakhya",
   "Randall Isenhour",
   "Ranjit Parva",
   "Raul Perez Pelaez",
   "Reinhard Sonnleitner",
   "Remi Delacourt",
   "Richard Kurle",
   "Rishi Shah",
   "Rob Romijnders",
   "Rohin Arora",
   "Romain Sauvestre",
   "Roman Soletskyi",
   "Rosalie Millner",
   "Rupert Menneer",
   "Sagar Vaze",
   "Samuel Barry",
   "Samuel Belkadi",
   "Samuel Humeau",
   "Sanchit Gandhi",
   "Sandeep Subramanian",
   "Sarthak Mittal",
   "Saskia Adaime",
   "Sean Cha",
   "Sebastian Kaltenbach",
   "Shashwat Dalal",
   "Shashwat Verma",
   "Sherif Waly",
   "Shrimai Prabhumoye",
   "Siddhant Waghjale",
   "Siddharth Gandhi",
   "Simon Lepage",
   "Simon Sorg",
   "Soham Ghosh",
   "Sophie Marbach",
   "Srijan Mishra",
   "Stanislas Lange",
   "Steve Hong",
   "Sumukh Aithal",
   "Szymon Antoniak",
   "Tarun Kumar Vangani",
   "Teven Le Scao",
   "Theo Cachet",
   "Thibaut Lavril",
   "Thomas Chabal",
   "Thomas Coste",
   "Thomas Defard",
   "Thomas Foubert",
   "Thomas Robert",
   "Thomas Wang",
   "Tianyu Zhang",
   "Tim Lawson",
   "Timothee Lacroix",
   "Tobias Kronlachner",
   "Tom Bewley",
   "Tom Edwards",
   "Tomas Hodan",
   "Tuhin Das",
   "Tyler Wang",
   "Ulrick BLE",
   "Umar Jamil",
   "Umberto Tomasini",
   "Valentin Mace",
   "Van Phung",
   "Vedant Nanda",
   "Victor Jouault",
   "Victor Letzelter",
   "Victor Paltz",
   "Victor Poucheret",
   "Vincent Maladiere",
   "Vincent Pfister",
   "Virgile Richard",
   "Vladislav Bataev",
   "Wassim Bouaziz",
   "Wen Ding Li",
   "William Havard",
   "William Marshall",
   "Xinghui Li",
   "Xingran Guo",
   "Xinyu Yang",
   "Yann Dreze",
   "Yassine El Ouahidi",
   "Yassir Bendou",
   "Yihan Wang",
   "Yimu Pan",
   "Yves Martin des Taillades",
   "Zaccharie Ramzi",
   "Zhenlin Xu",
   "Zsofia Csakany"
  ],
  "author_count": 276,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Robostral Navigate, an 8B vision-language model built around this scalability objective, is introduced, which consumes only a stream of monocular RGB images - the most ubiquitous sensor across robotic platforms and predicts waypoints by pointing to the next target location in the current camera view.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Arjun Majumdar",
    "id": "2905057",
    "h_index": 15,
    "papers": 28
   },
   {
    "name": "Avinash Sooriyarachchi",
    "id": "114898105",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Benjamin Tibi",
    "id": "2410358587",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Chris Bamford",
    "id": "2256994975",
    "h_index": 9,
    "papers": 30
   },
   {
    "name": "Elliot Chane-Sane",
    "id": "2117262319",
    "h_index": 7,
    "papers": 16
   },
   {
    "name": "Guillaume Lample",
    "id": "1830914",
    "h_index": 29,
    "papers": 55
   },
   {
    "name": "Khyathi Raghavi Chandu",
    "id": "37619618",
    "h_index": 23,
    "papers": 63
   },
   {
    "name": "Ludovic Ho Fuh",
    "id": "2452539142",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Mathieu Poir'ee",
    "id": "2404318304",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Olivier Duchenne",
    "id": "2096643450",
    "h_index": 14,
    "papers": 62
   },
   {
    "name": "R. Millner",
    "id": "2283583320",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Srijan Mishra",
    "id": "2366657923",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Th\u00e9o Cachet",
    "id": "2093910530",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Thomas Chabal",
    "id": "2162961195",
    "h_index": 3,
    "papers": 10
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "rl-control",
   "navigation",
   "data-teleop",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.20785v3",
  "pdf_url": "https://arxiv.org/pdf/2607.20785v3",
  "html_url": "https://arxiv.org/html/2607.20785v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.20748",
  "slug": "a-real-time-rgb-d-perception-pipeline-for-autonomous-impact-hammers-in",
  "title": "A real-time RGB-D perception pipeline for autonomous impact hammers in mining: self-filtering, rock segmentation and rock-breaking poses generation",
  "abstract": "Impact hammers, also known as rock-breakers, are essential machines in mining operations, where they perform secondary reduction. In underground mining, these machines are typically teleoperated, limiting operational efficiency. This paper presents a real-time RGB-D perception pipeline as a step towards automating the operation of hydraulic impact hammers used in mining. The proposed system simultaneously generates operationally feasible rock-breaking poses and a robot-free 3D representation of the workspace. The proposed approach combines image-based instance segmentation with geometric point cloud processing, and operates on embedded hardware at approximately 10 Hz with a total latency of around 675 ms, enabling responsive closed-loop behavior when integrated with a control system. Experimental results in a representative scaled scenario demonstrate that the proposed system is suitable for real-time autonomous impact hammer operation.",
  "published": "2026-07-22",
  "updated": "2026-07-22",
  "year": "2026",
  "authors": [
   "Mart\u00edn Gallegos",
   "Francisco Leiva",
   "Patricio Loncomilla",
   "Michelle Cort\u00e9s",
   "Javier Ruiz-del-Solar"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Mart\u00edn Gallegos",
    "id": "2452537452",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Francisco Leiva",
    "id": "2071821130",
    "h_index": 7,
    "papers": 14
   },
   {
    "name": "P. Loncomilla",
    "id": "2926000",
    "h_index": 14,
    "papers": 41
   },
   {
    "name": "Michelle Cort\u00e9s",
    "id": "2452538139",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Javier Ruiz-del-Solar",
    "id": "1399026483",
    "h_index": 31,
    "papers": 201
   }
  ],
  "comment": "25 pages, 20 figures",
  "topics": [
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.20748v1",
  "pdf_url": "https://arxiv.org/pdf/2607.20748v1",
  "html_url": "https://arxiv.org/html/2607.20748v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.20683",
  "slug": "felt-generating-tactile-signals-from-vision-for-visuo-tactile-manipula",
  "title": "FELT: Generating Tactile Signals from Vision for Visuo-Tactile Manipulation",
  "abstract": "The sense of touch is central to manipulation, especially when vision is occluded or ambiguous. Although combining vision and touch improves manipulation, learning robust visuo-tactile policies requires substantial tactile data. Such data remains scarcer than visual data, because tactile sensors are fragile, specialized, and hard to standardize. To address this, we present Feature-Extracted Latent Tactile (FELT), a learning-based framework that synthesizes per-finger pressure tactile images from RGB observations, reducing the need for tactile-equipped data collection. FELT uses a large frozen visual encoder and a lightweight query decoder to predict tactile signals in a single feed-forward pass. To respect the physical topology of dual-finger tactile sensors, FELT decodes the left and right tactile sensor panels through separate branches, capturing the asymmetric contact patterns during interactions such as wiping, insertion, and in-hand rotation. At inference time, FELT only requires RGB data, allowing us to augment existing vision-only data with tactile observations, either as generated tactile images or as latent tactile features. Experiments on four contact-rich manipulation tasks demonstrate that both generated tactile images and latent tactile features improve policy success over vision-only baselines, with latent feature requiring no real tactile sensor during policy training or deployment. Supplementary material is available on our anonymous website: https://felt-tactile.github.io/.",
  "published": "2026-07-22",
  "updated": "2026-07-22",
  "year": "2026",
  "authors": [
   "Zinan Li",
   "Yiyang Ling",
   "Yuming Gu",
   "Binghao Huang",
   "Chenhao Liang",
   "Sharfin Islam",
   "Hisham Bedri",
   "John Chirikjian",
   "Yunzhu Li",
   "Stefanos Nikolaidis",
   "Daniel Seita"
  ],
  "author_count": 11,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Experiments on four contact-rich manipulation tasks demonstrate that both generated tactile images and latent tactile features improve policy success over vision-only baselines, with latent feature requiring no real tactile sensor during policy training or deployment.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zinan Li",
    "id": "2452957024",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yiyang Ling",
    "id": "2349798230",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Yuming Gu",
    "id": "2452628253",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Binghao Huang",
    "id": "2287019710",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Chenhao Liang",
    "id": "2452609563",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Sharfin Islam",
    "id": "2302999693",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Hisham Bedri",
    "id": "2977909",
    "h_index": 2,
    "papers": 17
   },
   {
    "name": "John S. Chirikjian",
    "id": "1396382520",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Yunzhu Li",
    "id": "2374457025",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "S. Nikolaidis",
    "id": "37586303",
    "h_index": 34,
    "papers": 159
   },
   {
    "name": "Daniel Seita",
    "id": "2381734672",
    "h_index": 2,
    "papers": 10
   }
  ],
  "comment": "26 pages, including supplementary material",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.20683v1",
  "pdf_url": "https://arxiv.org/pdf/2607.20683v1",
  "html_url": "https://arxiv.org/html/2607.20683v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.20679",
  "slug": "towards-capability-aware-traversability-navigation-for-unstructured-en",
  "title": "Towards Capability-Aware Traversability Navigation for Unstructured Environments",
  "abstract": "Estimating traversability in unstructured environments requires conditioning on robot embodiment, as the same terrain can be traversable for one platform and unsafe for another. Existing methods often transfer predictions across morphologies through late-stage trajectory filtering rather than encoding platform constraints in the learned representation. We propose Capability-Aware Traversability (CAT), a framework that embeds physical limits directly into the spatial feature space. CAT grounds dense supervision masks in physical trajectories through an interactive annotation pipeline and modulates semantic terrain maps with robot-specific traversability vectors through Spatially-Adaptive Denormalization (SPADE) blocks. Across human-annotated and trajectory-aligned datasets, CAT leads all ranking-based metrics, improving AUROC by 11.0% on physically executed trajectories and AUPRC by 15.8% on human traces over the strongest baseline. Ablations show that spatial conditioning and per-robot prototypes produce capability sensitivity beyond generic path prediction. Deployments on a legged quadruped and a wheeled skid-steer demonstrate embodiment-aware obstacle avoidance on embedded hardware at 4.8 Hz.",
  "published": "2026-07-22",
  "updated": "2026-07-24",
  "year": "2026",
  "authors": [
   "Gianluca Capezzuto",
   "Felipe Tommaselli",
   "Matheus P. Angarola",
   "Ricardo V. Godoy",
   "Marcelo Becker"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Capability-Aware Traversability (CAT) is proposed, a framework that embeds physical limits directly into the spatial feature space and modulates semantic terrain maps with robot-specific traversability vectors through Spatially-Adaptive Denormalization blocks.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Gianluca Capezzuto",
    "id": "2345507719",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "F. Tommaselli",
    "id": "2270707642",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Matheus P. Angarola",
    "id": "2381986246",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Ricardo V. Godoy",
    "id": "2359447738",
    "h_index": 2,
    "papers": 16
   },
   {
    "name": "Marcelo Becker",
    "id": "2269949439",
    "h_index": 2,
    "papers": 7
   }
  ],
  "comment": "8 pages, 7 figures. Accepted to the 2026 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS 2026). Project page: https://capability-aware-traversability.github.io/",
  "topics": [
   "humanoids",
   "navigation",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.20679v2",
  "pdf_url": "https://arxiv.org/pdf/2607.20679v2",
  "html_url": "https://arxiv.org/html/2607.20679v2",
  "code_url": "https://capability-aware-traversability.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.5
 },
 {
  "id": "2607.20665",
  "slug": "safe-and-scalable-multi-drone-payload-transport-via-cbf-based-reinforc",
  "title": "Safe and Scalable Multi-Drone Payload Transport via CBF-based Reinforcement Learning with Zero-Shot Sim-to-Real Transfer",
  "abstract": "Multi-drone payload transportation has emerged as a promising research paradigm with potential applications in construction, logistics, and disaster response. However, the complex coupled dynamics among drones, cables, and payloads pose significant challenges, and existing approaches remain limited in safety and scalability, particularly in dynamic and unstructured environments. In this work, we propose a learning-based framework for safe and scalable multi-drone cooperative payload transport. We introduce a minimal 2D abstraction that preserves the task-relevant drone-payload coupling required for coordination and safety, while remaining computationally efficient for large-scale learning. Using domain randomization over team size and physical parameters, we train a fully distributed policy via Discrete Graph Control Barrier Function Proximal Policy Optimization (DGPPO), enabling robust zero-shot sim-to-real transfer without fine-tuning. Extensive real-world evaluations demonstrate that a single learned policy generalizes across varying team sizes and task scenarios. Furthermore, multi-group hardware experiments show that the same policy can safely operate in dynamic environments, where other drone teams act as moving obstacles. These results indicate that the proposed framework enables efficient, safe, and scalable multi-drone payload transportation with strong generalization to complex real-world conditions.",
  "published": "2026-07-22",
  "updated": "2026-07-22",
  "year": "2026",
  "authors": [
   "Jaeyoun Choi",
   "Oswin So",
   "Songyuan Zhang",
   "Cooper Taylor",
   "Chuchu Fan"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.MA",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "RA-L",
  "venue_source": "semantic-scholar",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Results indicate that the proposed framework enables efficient, safe, and scalable multi-drone payload transportation with strong generalization to complex real-world conditions.",
  "doi": "10.1109/LRA.2026.3715346",
  "oa_pdf": "https://arxiv.org/pdf/2607.20665",
  "s2_authors": [
   {
    "name": "Jaeyoun Choi",
    "id": "2403468738",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Oswin So",
    "id": "1923350663",
    "h_index": 14,
    "papers": 41
   },
   {
    "name": "Songyuan Zhang",
    "id": "2184424964",
    "h_index": 9,
    "papers": 24
   },
   {
    "name": "Cooper Taylor",
    "id": "2451955227",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Chuchu Fan",
    "id": "2326225950",
    "h_index": 6,
    "papers": 18
   }
  ],
  "comment": "Published in IEEE Robotics and Automation Letters (Early Access), 2026",
  "topics": [
   "sim2real",
   "rl-control",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.20665v1",
  "pdf_url": "https://arxiv.org/pdf/2607.20665v1",
  "html_url": "https://arxiv.org/html/2607.20665v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2607.20653",
  "slug": "physcore-physics-corrected-residual-world-models-for-material-aware-de",
  "title": "PhysCoRe: Physics-Corrected Residual World Models for Material-Aware Deformable Dynamics",
  "abstract": "Predicting how deformable objects evolve under robotic manipulation is a longstanding challenge. Existing approaches typically rely on per-object optimization to fit material parameters, which can be slow and cannot generalize, while end-to-end learned alternatives extrapolate poorly and often violate basic physical structure. We present PhysCoRe, a physics-corrected residual world model that couples a differentiable Material Point Method (MPM) simulator with two feed-forward neural networks. A material refinement module, Material from Motion (MfM), infers per-particle elasticity from visual observations, grounding the simulator in object-specific physics. A residual correction module, Residual from Dynamics (RfD), learns the discrepancy and predicts corrections to the simulator's internal dynamics, absorbing systematic biases that the analytical model cannot capture. This design also supports online material identification on novel objects. MfM adapts from limited interactions, and its predictive uncertainty steers further exploration toward the regions where its estimate is least confident. Experiments on real deformable-object manipulation sequences show that PhysCoRe outperforms state-of-the-art baselines in prediction accuracy, and that its predicted confidence forms a reliable distribution across the object's geometry, providing a natural signal for future confidence-guided exploration.",
  "published": "2026-07-22",
  "updated": "2026-07-22",
  "year": "2026",
  "authors": [
   "Haocheng Yin",
   "Shuohan Tao",
   "Yongsheng Chen",
   "Lu Gan"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Experiments on real deformable-object manipulation sequences show that PhysCoRe outperforms state-of-the-art baselines in prediction accuracy, and that its predicted confidence forms a reliable distribution across the object's geometry, providing a natural signal for future confidence-guided exploration.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Haocheng Yin",
    "id": "2325102521",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Shuohan Tao",
    "id": "2397377630",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Yongsheng Chen",
    "id": "2380646264",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Lu Gan",
    "id": "2333366947",
    "h_index": 4,
    "papers": 12
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "sim2real",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.20653v1",
  "pdf_url": "https://arxiv.org/pdf/2607.20653v1",
  "html_url": "https://arxiv.org/html/2607.20653v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.20399",
  "slug": "towards-miniature-humanoid-tele-loco-manipulation-using-virtual-realit",
  "title": "Towards Miniature Humanoid Tele-Loco-Manipulation Using Virtual Reality and Reinforcement Learning",
  "abstract": "Full-sized humanoid robot capabilities have grown exponentially in recent years, aiming towards general-purpose deployment in human environments. A popular control method used by manufacturers utilizes Virtual Reality for upper-body teleoperation and Reinforcement Learning for lower-body balance and locomotion control. As a result, a single remote operator can see, manipulate, and navigate about a real, distant physical environment. This powerful control stack is often relegated to expensive full-sized robots, many of which are inaccessible to the research community. Miniature humanoids are more prevalent, but employ less biomimicry in their design (e.g. fewer sensors, Degrees of Freedom, etc) and lack similar developments. This paper describes a compliant full-body telepresence control stack developed from the ground up for miniature humanoids. Framework experimentation on ROBOTIS OP3 hardware showcases walking at speeds up to 0.45 m/s independent of arm motions. Tele-loco-manipulation is demonstrated via a cube relocation experiment with an expert human operator. On average, the teleoperated system moved 2 different 40 g cubes within 10 mins, walking a total distance of 5 m. Overall, the developed system shows potential for miniature humanoid tele-loco-manipulation.",
  "published": "2026-07-22",
  "updated": "2026-07-22",
  "year": "2026",
  "authors": [
   "Nicolas Kosanovic",
   "Jordan Dowdy",
   "Jean Chagas Vaz"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.HC",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "Humanoids",
  "venue_source": "semantic-scholar",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A compliant full-body telepresence control stack developed from the ground up for miniature humanoids shows potential for miniature humanoid tele-loco-manipulation.",
  "doi": "10.1109/Humanoids65713.2025.11264861",
  "oa_pdf": "https://arxiv.org/pdf/2607.20399",
  "s2_authors": [
   {
    "name": "Nicolas Kosanovic",
    "id": "2174070902",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Jordan Dowdy",
    "id": "2308278983",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "J. Vaz",
    "id": "31067212",
    "h_index": 6,
    "papers": 28
   }
  ],
  "comment": "8 pages, 6 figures. Accepted manuscript. Published in the 2025 IEEE-RAS 24th International Conference on Humanoid Robots (Humanoids), pp. 1233-1240",
  "topics": [
   "humanoids",
   "rl-control",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.20399v1",
  "pdf_url": "https://arxiv.org/pdf/2607.20399v1",
  "html_url": "https://arxiv.org/html/2607.20399v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2607.20392",
  "slug": "distributed-acoustic-localization-array-deployed-using-a-soft-everting",
  "title": "Distributed Acoustic Localization Array Deployed Using a Soft Everting Vine Robot",
  "abstract": "Soft robot exteroception is increasingly being explored for a variety of field applications. In this work, we present a sound-based system for localizing disaster victims in confined and unstructured environments, based on a distributed acoustic sensing architecture embedded along the body of a soft everting vine robot. We propose a dynamic Steered Response Power with Phase Transform framework that supports both far-field direction-of-arrival estimation and near-field three-dimensional source localization as the robot approaches the sound source. To better understand the design and control space related to localizing sound using a soft, shape-morphing robot body, we conduct experiments measuring the accuracy of these methods for a five-microphone array attached to the robot body using three placements relative to the outer membrane of the robot (inside the pressurized body, inside the inner tail, and outside the outer wall) and in four robot configurations (linear, double linear, circular, and sinusoidal). We measure the change in accuracy as the signal-to-noise ratio, the direction of approach, and the distance of the sound source from the center of the array change. Finally, we demonstrate a vine robot growing into an arbitrary shape while carrying microphones along its outer wall, and show that a sound source located with the array's near field can be localized with high accuracy after only three microphones have everted from the robot body. These results highlight the potential of distributed acoustic sensing for reliable victim localization using soft growing robots.",
  "published": "2026-07-22",
  "updated": "2026-07-22",
  "year": "2026",
  "authors": [
   "Sebastian Lorca Godoy",
   "Ciera McFarland",
   "Michael Val",
   "Antonio Alvarez Valdivia",
   "Nathaniel Hanson",
   "Margaret McGuinness"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "S. Godoy",
    "id": "2069723031",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Ciera McFarland",
    "id": "2009402323",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Michael Val",
    "id": "2452298713",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Antonio Alvarez Valdivia",
    "id": "2139707144",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Nathaniel Hanson",
    "id": "2330191159",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Margaret McGuinness",
    "id": "2387216803",
    "h_index": 1,
    "papers": 7
   }
  ],
  "comment": "Sebastian Lorca Godoy, Ciera McFarland, Michael Val, Antonio Alvarez Valdivia, Nathaniel Hanson, and Margaret McGuinness, \"Distributed Acoustic Localization Array Deployed Using a Soft Everting Vine Robot\", in IEEE International Conference on Intelligent Robots and Systems, 2026",
  "topics": [
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.20392v1",
  "pdf_url": "https://arxiv.org/pdf/2607.20392v1",
  "html_url": "https://arxiv.org/html/2607.20392v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.20352",
  "slug": "distributed-motion-planning-with-safety-guarantees-for-self-reconfigur",
  "title": "Distributed Motion Planning with Safety Guarantees for Self-Reconfiguring Robotic Boats",
  "abstract": "Aquatic self-reconfigurable robots must assemble into desired shapes while ensuring safe interactions among multiple agents. This paper proposes a hybrid framework that combines distributed Model Predictive Control (MPC) with Control Barrier Functions (CBFs) for multi-agent shape formation and reconfiguration. Given a desired shape and target assignment, a distributed MPC scheme, solved via the Alternating Direction Method of Multipliers (ADMM), computes coordinated trajectories through local optimization and information exchange. To ensure safety in real time, distributed CBF-based filters are applied to enforce inter-agent collision avoidance. The proposed approach leverages the predictive capabilities of MPC to mitigate local minima, while CBFs provide formal safety guarantees despite the nonconvexity of the underlying optimization problem. Simulation results with up to 25 agents and experimental validation with four physical robots demonstrate the effectiveness and scalability of the framework.",
  "published": "2026-07-22",
  "updated": "2026-07-22",
  "year": "2026",
  "authors": [
   "Alejandro Gonzalez-Garcia",
   "Wei Wang",
   "Wei Xiao",
   "Wilm Decre",
   "Jan Swevers",
   "Carlo Ratti",
   "Daniela Rus"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A hybrid framework that combines distributed Model Predictive Control with Control Barrier Functions (CBFs) for multi-agent shape formation and reconfiguration that leverages the predictive capabilities of MPC to mitigate local minima, while CBFs provide formal safety guarantees despite the nonconvexity of the underlying optimization problem.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Alejandro Gonzalez-Garcia",
    "id": "2304420856",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Wei Wang",
    "id": "2158626783",
    "h_index": 11,
    "papers": 21
   },
   {
    "name": "Wei Xiao",
    "id": "2261736830",
    "h_index": 7,
    "papers": 18
   },
   {
    "name": "Wilm Decr\u00e9",
    "id": "2204649257",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Jan Swevers",
    "id": "2267996957",
    "h_index": 5,
    "papers": 40
   },
   {
    "name": "C. Ratti",
    "id": "2293624644",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Daniela Rus",
    "id": "2136540502",
    "h_index": 1,
    "papers": 7
   }
  ],
  "comment": "Submitted to IEEE",
  "topics": [
   "rl-control",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.20352v1",
  "pdf_url": "https://arxiv.org/pdf/2607.20352v1",
  "html_url": "https://arxiv.org/html/2607.20352v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.20345",
  "slug": "closing-the-lab-to-store-gap-a-data-efficient-post-training-and-experi",
  "title": "Closing the Lab-to-Store Gap: A Data-Efficient Post-Training and Experience-Driven Learning VLA Framework for Retail Humanoids",
  "abstract": "Closing the gap between benchmark performance and reliable real-world operation remains a central challenge for Vision-Language-Action (VLA) humanoid robots, which must handle execution errors, distribution shifts, and environmental variability. This paper presents DEED (Data-Efficient Post-Training and Experience-Driven Learning), a systems-level approach evaluated on a supermarket chip-restocking task using a Unitree G1-Edu humanoid robot and the GR00T N1.6 foundation model. DEED comprises three key components: (1) a data-efficient post-training pipeline with control-frequency alignment, data curation, task-relevant visual highlighting, and reduced VLA dependence; (2) a real-world study of experience-driven refinement, adapted from RECAP via a text-based advantage prefix and a vision-language value function; and (3) a latent-space analysis tool for studying in- and out-of-distribution behavior. Our results suggest that bridging the lab-to-store gap is primarily a systems integration challenge rather than an architectural one: careful data design and targeted post-training can transform a policy that fails under naive fine-tuning into a competent real-world system using only a single GPU.",
  "published": "2026-07-22",
  "updated": "2026-07-22",
  "year": "2026",
  "authors": [
   "Roger Sala Sis\u00f3",
   "Tiago Silv\u00e9rio",
   "Jakob Sand",
   "Tran Nguyen Le"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The results suggest that bridging the lab-to-store gap is primarily a systems integration challenge rather than an architectural one: careful data design and targeted post-training can transform a policy that fails under naive fine-tuning into a competent real-world system using only a single GPU.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Roger Sala Sis\u00f3",
    "id": "2452306952",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Tiago Silv\u00e9rio",
    "id": "1729358783",
    "h_index": 4,
    "papers": 14
   },
   {
    "name": "J. Sand",
    "id": "2121704746",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Tran Nguyen Le",
    "id": "2291487964",
    "h_index": 2,
    "papers": 9
   }
  ],
  "comment": "8 pages. This work has been submitted to the IEEE for possible publication",
  "topics": [
   "vla",
   "humanoids",
   "foundation-pretraining",
   "safety-eval"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2607.20345v1",
  "pdf_url": "https://arxiv.org/pdf/2607.20345v1",
  "html_url": "https://arxiv.org/html/2607.20345v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2607.20110",
  "slug": "extreme-rgmt-continual-learning-of-highly-dynamic-skills-for-robust-ge",
  "title": "Extreme-RGMT: Continual Learning of Highly Dynamic Skills for Robust Generalist Humanoid Control",
  "abstract": "Humans can progressively acquire highly dynamic motor skills while preserving reliable everyday motor abilities. In contrast, existing humanoid controllers face a trade-off between generalist and specialist capabilities: generalist motion tracking policies struggle to reliably execute rare highly dynamic motions, whereas specialist training can degrade previously acquired behaviors. We introduce Extreme-RGMT, a two-stage continual learning framework for robust generalist humanoid control. The method first learns a generalist motion-tracking base policy from diverse multi-source motion data, then employs an asymmetric skill acquisition and capability consolidation mechanism to constrain policy drift on mastered motions while emphasizing difficult dynamic segments. To address the scarcity of highly dynamic motions, their high failure rates, and the resulting shortage of informative samples, Extreme-RGMT combines difficulty-aware sampling with advantage-prioritized trajectory resampling to emphasize critical segments. Experiments show that Extreme-RGMT achieves state-of-the-art generalist whole-body motion-tracking performance, including substantially improved completion of challenging highly dynamic motions. The resulting controller directly executes diverse unseen highly dynamic motions under fixed references and online inertial motion-capture inputs, advancing generalist whole-body motion-tracking controllers toward highly dynamic motor capabilities at the human-expert level.",
  "published": "2026-07-22",
  "updated": "2026-07-22",
  "year": "2026",
  "authors": [
   "Yubiao Ma",
   "Han Yu",
   "Kai Guo",
   "Changtai Lv",
   "Zhengquan Mao",
   "Boyang Xing",
   "Xuemei Ren",
   "Dongdong Zheng"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Extreme-RGMT is introduced, a two-stage continual learning framework for robust generalist humanoid control that achieves state-of-the-art generalist whole-body motion-tracking performance, including substantially improved completion of challenging highly dynamic motions.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yubiao Ma",
    "id": "2408294522",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Han Yu",
    "id": "2309678337",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Kai Guo",
    "id": "2261538209",
    "h_index": 7,
    "papers": 27
   },
   {
    "name": "Changtai Lv",
    "id": "2408351917",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Zhengquan Mao",
    "id": "2296039196",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Boyang Xing",
    "id": "2381847878",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Xuemei Ren",
    "id": "2242066906",
    "h_index": 8,
    "papers": 43
   },
   {
    "name": "Dongdong Zheng",
    "id": "2321989481",
    "h_index": 1,
    "papers": 12
   }
  ],
  "comment": "",
  "topics": [
   "humanoids"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.20110v1",
  "pdf_url": "https://arxiv.org/pdf/2607.20110v1",
  "html_url": "https://arxiv.org/html/2607.20110v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.20061",
  "slug": "refertrack-referring-then-tracking-for-embodied-visual-tracking",
  "title": "ReferTrack: Referring Then Tracking for Embodied Visual Tracking",
  "abstract": "Embodied visual tracking (EVT) requires a mobile agent to continuously follow a specific target described in natural language using only onboard vision. While recent vision-language-action (VLA) policies unify target identification and trajectory planning, their chain-of-thought (CoT) reasoning often operates in abstract spatial latents that are difficult to supervise and weakly aligned with explicit image-space detections. To address this, we introduce ReferTrack, a referring-then-tracking paradigm that grounds EVT using a single forward-facing camera. Our model first selects the target from an indexed set of bounding boxes, then decodes tracking waypoints conditioned on this image-grounded decision. To preserve target motion cues over time, ReferTrack maintains a sliding-window queue of previously selected bounding boxes, injecting their geometric features into the visual history via temporal-viewpoint-bbox indicator (TVBI) tokens. We further enhance target identification by co-training on a custom Refer-QA dataset. On EVT-Bench, ReferTrack achieves state-of-the-art single-view performance with success rates of 89.4%, 73.3%, and 74.1% on the single-target, distracted, and ambiguity tracking splits, respectively -- matching or even surpassing several multi-camera baselines on identification-heavy tasks. Finally, real-world deployments on legged and humanoid robots validate its robust sim-to-real transfer capabilities. Code is available at https://github.com/MedlarTea/referTrack.",
  "published": "2026-07-22",
  "updated": "2026-07-22",
  "year": "2026",
  "authors": [
   "Hanjing Ye",
   "Tianle Zeng",
   "Jiazhao Zhang",
   "Shaoan Wang",
   "Zibo Zhang",
   "Weisi Situ",
   "Yuchen Zhou",
   "Yonggen Ling",
   "Hong Zhang"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "ReferTrack is introduced, a referring-then-tracking paradigm that grounds EVT using a single forward-facing camera, and real-world deployments on legged and humanoid robots validate its robust sim-to-real transfer capabilities.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hanjing Ye",
    "id": "2160747744",
    "h_index": 7,
    "papers": 27
   },
   {
    "name": "Tianle Zeng",
    "id": "2312328756",
    "h_index": 1,
    "papers": 11
   },
   {
    "name": "Jiazhao Zhang",
    "id": "2107990526",
    "h_index": 19,
    "papers": 43
   },
   {
    "name": "Shaoan Wang",
    "id": "2335081438",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Zibo Zhang",
    "id": "2452256550",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Weisi Situ",
    "id": "2452233045",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yuchen Zhou",
    "id": "2261892993",
    "h_index": 8,
    "papers": 26
   },
   {
    "name": "Yonggen Ling",
    "id": "2335590933",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Hong Zhang",
    "id": "2274082906",
    "h_index": 5,
    "papers": 21
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "humanoids",
   "sim2real"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.20061v1",
  "pdf_url": "https://arxiv.org/pdf/2607.20061v1",
  "html_url": "https://arxiv.org/html/2607.20061v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.20033",
  "slug": "host-robots-acquire-manipulation-skills-in-seconds-from-a-single-human",
  "title": "HOST:Robots Acquire Manipulation Skills in Seconds from a Single Human Video",
  "abstract": "The ability to acquire skills rapidly and effortlessly while retaining those already mastered is essential for robots. However, current methods still rely on a cumbersome training-time loop that is costly and slow, while eroding skills already mastered. In this paper, we introduce HOST (Human-to-robot One-Shot Skill AcquisiTion), a framework that enables a robot to acquire skills in seconds from a single human video while retaining previously mastered skills. HOST resolves skill acquisition through a cascade of self-grounded prediction. It first estimates the robot's progress within the demonstrated task, then translates the upcoming progression into the robot's own future observations, and finally derives actions from these predicted observations. This cascade is trained on targets coupled to the video demonstration, obtained by mapping the robot trajectory and the video demonstration onto a shared task progress manifold, then redefining each target to align with the future progression of the video. HOST thereby enables the robot to actively follow the demonstrated procedure and adapt it to the robot's embodiment. HOST acquires novel skills at inference time from a single human video in an average of 29 seconds and achieves a 62% average success rate. It exceeds the zero-shot baseline by 45% while retaining previously mastered skills. HOST even exceeds the baseline fine-tuned on 50 robot demonstrations per task while requiring 50 times fewer demonstrations and acquiring each skill 507 times faster. Additional information about HOST is available on the project website.",
  "published": "2026-07-22",
  "updated": "2026-08-20",
  "year": "2026",
  "authors": [
   "Guangyan Chen",
   "Meiling Wang",
   "Te Cui",
   "Zichen Zhou",
   "Qi Shao",
   "Shalfun Li",
   "Hang Su",
   "Roy Gan",
   "Hao Wang",
   "Mengyin Fu",
   "Yi Yang",
   "Yufeng Yue"
  ],
  "author_count": 12,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "HOST (Human-to-robot One-Shot Skill AcquisiTion), a framework that enables a robot to acquire skills in seconds from a single human video while retaining previously mastered skills.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Guangyan Chen",
    "id": "2221150662",
    "h_index": 6,
    "papers": 21
   },
   {
    "name": "Meiling Wang",
    "id": "2297340202",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Te Cui",
    "id": "2269734310",
    "h_index": 5,
    "papers": 16
   },
   {
    "name": "Zichen Zhou",
    "id": "2383067372",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Qi Shao",
    "id": "2397359612",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Shalfun Li",
    "id": "2380572085",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Hang Su",
    "id": "2275603083",
    "h_index": 1,
    "papers": 13
   },
   {
    "name": "Roy Gan",
    "id": "2380522450",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Hao Wang",
    "id": "2277425499",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Mengyin Fu",
    "id": "2240528626",
    "h_index": 6,
    "papers": 33
   },
   {
    "name": "Yi Yang",
    "id": "2270793657",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Yufeng Yue",
    "id": "2242944661",
    "h_index": 12,
    "papers": 79
   }
  ],
  "comment": "",
  "topics": [
   "egocentric-data"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.20033v4",
  "pdf_url": "https://arxiv.org/pdf/2607.20033v4",
  "html_url": "https://arxiv.org/html/2607.20033v4",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.19919",
  "slug": "diffusion-reroll-revisable-denoising-for-robotic-sequential-prediction",
  "title": "Diffusion ReRoll: Revisable Denoising for Robotic Sequential Prediction",
  "abstract": "We propose Diffusion ReRoll, a diffusion-based framework for robotic sequential prediction that enables revisable denoising over horizons. Existing diffusion-based sequence predictors typically perform a single monotonic denoising process. In contrast, Diffusion ReRoll selectively re-noises regions that have become locally stable while the remaining regions continue denoising, so the re-noised regions can be refined again using context from the rest of the horizon. This structured re-noising enables iterative cross-horizon revision, allowing earlier and later segments to revise one another, while maintaining local consistency. We evaluate Diffusion ReRoll against full-sequence diffusion and causal denoising based on Diffusion Forcing across long-horizon planning, policy learning, and unified video-action modeling. On OGBench PointMaze and AntMaze, Diffusion ReRoll achieves relative gains in average success rate of 21% over Diffusion Forcing in matched guidance-based planning and 23% over Diffuser in matched goal-inpainting. In diffusion-policy-style action prediction, Diffusion ReRoll improves average success by 56.5% relative to Diffusion Policy across different prediction horizons and history lengths on the LIBERO-10 multi-task benchmark. In unified video-action prediction, Diffusion ReRoll improves policy and inverse dynamics performance, especially under out-of-distribution evaluation, and achieves the best action-video consistency. These results support structured re-noising as an effective mechanism for revisable robotic sequence generation.",
  "published": "2026-07-22",
  "updated": "2026-07-22",
  "year": "2026",
  "authors": [
   "Seonsoo Kim",
   "Seongil Hong",
   "Jun-Gill Kang"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Results support structured re-noising as an effective mechanism for revisable robotic sequence generation and improves policy and inverse dynamics performance, especially under out-of-distribution evaluation, and achieves the best action-video consistency.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Seonsoo Kim",
    "id": "2378733728",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "S. Hong",
    "id": "2117087383",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Jun-Gill Kang",
    "id": "2274097952",
    "h_index": 2,
    "papers": 4
   }
  ],
  "comment": "Project Page: https://seonsoo-p1.github.io/DiffusionReRoll/",
  "topics": [
   "imitation-diffusion",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.19919v1",
  "pdf_url": "https://arxiv.org/pdf/2607.19919v1",
  "html_url": "https://arxiv.org/html/2607.19919v1",
  "code_url": "https://seonsoo-p1.github.io/DiffusionReRoll/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2607.19903",
  "slug": "what-matters-in-humanoid-general-motion-tracking-an-empirical-study",
  "title": "What Matters in Humanoid General Motion Tracking? An Empirical Study",
  "abstract": "Humanoid general motion tracking requires policies that can follow diverse whole-body references while maintaining balance. Building such policies involves many practical design choices, and their individual effects are often hard to assess. We address this issue with an empirical study of common modeling and training factors used in recent humanoid motion-imitation pipelines. To make the study controlled and reproducible, we developed YAHMP, an open-source modular framework for training, evaluating, and deploying whole-body motion tracking policies on the Unitree G1. Within YAHMP, we define a nominal configuration and compare variants that differ in motion-command representation, observation history, action representation, actuation profile, hand-force randomization during training, and training approach. We evaluate the resulting policies on a test set of retargeted human motions and compare the nominal policy with TWIST2 as an external baseline trained on the same motion set. The results distinguish choices with clear tracking effects from choices that mainly change actuation effort, training complexity, or physical interaction capability. Finally, we deploy YAHMP policies zero-shot on the real Unitree G1, demonstrating diverse whole-body motion tracking, balance under external perturbations, and forceful interaction.",
  "published": "2026-07-22",
  "updated": "2026-07-22",
  "year": "2026",
  "authors": [
   "Fabio Amadio",
   "Enrico Mingo Hoffman"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "An empirical study of common modeling and training factors used in recent humanoid motion-imitation pipelines, developing YAHMP, an open-source modular framework for training, evaluating, and deploying whole-body motion tracking policies on the Unitree G1.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Fabio Amadio",
    "id": "2336951383",
    "h_index": 7,
    "papers": 22
   },
   {
    "name": "Enrico Mingo Hoffman",
    "id": "2336951546",
    "h_index": 2,
    "papers": 8
   }
  ],
  "comment": "",
  "topics": [
   "humanoids"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2607.19903v1",
  "pdf_url": "https://arxiv.org/pdf/2607.19903v1",
  "html_url": "https://arxiv.org/html/2607.19903v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.8
 },
 {
  "id": "2607.19880",
  "slug": "ea-nav-learning-safe-visual-navigation-policies-with-embodiment-awaren",
  "title": "EA-Nav: Learning Safe Visual Navigation Policies with Embodiment Awareness",
  "abstract": "Cross-embodiment navigation is a key challenge in embodied intelligence. Due to differences in embodiment, the same visual observation may imply different actions for different agents, making prediction ambiguous when relying solely on vision. Existing studies mainly rely on reinforcement learning, which requires large-scale interaction and careful reward design, making it difficult to support scalable pretraining and real-world adaptation. In contrast, imitation-learning-based approaches remain limited. To address these challenges, we propose an imitation-learning-based embodiment-aware navigation framework with a modular multi-stage design. In pretraining, we construct a cross-embodiment navigation dataset from Internet videos and introduce embodiment geometry as conditional tokens to reduce action ambiguity under the same observation. In fine-tuning, we design a multimodal information injection mechanism based on a decoupled architecture. Specifically, we design a trajectory augmentation strategy to generate high-risk samples, which are used to train spatial perception and risk-aware correction separately, thereby explicitly incorporating embodiment geometry for safe navigation. Experimental results show that the proposed method effectively improves navigation performance across different embodiment settings, demonstrating the effectiveness of incorporating embodiment geometry into embodied navigation.",
  "published": "2026-07-22",
  "updated": "2026-08-05",
  "year": "2026",
  "authors": [
   "Jialu Zhang",
   "Yong Du",
   "Xianda Guo",
   "Shunwang Sun",
   "Xinqi Liu",
   "Yue Sun",
   "Guodong Lu",
   "Wei Sui",
   "Jituo Li"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Experimental results show that the proposed method effectively improves navigation performance across different embodiment settings, demonstrating the effectiveness of incorporating embodiment geometry into embodied navigation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jialu Zhang",
    "id": "2317130686",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Yong Du",
    "id": "2452294433",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Xianda Guo",
    "id": "2284674027",
    "h_index": 8,
    "papers": 23
   },
   {
    "name": "Shunwang Sun",
    "id": "2418665559",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Xinqi Liu",
    "id": "2137342124",
    "h_index": 4,
    "papers": 16
   },
   {
    "name": "Yue Sun",
    "id": "2317121008",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Guodong Lu",
    "id": "2283403456",
    "h_index": 4,
    "papers": 20
   },
   {
    "name": "Wei Sui",
    "id": "2382917759",
    "h_index": 3,
    "papers": 18
   },
   {
    "name": "Jituo Li",
    "id": "2767548",
    "h_index": 15,
    "papers": 86
   }
  ],
  "comment": "",
  "topics": [
   "rl-control",
   "navigation",
   "foundation-pretraining",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.19880v3",
  "pdf_url": "https://arxiv.org/pdf/2607.19880v3",
  "html_url": "https://arxiv.org/html/2607.19880v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.19876",
  "slug": "kinebench-benchmarking-embodied-world-models-via-idm-free-kinematic-gr",
  "title": "KineBench: Benchmarking Embodied World Models via IDM-Free Kinematic Grounding",
  "abstract": "Evaluating the physical consistency of embodied world models(EWMs) is a critical open challenge. While closed-loop evaluation via simulator rollouts offers a more faithful assessment of physical plausibility than open-loop alternatives, existing frameworks almost exclusively rely on Inverse Dynamics Models(IDMs) for action extraction. Due to the intricate mapping from 2D pixel space to 3D kinematic space, the learned IDMs can be brittle to data outside their training distribution, resulting in unreliable action extraction from the generated videos with novel objects and scenarios. This creates an unavoidable attribution ambiguity between world model inaccuracies and extractor errors. To reduce this ambiguity, we present KineBench, an IDM-free closed-loop benchmark for EWMs, built upon an explicit kinematic grounding pipeline. Given a generated video, KineBench employs cascaded visual foundation models to directly extract 6D end-effector poses from individual frames, which are then executed in a physics simulator for closed-loop validation. Beyond execution-based task success, KineBench incorporates two classical 3D kinematic metrics--Spectral Arc Length (SPARC) and the Maruyama Manipulability Index--to characterize trajectory smoothness and kinematic feasibility from a robot-centric perspective. Built on 20 diverse manipulation tasks in ManiSkill3, KineBench evaluates EWMs across four progressive suites: basic execution, task transfer, visual out-of-distribution generalization, and complexity-conditioned scaling. Evaluation across frontier models reveals task-complexity-bounded nonlinear scaling in embodied video generation, providing empirical guidance for future data-scaling strategies.",
  "published": "2026-07-22",
  "updated": "2026-07-22",
  "year": "2026",
  "authors": [
   "Zeyu Liu",
   "Zhangzhe Zhu",
   "Yang Zhang",
   "Chenyou Fan",
   "Chenjia Bai",
   "Xuelong Li"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "KineBench is an IDM-free closed-loop benchmark for EWMs, built upon an explicit kinematic grounding pipeline, and reveals task-complexity-bounded nonlinear scaling in embodied video generation, providing empirical guidance for future data-scaling strategies.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zeyu Liu",
    "id": "2312198055",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Zhangzhe Zhu",
    "id": "2452285819",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yang Zhang",
    "id": "2301230693",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Chenyou Fan",
    "id": "2320339180",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Chenjia Bai",
    "id": "2303257958",
    "h_index": 7,
    "papers": 18
   },
   {
    "name": "Xuelong Li",
    "id": "2295686463",
    "h_index": 10,
    "papers": 30
   }
  ],
  "comment": "Accept to ECCV2026",
  "topics": [
   "world-models",
   "sim2real",
   "foundation-pretraining",
   "video-generation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.19876v1",
  "pdf_url": "https://arxiv.org/pdf/2607.19876v1",
  "html_url": "https://arxiv.org/html/2607.19876v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2607.19804",
  "slug": "v2f-vision-informed-grasp-force-prediction-for-damage-aware-robotic-ha",
  "title": "V2F: Vision-Informed Grasp Force Prediction for Damage-Aware Robotic Handling of Date Fruits",
  "abstract": "This paper presents a vision-informed grasp force prediction framework for robotic handling of date fruits. Addressing the dual challenge of high detachment forces and low bruise thresholds, we first conduct mechanical characterization on date samples to define a safe grasping envelope and quantify the relationship between fruit geometry and bioyield stress. In this work, we develop a Vision-to-Force (V2F) pipeline that combines computer vision-based segmentation, active-contour refinement, and geometric feature extraction with a physics-informed residual neural network that augments a Hertz contact equation. The resulting model maps non-contact visual descriptors and cultivar metadata to predict a safe grasp force with mean validation performance of $R^2 \\approx 0.7$ across unseen cultivar groups, which is a good result given the inherent mechanical variability of biological tissue. Experimental validation using a gripper and load cell indicates that the predicted forces enable stable manipulation of different types of date fruits, with residual deformations below 1 mm and no observable damage. These results show that pre-emptive, vision-driven force estimation% can replace slow and potentially damaging tactile exploration , enabling safer robotic handling of fragile fruits.",
  "published": "2026-07-22",
  "updated": "2026-07-22",
  "year": "2026",
  "authors": [
   "Shahd Shami",
   "Obadah Wali",
   "Eric Feron",
   "Shinkyu Park"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Results show that pre-emptive, vision-driven force estimation% can replace slow and potentially damaging tactile exploration, enabling safer robotic handling of fragile fruits.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Shahd Shami",
    "id": "2452239926",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Obadah Wali",
    "id": "2122157388",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Eric Feron",
    "id": "2181701143",
    "h_index": 4,
    "papers": 19
   },
   {
    "name": "Shinkyu Park",
    "id": "2333520575",
    "h_index": 1,
    "papers": 6
   }
  ],
  "comment": "Accepted to IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS) 2026",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.19804v1",
  "pdf_url": "https://arxiv.org/pdf/2607.19804v1",
  "html_url": "https://arxiv.org/html/2607.19804v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2607.19745",
  "slug": "egorecovery-acquiring-failure-recovery-ability-through-human-recovery",
  "title": "EgoRecovery: Acquiring Failure Recovery Ability Through Human Recovery Demonstration",
  "abstract": "Robust embodied robots should be able to recover from failures and retry tasks in order to operate reliably in unstructured and noisy real-world environments. Achieving this capability requires training policies on data that captures recovery behaviors. However, collecting such data through robot teleoperation is difficult to scale, as it is time-consuming to induce diverse failure states, perform corrective actions, and reset the environment. This challenge is further exacerbated by the high diversity of failure modes, which demands substantially more recovery data than success demonstrations. In this work, we show that egocentric human data capturing failure recovery processes provides a scalable alternative. By efficiently arranging task-level failure configurations and recording short recovery segments, human operators can generate more than 10x as much valid recovery data per hour compared to robot teleoperation under our protocol. To address the embodiment gap between human and robot, we propose EgoRecovery, a co-training framework for learning recovery behavior, where human recovery demonstrations are aligned to a compact corrective-intent space shared with robot data, which captures the timing and magnitude of correction. Only a small number of robot recovery demonstrations are required to connect this intent to executable robot actions. At deployment, a learned recovery gate predicts when correction is needed from robot observations and activates the corrective intent only in recovery states. Experiments on real-world recovery tasks show that EgoRecovery improves success from failure starts over robot-only recovery, direct co-training with human recovery data, and direct intent-transfer baselines.",
  "published": "2026-07-22",
  "updated": "2026-07-23",
  "year": "2026",
  "authors": [
   "Zuhao Ge",
   "Yuchen Zhou",
   "Weitao Zhou",
   "Minglei Li",
   "Xinyu Li",
   "Chao Wu",
   "Hanwen Zhao",
   "Haotian Wang",
   "Zuxuan Wu",
   "Xiaosong Jia",
   "Yu-Gang Jiang"
  ],
  "author_count": 11,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work shows that egocentric human data capturing failure recovery processes provides a scalable alternative and proposes EgoRecovery, a co-training framework for learning recovery behavior, where human recovery demonstrations are aligned to a compact corrective-intent space shared with robot data, which captures the timing and magnitude of correction.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zuhao Ge",
    "id": "2032815387",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Yuchen Zhou",
    "id": "2375400254",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Weitao Zhou",
    "id": "2148943699",
    "h_index": 7,
    "papers": 29
   },
   {
    "name": "Minglei Li",
    "id": "2448263547",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Xinyu Li",
    "id": "2449465524",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Chao Wu",
    "id": "2385840798",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Hanwen Zhao",
    "id": "2452330422",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Haotian Wang",
    "id": "2452285484",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Zuxuan Wu",
    "id": "2384164233",
    "h_index": 2,
    "papers": 13
   },
   {
    "name": "Xiaosong Jia",
    "id": "1958998899",
    "h_index": 23,
    "papers": 53
   },
   {
    "name": "Yu-Gang Jiang",
    "id": "2409877655",
    "h_index": 3,
    "papers": 21
   }
  ],
  "comment": "",
  "topics": [
   "egocentric-data",
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.19745v2",
  "pdf_url": "https://arxiv.org/pdf/2607.19745v2",
  "html_url": "https://arxiv.org/html/2607.19745v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.19719",
  "slug": "koopman-dreamer-spectrally-constrained-latent-dynamics-for-stable-worl",
  "title": "Koopman Dreamer: Spectrally Constrained Latent Dynamics for Stable World-Model Imagination",
  "abstract": "Latent world models improve sample efficiency in continuous control by optimizing policies over imagined latent trajectories, but common neural transitions offer limited direct control over modal persistence and error accumulation in long rollouts. We propose Koopman Dreamer, a Dreamer-style world model with a spectrally constrained deterministic latent dynamics core. Its Koopman-inspired backbone uses two-dimensional rotation--scaling blocks with bounded radii to represent damping, rotation, and near-periodic modes. Linear and low-rank bilinear action terms capture global and state-dependent control effects, while stochastic-state modulation supplies local correction information. To reduce the mismatch between posterior-conditioned training and prior-only imagination, the model combines posterior-conditioned EMA teacher targets with one-step consistency, multi-step rollout, and open-loop observation-prediction objectives. We further derive a multi-step rollout-error bound that separates amplification by the spectral backbone and bilinear interaction from the additive effects of stochastic-state mismatch and modeling residuals, clarifying the trade-off between error attenuation and long-term information retention. Experimental results on proprioceptive continuous-control tasks from the DeepMind Control Suite and UAV-LiDAR autonomous navigation demonstrate that Koopman Dreamer improves the stability of long-horizon latent rollouts and achieves stronger closed-loop control performance on tasks that rely on high-quality multi-step imagination.",
  "published": "2026-07-22",
  "updated": "2026-08-01",
  "year": "2026",
  "authors": [
   "Jiaqi Li",
   "Xinglong Zhang",
   "Haibin Xie",
   "Yixing Lan",
   "Wei Pan",
   "Xin Xu"
  ],
  "author_count": 6,
  "categories": [
   "cs.LG",
   "cs.RO"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Koopman Dreamer is proposed, a Dreamer-style world model with a spectrally constrained deterministic latent dynamics core that improves the stability of long-horizon latent rollouts and achieves stronger closed-loop control performance on tasks that rely on high-quality multi-step imagination.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jiaqi Li",
    "id": "2445497343",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Xinglong Zhang",
    "id": "2258298525",
    "h_index": 5,
    "papers": 38
   },
   {
    "name": "Haibin Xie",
    "id": "2117713120",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Yixing Lan",
    "id": "2150277967",
    "h_index": 6,
    "papers": 30
   },
   {
    "name": "Wei Pan",
    "id": "2256993725",
    "h_index": 7,
    "papers": 15
   },
   {
    "name": "Xin Xu",
    "id": "2345004165",
    "h_index": 3,
    "papers": 13
   }
  ],
  "comment": "20 pages, 13 figures, 11 tables. Revised manuscript with a more concise and precise abstract and improved clarity and presentation throughout the main text. The main technical content, experimental results, and conclusions remain unchanged",
  "topics": [
   "world-models",
   "navigation"
  ],
  "orgs": [
   "Google DeepMind"
  ],
  "abs_url": "https://arxiv.org/abs/2607.19719v2",
  "pdf_url": "https://arxiv.org/pdf/2607.19719v2",
  "html_url": "https://arxiv.org/html/2607.19719v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2607.19714",
  "slug": "morphing-milr-design-and-control-of-a-cable-driven-limbless-robot-with",
  "title": "Morphing MILR: Design and control of a cable-driven limbless robot with rolling joints for maneuvering in complex environments",
  "abstract": "Limbless robots offer exceptional mobility in confined and cluttered environments due to their slender bodies and their ability to exploit body-terrain interactions. Recent designs incorporating compliance demonstrate robust locomotion without complex sensing or control; however, these systems typically rely on fixed body configurations, with each morphology specialized for a single locomotion mode or environment. This raises a key challenge: how can a single limbless robot achieve versatile locomotion while preserving the robustness of compliance-mediated locomotion? To address this challenge, we present a cable-driven limbless robot that reconfigures body morphology and compliance to enable diverse locomotion modes. Distributed cable actuation generates traveling body waves, while programmable passive compliance enables robust contact-rich locomotion without terrain knowledge or high-bandwidth feedback. Rolling joints reorient bending planes along the body, enabling rapid reconfiguration and smooth transitions between locomotion styles, and incorporate geared locking to maintain configuration without continuous power. By combining programmable bending compliance and morphology control, the platform achieves lateral undulation, sidewinding, rolling, and twisting within a single system. Experiments demonstrate reliable gait generation, traversal in obstacle-rich environments, and transitions between modes, establishing a versatile limbless platform for navigating complex environments with applications in search and rescue, environmental monitoring, and inspection.",
  "published": "2026-07-22",
  "updated": "2026-07-22",
  "year": "2026",
  "authors": [
   "Donoven Dortilus",
   "Tianyu Wang",
   "Galen Tunnicliffe",
   "Matthew Fernandez",
   "Daniel I. Goldman"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Donoven Dortilus",
    "id": "2323369851",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Tianyu Wang",
    "id": "2311376888",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Galen Tunnicliffe",
    "id": "2323369306",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Matthew Fernandez",
    "id": "2309922925",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Daniel I. Goldman",
    "id": "2309247267",
    "h_index": 4,
    "papers": 9
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "tactile",
   "hardware-codesign",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.19714v1",
  "pdf_url": "https://arxiv.org/pdf/2607.19714v1",
  "html_url": "https://arxiv.org/html/2607.19714v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.19708",
  "slug": "contact-persistent-full-actuation-for-aerial-physical-interaction",
  "title": "Contact-Persistent Full Actuation for Aerial Physical Interaction",
  "abstract": "Fully actuated unmanned aerial vehicles (UAVs) are usually certified through rank conditions on a control-allocation matrix or through free-flight tracking performance. For aerial physical interaction, this certification may be incomplete. During sustained contact, part of the available wrench is consumed by the interaction task, and only the residual wrench remains available for stabilization, disturbance rejection, and maneuvering. This paper introduces a control-theoretic framework for \\emph{contact-persistent full actuation}. A rigid-body model on $\\R^{3}\\times\\SO\\left(3\\right)$ is combined with a morphology-dependent wrench map that captures fixed-tilt, variable-tilt, coaxial, and overactuated multirotor architectures. We define feasible wrench sets under actuator limits, residual wrench sets under task loading, and residual authority margins that strengthen the usual rank-based notion of full actuation. The main result shows that contact-persistent full actuation is equivalent to interiority of the task wrench in the constrained feasible wrench polytope, and that the residual authority radius is exactly the distance to the polytope boundary. We further introduce a signed residual-margin certificate for infeasible and boundary cases, a slack-maximizing allocation certificate, and a robust implementability condition that can be used as a margin-aware safety filter. Numerical evaluation on an abstract tilted hexarotor shows that full row rank alone does not imply feasible contact operation. Intermediate tilt angles preserve residual authority during pushing, whereas small or excessive tilts fail because of lateral-force deficiency or hover-margin loss.",
  "published": "2026-07-22",
  "updated": "2026-07-22",
  "year": "2026",
  "authors": [
   "Abhimanyu Khadga",
   "Abhinav Sinha",
   "Shashi Ranjan Kumar"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "eess.SY",
   "math.DS",
   "math.OC"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A control-theoretic framework for contact-persistent full actuation is introduced, showing that contact-persistent full actuation is equivalent to interiority of the task wrench in the constrained feasible wrench polytope, and that the residual authority radius is exactly the distance to the polytope boundary.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Abhimanyu Khadga",
    "id": "2452255308",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Abhinav Sinha",
    "id": "143668057",
    "h_index": 19,
    "papers": 133
   },
   {
    "name": "S. R. Kumar",
    "id": "2109680801",
    "h_index": 24,
    "papers": 158
   }
  ],
  "comment": "",
  "topics": [
   "navigation",
   "hardware-codesign",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.19708v1",
  "pdf_url": "https://arxiv.org/pdf/2607.19708v1",
  "html_url": "https://arxiv.org/html/2607.19708v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.19633",
  "slug": "lens-llm-guided-environment-simplification-for-planning-and-control-in",
  "title": "LENS: LLM-guided Environment Simplification for Planning and Control in Clutter",
  "abstract": "Despite recent advances in general-purpose robotic manipulation, real-world multi-object clutter remains challenging to handle for today's prevalent approaches. The problem scales in complexity due to more objects and collisions, more unpredictable contact physics, distractors, and task ambiguity. Bridging this gap to real-world deployment requires effective scene abstractions; yet today, producing such abstractions requires extensive task-specific manual engineering, which does not scale. These abstractions are costly to generate and difficult to adjust or fine-tune. We instead propose a plug-and-play fix to automatically generate scene-specific, task-specific, adaptively updating abstractions on top of existing planning and control stacks. LLM-guided Environment Simplification (LENS) produces a de-cluttered abstracted scene representation by merging (e.g., stacked objects) or pruning (e.g., distant objects) scene entities in a closed loop in response to task progress. These dynamic, task-relevant abstractions are versatile and easy to use. In our experiments, we show that LENS improves classical planning, model-based control, and a vision-language-action model, across a diverse set of highly cluttered manipulation scenes. Project website: https://lens-2026.github.io/.",
  "published": "2026-07-22",
  "updated": "2026-07-22",
  "year": "2026",
  "authors": [
   "Aileen Liao",
   "Rachel Holladay",
   "Dinesh Jayaraman",
   "Michael Posa"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes a plug-and-play fix to automatically generate scene-specific, task-specific, adaptively updating abstractions on top of existing planning and control stacks, and shows that LENS improves classical planning, model-based control, and a vision-language-action model, across a diverse set of highly cluttered manipulation scenes.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Aileen Liao",
    "id": "2380609343",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Rachel Holladay",
    "id": "2486032",
    "h_index": 14,
    "papers": 22
   },
   {
    "name": "Dinesh Jayaraman",
    "id": "2352942932",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Michael Posa",
    "id": "2261351298",
    "h_index": 3,
    "papers": 7
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.19633v1",
  "pdf_url": "https://arxiv.org/pdf/2607.19633v1",
  "html_url": "https://arxiv.org/html/2607.19633v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.19534",
  "slug": "learning-personalized-safety-interventions-for-haptic-human-robot-shar",
  "title": "Learning Personalized Safety Interventions for Haptic Human-Robot Shared Control",
  "abstract": "Haptic feedback provides an implicit channel for communicating safety intentions during human-robot shared control. Existing haptic guidance systems typically employ predefined intervention strategies that cannot accommodate the diverse safety preferences of individual users or application scenarios. To address this limitation, we propose a Learning from Haptics (LfH) framework that learns user-preferred safety interventions from sparse demonstrations, eliminating the need for manual trial-and-error design. Our framework is built on a differentiable Control Barrier Function (CBF)-based optimization layer that automatically adjusts the underlying safety parameters to match the demonstrated haptic responses. Instead of tuning controller parameters directly, users teach the system how they expect it to intervene during teleoperation. The resulting haptic guidance reflects the demonstrated intervention preferences while preserving the intuitive interaction of haptic shared control. Simulation and hardware experiments demonstrate that the proposed framework can learn personalized safety interventions from sparse user input and reduce the mismatch between the generated haptic feedback and the demonstrated preferences.",
  "published": "2026-07-21",
  "updated": "2026-07-21",
  "year": "2026",
  "authors": [
   "Dawei Zhang",
   "Roberto Tron"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A Learning from Haptics (LfH) framework that learns user-preferred safety interventions from sparse demonstrations, eliminating the need for manual trial-and-error design is proposed.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Dawei Zhang",
    "id": "2274062015",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Roberto Tron",
    "id": "2273977308",
    "h_index": 2,
    "papers": 8
   }
  ],
  "comment": "",
  "topics": [
   "rl-control",
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.19534v1",
  "pdf_url": "https://arxiv.org/pdf/2607.19534v1",
  "html_url": "https://arxiv.org/html/2607.19534v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.19479",
  "slug": "modpack-an-extensible-teleoperation-interface-for-bimanual-mobile-mani",
  "title": "ModPack: An Extensible Teleoperation Interface for Bimanual Mobile Manipulation",
  "abstract": "Existing teleoperation systems are often tailored to specific robot hardware and task domains, limiting their scalability and adaptability. We present ModPack, a modular and extensible teleoperation system designed to support diverse robot embodiments and task requirements within a unified framework. At the core of ModPack is a self-contained wearable \"backpack\" that integrates onboard computation, power, communication, and data storage. Built on top of this shared interface, the system supports plug-and-play capability modules including joint-level teleoperation with haptic feedback, mobile manipulation, and active perception. Experiments across two distinct robot platforms and real-world mobile manipulation tasks demonstrate that ModPack provides a flexible and reusable framework for data collection and policy learning. To support future research, we open-source the complete hardware design and software stack. Project website: https://modpack-robotics.github.io/",
  "published": "2026-07-21",
  "updated": "2026-07-21",
  "year": "2026",
  "authors": [
   "Joshua Citron",
   "Renee Zbizika",
   "Zeyi Liu",
   "Shuran Song"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "ModPack is a modular and extensible teleoperation system designed to support diverse robot embodiments and task requirements within a unified framework that integrates onboard computation, power, communication, and data storage.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Joshua Citron",
    "id": "2284223401",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Renee Zbizika",
    "id": "2378980381",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Zeyi Liu",
    "id": "2176845464",
    "h_index": 9,
    "papers": 11
   },
   {
    "name": "Shuran Song",
    "id": "2289085682",
    "h_index": 7,
    "papers": 13
   }
  ],
  "comment": "",
  "topics": [
   "navigation",
   "data-teleop",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.19479v1",
  "pdf_url": "https://arxiv.org/pdf/2607.19479v1",
  "html_url": "https://arxiv.org/html/2607.19479v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.19343",
  "slug": "masked-visual-actions-for-unified-world-modeling",
  "title": "Masked Visual Actions for Unified World Modeling",
  "abstract": "Video models absorb rich priors over how the visual world moves, interacts, and responds to contact, making them promising substrates for robotic world modeling. The central challenge is how to communicate action to such models in a form aligned with the visual space in which they learned these interaction priors, yet still grounded in physical manipulation. We introduce Masked Visual Actions, a pixel-space control interface that expresses action as a partially revealed trajectory of an arbitrary entity in a video. Revealing robot motion makes the model act as a forward dynamics model that predicts the scene's response to low-level robot actions, while revealing desired object motion makes the same model recover robot behavior consistent with that outcome. Finetuned with only 15 hours of masked examples from real videos and simulation, a single checkpoint achieves strong visual fidelity and controllability across diverse scenes and multiple embodiments. In downstream manipulation settings, the model produces imagined rollouts whose outcomes correlate with real-world execution for policy evaluation, improves decision making by ranking candidate futures in model-based planning, and supports inverse modeling by synthesizing robot motion from desired object motion.",
  "published": "2026-07-21",
  "updated": "2026-07-21",
  "year": "2026",
  "authors": [
   "Hadi Alzayer",
   "Wenlong Huang",
   "Haonan Chen",
   "Christopher Luey",
   "Lvmin Zhang",
   "Maneesh Agrawala",
   "Gordon Wetzstein",
   "Li Fei-Fei",
   "Yilun Du",
   "Jiajun Wu",
   "Jia-Bin Huang"
  ],
  "author_count": 11,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "Masked Visual Actions is introduced, a pixel-space control interface that expresses action as a partially revealed trajectory of an arbitrary entity in a video that supports inverse modeling by synthesizing robot motion from desired object motion.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hadi Alzayer",
    "id": "2127598598",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Wenlong Huang",
    "id": "2319777572",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Haonan Chen",
    "id": "2309175413",
    "h_index": 6,
    "papers": 19
   },
   {
    "name": "Christopher Luey",
    "id": "2268829337",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Lvmin Zhang",
    "id": "2287850511",
    "h_index": 8,
    "papers": 24
   },
   {
    "name": "Maneesh Agrawala",
    "id": "1820412",
    "h_index": 78,
    "papers": 290
   },
   {
    "name": "Gordon Wetzstein",
    "id": "2297763521",
    "h_index": 7,
    "papers": 20
   },
   {
    "name": "Fei-Fei Li",
    "id": "2330589126",
    "h_index": 3,
    "papers": 15
   },
   {
    "name": "Yilun Du",
    "id": "2383299857",
    "h_index": 2,
    "papers": 14
   },
   {
    "name": "Jiajun Wu",
    "id": "2326444500",
    "h_index": 35,
    "papers": 214
   },
   {
    "name": "Jia-Bin Huang",
    "id": "2213332546",
    "h_index": 41,
    "papers": 108
   }
  ],
  "comment": "Project webpage: https://masked-visual-actions.github.io",
  "topics": [
   "world-models"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.19343v1",
  "pdf_url": "https://arxiv.org/pdf/2607.19343v1",
  "html_url": "https://arxiv.org/html/2607.19343v1",
  "code_url": "https://masked-visual-actions.github.io",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.3
 },
 {
  "id": "2607.19306",
  "slug": "from-distances-to-trajectories-real-time-signed-distance-function-mapp",
  "title": "From Distances to Trajectories: Real-Time Signed Distance Function Mapping and Distance-Accelerated Motion Planning for UAVs",
  "abstract": "Autonomous flight in cluttered environments requires a robot to build a geometric map of its surroundings and plan safe, dynamically feasible trajectories, all onboard and in real time. Conventional approaches treat mapping and planning as separate stages and often rely on binary occupancy for collision checking. We argue that these two stages should be co-designed around a single representation: a signed distance function (SDF). By encoding distance to the nearest obstacle, an SDF provides richer information for planning and trajectory optimization than occupancy alone. We develop an Octree REsidual Network (OREN) that pairs an explicit octree prior with an implicit neural residual to reconstruct SDFs online from point cloud observations with the efficiency of volumetric methods and the accuracy and differentiability of neural methods. In tandem, we develop Bubble$^\\star$, a search-based planner that exploits the distance information to grow maximal collision-free balls, which we call bubbles, with formal guarantees of termination, completeness, and failure detection. Planning over a graph of bubbles significantly reduces collision checks compared to a grid-based A$^\\star$ search and returns a bubble sequence that forms a safe corridor for trajectory optimization. We demonstrate the integrated OREN-Bubble$^\\star$ approach onboard a quadrotor, navigating unseen indoor environments in real time under tight compute constraints. OREN improves SDF estimation by $22$% compared to baselines, while Bubble$^\\star$ finds trajectories spanning $\\approx 90$ m through a cluttered environment in $1$-$3$ sec., whereas baselines take up to $10$ sec. in the same environment.",
  "published": "2026-07-21",
  "updated": "2026-07-21",
  "year": "2026",
  "authors": [
   "Jason Stanley",
   "Zhirui Dai",
   "Qihao Qian",
   "Tzu-Chin Ho",
   "Tianxing Fan",
   "Siddharth Saha",
   "Christopher Barngrover",
   "Ki Myung Brian Lee",
   "Nikolay Atanasov"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "An Octree REsidual Network (OREN) is developed that pairs an explicit octree prior with an implicit neural residual to reconstruct SDFs online from point cloud observations with the efficiency of volumetric methods and the accuracy and differentiability of neural methods.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jason Stanley",
    "id": "2242831838",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Zhirui Dai",
    "id": "2242895999",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Qihao Qian",
    "id": "107960746",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Tzu-Chin Ho",
    "id": "2451878634",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Tianxing Fan",
    "id": "2137863299",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Siddhartha Saha",
    "id": "2114848259",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "C. Barngrover",
    "id": "2436751",
    "h_index": 6,
    "papers": 19
   },
   {
    "name": "K. Lee",
    "id": "2316987578",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Nikolay Atanasov UC San Diego",
    "id": "2451878174",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "La Jolla",
    "id": "20499297",
    "h_index": 16,
    "papers": 80
   },
   {
    "name": "Usa",
    "id": "2238206301",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "AI Shield",
    "id": "2451864690",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "San Diego",
    "id": "2451878220",
    "h_index": 0,
    "papers": 1
   }
  ],
  "comment": "25 pages, 10 figures, 5 tables",
  "topics": [
   "spatial-3d",
   "navigation",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.19306v1",
  "pdf_url": "https://arxiv.org/pdf/2607.19306v1",
  "html_url": "https://arxiv.org/html/2607.19306v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.19194",
  "slug": "cognitive-dual-process-planning-for-autonomous-driving-with-structured",
  "title": "Cognitive Dual-Process Planning for Autonomous Driving with Structured Scene Knowledge and Verifiable Reasoning-Action Consistency",
  "abstract": "High-level planning for autonomous driving is a knowledge-intensive engineering decision task that requires accurate scene understanding, timely inference, and internally consistent action selection. Vision-language models (VLMs) can make intermediate reasoning explicit, but their use in deployed planners is constrained by costly structured supervision, unnecessary reasoning in routine scenes, and possible inconsistencies between generated rationales and driving actions. We present a cognitive dual-process planning framework that represents planning-relevant scene knowledge in a machine-parsable structured chain-of-thought (S-CoT) schema. An automated data engine integrates perception foundation models, critical-path filtering, and an expert VLM to generate S-CoT supervision without manual annotation of individual rationales. A lightweight visual Arbiter estimates scene complexity from multilevel vision-encoder features before language decoding and routes each input to either fast meta-action prediction or slow structured reasoning. For slow-path outputs, a deterministic rule-based validator checks whether the parsed S-CoT fields are consistent with the final meta-action and provides verifiable rewards for Group Relative Policy Optimization (GRPO). In a 195-scene manual audit, the generated annotations achieve 91.8\\% CoT accuracy and a 98.5\\% Logical Consistency Score (LCS). On 574 manually verified NAVSIM test samples, the planner achieves 80.14\\% planning accuracy and 97.20\\% LCS while reducing average latency by 17.39\\% relative to applying slow reasoning to every scene. Evaluation on external long-tail subsets further identifies conditions under which routing and planning performance degrade. Together, these results show how explicit scene knowledge can be operationalized through adaptive reasoning and rule-based verification to support high-level VLM planning decisions.",
  "published": "2026-07-21",
  "updated": "2026-07-22",
  "year": "2026",
  "authors": [
   "Zhongyao Yang",
   "Haoyu Li",
   "Yu Yan",
   "Zhuangxuan Yu",
   "Jiangfeng Nan",
   "Jinrui Nan"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A cognitive dual-process planning framework that represents planning-relevant scene knowledge in a machine-parsable structured chain-of-thought (S-CoT) schema and shows how explicit scene knowledge can be operationalized through adaptive reasoning and rule-based verification to support high-level VLM planning decisions.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhongyao Yang",
    "id": "9285386",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Haoyu Li",
    "id": "2283299897",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Yuchen Yan",
    "id": "2449502511",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Zhuangxuan Yu",
    "id": "2451958728",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Jiangfeng Nan",
    "id": "2346445251",
    "h_index": 0,
    "papers": 5
   },
   {
    "name": "Jinrui Nan",
    "id": "2150781360",
    "h_index": 10,
    "papers": 18
   }
  ],
  "comment": "",
  "topics": [
   "rl-control",
   "navigation",
   "foundation-pretraining",
   "data-teleop",
   "video-generation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.19194v2",
  "pdf_url": "https://arxiv.org/pdf/2607.19194v2",
  "html_url": "https://arxiv.org/html/2607.19194v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.19190",
  "slug": "agentic-real2sim-physics-based-world-modeling-with-vision-language-age",
  "title": "Agentic Real2Sim: Physics-based World Modeling with Vision-Language Agents",
  "abstract": "Real-to-sim conversion for robotic interaction with objects remains labor-intensive because it requires more than visual reconstruction: a streamlined real2sim process must recover scene geometries and object states, infer physical parameters, and assemble actors, objects, cameras, poses, and trajectories into a runnable physical simulation. Today this process still depends on manual tuning of visual foundation models, mesh cleanup, coordinate-frame alignment, and brittle workflow glue across visual perception tools and simulators. We introduce \\textit{Agentic Real2Sim}, a framework for generalized physical world modeling with vision-language agents, converting a real-world recording of object-robot interaction into a simulatable episodic twin which preserves observations, geometries, robot interactions, and object states. We evaluate Agentic Real2Sim on rigid-object manipulation, deformable-object interaction, and humanoid motion scenes, spanning domains that are usually handled by separate Real2Sim pipelines, marking a first step toward scalable conversion. The framework's agentic decisions can be driven by an open-weight VLM backend at a small fraction of the cost of frontier models, while attaining comparable conversion success rate. We aim to use the resulting real-world-aligned twins for downstream robotics tasks, specifically policy learning and evaluation. The project site is available at https://agentic-real2sim.github.io/.",
  "published": "2026-07-21",
  "updated": "2026-07-24",
  "year": "2026",
  "authors": [
   "Guanxiong Chen",
   "Qianjun Xia",
   "Jiawei Peng",
   "Heng Zhang",
   "Bole Ma",
   "Justin Qian",
   "Ziyi Jiao",
   "Bingyang Zhou",
   "Luoxin Ye",
   "Kaifeng Zhang",
   "Kunyi Wang",
   "Weijia Zeng",
   "Yunuo Chen",
   "Pengzhi Yang",
   "Ziqiu Zeng",
   "Siyuan Luo",
   "Huamin Wang",
   "Chao Liu",
   "Alan Yuille",
   "Fan Shi",
   "Changxi Zheng",
   "Yunzhu Li",
   "Chenfanfu Jiang",
   "Peter Yichen Chen"
  ],
  "author_count": 24,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Agentic Real2Sim is introduced, a framework for generalized physical world modeling with vision-language agents, converting a real-world recording of object-robot interaction into a simulatable episodic twin which preserves observations, geometries, robot interactions, and object states.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Guanxiong Chen",
    "id": "2149510317",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Qianjun Xia",
    "id": "2451873384",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Jiawei Peng",
    "id": "2269048911",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Heng Zhang",
    "id": "2294361958",
    "h_index": 0,
    "papers": 5
   },
   {
    "name": "Bole Ma",
    "id": "2451869117",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Justin Qian",
    "id": "2451956089",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Ziyi Jiao",
    "id": "2451885676",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Bingyang Zhou",
    "id": "2232926592",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Luoxin Ye",
    "id": "2309199912",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Kaifeng Zhang",
    "id": "2310649159",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Kunyi Wang",
    "id": "2305352582",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Weijia Zeng",
    "id": "2294361684",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Yunuo Chen",
    "id": "2344397292",
    "h_index": 4,
    "papers": 16
   },
   {
    "name": "Pengzhi Yang",
    "id": "49731662",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Ziqiu Zeng",
    "id": "2353345888",
    "h_index": 1,
    "papers": 11
   },
   {
    "name": "Siyuan Luo",
    "id": "2307566563",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Huamin Wang",
    "id": "2326880437",
    "h_index": 2,
    "papers": 11
   },
   {
    "name": "Chao Liu",
    "id": "2448705802",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Alan L. Yuille",
    "id": "2411931872",
    "h_index": 3,
    "papers": 16
   },
   {
    "name": "Fan Shi",
    "id": "2376206373",
    "h_index": 2,
    "papers": 12
   },
   {
    "name": "Changxi Zheng",
    "id": "2391887093",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Yunzhu Li",
    "id": "2352183976",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Chenfanfu Jiang",
    "id": "2267866705",
    "h_index": 16,
    "papers": 77
   },
   {
    "name": "P. Y. Chen",
    "id": "2294384470",
    "h_index": 4,
    "papers": 13
   }
  ],
  "comment": "Authorship change",
  "topics": [
   "world-models",
   "humanoids",
   "sim2real",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.19190v3",
  "pdf_url": "https://arxiv.org/pdf/2607.19190v3",
  "html_url": "https://arxiv.org/html/2607.19190v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.18855",
  "slug": "pose-parameterized-motion-planning-and-cbf-qp-self-collision-filtering",
  "title": "Pose-Parameterized Motion Planning and CBF-QP Self-Collision Filtering for a Long-Reach Drilling Boom",
  "abstract": "Long-reach drilling booms must reach successive poses without self-collision. Moving from operator-supervised control toward autonomy requires collision-aware motion planning and execution. For the Sandvik SB60, this study adapts established methods by integrating pose-parameterized planning with a capsule-based control barrier function quadratic program (CBF-QP) in measured-state inverse kinematics (IK). A fixed task-specific parameter set within each task generates waypoints, detours, timed references, and chained motion without target-specific retuning. The offline detour planner screens candidate waypoints using 23 selected rod-segment-to-body-region distances, whereas the online CBF-QP filters joint velocities using 14 configured capsule-pair constraints from a nine-primitive whole-body capsule model. Evaluation considers two drilling tasks in a manufacturer-developed SB60 Simscape Multibody model: a five-target restricted-orientation tour and a three-target full-pose tour. Across several hundred thousand samples, the method produced zero IK failures, generated several detour waypoints, achieved millimetre-level mean final-position error, and recorded no sampled CBF margins below the reported thresholds.",
  "published": "2026-07-21",
  "updated": "2026-07-21",
  "year": "2026",
  "authors": [
   "Mehdi Heydari Shahna",
   "Tuomo Kivel\u00e4",
   "Jouni Mattila"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Mehdi Heydari Shahna",
    "id": "1556780569",
    "h_index": 6,
    "papers": 20
   },
   {
    "name": "T. Kivel\u00e4",
    "id": "71846005",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "J. Mattila",
    "id": "143939480",
    "h_index": 23,
    "papers": 276
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.18855v1",
  "pdf_url": "https://arxiv.org/pdf/2607.18855v1",
  "html_url": "https://arxiv.org/html/2607.18855v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.18840",
  "slug": "worldscape-policy-2-0-empowering-steerable-world-action-modeling-with",
  "title": "WorldScape Policy 2.0: Empowering Steerable World Action Modeling with Reasoning-Augmented Memory",
  "abstract": "World Action Models (WAMs) offer a promising paradigm for robotic manipulation by jointly modeling visual state transitions and robot actions. However, existing WAMs are constrained by limited temporal context, coarse episode-level language supervision, and predominantly text-only conditioning, which hinder task-progress tracking and fine-grained language-video-action grounding while limiting visual-context reasoning and cross-embodiment transfer. In this paper, we introduce WorldScape Policy 2.0, a controllable WAM with reasoning-augmented long short-term memory. Its causal short-term visual memory supplies recent observations as DiT prefill to preserve local interaction dynamics, while its long short-term event memory organizes historical VLM outputs into global-history, local-active, and event-boundary representations for progress-aware retrieval. The retrieved history augments perception and autoregressively generated planning tokens, yielding an implicit subgoal condition for autonomous planning; semantic forcing further transfers event-level instruction semantics into this latent planning pathway. To establish fine-grained multimodal controllability, we construct ManipEvent-5M, an event-grounded embodied pretraining dataset containing nearly 5 million event segments with aligned action trajectories, episode-level task instructions, segment-level subtask captions, goal images, and video demonstrations. These designs provide a unified interface for autonomous planning from high-level instructions and controllable execution from fine-grained text, goal-image, or video-context prompts. Experiments in both simulation and real-world platforms demonstrate superior capabilities in long-horizon autonomous planning, fine-grained instruction following and in-context adaptation.",
  "published": "2026-07-21",
  "updated": "2026-07-21",
  "year": "2026",
  "authors": [
   "Haisheng Su",
   "Zongdai Liu",
   "Xin Jin",
   "Haoxuan Dou",
   "Chengming Hu",
   "Baorun Li",
   "Zhanwang Liu",
   "Ruiyan Xu",
   "Jianjie Fang",
   "Xin Zhang",
   "Zhenjie Yang",
   "Xue Yang",
   "Chen Gao",
   "Junchi Yan",
   "Yong Li",
   "Wei Wu"
  ],
  "author_count": 16,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "WorldScape Policy 2.0 is introduced, a controllable WAM with reasoning-augmented long short-term memory and fine-grained instruction following and in-context adaptation that demonstrates superior capabilities in long-horizon autonomous planning, fine-grained instruction following and in-context adaptation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Haisheng Su",
    "id": "2363482540",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Zongdai Liu",
    "id": "2452182668",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Xin Jin",
    "id": "2382888447",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Hao Dou",
    "id": "2298908307",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Chengming Hu",
    "id": "2452176405",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Baorun Li",
    "id": "2451767233",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Zhanwang Liu",
    "id": "2451838592",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Ruiyan Xu",
    "id": "2451953152",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Jianjie Fang",
    "id": "2326466107",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Xin Zhang",
    "id": "2333419153",
    "h_index": 4,
    "papers": 22
   },
   {
    "name": "Zhenjie Yang",
    "id": "2264665546",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Xue Yang",
    "id": "2353618153",
    "h_index": 4,
    "papers": 18
   },
   {
    "name": "Chen Gao",
    "id": "2351212267",
    "h_index": 7,
    "papers": 26
   },
   {
    "name": "Junchi Yan",
    "id": "2268719823",
    "h_index": 13,
    "papers": 29
   },
   {
    "name": "Yong Li",
    "id": "2300459663",
    "h_index": 6,
    "papers": 22
   },
   {
    "name": "Wei Wu",
    "id": "2302827202",
    "h_index": 5,
    "papers": 15
   }
  ],
  "comment": "",
  "topics": [
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.18840v1",
  "pdf_url": "https://arxiv.org/pdf/2607.18840v1",
  "html_url": "https://arxiv.org/html/2607.18840v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.18794",
  "slug": "beyond-transformers-linear-attention-policy-for-open-vocabulary-object",
  "title": "Beyond Transformers: Linear Attention Policy for Open-Vocabulary Object Goal Navigation",
  "abstract": "Open-Vocabulary Object Goal Navigation (OVON) requires agents to operate under partial observability, making effective internal state updates critical for navigation performance. This update is implemented by the policy network, where recent approaches adopt Transformer-based backbones with self-attention over a context window to integrate temporal information. However, our controlled experiments show that performance does not scale with context length under Transformer-based policies, questioning the suitability of self-attention for state integration in navigation. To this end, we propose Linear Attention-based Navigation (LANav), which adopts linear attention (LA) as the policy backbone to maintain a structured state update rather than self-attention over the context window. Across multiple LA variants evaluated under identical settings, LANav consistently outperforms Transformer-based baselines. Performance improves as state update mechanisms become more structured and regulated, highlighting the importance of state update design. To improve state update effectiveness, we introduce Weighted State-Expansion Linear Attention (WSLA), which expands each attention head's state into multiple sub-states and uses learnable weighted readout to aggregate expanded sub-states. Equipped with WSLA, LANav achieves 36.4% average success rate (SR) on HM3D-OVON, outperforming Transformer-based counterparts by 6.3 percentage points in macro-averaged SR, while maintaining computational efficiency. Distance-stratified results show larger gains in long-distance episodes, while HSSD transfer and fine-tuning demonstrate robustness across scene distributions. Real-world deployment on a Unitree Go2 further achieves an 82% success rate over 50 trials, supporting the practical feasibility and sim-to-real transfer of LANav.",
  "published": "2026-07-21",
  "updated": "2026-07-21",
  "year": "2026",
  "authors": [
   "Jiahong Zhang",
   "Yifan Lin",
   "Yandong Zhang",
   "Sijun Shen",
   "Kexin Wang",
   "Yuqi Pan",
   "Hongjuan Pei",
   "Wei Wang",
   "Guoqi Li"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes Linear Attention-based Navigation (LANav), which adopts linear attention (LA) as the policy backbone to maintain a structured state update rather than self-attention over the context window, and introduces Weighted State-Expansion Linear Attention (WSLA), which expands each attention head's state into multiple sub-states and uses learnable weighted readout to aggregate expanded sub-states.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jiahong Zhang",
    "id": "2314781146",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Yifan Lin",
    "id": "2451917912",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Yandong Zhang",
    "id": "2452194993",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Sijun Shen",
    "id": "2398661223",
    "h_index": 0,
    "papers": 7
   },
   {
    "name": "Kexin Wang",
    "id": "2316641286",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Yuqi Pan",
    "id": "2332050325",
    "h_index": 4,
    "papers": 14
   },
   {
    "name": "Hongjuan Pei",
    "id": "2342857546",
    "h_index": 1,
    "papers": 10
   },
   {
    "name": "Wei Wang",
    "id": "2303919444",
    "h_index": 7,
    "papers": 16
   },
   {
    "name": "Guoqi Li",
    "id": "2243952733",
    "h_index": 10,
    "papers": 31
   }
  ],
  "comment": "12 pages, 7 figures",
  "topics": [
   "sim2real",
   "navigation",
   "safety-eval"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2607.18794v1",
  "pdf_url": "https://arxiv.org/pdf/2607.18794v1",
  "html_url": "https://arxiv.org/html/2607.18794v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2607.18760",
  "slug": "koopman-dcm-unstable-eigenfunctions-as-data-driven-representations-for",
  "title": "Koopman DCM: Unstable Eigenfunctions as Data-driven Representations for Legged Balancing",
  "abstract": "In legged locomotion, divergent components of motion (DCMs) have emerged as characteristic states for balance control. They isolate the unstable mode of the dynamics but, in existing formulations, apply only to reduced models such as the linear inverted pendulum. In this study, we show how DCMs can be more generally formulated as Koopman eigenfunctions. Whereas Koopman analysis typically targets eigenvalues near zero, which capture conserved or slowly varying quantities, our investigation leads us to deliberately search for unstable eigenpairs with large eigenvalues. The resulting Koopman DCMs are data-driven observables trained using only real-robot data. On a real biped, DCMs learned from one hour of robot data improve tracking of reference walking patterns. We further show how learned DCMs provide state-based viability constraints when combined with model predictive control.",
  "published": "2026-07-21",
  "updated": "2026-07-21",
  "year": "2026",
  "authors": [
   "St\u00e9phane Caron"
  ],
  "author_count": 1,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "St\u00e9phane Caron",
    "id": "152715167",
    "h_index": 20,
    "papers": 40
   }
  ],
  "comment": "12 pages, 4 figures",
  "topics": [
   "humanoids",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.18760v1",
  "pdf_url": "https://arxiv.org/pdf/2607.18760v1",
  "html_url": "https://arxiv.org/html/2607.18760v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.18737",
  "slug": "motion-primitive-discovery-in-a-humanoid-robot-via-self-organising-map",
  "title": "Motion Primitive Discovery in a Humanoid Robot via Self-Organising Maps for Phase Recognition",
  "abstract": "Understanding the computational basis of action recognition is a central challenge in social cognition as well as in human-robot interaction. Inspired by the Mirror Neuron System (MNS), we propose a two-level architecture for motor primitive discovery and online phase recognition applied to the NICO humanoid robot. At the first level, two Self-Organising Maps (SOMs) learn topographic representations of arm kinematics (A-SOM) and hand kinematics (H-SOM) from simulated trials covering seven motor actions. The maps are trained on non-redundant features identified through hierarchical correlation analysis of motion trajectories. The results show that the two SOMs encode complementary aspects of motor behaviour. At the second level, an Echo State Network (ESN) evaluates whether temporal trajectories of SOM activations, represented by consecutive best-matching units, are sufficient for online recognition of the currently executed movement phase. The results show that SOM-based trajectories preserve the dominant phase-discriminative structure of the movement, while contextual information provides only a secondary refinement. Our contribution is the integration of established SOM and ESN methods within an MNS-inspired architecture for motor primitive representation and online phase recognition. The results are compatible with the computational hypothesis that self-organised motor representations, when temporally integrated, can support accurate online recognition of ongoing movement phases.",
  "published": "2026-07-21",
  "updated": "2026-07-21",
  "year": "2026",
  "authors": [
   "Radovan Gregor",
   "Igor Farka\u0161"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The results show that SOM-based trajectories preserve the dominant phase-discriminative structure of the movement, while contextual information provides only a secondary refinement, compatible with the computational hypothesis that self-organised motor representations, when temporally integrated, can support accurate online recognition of ongoing movement phases.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "R. Gregor",
    "id": "144815964",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Igor Farkas",
    "id": "2323367630",
    "h_index": 1,
    "papers": 3
   }
  ],
  "comment": "12 pages, 4 figures",
  "topics": [
   "humanoids",
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.18737v1",
  "pdf_url": "https://arxiv.org/pdf/2607.18737v1",
  "html_url": "https://arxiv.org/html/2607.18737v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.18709",
  "slug": "robointer1-5-a-holistic-intermediate-representation-suite-for-embodied",
  "title": "RoboInter1.5: A Holistic Intermediate Representation Suite for Embodied World Modeling and Robotic Manipulation",
  "abstract": "Existing robot datasets remain expensive to curate, embodiment-specific, and insufficiently annotated with the fine-grained structure required for generalizable reasoning, execution, or long-horizon environment dynamics simulation. Building on our prior work, RoboInter1.0, we present RoboInter1.5, an extended and holistic suite of intermediate representations for both robotic manipulation and embodied world modeling. RoboInter1.5 provides a unified resource of data, benchmarks, and models centered on dense manipulation-oriented intermediate representations. Specifically, RoboInter-Data contains over 230k manipulation episodes across 571 scenes with dense per-frame annotations covering more than ten types of intermediate representations, including subtasks, primitive skills, object and gripper grounding, segmentation, affordance, grasp poses, contact points, motion traces, etc. Built upon these annotations, RoboInter-VQA introduces spatial and temporal embodied VQA tasks to benchmark and improve the intermediate-representation reasoning capabilities of our RoboInter-VLM. RoboInter-VLA further studies how such representations benefit action execution through implicit, explicit, and modular plan-then-execute paradigms. To better model the physical world, we further introduce RoboInter-World, which leverages intermediate representations as structured conditioning signals for controllable prediction of future world states. Extensive evaluations demonstrate that RoboInter1.5 provides a unified spatiotemporal scaffolding for intermediate representations. Rather than treating intermediate representations merely as interpretable signals, RoboInter1.5 conceptualizes them as a bidirectional interface that both regularizes low-level action spaces and constrains the latent rollouts of open-world physical simulators.",
  "published": "2026-07-21",
  "updated": "2026-07-22",
  "year": "2026",
  "authors": [
   "Ziqin Wang",
   "Hao Li",
   "Weijun Wang",
   "Junhao Cai",
   "Jia Zeng",
   "Yilun Chen",
   "Jiangmiao Pang",
   "Si Liu"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work presents RoboInter1.5, an extended and holistic suite of intermediate representations for both robotic manipulation and embodied world modeling, and introduces RoboInter-World, which leverages intermediate representations as structured conditioning signals for controllable prediction of future world states.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ziqin Wang",
    "id": "2265646739",
    "h_index": 8,
    "papers": 13
   },
   {
    "name": "Hao Li",
    "id": "2358863858",
    "h_index": 8,
    "papers": 14
   },
   {
    "name": "Weijun Wang",
    "id": "2448918119",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Junhao Cai",
    "id": "2400499913",
    "h_index": 2,
    "papers": 11
   },
   {
    "name": "Jia Zeng",
    "id": "2337356727",
    "h_index": 10,
    "papers": 24
   },
   {
    "name": "Yilun Chen",
    "id": "2236662733",
    "h_index": 30,
    "papers": 75
   },
   {
    "name": "Jiangmiao Pang",
    "id": "2377561990",
    "h_index": 10,
    "papers": 28
   },
   {
    "name": "Si Liu",
    "id": "2290973397",
    "h_index": 3,
    "papers": 4
   }
  ],
  "comment": "28 pages. arXiv admin note: substantial text overlap with arXiv:2602.09973",
  "topics": [
   "world-models",
   "vla",
   "dexterous-manipulation",
   "sim2real",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.18709v2",
  "pdf_url": "https://arxiv.org/pdf/2607.18709v2",
  "html_url": "https://arxiv.org/html/2607.18709v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.18660",
  "slug": "mvp-tac-a-miniaturized-dual-modal-vision-and-photoelastic-tactile-sens",
  "title": "MVP-Tac: A Miniaturized Dual-Modal Vision and Photoelastic Tactile Sensor for Robot-Assisted Minimally Invasive Surgery",
  "abstract": "Robot-assisted minimally invasive surgery (RMIS) offers major benefits over open and conventional laparoscopic procedures, yet it still lacks tactile feedback for palpation while operating under strict requirements to preserve reliable vision for navigation and safety. In practice, visual feedback is indispensable, and tactile solutions that cannot coexist with vision are difficult to translate into RMIS tools. To address both needs, we introduce MVP-Tac, a compact, vision-based tactile sensor that provides co-located vision and tactile sensing. MVP-Tac uses reflective photoelastic imaging: a thin photoelastic elastomer produces stress-dependent interferograms under contact that are captured by an embedded camera through a miniaturized reflective polariscope. A semi-transparent membrane and controllable illumination enable switching between visual mode and tactile mode, enabling tactile perception without sacrificing vision. We validate MVP-Tac through force calibration in the 0 to 2 N range and demonstrate its potential for tumor palpation via video-based hardness classification on tissue phantoms, achieving 97% accuracy for exposed-tumor classification and 92% accuracy for subdermal-tumor classification. Finally, we conduct a simulated colonoscopy to validate both visual and tactile modalities in a constrained lumen, including vision-guided 3D photomapping of the luminal wall and in situ hardness classification of localized nodules. Overall, MVP-Tac provides a practical path toward restoring clinically useful palpation in RMIS while maintaining essential visual feedback. The design, fabrication, and firmware of MVP-Tac are open-sourced at https://mvp-tac.github.io/",
  "published": "2026-07-21",
  "updated": "2026-08-04",
  "year": "2026",
  "authors": [
   "Md Rakibul Islam Prince",
   "Jaeeun Kim",
   "Yuhao Zhou",
   "Mason Vrshek",
   "Shivani Reddy Sama",
   "Adyaa Khera",
   "Sheeraz Athar",
   "Zijie Xu",
   "Jiabin Liu",
   "Shaoting Lin",
   "Wei Li",
   "Yu She"
  ],
  "author_count": 12,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "MVP-Tac, a compact, vision-based tactile sensor that provides co-located vision and tactile sensing, is introduced and provides a practical path toward restoring clinically useful palpation in RMIS while maintaining essential visual feedback.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Md. Rakibul Islam Prince",
    "id": "2148542840",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Jaeeun Kim",
    "id": "2325114336",
    "h_index": 2,
    "papers": 17
   },
   {
    "name": "Yuhao Zhou",
    "id": "2314297484",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Mason Vrshek",
    "id": "2451746126",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Shivani Reddy Sama",
    "id": "2451746154",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Adyaa Khera",
    "id": "2451732536",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Sheeraz Athar",
    "id": "2238336847",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Zijie Xu",
    "id": "2328760626",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Jiabin Liu",
    "id": "2305531013",
    "h_index": 5,
    "papers": 19
   },
   {
    "name": "Shaoting Lin",
    "id": "2237737268",
    "h_index": 7,
    "papers": 25
   },
   {
    "name": "Wei Li",
    "id": "2330559171",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Yu She",
    "id": "2317112691",
    "h_index": 3,
    "papers": 12
   }
  ],
  "comment": "8 pages, 8 figures. To appear in IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS) 2026",
  "topics": [
   "tactile",
   "navigation",
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.18660v2",
  "pdf_url": "https://arxiv.org/pdf/2607.18660v2",
  "html_url": "https://arxiv.org/html/2607.18660v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2607.18641",
  "slug": "fabric-pneumatic-artificial-muscles-based-on-the-drawstring-principle",
  "title": "Fabric Pneumatic Artificial Muscles Based on the Drawstring Principle",
  "abstract": "Pneumatic artificial muscles have wide applications in robotics and industrial fields. Conventional pneumatic artificial muscles generate extra radial deformation during axial contraction, which severely wastes available working space. Inspired by the widely adopted drawstring principle in textile products, this paper proposes a novel drawstring fabric pneumatic artificial muscle (DPAM). Unlike traditional counterparts, the proposed DPAM produces no extra radial deformation during contraction, greatly improving structural compactness. The DPAM exhibits outstanding mechanical performance: a load capacity over 800 times its self-weight, a maximum contraction ratio of 44%, and a power density up to 4.98 kW/kg, alongside excellent scalability. Two representative application scenarios, bionic robots and industrial production lines, are demonstrated to validate its practicability. The DPAM can be easily expanded within a two-dimensional plane, as verified by the fabricated DPAM matrix. This work not only presents a high-performance novel pneumatic artificial muscle but also inspires researchers to draw design inspiration from conventional textile structures to address existing challenges in soft robotics.",
  "published": "2026-07-21",
  "updated": "2026-07-21",
  "year": "2026",
  "authors": [
   "Chendong Liu",
   "Dapeng Yang",
   "Yiming Dai",
   "Li Jiang",
   "Hong Liu"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Chendong Liu",
    "id": "2279450768",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Dapeng Yang",
    "id": "50773392",
    "h_index": 25,
    "papers": 112
   },
   {
    "name": "Yiming Dai",
    "id": "2279085630",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Li Jiang",
    "id": "2277450289",
    "h_index": 3,
    "papers": 15
   },
   {
    "name": "Hong Liu",
    "id": "2118903023",
    "h_index": 4,
    "papers": 14
   }
  ],
  "comment": "",
  "topics": [
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.18641v1",
  "pdf_url": "https://arxiv.org/pdf/2607.18641v1",
  "html_url": "https://arxiv.org/html/2607.18641v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.18637",
  "slug": "end-to-end-conditional-diffusion-for-realistic-and-controllable-visual",
  "title": "End-to-end Conditional Diffusion for Realistic and Controllable Visual Traffic Scenario Generation",
  "abstract": "Generating closed-loop traffic scenarios that are both realistic and controllable is crucial for evaluating autonomous driving systems, especially under rare safety-critical interactions. Existing learning-based methods often struggle to balance controllability and realism, offering either limited fine-grained control over traffic behavior or controllable scenarios at the expense of behavioral plausibility. This paper presents E2E-CDiff, an end-to-end conditional diffusion framework for controllable and realistic scenario generation. Conditioned on front-view visual observations, E2E-CDiff jointly denoises future motion states and executable low-level controls for route-interacting background vehicles. This unified state-action generation mitigates the planning-control mismatch in conventional two-stage trajectory-then-controller pipelines. Differentiable guidance further regulates speed, enforces drivable-area compliance, and supports collision-avoidance or collision-seeking behaviors, enabling both naturalistic and safety-critical scenario generation. Experiments on Bench2Drive show that E2E-CDiff achieves a favorable controllability-realism trade-off compared with representative reinforcement- and imitation-learning baselines, while its collision-guided variant induces challenging interactions across multiple autonomous driving systems. E2E-CDiff also performs competitively as a learning-based ego planner, demonstrating the generality of end-to-end state-action diffusion.",
  "published": "2026-07-21",
  "updated": "2026-07-21",
  "year": "2026",
  "authors": [
   "Jingzheng Li",
   "Yufei Ge",
   "Zhijun Chen",
   "Qianren Mao",
   "Zizhe Wang",
   "Binhang Qi",
   "Bing Li",
   "Keyu Chen",
   "Baochang Zhang",
   "Xianglong Liu",
   "Philip S Yu"
  ],
  "author_count": 11,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Conditioned on front-view visual observations, E2E-CDiff jointly denoises future motion states and executable low-level controls for route-interacting background vehicles, and mitigates the planning-control mismatch in conventional two-stage trajectory-then-controller pipelines.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jingzheng Li",
    "id": "2117959152",
    "h_index": 8,
    "papers": 42
   },
   {
    "name": "Yufei Ge",
    "id": "2451951078",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Zhijun Chen",
    "id": "2109387081",
    "h_index": 8,
    "papers": 23
   },
   {
    "name": "Qianren Mao",
    "id": "67081502",
    "h_index": 7,
    "papers": 33
   },
   {
    "name": "Zizhe Wang",
    "id": "2352010490",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Binhang Qi",
    "id": "2451754630",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Bing Li",
    "id": "2353039467",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Keyu Chen",
    "id": "2351709446",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Member Ieee Baochang Zhang",
    "id": "2451744911",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Xianglong Liu",
    "id": "2237942988",
    "h_index": 17,
    "papers": 57
   },
   {
    "name": "I. P. S. Y. Senior Member",
    "id": "2451758066",
    "h_index": 0,
    "papers": 1
   }
  ],
  "comment": "",
  "topics": [
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.18637v1",
  "pdf_url": "https://arxiv.org/pdf/2607.18637v1",
  "html_url": "https://arxiv.org/html/2607.18637v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.18606",
  "slug": "on-the-limits-of-sampling-based-reachability-geometry-dynamics-and-sam",
  "title": "On the Limits of Sampling-Based Reachability: Geometry, Dynamics, and Sample Complexity",
  "abstract": "Reachability analysis is central to safety-critical control, robotics, and neural network verification, but classical computational methods, such as Hamilton--Jacobi reachability and set propagation, scale poorly with state dimension. Sampling-based methods have emerged as a promising alternative, often providing finite-sample guarantees that bound the probability-mass left uncovered. However, an explicit account of how the geometry of the initial set, the dynamics, and the sampling law affect the accuracy of the estimator is not fully available in the literature. We study this by casting sampling-based reachable-set recovery as geometric support estimation over a family of problems specified by an initial set, its dynamics, and a sampling law. First, we identify two regularity properties, positive reach of the initial set's complement and Lipschitz continuity of the dynamics, that together make recovery well-posed: a probability-mass coverage guarantee can be upgraded to accuracy $r$ in Hausdorff distance. Second, we bound the resulting sample complexity: recovery is achievable with $\\tilde{\\mathcal{O}}\\big((e^{3LT}/r)^n\\big)$ samples, exponential in both the state dimension and the time horizon. Third, we show that neither can be removed: an minimax lower bound of $\u03a9\\big((e^{LT}/r)^n\\big)$ holds for every estimator, so the exponential dependence on dimension and the degradation over the horizon are both intrinsic, not artifacts of a particular method. Experiments on nonlinear systems confirm that adversarial sampling improves constants but not the scaling.",
  "published": "2026-07-21",
  "updated": "2026-07-21",
  "year": "2026",
  "authors": [
   "Jixian Liu",
   "Ihab Tabbara",
   "Hussein Sibai",
   "Enrique Mallada"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work casts sampling-based reachable-set recovery as geometric support estimation over a family of problems specified by an initial set, its dynamics, and a sampling law, and identifies two regularity properties that together make recovery well-posed.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jixian Liu",
    "id": "2383249921",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Ihab Tabbara",
    "id": "2302583556",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Hussein Sibai",
    "id": "2328309846",
    "h_index": 2,
    "papers": 13
   },
   {
    "name": "Enrique Mallada",
    "id": "1706983",
    "h_index": 30,
    "papers": 147
   }
  ],
  "comment": "",
  "topics": [
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.18606v1",
  "pdf_url": "https://arxiv.org/pdf/2607.18606v1",
  "html_url": "https://arxiv.org/html/2607.18606v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.18586",
  "slug": "bridging-the-sim-to-real-gap-under-real-time-constraints-in-autonomous",
  "title": "Bridging the Sim-to-Real Gap under Real-Time Constraints in Autonomous Racing",
  "abstract": "Autonomous racing exposes the sim-to-real gap under extreme operating conditions characterized by high speed, tight stability margins, and stringent real-time constraints. Although simulation is indispensable for development, controllers that perform well in simulation often degrade abruptly on physical platforms due to interacting effects of dynamics mismatch, estimation delay, and execution-layer latency. This paper frames sim-to-real transfer in autonomous racing as a full-stack, real-time systems problem. We introduce a structured three-layer perspective (Physical/Cyber/Execution) to analyze how mismatches propagate and amplify through closed-loop feedback. We present diagnostic metrics beyond nominal lap time, including performance flip, stability-oriented measures, sensitivity to delay and noise, and latency distribution characterization. Mitigation strategies are synthesized from a deployment-oriented viewpoint, emphasizing execution-aware and delay-aware design. Finally, we outline benchmarking guidelines that enable reproducible and fair sim-to-real evaluation under compute and timing constraints. The resulting framework clarifies cross-layer failure mechanisms and provides practical design principles for deployable autonomous racing systems operating near dynamic limits.",
  "published": "2026-07-20",
  "updated": "2026-07-20",
  "year": "2026",
  "authors": [
   "Hossein Maghsoumi",
   "Yaser P. Fallah"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper frames sim-to-real transfer in autonomous racing as a full-stack, real-time systems problem, and introduces a structured three-layer perspective (Physical/Cyber/Execution) to analyze how mismatches propagate and amplify through closed-loop feedback.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hossein Maghsoumi",
    "id": "2353608035",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Yaser P. Fallah",
    "id": "2303655012",
    "h_index": 3,
    "papers": 7
   }
  ],
  "comment": "Accepted for presentation at the 2026 IEEE 104th Vehicular Technology Conference (VTC2026-Fall). 6 pages, 2 figures, 3 tables",
  "topics": [
   "sim2real"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.18586v1",
  "pdf_url": "https://arxiv.org/pdf/2607.18586v1",
  "html_url": "https://arxiv.org/html/2607.18586v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.18580",
  "slug": "step-signal-temporal-logic-for-precise-specifications-for-action-gener",
  "title": "STeP: Signal Temporal Logic for Precise Specifications for Action Generation with Vision Language Models",
  "abstract": "Vision-language-action (VLA) models have shown impressive generalization, but often lack interpretability and can struggle to follow precise natural language instructions that encode spatial, temporal, and logical requirements. We propose a hierarchical framework that uses Signal Temporal Logic (STL) as a shared representation connecting high-level language understanding with low-level robot execution. A high-level policy leverages a VLM to decompose language instructions into high-level subtasks, generate STL specifications for each subtask, and choose a low-level policy for executing each subtask. The STL specifications translate language-derived intent into precise constraints, and the low-level policy selection determines whether those constraints are enforced directly through STL-guided model-predictive control or monitored during execution of a learned policy for perceptually complex, or contact-rich behaviors. By integrating STL into plan validation, low-level policy, subtask monitoring, and replanning, our framework enables language-derived plans to be checked, optimized, and revised at runtime using a common formal structure. We evaluate the approach on a real-world tabletop domain, demonstrating how formal specifications can improve the precision, reliability, and interpretability of language-conditioned robot planning.",
  "published": "2026-07-20",
  "updated": "2026-07-20",
  "year": "2026",
  "authors": [
   "Kasra Torshizi",
   "Anukriti Singh",
   "Sidharth Mathur",
   "Khuzema Habib",
   "Leo Du",
   "Pratap Tokekar"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A hierarchical framework that uses Signal Temporal Logic (STL) as a shared representation connecting high-level language understanding with low-level robot execution is proposed, demonstrating how formal specifications can improve the precision, reliability, and interpretability of language-conditioned robot planning.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Kasra Torshizi",
    "id": "2190043473",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Anukriti Singh",
    "id": "2272901800",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Sidharth Mathur",
    "id": "2451734800",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Khuzema Habib",
    "id": "2382926469",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Leo Du",
    "id": "2401259453",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Pratap Tokekar",
    "id": "2390456",
    "h_index": 31,
    "papers": 191
   }
  ],
  "comment": "14 pages, 6 figures",
  "topics": [
   "vla",
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.18580v1",
  "pdf_url": "https://arxiv.org/pdf/2607.18580v1",
  "html_url": "https://arxiv.org/html/2607.18580v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.18565",
  "slug": "integrity-gated-eco-cacc-epistemic-admissibility-for-cooperative-drivi",
  "title": "Integrity-Gated Eco-CACC: Epistemic Admissibility for Cooperative Driving at Signalized Intersections",
  "abstract": "Eco-Cooperative Adaptive Cruise Control (Eco-CACC) systems rely on accurate localization, signal timing, and interaction awareness to optimize energy consumption at signalized intersections. Existing approaches typically assume that the internal world model used for optimization remains valid, making them vulnerable when sensing outages or semantic inconsistencies invalidate planning premises. This letter proposes an Integrity-Gated Eco-CACC framework that explicitly monitors the consistency between internal vehicle beliefs and external sensing. A unified integrity metric is constructed by combining positional innovation, observability loss, and semantic inconsistencies. The resulting trust score regulates control authority, enabling a transition between nominal eco-driving and a safety-dominant fallback maneuver. Unlike robust control methods that attempt to preserve performance under uncertainty, the proposed framework regulates whether energy-optimal control remains admissible. Scenario-based simulations demonstrate that the method preserves nominal efficiency when model consistency is maintained, while enabling early and conservative responses under integrity degradation.",
  "published": "2026-07-20",
  "updated": "2026-07-20",
  "year": "2026",
  "authors": [
   "Lyes Saad Saoud",
   "Moussa Ayyash"
  ],
  "author_count": 2,
  "categories": [
   "eess.SY",
   "cs.RO"
  ],
  "primary_category": "eess.SY",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This letter proposes an Integrity-Gated Eco-CACC framework that explicitly monitors the consistency between internal vehicle beliefs and external sensing, and regulates whether energy-optimal control remains admissible.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "L. S. Saoud",
    "id": "2319631",
    "h_index": 12,
    "papers": 48
   },
   {
    "name": "Moussa Ayyash",
    "id": "2693447",
    "h_index": 21,
    "papers": 76
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.18565v1",
  "pdf_url": "https://arxiv.org/pdf/2607.18565v1",
  "html_url": "https://arxiv.org/html/2607.18565v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.18362",
  "slug": "faro-feasibility-aware-robot-motion-optimization",
  "title": "FARO: Feasibility-Aware Robot Motion Optimization",
  "abstract": "Fast planning of novel behaviors in unseen scenarios remains a fundamental challenge in robotics. The high-dimensional, hybrid, and underactuated nature of humanoid loco-manipulation continues to hinder the realization of this goal. In this paper, we address this challenge by proposing a nested kino-dynamic framework for rapid feasibility checking and dynamically consistent trajectory generation given a candidate contact sequence. By integrating this module with a feasibility-guided tree search and a Large Language Model (LLM)-based contact plan sampling strategy, we demonstrate that the proposed framework can substantially improve the search process. Furthermore, we show that the generated trajectories can be tracked using a reinforcement learning (RL)-based controller and show that the resulting trajectories are of sufficiently high quality for execution in real-world loco-manipulation scenarios. A supplementary video is available at: https://youtu.be/R6qCHoCormQ.",
  "published": "2026-07-20",
  "updated": "2026-07-20",
  "year": "2026",
  "authors": [
   "Michal Ciebielski",
   "Shafeef Omar",
   "Aaron Johnson",
   "Majid Khadiv"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper proposes a nested kino-dynamic framework for rapid feasibility checking and dynamically consistent trajectory generation given a candidate contact sequence and shows that the generated trajectories can be tracked using a reinforcement learning (RL)-based controller and are of sufficiently high quality for execution in real-world loco-manipulation scenarios.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Michal Ciebielski",
    "id": "2314694144",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Shafeef Omar",
    "id": "2224615527",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Aaron M. Johnson",
    "id": "2326841548",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "M. Khadiv",
    "id": "8134198",
    "h_index": 19,
    "papers": 89
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.18362v1",
  "pdf_url": "https://arxiv.org/pdf/2607.18362v1",
  "html_url": "https://arxiv.org/html/2607.18362v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.18231",
  "slug": "fm-vla-force-based-memory-for-vision-language-action-models-in-contact",
  "title": "FM-VLA: Force-based Memory for Vision-Language-Action Models in Contact-Rich Manipulation",
  "abstract": "Vision-language-action (VLA) models have achieved impressive generalization in robotic manipulation, and recent memory-augmented VLAs have relaxed the Markovian assumption by conditioning on past images or language summaries. Vision-based memory approaches address this by conditioning on sampled past image frames, but they are computationally expensive and fundamentally limited when temporal events are visually ambiguous, e.g., pushing a button multiple times with small movements. We propose FM-VLA, a VLA model with force-based memory, enabling temporal context reasoning for non-Markovian, contact-rich manipulation. We encode force histories into compact force memory tokens with a variational autoencoder (VAE) pretrained with force time series reconstruction. By projecting force latent representations and short state history as additional conditioning tokens to the action expert module, we enable VLAs to leverage accumulated contact event history to guide manipulation. We evaluate FM-VLA on three memory-dependent tasks, including finding a hidden block, pressing a button, and wiping a dish for a specific number of times. Our lightweight force memory achieves over 80% success rate with minimal inference overhead, significantly outperforming baseline approaches. Project page: https://qft-333.github.io/FM-VLA-Page/",
  "published": "2026-07-20",
  "updated": "2026-07-20",
  "year": "2026",
  "authors": [
   "Ruicheng Li",
   "Qixiu Li",
   "Ruichun Ma",
   "Yu Deng",
   "Lin Luo",
   "Zhiying Du",
   "Jianfeng Xiang",
   "Huizhi Liang",
   "Ruicheng Wang",
   "Jiaolong Yang",
   "Baining Guo"
  ],
  "author_count": 11,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "FM-VLA is proposed, a VLA model with force-based memory, enabling temporal context reasoning for non-Markovian, contact-rich manipulation, and achieves over 80% success rate with minimal inference overhead, significantly outperforming baseline approaches.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ruicheng Li",
    "id": "2451618637",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Qixiu Li",
    "id": "2282095267",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Ruichun Ma",
    "id": "2146355021",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Yu Deng",
    "id": "2387172823",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Lin Luo",
    "id": "2333975449",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Zhiying Du",
    "id": "2390621033",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Jianfeng Xiang",
    "id": "2147263432",
    "h_index": 11,
    "papers": 16
   },
   {
    "name": "Huizhi Liang",
    "id": "2391890100",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Ruicheng Wang",
    "id": "2145158793",
    "h_index": 7,
    "papers": 12
   },
   {
    "name": "Jiaolong Yang",
    "id": "2237946707",
    "h_index": 10,
    "papers": 22
   },
   {
    "name": "Baining Guo",
    "id": "2400505652",
    "h_index": 3,
    "papers": 6
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.18231v1",
  "pdf_url": "https://arxiv.org/pdf/2607.18231v1",
  "html_url": "https://arxiv.org/html/2607.18231v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.18210",
  "slug": "optimization-of-sim-to-real-transfer-in-the-humanoid-robot-nico",
  "title": "Optimization of sim-to-real transfer in the humanoid robot NICO",
  "abstract": "Robotic grasping requires accurate coordination between visual perception, object localization, inverse kinematics, and hand control. However, when movements planned in simulation are executed on a physical robot, the sim-to-real gap can cause small positioning errors that prevent successful grasping. In our previous work, we introduced a low-cost haptic calibration method that improved 2D reaching accuracy of the humanoid robot NICO. In this paper, we extend this approach from reaching to tabletop object grasping by adding YOLO-based object and hand detection, stereo vision-based localization using the robot's built-in low-resolution fisheye cameras, and task-specific corrections for grasp execution. Together, these components form a novel calibration-based grasping pipeline that does not require RGB-D cameras, motion capture, or external tracking systems. We also implemented a visual feedback model that aligns the robot hand with the detected object before grasping. Our results show that the fully nonlinear calibration model achieved the best performance inside the calibrated area, while the visual feedback model achieved the highest overall grasping success across the full tabletop workspace.",
  "published": "2026-07-20",
  "updated": "2026-07-20",
  "year": "2026",
  "authors": [
   "Juraj Gavura",
   "Igor Farka\u0161"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work added YOLO-based object and hand detection, stereo vision-based localization using the robot's built-in low-resolution fisheye cameras, and task-specific corrections for grasp execution to form a novel calibration-based grasping pipeline that does not require RGB-D cameras, motion capture, or external tracking systems.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "J. Gavura",
    "id": "2373466906",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Igor Farkas",
    "id": "2268765678",
    "h_index": 2,
    "papers": 7
   }
  ],
  "comment": "12 pages, 8 figures, accepted to International Conference on Artificial Neural Networks 2026, Neurorobotics workshop",
  "topics": [
   "dexterous-manipulation",
   "humanoids",
   "egocentric-data",
   "sim2real"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.18210v1",
  "pdf_url": "https://arxiv.org/pdf/2607.18210v1",
  "html_url": "https://arxiv.org/html/2607.18210v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.18200",
  "slug": "learning-adaptive-safety-margins-for-visual-navigation",
  "title": "Learning Adaptive Safety Margins for Visual Navigation",
  "abstract": "Robots in cluttered indoor spaces often fail not because they cannot generate collision-free paths, but because a fixed safety margin is mis-calibrated: conservative margins cause detours and timeouts, while permissive margins lead to near-boundary shortcuts under perception bias. Diffusion-based planners propose diverse trajectory candidates from egocentric RGB-D, yet reliable selection remains the bottleneck. We propose a context-conditioned safety critic that learns an adaptive clearance preference for ranking diffusion proposals, decomposed into three complementary terms: (i) a safety term with a clearance-budget penalty and a control-barrier-function residual for waypoint- and transition-wise safety, (ii) an efficiency term combining a smoothness penalty with a safety-gated detour-ratio penalty that avoids detours without incentivizing risky shortcuts, and (iii) a distance-constraint matching term that anchors the learned budget to realized ESDF clearances to prevent margin collapse. We train the critic with privileged ESDF geometry in simulation and distill it into a perception-only selector via a two-stage teacher-student procedure. On PointGoal navigation in HM3D and MP3D, including cross-dataset transfer, our method achieves the highest success rate (SR) and success weighted by path length (SPL) among strong diffusion, optimization, and RL baselines. Trained purely in simulation, it transfers to a Unitree G1 humanoid and navigates cluttered indoor scenes without task-specific tuning.",
  "published": "2026-07-20",
  "updated": "2026-07-20",
  "year": "2026",
  "authors": [
   "Junyi Hu",
   "Shuaihang Yuan",
   "Geeta Chandra Raju Bethala",
   "Anthony Tzes",
   "Yi Fang"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes a context-conditioned safety critic that learns an adaptive clearance preference for ranking diffusion proposals and trains the critic with privileged ESDF geometry in simulation and distill it into a perception-only selector via a two-stage teacher-student procedure.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Junyi Hu",
    "id": "2362652276",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Shuaihang Yuan",
    "id": "1491104638",
    "h_index": 9,
    "papers": 43
   },
   {
    "name": "Geeta Chandra Raju Bethala",
    "id": "2320308161",
    "h_index": 2,
    "papers": 12
   },
   {
    "name": "Anthony Tzes",
    "id": "2282297404",
    "h_index": 6,
    "papers": 58
   },
   {
    "name": "Yi Fang",
    "id": "2264345449",
    "h_index": 6,
    "papers": 36
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "egocentric-data",
   "navigation",
   "safety-eval"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2607.18200v1",
  "pdf_url": "https://arxiv.org/pdf/2607.18200v1",
  "html_url": "https://arxiv.org/html/2607.18200v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2607.18197",
  "slug": "imitation-of-arm-gestures-by-the-semi-humanoid-robot-nico",
  "title": "Imitation of Arm Gestures by the Semi-Humanoid Robot NICO",
  "abstract": "Seamless human-robot interaction (HRI) requires a number of perceptual and motor abilities from the robot, one of them being the imitation of human gestures. Humanoid robots have an advantage in HRI thanks to their anthropomorphic features. In this work, we develop a system for imitation of human arm gestures by the semi-humanoid robot NICO based on analytical geometry and a pretrained MediaPipe pose-estimation model. For each input RGB frame, 3D coordinates of relevant human body landmarks, including arm joints and hand keypoints, are obtained using the MediaPipe framework. Joint angles are then computed from these coordinates using derived geometric relations. Finally, the computed angles are properly mapped to NICO's motor configuration and executed in a predefined motion sequence. Preliminary experiments on several representative arm gestures with six participants of different height indicate that the proposed method can produce meaningful imitative motions from monocular RGB input only, while also highlighting limitations in more complex poses and wrist-related movements.",
  "published": "2026-07-20",
  "updated": "2026-08-19",
  "year": "2026",
  "authors": [
   "Anastasiya Ihnatovich",
   "Igor Farka\u0161"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Preliminary experiments on several representative arm gestures indicate that the proposed method can produce meaningful imitative motions from monocular RGB input only, while also highlighting limitations in more complex poses and wrist-related movements.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Anastasiya Ihnatovich",
    "id": "2451572435",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Igor Farkas",
    "id": "2268765678",
    "h_index": 2,
    "papers": 7
   }
  ],
  "comment": "15 pages, 7 figures, presented at Human-Friendly Robotics workshop 2026, Trento, Italy",
  "topics": [
   "humanoids",
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.18197v2",
  "pdf_url": "https://arxiv.org/pdf/2607.18197v2",
  "html_url": "https://arxiv.org/html/2607.18197v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.18154",
  "slug": "world-translation-minimizing-sim-to-real-gap-with-backward-dynamics-ex",
  "title": "World Translation: Minimizing Sim-to-Real Gap with Backward Dynamics Extraction and Unpaired Domain Translation",
  "abstract": "The gap between simulation and reality remains a fundamental challenge in deploying simulation-trained robotic policies in the real world. Real-to-sim methods narrow this gap from the real side, learning transition dynamics from real data to build a more realistic digital world. Learned dynamics models are their dominant instance. Such methods, however, face a partial observability problem: the same observation may branch to different transitions due to unobservable factors. Existing methods assume these factors can be recovered from observation history. However, this may fail whenever observation history is uninformative, such as a sudden contact event with no prior warning. To address this limitation, we propose \\textit{World Translation}, which exploits a complementary strength of simulators and learned dynamics. Simulators are deterministic but physically imperfect, while learned models are accurate but underdetermined under partial observability. Rather than predicting transitions forward from history, we extract the unobservable dynamics information backward from an observed transition, then translate this feature across simulation and reality as an unpaired domain-translation problem that preserves dynamics content while transferring domain style. Experiments across humanoid, quadruped, and manipulator platforms show that our method achieves more accurate dynamics modeling than baselines, with the largest gains when unobservable factors cannot be recovered from observation history. Real-robot deployment on Go2 quadruped confirms improved policy transfer.",
  "published": "2026-07-20",
  "updated": "2026-07-20",
  "year": "2026",
  "authors": [
   "Xinchen Yao",
   "Leixin Chang",
   "Hua Chen"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes World Translation, which exploits a complementary strength of simulators and learned dynamics to extract the unobservable dynamics information backward from an observed transition, then translates this feature across simulation and reality as an unpaired domain-translation problem that preserves dynamics content while transferring domain style.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xincheng Yao",
    "id": "2295954191",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Leixin Chang",
    "id": "2291142443",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Hua Chen",
    "id": "2331031298",
    "h_index": 4,
    "papers": 11
   }
  ],
  "comment": "8 pages, 8 figures",
  "topics": [
   "humanoids",
   "sim2real"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.18154v1",
  "pdf_url": "https://arxiv.org/pdf/2607.18154v1",
  "html_url": "https://arxiv.org/html/2607.18154v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.18135",
  "slug": "isaac-sim-to-real-reinforcement-learning-based-locomotion-for-quadrupe",
  "title": "Isaac Sim-to-Real: Reinforcement Learning based Locomotion for Quadrupeds",
  "abstract": "Learning-based approaches to locomotion have risen in popularity in recent years, showing the capability for complex legged locomotion and whole-body control. Reinforcement learning (RL), the primary learning-based approach for locomotion, often utilizes a high-performance simulation tool, providing a controlled and efficient training and development environment. However, policies that perform well in simulation frequently encounter unexpected challenges when deployed on a physical system, known as the sim-to-real gap. This work presents a robust RL locomotion framework capable of whole-body control. The proposed RL framework utilizes Nvidia's new set of simulation tools, Isaac Sim, and its companion RL framework, Isaac Lab, for training, achieving a zero-shot sim-to-real policy. The performance of our policy is validated on physical hardware using the Unitree Go1, with experimental results showing similar velocity tracking performance to the quadruped's integrated controller, with a greater ability to recover from large disturbances, and achieve linear velocities of 2.0 m/s and angular velocities of 1.8 rad/s.",
  "published": "2026-07-20",
  "updated": "2026-07-20",
  "year": "2026",
  "authors": [
   "Jordan Dowdy",
   "Jean Chagas Vaz"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 3,
  "influential_citations": 0,
  "tldr": "This work presents a robust RL locomotion framework capable of whole-body control that utilizes Nvidia\u2019s new set of simulation tools, Isaac Sim, and its companion RL framework, Isaac Lab, for training, achieving a zero-shot sim-to-real policy.",
  "doi": "10.1109/CASE58245.2025.11163761",
  "oa_pdf": "https://arxiv.org/pdf/2607.18135",
  "s2_authors": [
   {
    "name": "Jordan Dowdy",
    "id": "2308278983",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "J. Vaz",
    "id": "31067212",
    "h_index": 6,
    "papers": 28
   }
  ],
  "comment": "6 pages, 5 figures. Accepted manuscript. Published in the 2025 IEEE 21st International Conference on Automation Science and Engineering (CASE), pp. 2194-2199",
  "topics": [
   "humanoids",
   "sim2real",
   "rl-control"
  ],
  "orgs": [
   "NVIDIA",
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2607.18135v1",
  "pdf_url": "https://arxiv.org/pdf/2607.18135v1",
  "html_url": "https://arxiv.org/html/2607.18135v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.1
 },
 {
  "id": "2607.18016",
  "slug": "closing-the-loop-in-humanoid-vla-persistent-3d-object-tokens-for-verif",
  "title": "Closing the Loop in Humanoid VLA: Persistent 3D Object Tokens for Verifiable Loco-Manipulation",
  "abstract": "Vision-language-action policies are a promising foundation for general robot control, but long-horizon humanoid loco-manipulation requires the robot to treat task objects as persistent physical entities across movement, contact, occlusion, and recovery. We study this problem as object-state divergence: the object state used to condition a whole-body action can differ from the state used to decide whether the action achieved the intended physical relation. We propose \\emph{Persistent Object Tokenization} (POT), which maintains role-indexed 3D object records from RGB-D observations and converts them into object tokens for a whole-body action expert. Instantiated as \\emph{POT-VLA}, the same object records condition action generation and support geometric predicate checks, yielding a closed-loop execution system in which object state is both actionable and verifiable. On a Unitree G1, POT-VLA improves a matched direct GR00T-N1.7 baseline from 39/80 to 71/80 successes over eight real-world task families. In an external Being-0-aligned reference, POT-VLA achieves 44/50 successes on aligned service tasks, compared with the 37/50 success reported by the Being-0 paper. The largest gains occur on tasks requiring maintained 3D relations, suggesting that persistent object-centered state is a useful abstraction for verifiable humanoid VLA execution.",
  "published": "2026-07-20",
  "updated": "2026-07-26",
  "year": "2026",
  "authors": [
   "Peng Ren",
   "Haoyang Ge",
   "Jiang Zhao",
   "Cong Huang",
   "Yukun Shi",
   "Pei Chi",
   "Kai Chen"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Persistent Object Tokenization (POT), which maintains role-indexed 3D object records from RGB-D observations and converts them into object tokens for a whole-body action expert, yielding a closed-loop execution system in which object state is both actionable and verifiable.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Penghua Ren",
    "id": "2391246244",
    "h_index": 0,
    "papers": 5
   },
   {
    "name": "Haoyang Ge",
    "id": "2378296295",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Jiang-Lin Zhao",
    "id": "2290846877",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Cong Huang",
    "id": "2399268226",
    "h_index": 4,
    "papers": 22
   },
   {
    "name": "Yukun Shi",
    "id": "2404257413",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Pei-I. Chi",
    "id": "2368677582",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Kaiyang Chen",
    "id": "2323841814",
    "h_index": 0,
    "papers": 4
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "humanoids"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2607.18016v2",
  "pdf_url": "https://arxiv.org/pdf/2607.18016v2",
  "html_url": "https://arxiv.org/html/2607.18016v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2607.17984",
  "slug": "distilling-global-traversability-priors-for-image-based-affordance-pre",
  "title": "Distilling Global Traversability Priors for Image-based Affordance Prediction in Off-road Environments",
  "abstract": "Standard methods for autonomous navigation in unstructured terrain are prone to myopic behaviors in long-horizon scenarios. The use of metric maps built from LiDAR or cameras provides necessary local geometry and semantic information but is strictly limited by depth sensing range. By discarding data beyond the mapping horizon robots suffer from suboptimal, short-sighted decisions. To recover this lost information, we focus on extracting long-range traversability-aware frontiers directly from first-person-view (FPV) images. By leveraging satellite imagery, we compute the set of feasible navigation paths for a dataset of image/pose pairs and use them to supervise our network, reducing the need for extensive human demonstration data. We demonstrate that this approach improves performance in long-range off-road navigation over existing methods by more than 10% in various offline benchmarks and reduces the number of human interventions incurred in a set of real-world experiments. More details can be found at https://theairlab.org/ss_frontiers_iros .",
  "published": "2026-07-20",
  "updated": "2026-07-20",
  "year": "2026",
  "authors": [
   "Matthew Sivaprakasam",
   "Samuel Triest",
   "Micah Nye",
   "Deegan Atha",
   "Shehryar Khattak",
   "David Fan",
   "Wenshan Wang",
   "Sebastian Scherer"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work focuses on extracting long-range traversability-aware frontiers directly from first-person-view (FPV) images and leveraging satellite imagery to compute the set of feasible navigation paths for a dataset of image/pose pairs and use them to supervise the network, reducing the need for extensive human demonstration data.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Matthew Sivaprakasam",
    "id": "2133414157",
    "h_index": 7,
    "papers": 12
   },
   {
    "name": "S. Triest",
    "id": "2068071355",
    "h_index": 9,
    "papers": 18
   },
   {
    "name": "Micah Nye",
    "id": "2220906253",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Deegan Atha",
    "id": "101580503",
    "h_index": 10,
    "papers": 20
   },
   {
    "name": "Shehryar Khattak",
    "id": "40186667",
    "h_index": 28,
    "papers": 70
   },
   {
    "name": "David D. Fan",
    "id": "2294057796",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Wenshan Wang",
    "id": "2264408535",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Sebastian A. Scherer",
    "id": "2264205708",
    "h_index": 8,
    "papers": 14
   }
  ],
  "comment": "",
  "topics": [
   "egocentric-data",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.17984v1",
  "pdf_url": "https://arxiv.org/pdf/2607.17984v1",
  "html_url": "https://arxiv.org/html/2607.17984v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.17977",
  "slug": "rynnbrain-1-1-towards-more-capable-and-generalizable-embodied-foundati",
  "title": "RynnBrain 1.1: Towards More Capable and Generalizable Embodied Foundation Model",
  "abstract": "We present RynnBrain 1.1, a family of embodied foundation models spanning 2B, 9B, and 122B-A10B scales. Trained with a unified spatio-temporal and physically grounded framework, RynnBrain 1.1 supports embodied perception, spatial reasoning, localization, and planning. Compared with RynnBrain 1.0, it further introduces contact-point prediction across the model family and native 3D grounding for the 2B and 9B models, yielding representations and outputs that are more directly aligned with robot manipulation. We also develop RynnBrain-VLA with a unified cross-embodiment action space and embodiment-specific masking, and deploy it on Unitree G1, Astribot-S1, and Tianji-Wuji. RynnBrain 1.1 achieves strong results on embodied cognition, localization, and 3D grounding, with the 122B-A10B model outperforming all evaluated proprietary and open-source models on VSI-Bench, MMSI, and RefSpatial-Bench. Real-robot experiments show that RynnBrain-initialized policies outperform Qwen-based and representative generalist VLAs, while joint multi-task and multi-embodiment training improves process scores and success rates over per-task training.",
  "published": "2026-07-20",
  "updated": "2026-07-31",
  "year": "2026",
  "authors": [
   "Kehan Li",
   "Bohan Hou",
   "Minghao Zhu",
   "Tianyi Zhang",
   "Zesen Cheng",
   "Zhikai Wang",
   "Sicong Leng",
   "Xin Li",
   "Xiao Lin",
   "Biying Yao",
   "Minghua Zeng",
   "Jiangpin Liu",
   "Ronghao Dang",
   "Jiayan Guo",
   "Siteng Huang",
   "Haoyu Zhao",
   "Heng Ping",
   "Yaxi Zhao",
   "Tong Zhao",
   "Kexiang Wang",
   "Tong Lu",
   "Shengke Xue",
   "Jiahao Tang",
   "Yulei Wang",
   "Zejing Wang",
   "Jianwei Gao",
   "Shijian Lu",
   "Chengju Liu",
   "Jianfei Yang",
   "Mingxiu Chen",
   "Deli Zhao"
  ],
  "author_count": 31,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 2,
  "influential_citations": 0,
  "tldr": "RynnBrain 1.1 achieves strong results on embodied cognition, localization, and 3D grounding, with the 122B-A10B model outperforming all evaluated proprietary and open-source models on VSI-Bench, MMSI, and RefSpatial-Bench.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Kehan Li",
    "id": "2371267604",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Bohan Hou",
    "id": "2310338711",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Minghao Zhu",
    "id": "2237478018",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Tianyi Zhang",
    "id": "2327147044",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Zesen Cheng",
    "id": "79673589",
    "h_index": 16,
    "papers": 36
   },
   {
    "name": "Zhikai Wang",
    "id": "2357980344",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Sicong Leng",
    "id": "2113966687",
    "h_index": 14,
    "papers": 32
   },
   {
    "name": "Xin Li",
    "id": "2376124596",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Xiao Lin",
    "id": "2237435197",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Biying Yao",
    "id": "2451635541",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Min Zeng",
    "id": "2348797867",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Jiangpin Liu",
    "id": "2376501614",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Ronghao Dang",
    "id": "2131077260",
    "h_index": 13,
    "papers": 25
   },
   {
    "name": "Jiayan Guo",
    "id": "2451656274",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Siteng Huang",
    "id": "2371084328",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Haoyu Zhao",
    "id": "2316659455",
    "h_index": 5,
    "papers": 15
   },
   {
    "name": "Heng Ping",
    "id": "2273557590",
    "h_index": 7,
    "papers": 20
   },
   {
    "name": "Yaxi Zhao",
    "id": "2291065642",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Kexiang Wang",
    "id": "2375802414",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Tong Lu",
    "id": "2440105974",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Shengke Xue",
    "id": "2381083694",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Jiahao Tang",
    "id": "2115856143",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Yulei Wang",
    "id": "2447657071",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Ze-Hui Wang",
    "id": "2451504742",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Jia-Ning Gao",
    "id": "2448647465",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Shijian Lu",
    "id": "2372492043",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Chengju Liu",
    "id": "2920326",
    "h_index": 22,
    "papers": 151
   },
   {
    "name": "Jianfei Yang",
    "id": "2398616773",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Mingxiu Chen",
    "id": "2381065591",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Deli Zhao",
    "id": "2303980061",
    "h_index": 13,
    "papers": 24
   }
  ],
  "comment": "KL,BH,MZ,TZ,ZC,ZW,SL,XL,XL,BY,MZ,JL,RD contribute equally. Project Lead: Kehan Li and Xin Li project: https://alibaba-damo-academy.github.io/RynnBrain github: https://github.com/alibaba-damo-academy/RynnBrain huggingface: https://huggingface.co/collections/Alibaba-DAMO-Academy/rynnbrain-11 modelscope: https://modelscope.cn/collections/DAMO_Academy/RynnBrain-11",
  "topics": [
   "vla",
   "spatial-3d",
   "foundation-pretraining"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2607.17977v2",
  "pdf_url": "https://arxiv.org/pdf/2607.17977v2",
  "html_url": "https://arxiv.org/html/2607.17977v2",
  "code_url": "https://alibaba-damo-academy.github.io/RynnBrain",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.98
 },
 {
  "id": "2607.17970",
  "slug": "mevion-low-cost-open-source-data-collection-system-for-powerful-and-hi",
  "title": "MEVION: Low-Cost Open-Source Data Collection System for Powerful and High-Speed Dual-Arm Manipulation",
  "abstract": "The global competition for developing robotic foundation models is intensifying. Among the data collection systems used for dual-arm robots, ALOHA is representative of being low-cost and open-source, and is widely adopted by researchers as a de facto standard. However, due to its limited ability to generate high forces and speeds, it is difficult to handle heavy objects or perform fast manipulations. To address this, we developed MEVION, a low-cost and open-source dual-arm robot data collection system capable of generating greater force and speed. All parts of this robot can be sourced through e-commerce, and by extensively utilizing sheet metal welding, its large body structure is constructed with a small number of components at low cost, while also simplifying assembly. MEVION is equipped with four 6-DoF arms with parallel grippers. Each arm weighs 7.0 kg and has a maximum torque of 60 Nm, and the entire system can be constructed for about USD 14,000. The elbow joint adopts a closed-link mechanism similar to those used in quadruped robots, which reduces the distal mass and enables higher force and speed output at the end-effector. We demonstrate that MEVION enables data collection for object manipulation tasks not previously possible and supports imitation learning-based motion generation. All hardware and software of this work are included in the Supplementary Materials or https://github.com/haraduka/mevion.",
  "published": "2026-07-20",
  "updated": "2026-07-20",
  "year": "2026",
  "authors": [
   "Kento Kawaharazuka",
   "Yoshiki Obinata",
   "Hirokazu Ishida",
   "Jihoon Oh",
   "Temma Suzuki",
   "Shintaro Inoue",
   "Keita Yoneda",
   "Ayumu Iwata",
   "Kei Okada"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "MEVION enables data collection for object manipulation tasks not previously possible and supports imitation learning-based motion generation and it is demonstrated that MEVION enables data collection for object manipulation tasks not previously possible.",
  "doi": "10.1109/RAP.2026.3716202",
  "oa_pdf": "https://arxiv.org/pdf/2607.17970",
  "s2_authors": [
   {
    "name": "Kento Kawaharazuka",
    "id": "8308607",
    "h_index": 17,
    "papers": 220
   },
   {
    "name": "Yoshiki Obinata",
    "id": "2147152545",
    "h_index": 7,
    "papers": 30
   },
   {
    "name": "Hirokazu Ishida",
    "id": "2295397825",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Jihoon Oh",
    "id": "2256484322",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Temma Suzuki",
    "id": "83369366",
    "h_index": 5,
    "papers": 45
   },
   {
    "name": "Shintaro Inoue",
    "id": "2295913150",
    "h_index": 3,
    "papers": 17
   },
   {
    "name": "Keita Yoneda",
    "id": "2365839163",
    "h_index": 1,
    "papers": 15
   },
   {
    "name": "Ayumu Iwata",
    "id": "2376541386",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Kei Okada",
    "id": "2248244895",
    "h_index": 5,
    "papers": 68
   }
  ],
  "comment": "Accepted to IEEE Robotics and Automation Practice, website: https://haraduka.github.io/mevion-hardware/",
  "topics": [
   "humanoids",
   "imitation-diffusion",
   "foundation-pretraining",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.17970v1",
  "pdf_url": "https://arxiv.org/pdf/2607.17970v1",
  "html_url": "https://arxiv.org/html/2607.17970v1",
  "code_url": "https://haraduka.github.io/mevion-hardware/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.3
 },
 {
  "id": "2607.17839",
  "slug": "receiver-centered-robot-to-human-handover-with-grasp-aware-object-orie",
  "title": "Receiver-Centered Robot-to-Human Handover with Grasp-Aware Object Orientation",
  "abstract": "Collaborative robots are increasingly sharing workspaces with human operators, making tool handover a frequent and safety-critical micro-interaction. However, traditional static handovers often lead to awkward grasps when handling asymmetric industrial tools. This paper presents a receiver-centered voice-driven adaptive handover system for mechanical tools, built on a Franka cobot. Using an LLM for intention recognition and MediaPipe for real-time 3D hand tracking, the framework dynamically adjusts the end-effector's orientation to present tools in an ergonomically optimal, handle-first pose. A within-subjects study compared this adaptive approach with an object-agnostic static baseline. The results demonstrate that the adaptive system reduces the grasp delay for asymmetric tools, improving the fluency of the interaction. Furthermore, the adaptive strategy improved specific trust-related perceptions, particularly motion predictability and perceived task simplicity.",
  "published": "2026-07-20",
  "updated": "2026-07-20",
  "year": "2026",
  "authors": [
   "Federico Biagi",
   "Dario Onfiani",
   "Simone Silenzi",
   "Luigi Biagiotti"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A receiver-centered voice-driven adaptive handover system for mechanical tools, built on a Franka cobot, that reduces the grasp delay for asymmetric tools, improving the fluency of the interaction and specific trust-related perceptions.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "F. Biagi",
    "id": "2383749435",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Dario Onfiani",
    "id": "2193897019",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Simone Silenzi",
    "id": "2256991075",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Luigi Biagiotti",
    "id": "2403208415",
    "h_index": 1,
    "papers": 5
   }
  ],
  "comment": "Accepted for presentation at the 19th International Workshop on Human-Friendly Robotics (HFR 2026), Trento, Italy. The paper will appear in Springer's Proceedings in Advanced Robotics",
  "topics": [
   "dexterous-manipulation",
   "hri",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.17839v1",
  "pdf_url": "https://arxiv.org/pdf/2607.17839v1",
  "html_url": "https://arxiv.org/html/2607.17839v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.17769",
  "slug": "from-sign-language-generation-to-humanoid-execution-vision-language-gu",
  "title": "From Sign Language Generation to Humanoid Execution: Vision-Language Guided Retargeting with Collision Mitigation",
  "abstract": "Recent sign language generation (SLG) systems increasingly output dense 3D body representations, which better preserve full-body kinematics and geometry for downstream embodiment on humanoid robots. However, these generated motions frequently exhibit self-intersections such as hand-hand and hand-torso penetration. While such artifacts may be tolerated in offline rendering, they become critical in humanoid execution as they lead to infeasible inverse-kinematics (IK) solutions, collisions, and unstable retargeted trajectories. We present a system-level framework that bridges SLG outputs to humanoid joint-space execution via two components. First, we introduce a volumetric SMPL-X collision-mitigation module that projects generated signing motions toward physically plausible configurations while minimally deviating from the original trajectory. Second, we propose a vision-language-guided retargeting algorithm built on an IK backbone: a VLM serves as a visual critic over rendered humanoid motion, identifies embodiment-specific failure modes, and triggers targeted task-space corrections. Our results highlight collision handling and perception-guided refinement as key missing components for reliable humanoid signing.",
  "published": "2026-07-20",
  "updated": "2026-07-20",
  "year": "2026",
  "authors": [
   "Nabeela Khan",
   "Bowen Wu",
   "Runwu Shi",
   "Benjamin Yen",
   "Takeshi Ashizawa",
   "Carlos Toshinori Ishi",
   "Takashi Minato",
   "Kazuhiro Nakadai"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO",
   "cs.CV",
   "cs.HC"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work introduces a volumetric SMPL-X collision-mitigation module that projects generated signing motions toward physically plausible configurations while minimally deviating from the original trajectory, and proposes a vision-language-guided retargeting algorithm built on an IK backbone that serves as a visual critic over rendered humanoid motion and triggers targeted task-space corrections.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Nabeela Khan",
    "id": "2334919377",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Bowen Wu",
    "id": "2257968000",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Runwu Shi",
    "id": "2337799764",
    "h_index": 2,
    "papers": 15
   },
   {
    "name": "Benjamin Yen",
    "id": "2269269202",
    "h_index": 3,
    "papers": 30
   },
   {
    "name": "Takeshi Ashizawa",
    "id": "2372459021",
    "h_index": 2,
    "papers": 13
   },
   {
    "name": "C. Ishi",
    "id": "1798428",
    "h_index": 25,
    "papers": 200
   },
   {
    "name": "T. Minato",
    "id": "1678393",
    "h_index": 24,
    "papers": 159
   },
   {
    "name": "Kazuhiro Nakadai",
    "id": "2334862638",
    "h_index": 3,
    "papers": 12
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.17769v1",
  "pdf_url": "https://arxiv.org/pdf/2607.17769v1",
  "html_url": "https://arxiv.org/html/2607.17769v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.17651",
  "slug": "hcpg-flow-hierarchical-contact-progress-guidance-for-flow-policy-robot",
  "title": "HCPG-Flow:Hierarchical Contact-Progress Guidance for Flow-Policy Robot Manipulation",
  "abstract": "Flow policies can represent multimodal action distributions for robot manipulation, yet a robot must execute one action at each control step. When several proposals are sampled, critic-based ranking makes data collection depend on value estimates over candidate actions that may be weakly represented in replay. We introduce HCPG-Flow, an analytic rollout-time selector that augments SAC-Flow with hierarchical, object-centric contact-progress guidance while preserving its actor and critic objectives. HCPG switches from end-effector approach to task progress after contact, scores each proposal by the first-order reduction of a task-relevant distance, standardizes scores within the candidate set, and executes a temperature-controlled action embedding. Across ten simulated tasks, HCPG improves mean success over SAC-Flow on both benchmarks, including a 9.5 percentage-point gain on Maniskill. Four physical tasks further show high success with a 17.4% reduction in successful completion steps.Project page: https://hitxraz.github.io/HCPG-Flow/",
  "published": "2026-07-20",
  "updated": "2026-07-20",
  "year": "2026",
  "authors": [
   "Guanghu Xie",
   "Mingxu Li",
   "Shuo Zhang",
   "Yonglong Zhang",
   "Yifan Yang",
   "Yang Liu",
   "Zongwu Xie",
   "Baoshi Cao"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "HCPG-Flow is introduced, an analytic rollout-time selector that augments SAC-Flow with hierarchical, object-centric contact-progress guidance while preserving its actor and critic objectives.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Guanghu Xie",
    "id": "2228022389",
    "h_index": 3,
    "papers": 17
   },
   {
    "name": "Mingxue Li",
    "id": "2198464720",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Shuo Zhang",
    "id": "2370080398",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Yonglong Zhang",
    "id": "2363598875",
    "h_index": 2,
    "papers": 11
   },
   {
    "name": "Yifan Yang",
    "id": "2279780423",
    "h_index": 1,
    "papers": 9
   },
   {
    "name": "Yang Liu",
    "id": "2292155018",
    "h_index": 4,
    "papers": 23
   },
   {
    "name": "Zongwu Xie",
    "id": "2276055560",
    "h_index": 5,
    "papers": 37
   },
   {
    "name": "Baoshi Cao",
    "id": "9242444",
    "h_index": 7,
    "papers": 39
   }
  ],
  "comment": "",
  "topics": [
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.17651v1",
  "pdf_url": "https://arxiv.org/pdf/2607.17651v1",
  "html_url": "https://arxiv.org/html/2607.17651v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.17574",
  "slug": "predictive-training-with-latent-imagination-for-visual-quadruped-navig",
  "title": "Predictive Training with Latent Imagination for Visual Quadruped Navigation",
  "abstract": "Reinforcement-learning navigation policies for legged robots select actions reactively from current observations and short-term memory, with limited capacity to anticipate how moving obstacles will evolve in the near future. In dynamic environments, this reactivity causes the robot to respond too late because collision risk depends on short-horizon scene structure rather than on current obstacle positions alone. Lightweight predictive supervision applied to the policy's recurrent state during training can encode anticipatory obstacle dynamics without modifying the inference-time controller. We augment a reactive LSTM-SRU navigation backbone with an auxiliary JEPA-style predictor and SIGReg regularization: during training, the predictor supervises the deterministic hidden state to anticipate its own next state; at inference, it is fully discarded, incurring zero additional computational cost. On simulated and real-world navigation benchmarks with dynamic obstacles, our method substantially improves navigation success while reducing collision rates through the predictive training signal alone, without additional inference-time parameters. Real-robot deployment on a Unitree Go2 demonstrates zero-shot sim-to-real transfer: the controller navigates cluttered indoor and dynamic outdoor environments without fine-tuning, with evasive behavior consistent with the collision reduction observed in simulation.",
  "published": "2026-07-20",
  "updated": "2026-07-20",
  "year": "2026",
  "authors": [
   "Yancheng Zhu",
   "Wanli Ma",
   "Chen Han",
   "Irvin Haozhe Zhan",
   "Bingfeng Qin",
   "Yixin Xu"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work augments a reactive LSTM-SRU navigation backbone with an auxiliary JEPA-style predictor and SIGReg regularization, and substantially improves navigation success while reducing collision rates through the predictive training signal alone, without additional inference-time parameters.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yanchen Zhu",
    "id": "2447307635",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Wanli Ma",
    "id": "2451964433",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Chen Han",
    "id": "2451672914",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Irvin Haozhe Zhan",
    "id": "72745477",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Bingfeng Qin",
    "id": "2451539819",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yixin Xu",
    "id": "2348306654",
    "h_index": 3,
    "papers": 7
   }
  ],
  "comment": "10 pages, 9 figures",
  "topics": [
   "world-models",
   "humanoids",
   "sim2real",
   "navigation",
   "safety-eval"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2607.17574v1",
  "pdf_url": "https://arxiv.org/pdf/2607.17574v1",
  "html_url": "https://arxiv.org/html/2607.17574v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2607.17541",
  "slug": "predicting-grasping-compliance-in-robotic-hands-through-analytical-mod",
  "title": "Predicting Grasping Compliance in Robotic Hands through Analytical-Model-Informed Neural Networks",
  "abstract": "In robotic manipulation studies, grasping is often treated as a binary success or failure problem, usually defined by whether the object simply stays in the hand. For forceful tool use, however, this view is insufficient because grasp compliance becomes a critical factor governing how the hand and tool behave under load. Compliance arises from coupled kinematics, grasp configuration, passive mechanics, and contact conditions, producing nonlinear behavior in which deformation and interaction forces influence each other. Understanding this relationship is essential for predictive models of how a grasped tool and a compliant hand jointly respond to external loading. In underactuated hands, these effects are amplified: such designs offer low cost and adaptive grasping, but make compliance behavior more difficult to model and predict. Our goal is therefore to develop a predictive model for grasped tool behavior during forceful interactions. To address this challenge, we introduce an analytical model informed neural network (AMINN), a hybrid predictive model that combines an analytical mechanics layer with data driven learning to estimate grasp stability and in hand tool displacement under external loading. The model is evaluated on a three finger underactuated robotic hand and shows strong predictive capability with mechanically meaningful outputs across diverse loading conditions. Compared with a black box multilayer perceptron baseline, AMINN also achieves better energy based physical consistency. Beyond prediction accuracy alone, this framework advances physically interpretable learning for robotic manipulation and supports more reliable, safer, and more trustworthy autonomous tool use in safety critical settings during forceful interaction.",
  "published": "2026-07-20",
  "updated": "2026-07-20",
  "year": "2026",
  "authors": [
   "Qianwen Zhao",
   "Long Wang"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "An analytical model informed neural network (AMINN) is introduced, a hybrid predictive model that combines an analytical mechanics layer with data driven learning to estimate grasp stability and in hand tool displacement under external loading.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Qianwen Zhao",
    "id": "2287531347",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Long Wang",
    "id": "2208971102",
    "h_index": 2,
    "papers": 9
   }
  ],
  "comment": "Preprint. 9 pages, 8 figures, 1 table",
  "topics": [
   "dexterous-manipulation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.17541v1",
  "pdf_url": "https://arxiv.org/pdf/2607.17541v1",
  "html_url": "https://arxiv.org/html/2607.17541v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.17521",
  "slug": "geoworldad-geometry-world-action-model-for-autonomous-driving",
  "title": "GeoWorldAD: Geometry World Action Model for Autonomous Driving",
  "abstract": "Autonomous driving requires both safe and efficient planning decisions in dynamic 3D environments. Although recent Vision/Video-Action models learn policies directly from visual observations and scale well with advances in vision transformers and large-scale training data, they often lack explicit geometric grounding and future-aware spatial guidance, limiting their ability to balance collision avoidance and driving progress. In this work, we propose GeoWorldAD, a geometry world action model that grounds trajectory planning in ego-aligned 3D space and anticipates short-horizon scene evolution with latent future geometry tokens. Present geometry provides essential spatial constraints for safe planning, while future geometry reveals how surrounding agents and ego-centric free space may evolve, reducing overly conservative decisions without sacrificing safety. To efficiently exploit these geometric cues, GeoWorldAD progressively aggregates multi-scale present geometry and latent future geometry through iterative trajectory refinement. Experiments on NAVSIM v1 and v2 demonstrate state-of-the-art performance, highlighting the effectiveness of explicit 3D geometry grounding and future geometry world modeling for safe and efficient autonomous driving.",
  "published": "2026-07-20",
  "updated": "2026-07-23",
  "year": "2026",
  "authors": [
   "Songyan Zhang",
   "Jinyuan Tian",
   "Hanbing Li",
   "Daqi Liu",
   "Hao Chen",
   "Wenhui Huang",
   "Fang Li",
   "Guang Chen",
   "Hangjun Ye",
   "Long Chen",
   "Kuiyuan Yang",
   "Chen Lv"
  ],
  "author_count": 12,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "GeoWorldAD is proposed, a geometry world action model that grounds trajectory planning in ego-aligned 3D space and anticipates short-horizon scene evolution with latent future geometry tokens and progressively aggregates multi-scale present geometry and latent future geometry through iterative trajectory refinement.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Songyan Zhang",
    "id": "2259954888",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Ji Tian",
    "id": "2455452914",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Hanbing Li",
    "id": "2393221599",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Daqi Liu",
    "id": "34776599",
    "h_index": 11,
    "papers": 34
   },
   {
    "name": "Hao Chen",
    "id": "2340552141",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Wenhui Huang",
    "id": "49015886",
    "h_index": 13,
    "papers": 28
   },
   {
    "name": "Fang Li",
    "id": "2328073322",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Guang Chen",
    "id": "2366092163",
    "h_index": 8,
    "papers": 36
   },
   {
    "name": "Hangjun Ye",
    "id": "2367554550",
    "h_index": 8,
    "papers": 35
   },
   {
    "name": "Long Chen",
    "id": "2366273398",
    "h_index": 6,
    "papers": 27
   },
   {
    "name": "Kuiyuan Yang",
    "id": "2976163",
    "h_index": 28,
    "papers": 67
   },
   {
    "name": "Chen Lv",
    "id": "2335448886",
    "h_index": 4,
    "papers": 8
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "navigation",
   "video-generation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.17521v2",
  "pdf_url": "https://arxiv.org/pdf/2607.17521v2",
  "html_url": "https://arxiv.org/html/2607.17521v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.17454",
  "slug": "test-time-scaling-for-world-action-models-via-zero-shot-geometric-eval",
  "title": "Test-Time Scaling for World Action Models via Zero-Shot Geometric Evaluation",
  "abstract": "Test-time scaling improves foundation-model inference by spending additional computation, but robot control requires deciding whether extra compute is useful before executing an action. World Action Models (WAMs) make this decision natural: each rollout exposes both an action chunk and predicted future observations. We propose \\methodgated, a training-free selective test-time scaling framework for WAMs. We first instantiate \\method, a fixed-budget Best-of-$N$ selector that ranks sampled rollouts by cross-view depth reprojection consistency of their predicted futures, computed with a frozen geometry foundation model. \\methodgated\\ adds a lightweight action--future consistency gate that invokes \\method\\ only when the initial rollout appears internally inconsistent. Across five benchmark--backbone settings on RoboCasa, LIBERO Long, and RoboTwin~2.0, fixed-budget \\method\\ improves $N{=}8$ task success in every setting, e.g., raising the RoboCasa group average from $66.3\\%$ to $68.4\\%$ with Cosmos Policy and from $80.8\\%$ to $82.5\\%$ with X-WAM. With gating enabled, \\methodgated\\ recovers on average $74.8\\%$ of the always-on success gain while triggering additional sampling on only $26.2\\%$ of decision points. Offline diagnostics show that cross-view reprojection is a strong task-label-free selector, and we identify false low-score selections as a failure mode that helps explain why performance can saturate or degrade as $N$ increases.",
  "published": "2026-07-20",
  "updated": "2026-07-20",
  "year": "2026",
  "authors": [
   "Zesen Zhao",
   "Minkyoung Cho",
   "Hui shen",
   "Boyuan Zheng",
   "Kunxiao Gao",
   "Yulong Cao",
   "Z. Morley Mao"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "CVPR 2026",
  "venue_source": "arxiv-comment",
  "citations": 2,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zesen Zhao",
    "id": "2319584762",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Minkyoung Cho",
    "id": "2261283074",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Hui Shen",
    "id": "2342472659",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Boyuan Zheng",
    "id": "2114719042",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Kunxiao Gao",
    "id": "2451650621",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yulong Cao",
    "id": "2146176464",
    "h_index": 17,
    "papers": 33
   },
   {
    "name": "Z. Mao",
    "id": "2362724454",
    "h_index": 2,
    "papers": 5
   }
  ],
  "comment": "Extened version of CVPR 2026 EAI workshop",
  "topics": [
   "foundation-pretraining",
   "safety-eval"
  ],
  "orgs": [
   "NVIDIA"
  ],
  "abs_url": "https://arxiv.org/abs/2607.17454v1",
  "pdf_url": "https://arxiv.org/pdf/2607.17454v1",
  "html_url": "https://arxiv.org/html/2607.17454v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.48
 },
 {
  "id": "2607.17351",
  "slug": "deeperradar-end-to-end-mimo-radar-design-and-multi-modal-fusion-for-au",
  "title": "DeeperRadar: End-to-End MIMO Radar Design and Multi-Modal Fusion for Autonomous Vehicle Perception",
  "abstract": "DeeperRadar is a radar-centric, sensor-stack-conditioned framework that co-designs radar sensing and multi-modal 3D detection for autonomous mobility by learning a sparse acquisition pattern end-to-end with the fusion model. A learnable MIMO design module is trained end-to-end within a fusion network that operates directly on raw radar ADC data together with camera images and LiDAR point clouds. During training, the design module is supervised by the other sensors, enabling the system to learn both which receiver antennas to activate and the effective number of them. At deployment, the design module is removed and replaced by the learned sparse subsampling mask, leaving the downstream model architecture unchanged. Evaluated on the RADIal dataset, DeeperRadar discovers sparse, task-aware radar configurations that match or exceed full-array baselines while using fewer receivers, potentially reducing radar cost and integration complexity. These results show that learned optimal MIMO radar design depends on the fusion stack and the downstream perception task.",
  "published": "2026-07-19",
  "updated": "2026-07-19",
  "year": "2026",
  "authors": [
   "Eli Goldenshluger",
   "Barak Pinkovich",
   "Chaim Baskin"
  ],
  "author_count": 3,
  "categories": [
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.AI",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Results show that learned optimal MIMO radar design depends on the fusion stack and the downstream perception task, potentially reducing radar cost and integration complexity.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Eli Goldenshluger",
    "id": "2451558962",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Barak Pinkovich",
    "id": "31083026",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Chaim Baskin",
    "id": "46906102",
    "h_index": 15,
    "papers": 75
   }
  ],
  "comment": "Accepted for publication at the IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS 2026)",
  "topics": [
   "spatial-3d",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.17351v1",
  "pdf_url": "https://arxiv.org/pdf/2607.17351v1",
  "html_url": "https://arxiv.org/html/2607.17351v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2607.17326",
  "slug": "rethinking-the-suitability-of-reinforcement-learning-algorithms-under",
  "title": "Rethinking the Suitability of Reinforcement Learning Algorithms Under Practical Transfer Constraints",
  "abstract": "Transfer-oriented reinforcement learning requires evaluating algorithms along dimensions that go beyond standard sample efficiency. We focus on two dimensions: practical efficiency, which asks whether conclusions about algorithm suitability change under wall-clock rather than interaction-based budgets, and robustness under dynamics mismatch, which asks how different learning paradigms respond to variability in the training distribution induced by domain randomization. We provide two insights to reinforcement-learning practitioners. First, comparing the sample efficiency of different algorithms is often an insufficient criterion in transfer-oriented settings. The wall-clock time required to train a decent policy is an important consideration for practitioners, and we find that the sample-inefficient PPO algorithm can produce a performant policy faster than relatively more sample-efficient algorithms such as SAC and TD-MPC2, validating the common understanding of massively parallel training paradigms. Second, domain randomization can help different kinds of algorithms learn robust policies. In particular, although PPO, SAC, and TD-MPC2 represent different RL paradigms - on-policy, off-policy, and model-based learning and planning, respectively - we find that domain randomization affects all three algorithms in a similar way. To the best of our knowledge, this is the first controlled comparison of the effect of domain-randomization coverage on PPO, SAC, and TD-MPC2 under the same transfer protocol. Taken together, these two insights highlight the importance of evaluating RL algorithms not only by sample efficiency, but also by practical considerations such as training time and the algorithms' ability to produce usable policies.",
  "published": "2026-07-19",
  "updated": "2026-07-19",
  "year": "2026",
  "authors": [
   "Hany Hamed",
   "Abhishek Naik",
   "Colin Bellinger",
   "A. Rupam Mahmood"
  ],
  "author_count": 4,
  "categories": [
   "cs.LG",
   "cs.RO"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The wall-clock time required to train a decent policy is an important consideration for practitioners, and it is found that the sample-inefficient PPO algorithm can produce a performant policy faster than relatively more sample-efficient algorithms such as SAC and TD-MPC2, validating the common understanding of massively parallel training paradigms.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hany Hamed",
    "id": "2288262500",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "A. Naik",
    "id": "5559109",
    "h_index": 5,
    "papers": 15
   },
   {
    "name": "Colin Bellinger",
    "id": "2332097212",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "A. Mahmood",
    "id": "2357289903",
    "h_index": 1,
    "papers": 3
   }
  ],
  "comment": "",
  "topics": [
   "sim2real",
   "rl-control",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.17326v1",
  "pdf_url": "https://arxiv.org/pdf/2607.17326v1",
  "html_url": "https://arxiv.org/html/2607.17326v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.17323",
  "slug": "from-perception-to-assistance-open-vocabulary-shared-autonomy-for-robo",
  "title": "From Perception to Assistance: Open-Vocabulary Shared Autonomy for Robotic Manipulation",
  "abstract": "Teleoperating a robotic manipulator in industrial environments demands precision that camera-based interfaces alone struggle to deliver. The operator must align the end-effector with a target in clutter, under limited depth perception, and without colliding with the surrounding structures. This paper presents a shared-autonomy framework that assists the operator throughout this process. A single RGB-D camera captures the operator's arm motion and hand gestures without wearables, fiducials, or a calibration stage. The intended target is specified by a free-form text prompt, grounded by a vision-language model in the robot's gripper camera, and tracked across its onboard cameras by a promptable video-segmentation model, resulting in a grasp frame continuously separated from the obstacle map. Every commanded motion is executed by a GPU-accelerated model-predictive controller that enforces self- and environment-collision avoidance against an online volumetric reconstruction, while a potential field corrects the operator's reference toward the grounded target during the final approach. An autonomous mode can be gesture-triggered to complete the grasp on the same target without a separate perception pipeline. The framework is validated on a quadruped mobile manipulator. The interface achieves a positional RMSE of 59 mm relative to motion-capture ground truth, and the controller keeps the arm at least 18 cm from obstacles while the operator deliberately commands the arm into them by 6 cm. In an industrial valve manipulation and a pick-and-place task, the full framework succeeded in all trials, while ablating either the collision or the assistance module produced failures through complementary mechanisms, and autonomous execution succeeded in four of five trials per task.",
  "published": "2026-07-19",
  "updated": "2026-07-19",
  "year": "2026",
  "authors": [
   "Murilo Vinicius da Silva",
   "Ricardo V. Godoy",
   "Juliano Negri",
   "Gustavo J. G. Lahr",
   "Ranulfo Bezerra",
   "Marcelo Becker"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.CV",
   "cs.HC",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A shared-autonomy framework that assists the operator throughout this process ofTeleoperating a robotic manipulator in industrial environments demands precision that camera-based interfaces alone struggle to deliver and is validated on a quadruped mobile manipulator.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Murilo Vinicius da Silva",
    "id": "2443244552",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Ricardo V. Godoy",
    "id": "2359447738",
    "h_index": 2,
    "papers": 16
   },
   {
    "name": "Juliano Negri",
    "id": "2348440135",
    "h_index": 2,
    "papers": 12
   },
   {
    "name": "G. J. Lahr",
    "id": "50772187",
    "h_index": 7,
    "papers": 31
   },
   {
    "name": "Ranulfo Bezerra",
    "id": "2148645796",
    "h_index": 3,
    "papers": 41
   },
   {
    "name": "Marcelo Becker",
    "id": "2185374183",
    "h_index": 1,
    "papers": 12
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "humanoids",
   "hri",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.17323v1",
  "pdf_url": "https://arxiv.org/pdf/2607.17323v1",
  "html_url": "https://arxiv.org/html/2607.17323v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.17257",
  "slug": "asynchronous-multimodal-diffusion-policy-composition-via-latency-aware",
  "title": "Asynchronous Multimodal Diffusion Policy Composition via Latency-Aware Guidance Fusion",
  "abstract": "Diffusion policies have shown strong potential for robotic imitation learning, and recent extensions incorporate additional modalities to improve manipulation performance. However, these modalities often differ not only in information content but also in sensing rates and inference latencies. Existing multimodal diffusion policies typically rely on synchronous fusion or manually designed multi-frequency architectures, which either slow down high-frequency feedback or limit extensibility to new modality combinations. We propose LAG-Fusion, a latency-aware guidance fusion framework for asynchronous multimodal diffusion policy composition. LAG-Fusion allows modality-specific policies to operate at their native inference rates and contribute denoising guidance whenever available. To make asynchronous composition consistent, we derive a reference-frame rebasing rule for diffusion variables under relative action representations, enabling delayed guidance to be aligned before fusion. We instantiate LAG-Fusion in contact-rich manipulation by composing a low-frequency vision policy with a high-frequency force policy. Experiments under heterogeneous modality latencies show that LAG-Fusion improves policy responsiveness and task performance over synchronous fusion and specially designed force-aware baselines.",
  "published": "2026-07-19",
  "updated": "2026-07-19",
  "year": "2026",
  "authors": [
   "Zihao He",
   "Hongjie Fang",
   "Shirun Tang",
   "Cewu Lu",
   "Haoshu Fang"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "LAG-Fusion allows modality-specific policies to operate at their native inference rates and contribute denoising guidance whenever available and improves policy responsiveness and task performance over synchronous fusion and specially designed force-aware baselines.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zihao He",
    "id": "2332316803",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Hongjie Fang",
    "id": "2152115958",
    "h_index": 16,
    "papers": 31
   },
   {
    "name": "Shirun Tang",
    "id": "2397957257",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Cewu Lu",
    "id": "2301174899",
    "h_index": 7,
    "papers": 29
   },
   {
    "name": "Haoshu Fang",
    "id": "122851212",
    "h_index": 34,
    "papers": 61
   }
  ],
  "comment": "",
  "topics": [
   "tactile",
   "imitation-diffusion"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.17257v1",
  "pdf_url": "https://arxiv.org/pdf/2607.17257v1",
  "html_url": "https://arxiv.org/html/2607.17257v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.17171",
  "slug": "vidar-visual-inertial-dense-alignment-and-reconstruction-via-a-geometr",
  "title": "VIDAR: Visual-Inertial Dense Alignment and Reconstruction via a Geometric Foundation Model",
  "abstract": "Monocular foundation models provide dense geometry but usually lack a stable metric scale. This paper presents VIDAR, a visual-inertial dense reconstruction framework that couples SVO+IMU odometry with Depth Anything 3. VIDAR uses the visual-inertial front end as a metric anchor: it provides camera poses, scale, and a consistent world frame for aligning dense foundation-model predictions across time. The foundation model then contributes detailed local geometry that is fused into a global reconstruction. We study both pose-conditioned DA3 and a decoupled alignment strategy. On EuRoC, pose injection reduces scale error to about 1\\% and reaches 0.463 mean F@0.10; the decoupled hybrid improves this to 0.676 without ground-truth poses. Results on EuRoC and TUM RGB-D show that VIDAR is a practical route to metric dense monocular reconstruction.",
  "published": "2026-07-19",
  "updated": "2026-07-19",
  "year": "2026",
  "authors": [
   "Diyari Mohammed Salih",
   "Lingxiang Hu",
   "Naima AitOufroukh-Mammar",
   "Fabien Bonardi"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "VIDAR is a visual-inertial dense reconstruction framework that couples SVO+IMU odometry with Depth Anything 3 to show that VIDAR is a practical route to metric dense monocular reconstruction.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Diyari Mohammed Salih",
    "id": "2451567815",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Lingxiang Hu",
    "id": "2323193593",
    "h_index": 2,
    "papers": 11
   },
   {
    "name": "Naima AitOufroukh-Mammar",
    "id": "2451543014",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Fabien Bonardi",
    "id": "2323043612",
    "h_index": 2,
    "papers": 3
   }
  ],
  "comment": "",
  "topics": [
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.17171v1",
  "pdf_url": "https://arxiv.org/pdf/2607.17171v1",
  "html_url": "https://arxiv.org/html/2607.17171v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.17042",
  "slug": "articulated-humanoid-head-for-a-robot-receptionist-capable-of-natural",
  "title": "Articulated Humanoid Head for a Robot Receptionist Capable of Natural Human Interaction",
  "abstract": "Humanoid robots have become increasingly popular in applications such as social interaction, education, and service roles, which drives the need for more natural and efficient human-robot interactions. However, currently available humanoid heads often face limitations, including high costs, mechanical complexity, and limited adaptability across diverse environments. To address these challenges, we present an articulated humanoid robot head designed for a receptionist role, integrating a mechanical structure with 21 degrees of freedom (DoF), including mechanisms for the mouth, eyes, eyebrows, and neck, and covered with realistic silicone skin to achieve a human-like appearance and expression. The system integrates a model-based architecture that combines SCRFD, ArcFace, and ByTetrack for face recognition and Llama and Whisper for natural language processing, with hardware support enabling real-time operations and human re-identification. The conversational ability and re-identification capabilities of the humanoid robot head were quantitatively measured, while its emotional expressiveness and human likeness were evaluated through a user study, achieving an average human likeness score of 4.13 out of 5.",
  "published": "2026-07-19",
  "updated": "2026-07-19",
  "year": "2026",
  "authors": [
   "Tharusha Fonseka",
   "Charuka Bandara",
   "Moshintha Hewavitharana",
   "Melisa Arukgoda",
   "Wageesha N. Manamperi",
   "Udaya S. K. Perera Miriya Thanthrige",
   "Peshala Jayasekara"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "cs.CV",
   "eess.IV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work presents an articulated humanoid robot head designed for a receptionist role, integrating a mechanical structure with 21 degrees of freedom (DoF), including mechanisms for the mouth, eyes, eyebrows, and neck, and covered with realistic silicone skin to achieve a human-like appearance and expression.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Tharusha Fonseka",
    "id": "2451575694",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Charuka Bandara",
    "id": "2451575625",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Moshintha Hewavitharana",
    "id": "2451566181",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Melisa Arukgoda",
    "id": "2422773167",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Wageesha Manamperi",
    "id": "51129009",
    "h_index": 4,
    "papers": 19
   },
   {
    "name": "U. M. Thanthrige",
    "id": "1417818414",
    "h_index": 7,
    "papers": 24
   },
   {
    "name": "P. Jayasekara",
    "id": "40529716",
    "h_index": 9,
    "papers": 62
   }
  ],
  "comment": "This work is accepted at IEEE/ASME International Conference on Advanced Intelligent Mechatronics (AIM2026)",
  "topics": [
   "humanoids",
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.17042v1",
  "pdf_url": "https://arxiv.org/pdf/2607.17042v1",
  "html_url": "https://arxiv.org/html/2607.17042v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.17032",
  "slug": "optimal-safety-control-using-high-order-control-barrier-functions",
  "title": "Optimal Safety Control using High-Order Control Barrier Functions",
  "abstract": "This paper investigates the optimal safety control problem of nonlinear control systems by proposing novel high-order control barrier functions (HOCBFs). Different from zeroing HOCBFs, two novel HOCBFs are derived and the safety controllers are designed in an explicit way. Next, we implement vector Lyapunov function approach to propose a novel high-order control Lyapunov function (HOCLF) for the stabilization control problem. The relations between the proposed and existing HOCBFs are discussed. Afterwards, the compatibility of the proposed HOCLF and HOCBF is addressed to guarantee the stabilization and safety control objectives simultaneously, and thus the optimal controller is established. Finally, a numerical example from the navigation problem of quadrotors is presented to illustrate the efficacy of the derived results.",
  "published": "2026-07-19",
  "updated": "2026-07-19",
  "year": "2026",
  "authors": [
   "Neng Li",
   "Zuodong Pan",
   "Jiaxing Wang",
   "Weiguo Xia",
   "Wei Ren"
  ],
  "author_count": 5,
  "categories": [
   "eess.SY",
   "cs.RO"
  ],
  "primary_category": "eess.SY",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Nengzhuo Li",
    "id": "2049036406",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Zuodong Pan",
    "id": "2432035609",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Jiaxing Wang",
    "id": "2386778663",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Weiguo Xia",
    "id": "2240011808",
    "h_index": 4,
    "papers": 18
   },
   {
    "name": "Wei Ren",
    "id": "2257220708",
    "h_index": 3,
    "papers": 23
   }
  ],
  "comment": "8 pages, 3 figures, Accepted by ASCC2026",
  "topics": [
   "rl-control",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.17032v1",
  "pdf_url": "https://arxiv.org/pdf/2607.17032v1",
  "html_url": "https://arxiv.org/html/2607.17032v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.18325",
  "slug": "hazard-or-anomaly-evaluating-vlms-for-understanding-dangers-and-discre",
  "title": "Hazard or Anomaly? Evaluating VLMs for Understanding Dangers and Discrepancies",
  "abstract": "Modern safety-critical systems increasingly rely on human-robot interaction to reduce disaster risk and support decision-making during emergencies. Vision-Language Models (VLMs) are promising for these settings because they can interpret complex scenes and communicate safety-relevant information, but they still require careful evaluation to ensure reliable safety reasoning. In particular, current evaluations often frame danger recognition as a binary decision (Safe/Unsafe), making it unclear whether a model is identifying true physical hazards or merely reacting to unusual scene elements. We address this limitation by introducing an explicit distinction between hazard and anomaly, and by separately recognizing hazardous and anomalous states. We evaluate several state-of-the-art VLMs across two datasets and multiple prompting strategies to test whether this distinction changes model behavior. Our results show that VLMs frequently misinterpret anomalousness as hazardousness, revealing an over-reliance on contextual irregularity as a proxy for danger. We further show that explicitly separating anomaly from hazard provides a more informative evaluation of VLM safety reasoning and exposes failure modes that binary safety judgments can obscure. Our public dataset is available on Roboflow https://app.roboflow.com/vlm-in-context-anomaly-and-hazard-detection/camera-ready-roman-ds.",
  "published": "2026-07-18",
  "updated": "2026-07-18",
  "year": "2026",
  "authors": [
   "Murali Indukuri",
   "Mohammad Eskandari",
   "Sree Nitya Kollu",
   "Stephanie Lukin",
   "Cynthia Matuszek"
  ],
  "author_count": 5,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work evaluates several state-of-the-art VLMs across two datasets and multiple prompting strategies to test whether an explicit distinction between hazard and anomaly changes model behavior, and shows that explicitly separating anomaly from hazard provides a more informative evaluation of VLM safety reasoning and exposes failure modes that binary safety judgments can obscure.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "M. Indukuri",
    "id": "2451296034",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Mohammad Eskandari",
    "id": "2451302778",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Sree Nitya Kollu",
    "id": "2451768413",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Stephanie M. Lukin",
    "id": "2333829640",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Cynthia Matuszek",
    "id": "2127879703",
    "h_index": 7,
    "papers": 44
   }
  ],
  "comment": "8 pages, accepted to RO-MAN 2026",
  "topics": [
   "hri",
   "video-generation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.18325v1",
  "pdf_url": "https://arxiv.org/pdf/2607.18325v1",
  "html_url": "https://arxiv.org/html/2607.18325v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.17007",
  "slug": "admm-based-safety-critical-distributed-nmpc-for-cooperative-transporta",
  "title": "ADMM-Based Safety-Critical Distributed NMPC for Cooperative Transportation by Quadrupedal Robots",
  "abstract": "This paper presents a safety-critical distributed nonlinear model predictive control (DNMPC) framework for cooperative payload transportation by teams of quadrupedal robots. The proposed approach models the robotic team and the shared payload as a dynamically coupled networked system with rigid holonomic coupling constraints arising from cooperative transportation. To enable distributed real-time optimization, the centralized finite-horizon optimal control problem is decomposed into parallel local NMPC subproblems coordinated through the alternating direction method of multipliers (ADMM). The resulting distributed framework enforces consensus over both payload-state and interaction-wrench trajectories while explicitly incorporating acceleration-level holonomic coupling constraints within the distributed predictive control formulation. Safety-critical obstacle avoidance constraints for both the robotic agents and payload are enforced using higher-order control barrier functions (HOCBFs). The framework is validated through numerical simulations with teams of two, three, and four quadrupedal robots transporting shared payloads in cluttered environments. Real-time experiments on two- and three-robot teams demonstrate safe and robust transportation under payload uncertainty and external disturbances. Compared with centralized NMPC, the proposed framework achieves up to 23% reduction in average NLP solve time while maintaining comparable closed-loop performance. Ablation studies further demonstrate robustness to communication delays and show that explicit payload-state consensus and holonomic constraints substantially improve payload tracking and distributed coordination over existing wrench-only consensus formulations.",
  "published": "2026-07-18",
  "updated": "2026-07-18",
  "year": "2026",
  "authors": [
   "Ruturaj S. Sambhus",
   "Kapi Ketan Mehta",
   "Yicheng Zeng",
   "Kaveh Akbari Hamed"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "math.OC"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper presents a safety-critical distributed nonlinear model predictive control framework for cooperative payload transportation by teams of quadrupedal robots that achieves up to 23% reduction in average NLP solve time while maintaining comparable closed-loop performance.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ruturaj S. Sambhus",
    "id": "2214109577",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Kapi Ketan Mehta",
    "id": "2373813997",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Yicheng Zeng",
    "id": "2309795508",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "K. Hamed",
    "id": "2586637",
    "h_index": 19,
    "papers": 76
   }
  ],
  "comment": "Supplementary video available at: https://youtu.be/w8hg52T8Luc?si=nzQrGsqP5ZBNFlGp",
  "topics": [
   "humanoids",
   "rl-control",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.17007v1",
  "pdf_url": "https://arxiv.org/pdf/2607.17007v1",
  "html_url": "https://arxiv.org/html/2607.17007v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.16943",
  "slug": "sind-2-0-a-multi-city-uav-dataset-with-semantic-risk-annotations-for-s",
  "title": "SinD 2.0: A Multi-City UAV Dataset with Semantic Risk Annotations for SOTIF-Oriented Safety Validation at Signalized Intersections",
  "abstract": "Safety validation at signalized intersections remains a critical bottleneck for the deployment of autonomous driving systems (ADS), as these scenarios involve dense heterogeneous traffic, contested right of way, and long-tail safety-critical interactions, posing significant challenges to the Safety of the Intended Functionality (SOTIF). Existing naturalistic driving datasets often suffer from geographical homogeneity, sparsity of safety-critical events, and lack of semantic risk annotations, which limit the evaluation of algorithmic generalizability and targeted SOTIF verification. To address these gaps, this paper introduces SinD 2.0, a large-scale drone-based intersection dataset dedicated to cross-domain ADS safety analysis. The main contributions of SinD 2.0 are: (1) Cross-domain diversity: It covers six signalized intersections across four Chinese cities, capturing distinct intersection topologies and regional driving behavior characteristics; (2) High-density risk interactions: A total of 32,682 safety-critical events are extracted via surrogate safety measures, significantly enriching the density of boundary test scenarios; (3) Hierarchical semantic annotations: Besides integration with high-definition (HD) maps and Signal Phase and Timing (SPaT) data, it provides multi-dimensional semantic labels including traffic violations, high-risk interactions, visual shielding, and narrow feasible areas; (4) Full-stack testing toolchain: It supports automated scenario extraction, prediction-only evaluation, open-loop replay, reactive closed-loop testing, and photorealistic rendering. Benchmark experiments demonstrate that SinD 2.0 exhibits significant domain shifts across cities, and the semantic risk subsets can effectively expose the performance limitations of ADS algorithms. The dataset, annotations, and testing toolchain are available at https://github.com/SOTIF-AVLab/SinD/tree/main.",
  "published": "2026-07-18",
  "updated": "2026-08-11",
  "year": "2026",
  "authors": [
   "Yunwei Li",
   "Shengjie Fu",
   "Chunrong Chen",
   "Chengxiang Zhao",
   "Yuchen Fan",
   "Mingyu Zhu",
   "Yanchao Xu",
   "Jiahui Xu",
   "Anran Wang",
   "Huanan Wang",
   "Yuxin Zhang",
   "Lan Yang",
   "Chuzhao Li",
   "Jie Ji",
   "Yi He",
   "Abhijit Sarkar",
   "Akash Sonth",
   "Hong Wang",
   "Jun Li"
  ],
  "author_count": 19,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yunwei Li",
    "id": "2311994944",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "S. Fu",
    "id": "2451957229",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Chun Chen",
    "id": "2449155506",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Chengxiang Zhao",
    "id": "2284611441",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Yuchen Fan",
    "id": "2451508783",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Mingyu Zhu",
    "id": "2453506344",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yanchao Xu",
    "id": "2284127160",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Jiahui Xu",
    "id": "2281610856",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Anran Wang",
    "id": "2351479867",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Huanan Wang",
    "id": "2113270488",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Yuxin Zhang",
    "id": "2451635673",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Lan Yang",
    "id": "2243109405",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Chuzhao Li",
    "id": "2266136565",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Jie Ji",
    "id": "2451588591",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yi He",
    "id": "2156156836",
    "h_index": 7,
    "papers": 17
   },
   {
    "name": "Abhijit Sarkar",
    "id": "2310098043",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Akash Sonth",
    "id": "1397314448",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Hong Wang",
    "id": "2260009763",
    "h_index": 6,
    "papers": 17
   },
   {
    "name": "Jun Li",
    "id": "2260336080",
    "h_index": 6,
    "papers": 13
   }
  ],
  "comment": "",
  "topics": [
   "navigation",
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.16943v2",
  "pdf_url": "https://arxiv.org/pdf/2607.16943v2",
  "html_url": "https://arxiv.org/html/2607.16943v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.16938",
  "slug": "what-do-they-see-interpreting-complex-road-scenarios-through-the-eyes",
  "title": "What Do They See? Interpreting Complex Road Scenarios Through the Eyes of Vision-Language-Action Models for Safe and Trustworthy Autonomous Vehicle Learning",
  "abstract": "End-to-end autonomous driving models are now able to navigate complex road scenarios, mapping raw sensor observations directly to observed paths for open-loop evaluation and often effective driving in closed-loop evaluation. Yet the internal logic of these safety-critical systems remains largely opaque, due to the complexity of traffic scenes. We propose a counterfactual ablation framework called Counterfactual Vision Action Analysis (CVAA) that systematically removes individual detected objects from front-camera images using photorealistic generative inpainting to prepare counterfactual sets to evaluate the difference in the model's response. This isolates the causal effect of each object's presence on the model's planning behaviour. Applied to the Alpamayo 1 trajectory predictor across 210 nuScenes driving scenes, we create a dataset Counter -nuScenes, using which we see that vehicles and pedestrians within the model's 'path' dominate causal influence as expected, while traffic lights, as expected, exert disproportionate effect relative to their image footprint. However, we also find cases where the model responds strongly to objects a human driver would consider irrelevant. This brings forth a deeper question: does the model itself view the scene as a sum of individual objects influencing the outcome, or does it encode an entirely different set of internal features that do not correspond to human-legible scene elements? To further understand this, we compare intermediate representations of original and inpainted image pairs using mechanistic interpretability techniques and examine the effect of the removal through the various model layers. Together, these two stages offer a path from behavioral auditing to representational understanding, creating explainable driving systems and solidifying human-AI trust.",
  "published": "2026-07-18",
  "updated": "2026-07-18",
  "year": "2026",
  "authors": [
   "Kalpana Panda",
   "Wesley Maia",
   "Vinti Agarwal",
   "Ross Greer"
  ],
  "author_count": 4,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.LG",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A counterfactual ablation framework called Counterfactual Vision Action Analysis (CVAA) is proposed that systematically removes individual detected objects from front-camera images using photorealistic generative inpainting to prepare counterfactual sets to evaluate the difference in the model's response.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Kalpana Panda",
    "id": "82324502",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "W. Maia",
    "id": "2325325078",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Vinti Agarwal",
    "id": "2372441683",
    "h_index": 1,
    "papers": 9
   },
   {
    "name": "Ross Greer",
    "id": "2334481082",
    "h_index": 4,
    "papers": 12
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.16938v1",
  "pdf_url": "https://arxiv.org/pdf/2607.16938v1",
  "html_url": "https://arxiv.org/html/2607.16938v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.16806",
  "slug": "token-wise-latent-streaming-from-slow-reasoners-to-fast-planners-for-d",
  "title": "Token-Wise Latent Streaming from Slow Reasoners to Fast Planners for Dynamic Vision Language Navigation",
  "abstract": "Vision-Language Navigation in dynamic, human-centric environments exposes a fundamental tension: linguistic reasoning is slow and deliberative, whereas safe, socially compliant planning should be instant and reactive. The resulting observation staleness is safety-critical: a maneuver chosen during inference can already be unsafe by the time it executes. We observe that, long before a VLM finishes its inference, its intermediate hidden states already encode action-relevant intent. We propose SPARK-VLN, a dual-system framework for dynamic social VLN that streams the slow VLM reasoner's knowledge to a fast flow-matching expert planner throughout token generation, providing fresh and evolving guidance during inference. This design is realized by three modules: a Token-Wise Hidden Streamer that extracts intermediate hidden states along the token generation process, a Sequence-to-Slot Latent Bridge that projects them into fixed-size latent slots, and an Evolving Latent Conditioner that infuses them into the expert planner. We also introduce a human-centric benchmark suite for dynamic social vision-language navigation that keeps pedestrians and the robot active throughout inference and reports navigation success, social compliance, human collisions, and explicit staleness statistics. Across these settings, SPARK-VLN mproves navigation success and social compliance while sustaining inference efficiency. Webpage: https://hutslib.github.io/SPARK-VLN/.",
  "published": "2026-07-18",
  "updated": "2026-07-18",
  "year": "2026",
  "authors": [
   "Tianshuai Hu",
   "Yangyi Zhong",
   "Zeying Gong",
   "Lingdong Kong",
   "Xiaodong Mei",
   "Guoyang Zhao",
   "Xiaolu Liu",
   "Song Wang",
   "Rong Li",
   "Junwei Liang"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "SPARK-VLN is proposed, a dual-system framework for dynamic social VLN that streams the slow VLM reasoner's knowledge to a fast flow-matching expert planner throughout token generation, providing fresh and evolving guidance during inference.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Tianshuai Hu",
    "id": "2374364190",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Yangyi Zhong",
    "id": "2216652512",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Zeying Gong",
    "id": "2249762165",
    "h_index": 6,
    "papers": 21
   },
   {
    "name": "Lingdong Kong",
    "id": "2334333720",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Xiaodong Mei",
    "id": "97913046",
    "h_index": 8,
    "papers": 19
   },
   {
    "name": "Guoyang Zhao",
    "id": "2382081267",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Xiaolu Liu",
    "id": "2294386367",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Song Wang",
    "id": "2399757541",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Rong Li",
    "id": "2372403246",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Junwei Liang",
    "id": "2333968399",
    "h_index": 5,
    "papers": 13
   }
  ],
  "comment": "",
  "topics": [
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.16806v1",
  "pdf_url": "https://arxiv.org/pdf/2607.16806v1",
  "html_url": "https://arxiv.org/html/2607.16806v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.16636",
  "slug": "phyagentos-a-self-evolving-operating-system-for-embodied-agents-with-d",
  "title": "PhyAgentOS: A Self-Evolving Operating System for Embodied Agents with Decoupled Cognitive Planning and Physical Execution",
  "abstract": "Vision-language-action models, world models, and agentic planners each advance physical intelligence, yet their composition lacks a common execution abstraction, shared state, semantic verification, and persistent experience across heterogeneous embodiments. We present PhyAgentOS, a runtime foundation delivering scheduling, verification, memory, benchmarking, and safety as system-level services. Its Session-Centered Runtime treats a session, not an action, as the minimum unit of scheduling, compatibility preflight, supervised execution, evidence collection, and acceptance. To decouple cognition from physical execution, the cognition-physics boundary is a file system: the State-as-a-File protocol materializes cross-layer state as Markdown with YAML, yielding inspectable, versionable records without code dependencies between Agent and Runtime layers. These views form a unified cognitive state space aligning intent, capabilities, environment, execution, and experience. The SessionVerifier distinguishes execution termination from semantic task completion via evidence-grounded verdicts of success, failure, or replan. Verified outcomes are consolidated through epistemic memory into reusable knowledge and corrective lessons, closing a trial-and-error loop without retraining. Benchmarking reuses the deployment session and verification path, so results trace to real execution. Layered safety constrains both policy-driven and agent-driven execution: preflight, action bridges, SafetyGuard, heartbeat monitoring, and target-local constraints. Validation is progressive: games test cognitive planning, simulation adds dynamics and control, real robots add hardware noise, with the cognitive layer held constant. PhyAgentOS is benchmarked on Optimus-67, StarDojo, and DST-Dojo, validated on 19+ simulated and physical embodiments, and gains on LIBERO, Calvin, and RoboCasa365 across multiple VLA models.",
  "published": "2026-07-18",
  "updated": "2026-07-18",
  "year": "2026",
  "authors": [
   "Yang Liu",
   "Weixing Chen",
   "Xinshuai Song",
   "Tao Pu",
   "Siwen Mo",
   "Yongjie Bai",
   "Zihao Chen",
   "Qianran Sun",
   "Liruo Zhong",
   "Ying Shen",
   "Liang Lin"
  ],
  "author_count": 11,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "This work presents PhyAgentOS, a runtime foundation delivering scheduling, verification, memory, benchmarking, and safety as system-level services, and distinguishes execution termination from semantic task completion via evidence-grounded verdicts of success, failure, or replan.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yang Liu",
    "id": "2257387082",
    "h_index": 10,
    "papers": 22
   },
   {
    "name": "Weixing Chen",
    "id": "2108946830",
    "h_index": 11,
    "papers": 30
   },
   {
    "name": "Xinshuai Song",
    "id": "2282346107",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Tao Pu",
    "id": "50325149",
    "h_index": 11,
    "papers": 23
   },
   {
    "name": "Siwen Mo",
    "id": "2297935285",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Yongjie Bai",
    "id": "2310523071",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Zihao Chen",
    "id": "2451302604",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Qianran Sun",
    "id": "2451373739",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Liruo Zhong",
    "id": "2408404540",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Ying Shen",
    "id": "2267387066",
    "h_index": 6,
    "papers": 19
   },
   {
    "name": "Liang Lin",
    "id": "2332478699",
    "h_index": 8,
    "papers": 27
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "vla",
   "safety-eval"
  ],
  "orgs": [
   "Physical Intelligence"
  ],
  "abs_url": "https://arxiv.org/abs/2607.16636v1",
  "pdf_url": "https://arxiv.org/pdf/2607.16636v1",
  "html_url": "https://arxiv.org/html/2607.16636v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.8
 },
 {
  "id": "2607.16630",
  "slug": "ai-augmented-model-predictive-control-for-safe-and-adaptive-rendezvous",
  "title": "AI-Augmented Model Predictive Control for Safe and Adaptive Rendezvous and Proximity Operations",
  "abstract": "Autonomous rendezvous and proximity operations (RPO) in adversarial orbital environments require guidance architectures balancing target pursuit, safety preservation, and real-time adaptability under dynamically evolving interaction conditions. Although learning-based approaches show promise, their application to safety-critical orbital robotics remains limited by concerns regarding interpretability, robustness, and constraint awareness. This work presents an adaptive Model Predictive Control (MPC) framework for autonomous spacecraft RPO in multi-agent adversarial scenarios. The proposed architecture combines a constrained receding-horizon MPC formulation with a data-driven supervisory tuning layer that adjusts controller parameters from offline closed-loop evaluation and online interaction geometry. Relative motion follows Clohessy-Wiltshire (CW) dynamics, enabling computationally efficient finite-horizon prediction and real-time quadratic optimization. The MPC formulation incorporates actuator limits, predictive keep-out-zone constraints, slack-variable feasibility handling, and optional Control Barrier Function (CBF) safety filtering. Rather than generating thrust commands directly, the adaptive layer modifies interpretable MPC parameters, including tracking weights, safety penalties, minimum-separation objectives, and keep-out-zone objectives. The framework was evaluated in the official Kerbal Space Program Differential Game (KSPDG) Capture-the-Satellite environment through Monte Carlo simulations. Results demonstrate improved closed-loop robustness, adaptive maneuvering behavior, and rendezvous performance compared with fixed-parameter MPC while preserving safety-aware operation and real-time feasibility, providing a modular, interpretable foundation for adaptive spacecraft RPO.",
  "published": "2026-07-18",
  "updated": "2026-07-18",
  "year": "2026",
  "authors": [
   "Luca Sportelli",
   "Tyler Barr",
   "Cagri Kilic",
   "Di Wu"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Results demonstrate improved closed-loop robustness, adaptive maneuvering behavior, and rendezvous performance compared with fixed-parameter MPC while preserving safety-aware operation and real-time feasibility, providing a modular, interpretable foundation for adaptive spacecraft RPO.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Luca Sportelli",
    "id": "2451292267",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Tyler Barr",
    "id": "2451279513",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Cagri Kilic",
    "id": "31014585",
    "h_index": 8,
    "papers": 24
   },
   {
    "name": "Di Wu",
    "id": "2283906616",
    "h_index": 5,
    "papers": 28
   }
  ],
  "comment": "34 pages, 7 figures",
  "topics": [
   "rl-control",
   "hardware-codesign",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.16630v1",
  "pdf_url": "https://arxiv.org/pdf/2607.16630v1",
  "html_url": "https://arxiv.org/html/2607.16630v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.16582",
  "slug": "autonomous-vr-based-risk-detection-for-situational-awareness-in-danger",
  "title": "Autonomous VR-Based Risk Detection for Situational Awareness in Dangerous Settings",
  "abstract": "In high-risk environments such as disaster response, situational awareness depends not only on detecting hazards but also on communicating them clearly to human operators. Vision Language Models (VLMs) have shown strong potential for scene understanding in safety-critical settings, yet their value as part of human-facing robotic systems remains underexplored. We present a VR-based Human Robot Interaction framework for studying how VLM-assisted robots can support situational awareness in simulated hazardous environments. In our system, a robot explores a virtual scene and queries a VLM to identify potential hazards and annotate user-facing points of interest. These annotations are presented to a human operator through an immersive VR interface. This framework enables controlled evaluation of both robotic hazard identification and the communication of safety-critical information to users. Results from our study indicate that the annotated VR interface was preferred over the unannotated baseline and that participants reported high clarity, usefulness, and comfort when interacting with the system. These findings suggest that combining VLM-based robotic perception with immersive visualization is a promising approach for supporting situational awareness in hazardous settings.",
  "published": "2026-07-18",
  "updated": "2026-07-18",
  "year": "2026",
  "authors": [
   "Mohammad Eskandari",
   "Murali Krishna Varma Indukuri",
   "Stephanie M. Lukin",
   "Cynthia Matuszek"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A VR-based Human Robot Interaction framework for studying how VLM-assisted robots can support situational awareness in simulated hazardous environments suggests that combining VLM-based robotic perception with immersive visualization is a promising approach for supporting situational awareness in hazardous settings.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Mohammad Eskandari",
    "id": "2451302778",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "M. Indukuri",
    "id": "2451296034",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Stephanie M. Lukin",
    "id": "2333829640",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Cynthia Matuszek",
    "id": "2127879703",
    "h_index": 7,
    "papers": 44
   }
  ],
  "comment": "7 Pages, Accepted to RO-MAN 2026",
  "topics": [
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.16582v1",
  "pdf_url": "https://arxiv.org/pdf/2607.16582v1",
  "html_url": "https://arxiv.org/html/2607.16582v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.16506",
  "slug": "foresight-residual-rl-for-long-horizon-robot-manipulation-with-vision",
  "title": "Foresight Residual RL for Long-Horizon Robot Manipulation with Vision-Language-Action Models",
  "abstract": "Vision-Language-Action (VLA) policies offer strong general-purpose manipulation priors, but often fail on tight-tolerance, contact-rich assembly due to long-horizon credit assignment and subtask coupling: a state that is geometrically successful for the current skill can be brittle for downstream skills. We show this failure mode in residual reinforcement learning (RL) over a frozen VLA base policy: constant sparse success rewards improve each subtask in isolation yet yield little or no gain when skills are chained, because terminal state quality is uncontrolled. We propose Foresight Residual RL, which optimizes handoff quality by augmenting each subtask's sparse success reward with an offline-estimated foresight value -- the probability of future subtask success conditioned on the terminal state of the current subtask. Concretely, we (i) train a visual foresight predictor from images of terminal states of the base policy, labeled using downstream rollout statistics, and (ii) train residual policies via backward foresight induction, using the predictor output as a reward multiplier. On a three-phase wrench-based nut-tightening assembly task in Isaac Gym (grasp, move-insert, rotate), our method achieves 85.6% full-task success, outperforming standard subtask residual RL (54.5%) and VLA baselines, while leaving per-subtask success unchanged. These results highlight that improving long-horizon performance requires shaping which successful states are produced at each sub-task, not only whether success occurs.",
  "published": "2026-07-17",
  "updated": "2026-07-17",
  "year": "2026",
  "authors": [
   "Yuhan Liu",
   "Xinyu Zhang",
   "Litao Liu",
   "Abdeslam Boularias"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "Foresight Residual RL is proposed, which optimizes handoff quality by augmenting each subtask's sparse success reward with an offline-estimated foresight value -- the probability of future subtask success conditioned on the terminal state of the current subtask.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yuhan Liu",
    "id": "2305664627",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Xinyu Zhang",
    "id": "2244773454",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Litao Liu",
    "id": "2323532361",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Abdeslam Boularias",
    "id": "2288336431",
    "h_index": 6,
    "papers": 22
   }
  ],
  "comment": "Accepted at IROS2026. Project website: https://jaysparrow.github.io/foresight-residual-rl",
  "topics": [
   "vla",
   "dexterous-manipulation",
   "tactile",
   "rl-control",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.16506v1",
  "pdf_url": "https://arxiv.org/pdf/2607.16506v1",
  "html_url": "https://arxiv.org/html/2607.16506v1",
  "code_url": "https://jaysparrow.github.io/foresight-residual-rl",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.3
 },
 {
  "id": "2607.16501",
  "slug": "certifiable-safe-model-based-reinforcement-learning-with-control-affin",
  "title": "Certifiable Safe Model-Based Reinforcement Learning with Control-Affine Dynamics Approximation",
  "abstract": "Safe model-based reinforcement learning (RL) often bridges control-theoretic analysis and RL for robots to safely explore (partially) unknown system dynamics while deriving control actions for task efficiency. The control performance and safety assurance typically rely on prior knowledge of partially modeled nominal system dynamics and the data-driven models that compensate for residual model uncertainties. However, existing methods often overlook the structure of residual model uncertainties (e.g., components affine in control), which could lead to overly conservative robot behaviors or invalid safety guarantees under the safe learning-based controllers. This paper proposes a safe reinforcement learning framework that learns control-affine dynamics with a certifiable data-driven safe policy using control barrier functions (CBF). Specifically, we first use Control-Affine Random Fourier Features (ARFF) to model robot dynamics in a control-affine form, which offers computational efficiency that scales with dataset size and reduces potential model bias for model-based reinforcement learning. Then, a model-free, efficient uncertainty quantification method using adaptive conformal prediction (ACP) is applied to quantify the uncertainty in the safety constraint arising from the learned control-affine dynamics. This allows for data-driven safety assurance amenable to principled and efficient controller synthesis with CBF. Simulation results on the cartpole and the 3D quadrotor platforms demonstrate the effectiveness of the proposed framework.",
  "published": "2026-07-17",
  "updated": "2026-07-17",
  "year": "2026",
  "authors": [
   "Hao Zhou",
   "Yanze Zhang",
   "Cameron Reid",
   "Wenhao Luo"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A model-free, efficient uncertainty quantification method using adaptive conformal prediction (ACP) is applied to quantify the uncertainty in the safety constraint arising from the learned control-affine dynamics, which allows for data-driven safety assurance amenable to principled and efficient controller synthesis with CBF.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hao Zhou",
    "id": "2310362592",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Yanze Zhang",
    "id": "2238390942",
    "h_index": 4,
    "papers": 16
   },
   {
    "name": "Cameron Reid",
    "id": "2370788056",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Wenhao Luo",
    "id": "2238259595",
    "h_index": 4,
    "papers": 20
   }
  ],
  "comment": "8 pages, accepted to IROS 2026",
  "topics": [
   "rl-control",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.16501v1",
  "pdf_url": "https://arxiv.org/pdf/2607.16501v1",
  "html_url": "https://arxiv.org/html/2607.16501v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2607.16187",
  "slug": "handroid-bridging-dexterous-hand-and-humanoid",
  "title": "Handroid: Bridging Dexterous Hand and Humanoid",
  "abstract": "Dexterous hands and humanoid robots are typically developed as distinct embodiments: the former enable contact-rich manipulation at the object scale, whereas the latter provide mobility and whole-body interaction in human-centered environments. We introduce \\textbf{Handroid}, a desktop-scale dual-embodiment robot that integrates both capabilities within a single reconfigurable platform. Handroid reuses one 27-DoF electromechanical body as either a dexterous hand or a desktop humanoid, measuring 0.33 m in height and 2.05 kg in weight. In the dexterous hand embodiment, 20 DoFs form an anthropomorphic hand closely matching the kinematic structure of the human hand. In the humanoid embodiment, the same articulated modules are reconfigured into a humanoid with a head, arms, and legs, including a 12-DoF lower-limb structure for locomotion and whole-body motion. Handroid further provides a unified control and learning framework supporting hand teleoperation, dexterous grasping, in-hand manipulation, humanoid locomotion, gait generation, and interactive motion authoring. We validate the platform through real-world dexterous manipulation, reinforcement-learning-based locomotion, keyframe motion deployment, and a long-horizon task involving embodiment reconfiguration, locomotion, docking, and dexterous pick-and-place. These results position Handroid as a compact and reproducible platform for advancing morphology-reconfigurable robotics and cross-embodiment robot learning.",
  "published": "2026-07-17",
  "updated": "2026-07-17",
  "year": "2026",
  "authors": [
   "Ruogu Li",
   "Chenyang Ma",
   "Sikai Li",
   "Zhenyu Wei",
   "Yunchao Yao",
   "Haochen Shi",
   "C. Karen Liu",
   "Shuran Song",
   "Mingyu Ding"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Handroid is introduced, a desktop-scale dual-embodiment robot that integrates both capabilities within a single reconfigurable platform and provides a unified control and learning framework supporting hand teleoperation, dexterous grasping, in-hand manipulation, humanoid locomotion, gait generation, and interactive motion authoring.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ruogu Li",
    "id": "2448107268",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Chenyang Ma",
    "id": "2292350179",
    "h_index": 5,
    "papers": 19
   },
   {
    "name": "Sikai Li",
    "id": "2283135687",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Zhenyu Wei",
    "id": "2394124879",
    "h_index": 4,
    "papers": 16
   },
   {
    "name": "Yunchao Yao",
    "id": "2352910959",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Haochen Shi",
    "id": "2325144259",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "C. K. Liu",
    "id": "2325109818",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Shuran Song",
    "id": "2336463656",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Mingyu Ding",
    "id": "2346837065",
    "h_index": 4,
    "papers": 15
   }
  ],
  "comment": "Project website: https://handroid.org",
  "topics": [
   "dexterous-manipulation",
   "humanoids",
   "tactile",
   "foundation-pretraining",
   "data-teleop",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.16187v1",
  "pdf_url": "https://arxiv.org/pdf/2607.16187v1",
  "html_url": "https://arxiv.org/html/2607.16187v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.16173",
  "slug": "vision-language-motion-maps-an-open-vocabulary-uncertainty-aware-query",
  "title": "Vision-Language-Motion Maps: An Open-Vocabulary, Uncertainty-Aware, Queryable Motion Attribute for 3D Scene Maps",
  "abstract": "Open-vocabulary 3D maps let robots answer language queries about what and where, but they assume a static world and cannot answer queries about how scene elements behave. We introduce Vision-Language-Motion Maps (VLMM), an open-vocabulary, language-queryable 3D map - queried through a rule-based intent router over open-vocabulary object nouns, not a general natural-language interface - in which each element carries a fused motion attribute: a VLM/LLM semantic movability prior combined with geometrically observed cross-frame motion, together with a per-element uncertainty. Queries reduce to attribute filters that distinguish what has been seen to move, what could move but has not, and what stays still. On a controlled simulator benchmark with exact ground truth (AI2-THOR, three scene types) we show through ablation that the schema fields are non-substitutable: a semantic-only baseline fails motion queries even with strong features, and neither motion field substitutes for the other (the prior cannot answer \"what is moving,\" observed motion cannot answer \"what could move\"). On real dynamic RGB-D (TUM and Bonn, six sequences) we show the uncertainty channel - our key difference from prior fused-motion work - consistently improves moving-vs-static average precision and reduces false motion flags, and that it is robust to estimated (noisy) poses. The raw confidence is not calibrated, but post-hoc isotonic calibration reaches an expected calibration error of 0.10. VLMM is a representation contribution: the closest prior maps each lack at least one of the four properties - open-vocabulary, language-queryable, fused prior-and-observed motion, and per-element uncertainty - that our combination provides.",
  "published": "2026-07-17",
  "updated": "2026-08-11",
  "year": "2026",
  "authors": [
   "Dibyendu Ghosh",
   "Ayushi Shakya"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "Vision-Language-Motion Maps is introduced, an open-vocabulary, language-queryable 3D map queried through a rule-based intent router over open-vocabulary object nouns, in which each element carries a fused motion attribute: a VLM/LLM semantic movability prior combined with geometrically observed cross-frame motion, together with a per-element uncertainty.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Dibyendu Ghosh",
    "id": "2237899516",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Ayushi Shakya",
    "id": "2451068869",
    "h_index": 0,
    "papers": 1
   }
  ],
  "comment": "8 pages, 5 figures, 3 tables. v2: corrected Eq. (7) (residual covariance;implementation unaffected), added the persistent-map update rule, the query-parser grammar, and an aggregation-quantile ablation",
  "topics": [
   "sim2real"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.16173v2",
  "pdf_url": "https://arxiv.org/pdf/2607.16173v2",
  "html_url": "https://arxiv.org/html/2607.16173v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2607.16146",
  "slug": "vtloc-learning-based-tactile-contact-localization-in-visual-point-clou",
  "title": "VTLoc: Learning-based Tactile Contact Localization in Visual Point Clouds",
  "abstract": "Vision and touch are complementary modalities essential for robotic perception and manipulation. While vision provides global object context, touch offers precise local information at contact points. Integrating these modalities for contact localization, i.e., predicting the location of touch on an object's surface, poses significant challenges due to the need for accurate spatial alignment between tactile data and visual geometry. To address this challenge, we propose VTLoc, a novel visual-tactile framework that localizes contact points from tactile readings using a 3D point cloud as visual input. VTLoc introduces two key components: a geometric multi-modal alignment module, which reconstructs a pseudo-point cloud from fused visual-tactile features and aligns it with the visual point cloud to enforce spatial consistencies across modalities; and an iterative localizing updater, which iteratively refines the predicted contact location using fused visual-tactile features. Evaluated on a new benchmark of 100 real-world objects, VTLoc improves single-touch contact localization by reducing local-to-global correspondence ambiguity.",
  "published": "2026-07-17",
  "updated": "2026-07-17",
  "year": "2026",
  "authors": [
   "Zhiyuan Wu",
   "Zhuo Chen",
   "Shan Luo"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "RA-L",
  "venue_source": "semantic-scholar",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "VTLoc is proposed, a novel visual-tactile framework that localizes contact points from tactile readings using a 3D point cloud as visual input and improves single-touch contact localization by reducing local-to-global correspondence ambiguity.",
  "doi": "10.1109/LRA.2026.3715029",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhiyuan Wu",
    "id": "2331572062",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Zhuo Chen",
    "id": "2312270976",
    "h_index": 5,
    "papers": 16
   },
   {
    "name": "Shan Luo",
    "id": "2349755470",
    "h_index": 3,
    "papers": 8
   }
  ],
  "comment": "",
  "topics": [
   "tactile",
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.16146v1",
  "pdf_url": "https://arxiv.org/pdf/2607.16146v1",
  "html_url": "https://arxiv.org/html/2607.16146v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2607.16123",
  "slug": "bayescontact-uncertain-pose-estimation-via-visuo-tactile-proposals-and",
  "title": "BayesContact: Uncertain Pose Estimation via Visuo-Tactile Proposals and Simulation-based Inference",
  "abstract": "Contact-rich manipulation requires pose estimates that are often more accurate than what depth-only sensing provides. Existing methods, relying on vision and contact, employ costly offline training procedures that need to be retrained for new environments and geometries. We propose BayesContact, a Simulation-Based Inference framework for visuo-tactile pose estimation in peg-in-hole insertion. BayesContact maintains a particle belief over object pose and fuses depth observations with force/torque-derived contact evidence. We employ simulation based forward models to approximate these observation likelihoods. For each pose hypothesis, a renderer predicts depth measurements and a physics simulator predicts contact outcomes under guarded probing actions; both are scored against real observations to update the belief. The resulting multimodal belief also enables information-gain-based probing for active disambiguation. Across simulated geometries and real-robot experiments, BayesContact improves pose observability and insertion success over vision-only inference by 30%",
  "published": "2026-07-17",
  "updated": "2026-07-28",
  "year": "2026",
  "authors": [
   "Aditya Kamireddypalli",
   "Matias Mattamala",
   "Joao Moura",
   "Russell Buchanan",
   "Sethu Vijayakumar",
   "Subramanian Ramamoorthy"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes BayesContact, a Simulation-Based Inference framework for visuo-tactile pose estimation in peg-in-hole insertion, which maintains a particle belief over object pose and fuses depth observations with force/torque-derived contact evidence.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Aditya Kamireddypalli",
    "id": "2351607537",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Mat\u00edas Mattamala",
    "id": "2380608749",
    "h_index": 0,
    "papers": 7
   },
   {
    "name": "Jo\u00e3o Moura",
    "id": "144968628",
    "h_index": 13,
    "papers": 38
   },
   {
    "name": "Russell Buchanan",
    "id": "2247808219",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "S. Vijayakumar",
    "id": "144575699",
    "h_index": 50,
    "papers": 459
   },
   {
    "name": "S. Ramamoorthy",
    "id": "144826759",
    "h_index": 27,
    "papers": 232
   }
  ],
  "comment": "Update funding sources",
  "topics": [
   "tactile",
   "sim2real"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.16123v2",
  "pdf_url": "https://arxiv.org/pdf/2607.16123v2",
  "html_url": "https://arxiv.org/html/2607.16123v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.16095",
  "slug": "let-the-body-follow-coupled-egocentric-control-for-whole-body-robot-te",
  "title": "Let the Body Follow: Coupled Egocentric Control for Whole-Body Robot Teleoperation",
  "abstract": "Whole-body teleoperation requires users to coordinate perception, manipulation, posture, and mobility across multiple robot components. This coordination is difficult because users must simultaneously control the robot's head, arms, torso, and base while maintaining task awareness and avoiding kinematic or environmental constraints. In this paper, we propose coupled egocentric control, a body-following teleoperation approach in which the robot's torso and base automatically respond to the operator's head and arm motions. Rather than requiring explicit touchpad commands for every torso or base adjustment, the system lets users focus on gaze and hand control: head pitch adjusts torso height, head yaw drives base rotation, end-effector height adjusts torso motion, and end-effector workspace boundaries trigger base translation. We evaluate this approach in a user study on whole-body teleoperation of a TIAGo mobile manipulator for home-care-inspired tasks. Compared with a baseline hybrid interface, coupled egocentric control improves object manipulation efficiency, reduces button-based control effort and arm singularities, lowers mental demand and overall workload, and increases ease of use, ease of learning, confidence, and user preference for torso and base control.",
  "published": "2026-07-17",
  "updated": "2026-07-17",
  "year": "2026",
  "authors": [
   "Tsung-Chi Lin",
   "Yichen Xie",
   "Chien-Ming Huang"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "Humanoids 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Coupled egocentric control is proposed, a body-following teleoperation approach in which the robot's torso and base automatically respond to the operator's head and arm motions, which improves object manipulation efficiency, reduces button-based control effort and arm singularities, lowers mental demand and overall workload, and increases ease of use, ease of learning, confidence, and user preference for torso and base control.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Tsung-Chi Lin",
    "id": "2143765881",
    "h_index": 7,
    "papers": 14
   },
   {
    "name": "Yichen Xie",
    "id": "2451364720",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Chien-Ming Huang",
    "id": "2290396120",
    "h_index": 4,
    "papers": 9
   }
  ],
  "comment": "8 pages, 11 figures, 1 table; submitted to the 2026 IEEE-RAS International Conference on Humanoid Robots (Humanoids 2026)",
  "topics": [
   "humanoids",
   "egocentric-data",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.16095v1",
  "pdf_url": "https://arxiv.org/pdf/2607.16095v1",
  "html_url": "https://arxiv.org/html/2607.16095v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2607.16012",
  "slug": "dpnext-a-lightweight-multi-scale-feature-fusion-framework-for-efficien",
  "title": "DPNeXt: A Lightweight Multi-Scale Feature Fusion Framework for Efficient ViT-Based Multi-Task Dense Prediction",
  "abstract": "Multi-Task Learning (MTL) in robotics perception systems supports comprehensive 3D spatial scene understanding by integrating semantic segmentation and depth estimation. While Vision Foundation Models (VFMs) are increasingly adopted as robust feature encoders, existing decoding strategies present a critical bottleneck. To address this, we propose DPNeXt, a streamlined multi-scale feature fusion decoder and efficient alternative to the standard Dense Prediction Transformer (DPT). DPNeXt uses dual depthwise separable inverted bottlenecks to improve frozen VFM utilization through fusion-centric decoding and independent task modularization. To further mitigate negative inductive transfer between tasks, we introduce the Multi-Task Boundary Guidance (MTBG) strategy. Unlike prior boundary-aware methods that add fusion modules or gating, MTBG applies symmetric boundary-focused supervision to encourage geometric consistency without extra annotation or inference cost. Experiments on Cityscapes show that DPNeXt-S outperforms prior state-of-the-art (SOTA) MTL models, while DPNeXt-B further improves the overall performance and achieves the best results among the compared methods. On NYUv2, DPNeXt-B also achieves the best semantic segmentation and depth estimation results among the compared methods while requiring substantially fewer trainable parameters than prior large-scale MTL models. Compared with the standard DPT, DPNeXt-S reduces trainable parameters by 78.6% and achieves the fastest inference speed among the compared models on resource-constrained laptop hardware. The source code, model checkpoints, and a demo video will be made available at https://github.com/kangjehun/DPNeXt.",
  "published": "2026-07-17",
  "updated": "2026-07-17",
  "year": "2026",
  "authors": [
   "Jehun Kang",
   "Jungha Wang",
   "Youngjun Hwang",
   "David Hyunchul Shim"
  ],
  "author_count": 4,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "DPNeXt, a streamlined multi-scale feature fusion decoder and efficient alternative to the standard Dense Prediction Transformer, and the Multi-Task Boundary Guidance (MTBG) strategy, which applies symmetric boundary-focused supervision to encourage geometric consistency without extra annotation or inference cost.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jehun Kang",
    "id": "2392912366",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Jungha Wang",
    "id": "2451253894",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Young-Joon Hwang",
    "id": "1782975067",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "D. H. Shim",
    "id": "2241208122",
    "h_index": 4,
    "papers": 12
   }
  ],
  "comment": "8 pages, 5 figures. Accepted to the 2026 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS 2026)",
  "topics": [
   "spatial-3d",
   "foundation-pretraining",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.16012v1",
  "pdf_url": "https://arxiv.org/pdf/2607.16012v1",
  "html_url": "https://arxiv.org/html/2607.16012v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2607.15982",
  "slug": "data-and-learning-where-it-matters-for-contact-rich-manipulation",
  "title": "Data and Learning Where it Matters for Contact-Rich Manipulation",
  "abstract": "Learned policies trained end-to-end on large datasets often remain brittle in high-precision tasks and struggle with generalization. We find that these limitations largely stem from a lack of structure and focus in data collection. Our key insight is to leverage dense data collection only for the critical segment of contact-rich tasks and to rely on traditional planning during simple free-space motion. We propose an automated data-collection scheme in combination with offline deep reinforcement learning for the critical segment of the task, eliminating reliance on a teleoperator's skill and on online policy updates. Across four challenging real-world tasks, using only 2 to 2.5 hours of autonomous data collection, we achieve an average success rate of 96%, compared to the strongest baseline at 55%. Notably, performance remains high in out-of-distribution scenarios where end-to-end approaches struggle. Our results pave the way for targeted data collection for contact-rich tasks and for high success rates in precision applications.",
  "published": "2026-07-17",
  "updated": "2026-07-17",
  "year": "2026",
  "authors": [
   "Oliver Hausd\u00f6rfer",
   "Linus Schwarz",
   "Gabor Marko",
   "Christian Dietz",
   "Timo Class",
   "Luka Hofer",
   "Jim Yun-Jin Li",
   "Johannes Hechtl",
   "Ralf R\u00f6mer",
   "Angela P. Schoellig"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes an automated data-collection scheme in combination with offline deep reinforcement learning for the critical segment of the task, eliminating reliance on a teleoperator's skill and on online policy updates and paving the way for targeted data collection for contact-rich tasks and for high success rates in precision applications.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Oliver Hausd\u00f6rfer",
    "id": "2330411916",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Linus Schwarz",
    "id": "2451067786",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Gabor Marko",
    "id": "2451068034",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Christian Dietz",
    "id": "2292398400",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Timo Class",
    "id": "2451067602",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Luka Hofer",
    "id": "2451067782",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Jim Yun-Jin Li",
    "id": "2451693220",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Johannes Hechtl",
    "id": "2381058197",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Ralf R\u00f6mer",
    "id": "2332358950",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Angela P. Schoellig",
    "id": "2321572233",
    "h_index": 4,
    "papers": 20
   }
  ],
  "comment": "Project webpage: https://anonymous.4open.science/w/data_and_learning_where_it_matters/",
  "topics": [
   "tactile",
   "rl-control",
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.15982v1",
  "pdf_url": "https://arxiv.org/pdf/2607.15982v1",
  "html_url": "https://arxiv.org/html/2607.15982v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.15898",
  "slug": "orbis-2-a-hierarchical-world-model-for-driving",
  "title": "Orbis 2: A Hierarchical World Model for Driving",
  "abstract": "Current world models operate at a single level of abstraction, with most prioritizing perceptual fidelity while lacking the spatial reasoning and semantic understanding required for real-world downstream tasks. We present a hierarchical driving world model that factorizes future prediction across two levels operating at distinct temporal and abstraction scales: a high-level predictor that forecasts coarse scene structure over extended temporal horizons, and a low-level generator that produces detailed predictions conditioned on the high-level output. This decomposition yields high perceptual fidelity while also capturing strong spatial and semantic representations. We further show that pretraining with a diffusion forcing objective yields substantially richer internal representations than the standard teacher forcing objective, while teacher forcing -- predicting only the next frame from clean context -- produces more stable autoregressive rollouts. We therefore introduce a generic two-stage training paradigm that pretrains the model with diffusion forcing and fine-tunes with teacher forcing, combining the representational benefits of the former with the rollout stability of the latter. Our approach achieves state-of-the-art results across the standard suite of driving world model evaluations on established benchmarks, including long-horizon generation fidelity, steering responsiveness evaluated on counterfactual scenarios, and internal representation quality. Project page with code, demo, checkpoints and qualitative results: https://lmb-freiburg.github.io/orbis2.github.io/",
  "published": "2026-07-17",
  "updated": "2026-07-17",
  "year": "2026",
  "authors": [
   "Sudhanshu Mittal",
   "Arian Mousakhan",
   "Silvio Galesso",
   "Karim Farid",
   "Jonannes Dienert",
   "Rajat Sahay",
   "Thomas Brox"
  ],
  "author_count": 7,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.LG",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A generic two-stage training paradigm that pretrains the model with diffusion forcing and fine-tunes with teacher forcing, combining the representational benefits of the former with the rollout stability of the latter, achieves state-of-the-art results across the standard suite of driving world model evaluations on established benchmarks.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Sudhanshu Mittal",
    "id": "9452482",
    "h_index": 10,
    "papers": 24
   },
   {
    "name": "Arian Mousakhan",
    "id": "2218778107",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Silvio Galesso",
    "id": "36011789",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Karim Farid",
    "id": "2257001639",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Jonannes Dienert",
    "id": "2451040505",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Rajat Sahay",
    "id": "2258115297",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Thomas Brox",
    "id": "2239727176",
    "h_index": 13,
    "papers": 29
   }
  ],
  "comment": "Project page: https://lmb-freiburg.github.io/orbis2.github.io/",
  "topics": [
   "world-models",
   "spatial-3d",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.15898v1",
  "pdf_url": "https://arxiv.org/pdf/2607.15898v1",
  "html_url": "https://arxiv.org/html/2607.15898v1",
  "code_url": "https://lmb-freiburg.github.io/orbis2.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2607.15880",
  "slug": "dynamics-aware-meta-imitation-for-generalization-to-unseen-robotic-man",
  "title": "Dynamics-Aware Meta-Imitation for Generalization to Unseen Robotic Manipulation",
  "abstract": "Imitation Learning aims to learn skills from extensive observations and demonstrations for robots, so it suffers from data scarcity and environment generalization. The existing methods predominantly focus on imitation from in-domain tasks and consequently struggle with generalization to unseen tasks. To bridge this generalization gap, we propose the \\textbf{D}ynamics-\\textbf{A}ware \\textbf{M}eta-\\textbf{I}mitation (DAMI) framework. By integrating meta-learning to construct a shared skill space, DAMI equips agents for rapid adaptation to novel tasks. We introduce the Visual-Motor Trajectory (VMT) module to capture complex spatio-temporal dynamics within the task latent space. Furthermore, we propose the Unpaired Unified Task (U2T) block to fuse unstructured multimodal observations. To coordinate these representations, we integrate a Task-Conditioned Feature Modulation (TCFM) mechanism customized for modulating low-level 3D features. By capturing intrinsic dynamics from a random complete reference demonstration, our framework learns the underlying task logic rather than memorizing static cues, ensuring effective generalization. Extensive experiments in both simulation and real-world settings demonstrate that our approach outperforms state-of-the-art baselines regarding direct inference on seen tasks and adaptation to unseen tasks via few-shot fine-tuning.",
  "published": "2026-07-17",
  "updated": "2026-07-17",
  "year": "2026",
  "authors": [
   "Zhenduo Shang",
   "Xiyao Liu",
   "Bohan Li",
   "Xudong Wang",
   "Teng Ren",
   "Lianqing Liu",
   "Zhi Han"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "By capturing intrinsic dynamics from a random complete reference demonstration, the DAMI framework learns the underlying task logic rather than memorizing static cues, ensuring effective generalization.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhenduo Shang",
    "id": "2330824785",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Xiyao Liu",
    "id": "2331001920",
    "h_index": 2,
    "papers": 11
   },
   {
    "name": "Bohan Li",
    "id": "2381415958",
    "h_index": 9,
    "papers": 20
   },
   {
    "name": "Xudong Wang",
    "id": "2397434474",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Teng Ren",
    "id": "2451042853",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Lianqing Liu",
    "id": "2376105249",
    "h_index": 3,
    "papers": 13
   },
   {
    "name": "Zhi Han",
    "id": "2376471791",
    "h_index": 3,
    "papers": 12
   }
  ],
  "comment": "",
  "topics": [
   "imitation-diffusion"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.15880v1",
  "pdf_url": "https://arxiv.org/pdf/2607.15880v1",
  "html_url": "https://arxiv.org/html/2607.15880v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.15868",
  "slug": "egoexomocap-distributed-ego-exo-human-motion-capture",
  "title": "EgoExoMoCap: Distributed Ego-Exo Human Motion Capture",
  "abstract": "Human motion capture from head-mounted devices (HMDs) offers a scalable way to acquire real-world human motion and interaction data, which is crucial for applications in embodied AI and VR/AR. Existing approaches focus on either egocentric body tracking, estimating the motion of the subject wearing the device, or exocentric tracking, capturing the movements of people in the wearer's surroundings. So far, these two paradigms have largely been explored in isolation. In this paper, we propose a novel distributed framework that jointly leverages ego- and exocentric multi-modal signals for human motion estimation from HMDs. Unlike traditional motion capture systems requiring bulky multi-camera setups or obtrusive mocap suits, our approach, EgoExoMoCap, is as simple as two (or more) people, each wearing a pair of smart glasses. The method leverages head (plus potentially wrist) tracking signals for accurate estimation of global motion in the 3D world and combines context-aware image features based on DINOv3 to achieve robustness in the presence of noise and occlusions. Extensive experiments on two in-the-wild datasets show that our approach can robustly reconstruct motion even in challenging scenarios.",
  "published": "2026-07-17",
  "updated": "2026-07-17",
  "year": "2026",
  "authors": [
   "Jiaxi Jiang",
   "Bharat Lal Bhatnagar",
   "Nan Yang",
   "Lingni Ma",
   "Sebastian Starke",
   "Robin Kips",
   "Nadine Bertsch",
   "Christian Holz",
   "Federica Bogo"
  ],
  "author_count": 9,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.GR",
   "cs.HC",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "ECCV 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper proposes a novel distributed framework that jointly leverages ego- and exocentric multi-modal signals for human motion estimation from HMDs and shows that the approach can robustly reconstruct motion even in challenging scenarios.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jiaxi Jiang",
    "id": "11324422",
    "h_index": 10,
    "papers": 16
   },
   {
    "name": "Bharat Lal Bhatnagar",
    "id": "48046803",
    "h_index": 17,
    "papers": 30
   },
   {
    "name": "Nan Yang",
    "id": "2306955572",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Lingni Ma",
    "id": "2284187142",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Sebastian Starke",
    "id": "2292029902",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Robin Kips",
    "id": "2332536023",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Nadine Bertsch",
    "id": "2332535976",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Christian Holz",
    "id": "2319602671",
    "h_index": 6,
    "papers": 19
   },
   {
    "name": "Federica Bogo",
    "id": "2279756657",
    "h_index": 3,
    "papers": 9
   }
  ],
  "comment": "Accepted by ECCV 2026, Project page and code: https://siplab.org/projects/EgoExoMoCap",
  "topics": [
   "egocentric-data",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.15868v1",
  "pdf_url": "https://arxiv.org/pdf/2607.15868v1",
  "html_url": "https://arxiv.org/html/2607.15868v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2607.15828",
  "slug": "beyond-frontiers-scene-anomaly-guided-autonomous-exploration",
  "title": "Beyond Frontiers: Scene-Anomaly Guided Autonomous Exploration",
  "abstract": "Autonomous exploration of unknown 3D environments is traditionally driven by coverage-maximizing geometric heuristics. However, these methods typically determine exploration targets without considering the underlying structural context. This leads to inefficient trajectories often limiting the fidelity of the final 3D reconstruction. To bridge the gap between spatial coverage and reconstruction quality, we introduce a novel paradigm: reframing exploration as a geometric anomaly minimization problem. We present SCAGE: SCene Anomaly Guided Exploration, a novel autonomous exploration framework that operates directly on unstructured 3D point clouds. Instead of blindly chasing volumetric boundaries, we equip the robot with a foundational understanding of standard indoor architecture. As the robot navigates, it continuously evaluates its live 3D observations against these learned expectations. When the incoming geometry contradicts the learned priors of a typical indoor environment, such as a fragmented wall or a partial table, the system flags these regions as scene anomalies. These geometric inconsistencies act as a guiding signal, naturally drawing the robot to investigate and resolve these structural anomalies from optimal vantage points. By actively targeting poorly reconstructed regions rather than just empty space, our approach seamlessly couples spatial discovery with high-fidelity mapping. Extensive evaluations demonstrate that SCAGE achieves superior volumetric coverage (~90% in all scenes) and higher 3D reconstruction quality compared to state-of-the-art baselines.",
  "published": "2026-07-17",
  "updated": "2026-07-17",
  "year": "2026",
  "authors": [
   "Akash Kumbar",
   "Abhinav Raundhal",
   "Madhava Krishna"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work presents SCAGE: SCene Anomaly Guided Exploration, a novel autonomous exploration framework that operates directly on unstructured 3D point clouds that achieves superior volumetric coverage and higher 3D reconstruction quality compared to state-of-the-art baselines.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Akash Kumbar",
    "id": "2276472089",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Abhinav Raundhal",
    "id": "2397376112",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Madhava Krishna",
    "id": "2245706022",
    "h_index": 2,
    "papers": 6
   }
  ],
  "comment": "Accepted in IEEE/RSJ IROS 2026. Project page: https://beyondfrontiers.github.io/",
  "topics": [
   "spatial-3d",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.15828v1",
  "pdf_url": "https://arxiv.org/pdf/2607.15828v1",
  "html_url": "https://arxiv.org/html/2607.15828v1",
  "code_url": "https://beyondfrontiers.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.5
 },
 {
  "id": "2607.15746",
  "slug": "towards-artificial-nerves-biomimetic-optical-fiber-tactile-sensing-for",
  "title": "Towards Artificial Nerves: Biomimetic Optical-Fiber Tactile Sensing for Robots",
  "abstract": "Robotic systems increasingly demand tactile sensing that approaches the adaptability and resolution of human skin to enable dexterous manipulation and safe interaction. OptiTac is a biomimetic tactile sensor that emulates the mechanoreceptor-to-nerve architecture of human touch by pairing each mechanical pin on a soft skin with an optical fiber acting as an artificial nerve. This design demonstrates an architectural principle for routing tactile information away from the sensing surface while preserving high spatial resolution, establishing a practical route toward distributed tactile sensing in future robotic systems. By treating tactile signals as images, simple analytical methods, rather than opaque deep-learning models, are used to infer contact location, size, and shape, providing interpretable and scalable tactile intelligence. This work demonstrates how evolutionary principles from biology can guide the development of artificial nerve systems for robots, offering a pathway toward human-like tactile perception in next-generation robotic platforms. More broadly, OptiTac establishes an artificial nerve-inspired sensing framework for interpretable robotic touch and a scalable route toward future distributed tactile systems.",
  "published": "2026-07-17",
  "updated": "2026-07-17",
  "year": "2026",
  "authors": [
   "Laura E. Butcher",
   "Chris J. Ford",
   "Nathan F. Lepora",
   "Efi Psomopoulou"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Laura E. Butcher",
    "id": "2451042792",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Christopher J. Ford",
    "id": "2173605714",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Nathan F. Lepora",
    "id": "2276954856",
    "h_index": 8,
    "papers": 30
   },
   {
    "name": "Efi Psomopoulou",
    "id": "1910114",
    "h_index": 11,
    "papers": 36
   }
  ],
  "comment": "19 pages, 6 figures",
  "topics": [
   "dexterous-manipulation",
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.15746v1",
  "pdf_url": "https://arxiv.org/pdf/2607.15746v1",
  "html_url": "https://arxiv.org/html/2607.15746v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.15733",
  "slug": "a-task-space-receding-horizon-controller-for-fast-collision-avoidance",
  "title": "A Task-Space Receding Horizon Controller for Fast Collision Avoidance",
  "abstract": "Real-time collision avoidance for robotic manipulators requires fast reactions to unexpected obstacle motion and lookahead to avoid becoming trapped by near-future constraints. Full model predictive control can provide this foresight, but its online cost may grow quickly with horizon length, model fidelity, and the number of active geometric constraints. Conversely, horizon-free reactive methods are computationally efficient but can be short-sighted in dynamic clutter. We present a task-space receding-horizon controller that uses a short contact-consistent rollout to generate a terminal kinematic reference satisfying internal non-penetration constraints, then computes only the first input of a smooth minimum-acceleration transition toward that reference. Starting from a closed-loop inverse-kinematics regulation law, the rollout is performed with an iterative dynamics solver operating on inflated convex robot and obstacle geometries, so that robot-obstacle contacts, dynamic obstacle motion, and self-collisions can shape the terminal reference without requiring full constrained trajectory optimization. We analyze the contact-inactive closed loop and show local exponential task-space regulation under standard regularity assumptions. For contacts activated inside the rollout, we characterize the corresponding discrete updates and bound the effect of moving obstacles on regular operating sets. Simulations on a 40-DOF multi-chain system show that intermediate horizons balance anticipation, responsiveness, and computational cost. Hardware experiments on a 6-DOF platform demonstrate consistent sim-to-real behavior without accurate inertial parameter estimation, and comparisons against dynamic optimization fabrics and model predictive control (MPC) baselines show improved success rates in dynamic clutter while preserving solve times compatible with real-time execution in the tested regimes.",
  "published": "2026-07-17",
  "updated": "2026-07-17",
  "year": "2026",
  "authors": [
   "Mattia Penzotti",
   "Marco Controzzi"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A task-space receding-horizon controller that uses a short contact-consistent rollout to generate a terminal kinematic reference satisfying internal non-penetration constraints, then computes only the first input of a smooth minimum-acceleration transition toward that reference.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "M. Penzotti",
    "id": "2199444096",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "M. Controzzi",
    "id": "2263255479",
    "h_index": 3,
    "papers": 16
   }
  ],
  "comment": "",
  "topics": [
   "sim2real",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.15733v1",
  "pdf_url": "https://arxiv.org/pdf/2607.15733v1",
  "html_url": "https://arxiv.org/html/2607.15733v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.15701",
  "slug": "raven-reinforcement-adaptive-visibility-graph-planning-for-robust-huma",
  "title": "RAVEN: Reinforcement-Adaptive Visibility-Graph Planning for Robust Humanoid Navigation with Collision-Free MPC",
  "abstract": "Humanoid navigation in dynamic environments requires long-horizon planning while respecting short-horizon dynamic and safety constraints. Classical visibility-graph planners combined with model predictive control (MPC) can efficiently generate collision-free trajectories, but their performance depends on manually tuned parameters and accurate system modeling. In real robotic systems, control delays, state-estimation noise, and locomotion uncertainties can cause overshoot and constraint violations even when the nominal path is geometrically optimal. We propose RAVEN, a hierarchical reinforcement learning (RL)-MPC framework for robust humanoid navigation. Unlike prior approaches that use learning to tune cost weights or replace planning entirely, RAVEN employs RL to adapt the geometric construction of a visibility-graph planner by modifying obstacle inflation and related graph parameters. By directly reshaping the free-space geometry, the learned planner alters the topology of the global path to compensate for delay and tracking imperfections. A collision-free MPC layer then tracks the planned trajectory while explicitly enforcing velocity bounds and obstacle-avoidance constraints. By training under realistic delays and observation noise, RAVEN learns planning adaptations that improve robustness while retaining explicit long-horizon geometric planning and constrained optimization, in contrast to end-to-end learning approaches. We evaluate RAVEN against a manually tuned visibility-graph MPC baseline and a pure RL navigation policy. Results demonstrate reduced overshoot near obstacles, improved robustness in narrow passages, and more reliable navigation under delay and noise. These findings indicate that reinforcement-adaptive graph construction combined with constrained MPC provides an effective and interpretable alternative to end-to-end learning for robust humanoid navigation.",
  "published": "2026-07-17",
  "updated": "2026-07-17",
  "year": "2026",
  "authors": [
   "Ruochen Hou",
   "Shiqi Wang",
   "Beom Jun Kim",
   "Hanzhang Fang",
   "Mehak Singal",
   "Dennis W. Hong"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Results demonstrate reduced overshoot near obstacles, improved robustness in narrow passages, and more reliable navigation under delay and noise, indicating that reinforcement-adaptive graph construction combined with constrained MPC provides an effective and interpretable alternative to end-to-end learning for robust humanoid navigation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ruochen Hou",
    "id": "2316060534",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Shiqi Wang",
    "id": "2372435173",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Beomdo Kim",
    "id": "2345408475",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Hanzhang Fang",
    "id": "2451066111",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Mehak Singal",
    "id": "2451045027",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Dennis W. Hong",
    "id": "2376201986",
    "h_index": 1,
    "papers": 8
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "rl-control",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.15701v1",
  "pdf_url": "https://arxiv.org/pdf/2607.15701v1",
  "html_url": "https://arxiv.org/html/2607.15701v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.15659",
  "slug": "continuously-stable-structure-through-plastic-deformation",
  "title": "Continuously Stable Structure through Plastic Deformation",
  "abstract": "Soft robots have seen widespread adoption in interactive tasks due to their inherent compliance and adaptability. However, these advantages often come at the cost of stability, posing challenges in a dynamic environment. This limitation is especially critical in soft grippers, where instability under acceleration or external disturbances can result in grasp failure. In this study, we present a continuously stable structure through plastic deformation (CSSPD), integrated into a soft gripper. By leveraging the mechanism of plastic deformation, the gripper maintains continuous configurations without energy input, while the added stiffness ensures both static and dynamic stability. We introduce a bioinspired paw pad that significantly enhances stability and enables sensing-based rapid object grasping. Then we develop the mathematical model and optimize the kirigami structure of the metal layer. Experimental results show that the gripper can sustain a passive holding force of up to 16 N without energy input, achieving performance comparable to pneumatic actuation at 0.3 MPa. When combined with pneumatic actuation, it remains stable under pulsed accelerations of up to 400 m/s^2. It can also passively perch on tree branches for extended periods without power, demonstrating promise for mobile robotic applications.",
  "published": "2026-07-17",
  "updated": "2026-07-20",
  "year": "2026",
  "authors": [
   "Junlong Xiao",
   "Yaoqiang Pan",
   "Xuan Zhang",
   "Michael Yu Wang",
   "Chao Chen"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Junlong Xiao",
    "id": "144184079",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Yaoqiang Pan",
    "id": "2451262151",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Xuanhao Zhang",
    "id": "2451035039",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "M. Wang",
    "id": "2388126997",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Chao Chen Faculty of Engineering",
    "id": "2451041529",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Monash University",
    "id": "2450376161",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Clayton",
    "id": "2451040558",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Victoria 3800",
    "id": "2162836835",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Australia",
    "id": "2268974249",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Michael Yu Wang. School of Engineering",
    "id": "2451041421",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Great Bay University",
    "id": "2451040520",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Songshan Lake",
    "id": "2451041407",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Dongguan",
    "id": "81088867",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Guangdong",
    "id": "2084760702",
    "h_index": 4,
    "papers": 18
   },
   {
    "name": "China",
    "id": "2452536298",
    "h_index": 0,
    "papers": 3
   }
  ],
  "comment": "24 pages, 8 figures",
  "topics": [
   "dexterous-manipulation",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.15659v2",
  "pdf_url": "https://arxiv.org/pdf/2607.15659v2",
  "html_url": "https://arxiv.org/html/2607.15659v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.15642",
  "slug": "difference-based-relational-learning-for-zero-shot-object-goal-visual",
  "title": "Difference-Based Relational Learning for Zero-Shot Object-Goal Visual Navigation With Direct Sim-to-Real Transfer",
  "abstract": "End-to-end deep reinforcement learning (DRL) for zero-shot object-goal visual navigation remains challenged by the sim-to-real gap, particularly variations in object appearance and restricted camera field-of-view (FoV). This letter proposes a Temporal Difference-Relational Network (T-DRN) for robust zero-shot sim-to-real transfer. T-DRN combines a Siamese difference-based feature extractor, which computes relational difference between the target and observed objects to produce domain-independent representations, with a dual-frame temporal buffer that preserves short-term object continuity under narrow FoV. Extensive experiments in AI2-THOR demonstrate that T-DRN improves zero-shot generalization in terms of success rates over strong baselines. Furthermore, T-DRN is systematically validated on a physical wheeled robot, demonstrating robust performance under real sensing and actuation constraints and supporting the feasibility of direct sim-to-real transfer.",
  "published": "2026-07-17",
  "updated": "2026-07-17",
  "year": "2026",
  "authors": [
   "Guolei Qi",
   "Feitian Zhang"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A Temporal Difference-Relational Network (T-DRN) is proposed for robust zero-shot sim-to-real transfer, combining a Siamese difference-based feature extractor, which computes relational difference between the target and observed objects to produce domain-independent representations, with a dual-frame temporal buffer that preserves short-term object continuity under narrow FoV.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Guolei Qi",
    "id": "2387977889",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Feitian Zhang",
    "id": "2298893540",
    "h_index": 3,
    "papers": 14
   }
  ],
  "comment": "",
  "topics": [
   "sim2real",
   "rl-control",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.15642v1",
  "pdf_url": "https://arxiv.org/pdf/2607.15642v1",
  "html_url": "https://arxiv.org/html/2607.15642v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.15641",
  "slug": "imbench-a-benchmark-for-intuitive-robotic-manipulation",
  "title": "IMBench: A Benchmark for Intuitive Robotic Manipulation",
  "abstract": "Humans combine reasoning and motor control to solve complex manipulation tasks under diverse constraints. They build an understanding of the physical world that helps them convert reasoning into actions and quickly adapt to new scenes, tasks, and rules. We refer to this capability as intuitive manipulation. Existing benchmarks fail to capture this integration: they evaluate physical reasoning in isolation from execution, or measure policy performance without requiring explicit reasoning. We introduce IMBENCH, a benchmark designed to evaluate intuitive manipulation as an integrated capability spanning perception, physical reasoning, action generation, and iterative execution. Our tasks require models to infer task-relevant physical structure and generate feasible action sequences under explicit constraints, including contact-rich manipulation, tool use, and multi-stage dependencies. We introduce a benchmark of 35 tasks, 14K filtered trajectories, and scalable tools for generating diverse scenarios. Experiments reveal a consistent gap: vision language models show partial physical reasoning ability but fail to produce executable plans, while state-of-the-art vision-language-action models struggle to satisfy task constraints and generalize across scenarios. These results identify intuitive manipulation as a missing axis in current foundation models and generalist robot policies, and position IMBENCH as a step toward evaluating and enabling more integrated, adaptive physical intelligence.",
  "published": "2026-07-17",
  "updated": "2026-07-17",
  "year": "2026",
  "authors": [
   "Anurag Maurya",
   "Sukhvansh Jain",
   "Prajwal Avhad",
   "Gautham Balachandran",
   "Ziyi Zhou",
   "Atharva Kshirsagar",
   "Satyam Singh",
   "Bowen Li. Rishabh Mukund",
   "Ritul Singh",
   "Jatin Vira",
   "Suvonil Chatterjee",
   "Devesh K. Jha"
  ],
  "author_count": 12,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "RSS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "IMBENCH is introduced, a benchmark designed to evaluate intuitive manipulation as an integrated capability spanning perception, physical reasoning, action generation, and iterative execution, and position IMBENCH as a step toward evaluating and enabling more integrated, adaptive physical intelligence.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Anurag Maurya",
    "id": "2355872634",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Sukhvansh Jain",
    "id": "2451204523",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Prajwal Avhad",
    "id": "2451040626",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "G. Balachandran",
    "id": "2084024116",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Ziyi Zhou",
    "id": "2121298138",
    "h_index": 10,
    "papers": 30
   },
   {
    "name": "Atharva Kshirsagar",
    "id": "2078974942",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Satyam Singh",
    "id": "2451075359",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Bowen Li. Rishabh Mukund",
    "id": "2451044199",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Ritul Singh",
    "id": "2451131942",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Jatin Vira",
    "id": "2451042019",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Suvonil Chatterjee",
    "id": "2451045466",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Devesh K. Jha",
    "id": "2293723730",
    "h_index": 6,
    "papers": 24
   }
  ],
  "comment": "Accepted to SemRob Workshop, RSS 2026. Project Website: https://imbench.org/",
  "topics": [
   "vla",
   "tactile",
   "foundation-pretraining"
  ],
  "orgs": [
   "Physical Intelligence"
  ],
  "abs_url": "https://arxiv.org/abs/2607.15641v1",
  "pdf_url": "https://arxiv.org/pdf/2607.15641v1",
  "html_url": "https://arxiv.org/html/2607.15641v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.0
 },
 {
  "id": "2607.15633",
  "slug": "scalable-open-source-visuotactile-sensor-for-6-axis-contact-wrench-est",
  "title": "Scalable Open-Source Visuotactile Sensor for 6-Axis Contact Wrench Estimation in Tensegrity Robots",
  "abstract": "This paper presents a scalable, open-source visuotactile sensing system for tensegrity robots that enables six-axis wrench estimation and contact detection. The proposed endcap sensor integrates an elastomeric shell, a 3D-printed thermoplastic polyurethane (TPU) interface, and a rigid base housing an embedded camera and LED illumination ring. A novel gyroid-infill bonding technique is introduced to form a durable elastomer-TPU interface without adhesives, yielding a lightweight and modular design compatible with large-scale tensegrity structures. A tactile-to-wrench neural network maps shear vector fields to six-dimensional force and torque measurements. Experimental results demonstrate accurate and stable wrench estimation with a mean squared error (MSE) of 0.1531 on static validation data and out-of-domain generalization under dynamic motion. Furthermore, full-system integration on a 12 kg tensegrity robot confirms the sensor's ability to reliably identify ground contacts. The system substantially improves the practicality of tactile feedback for tensegrity robots, offering a low-cost, reproducible, and physically interpretable pathway toward contact-aware proprioception and state estimation. Open source files are available at \\href{https://github.com/Jonathan-Twz/tensegrity-gelfoot}{github.com/Jonathan-Twz/tensegrity-gelfoot}",
  "published": "2026-07-17",
  "updated": "2026-07-17",
  "year": "2026",
  "authors": [
   "Wenzhe Tong",
   "Jonathan Mi",
   "Xili Yi",
   "Nima Fazeli",
   "Xiaonan Huang"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Wenzhe Tong",
    "id": "2320303624",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Jonathan Mi",
    "id": "2320309040",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Xili Yi",
    "id": "51147397",
    "h_index": 4,
    "papers": 16
   },
   {
    "name": "Nima Fazeli",
    "id": "2718552",
    "h_index": 19,
    "papers": 83
   },
   {
    "name": "Xiaonan Huang",
    "id": "2320347568",
    "h_index": 4,
    "papers": 11
   }
  ],
  "comment": "8 pages, 10 figures, IROS 2026",
  "topics": [
   "tactile",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.15633v1",
  "pdf_url": "https://arxiv.org/pdf/2607.15633v1",
  "html_url": "https://arxiv.org/html/2607.15633v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2607.15620",
  "slug": "aegis-assay-aware-protocol-validation-and-runtime-monitoring-for-open",
  "title": "AEGIS: Assay-Aware Protocol Validation and Runtime Monitoring for Open-Source Liquid Handling Robots",
  "abstract": "Self-driving laboratories increasingly rely on low-cost liquid handlers such as the Opentrons OT-2, which ship without the pressure-based aspiration monitoring of Hamilton or Tecan systems and are typically run open-loop. Two failure modes go undetected: protocols that are syntactically valid but violate assay-specific invariants (e.g., tip reuse between a PCR template and a no-template control), and physical execution failures (partial dispense, air bubbles, missing tips) at runtime. We present AEGIS, a two-layer guardian for both. Layer 1 pairs a curated machine-readable assay rule database with an LLM that reasons over OT-2 Python code, reaching an adjusted F1 of 0.97 on a 24-protocol benchmark across five assay families and beating rules-only and LLM-only ablations across five backends; a free open-weight model ties the best proprietary one, so no paid API is required. Layer 2 fits a PCA world model to YOLO-cropped four-frame pipette trajectories; under a leakage-free leave-one-plate-out evaluation it reaches average precision 0.89 and operating-point F1 0.71 (AUROC 0.80), a deployment-faithful number that matches the live demonstration, and we characterize the small-pipette (p20) resolution limit (F1 0.47). A live demonstration on a physical OT-2 (five replicates per condition) catches planted no-tip failures deterministically and partial dispense on coloured dyes, with an always-VLM self-vote gate lifting partial-dispense recall to 5/5; transparent water is a principled limit of any front-view-only monitor, which AEGIS surfaces as low-confidence VLM reasoning rather than a wrong verdict. Cascade triage holds VLM cost near $1.63 per plate versus $10.33 for an always-VLM baseline. AEGIS is open source and, to our knowledge, the first system to unify pre-flight assay-aware validation with runtime visual monitoring for an open-source liquid handler.",
  "published": "2026-07-17",
  "updated": "2026-07-17",
  "year": "2026",
  "authors": [
   "Priyanka V. Setty",
   "Arvind Ramanathan",
   "Ian Foster",
   "Rick Stevens"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "AEGIS is open source and, to the authors' knowledge, the first system to unify pre-flight assay-aware validation with runtime visual monitoring for an open-source liquid handler.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Priyanka V.Setty",
    "id": "2451042192",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Arvind Ramanathan",
    "id": "47941567",
    "h_index": 8,
    "papers": 34
   },
   {
    "name": "Ian T. Foster",
    "id": "2305622167",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Rick L. Stevens",
    "id": "2275055526",
    "h_index": 5,
    "papers": 10
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.15620v1",
  "pdf_url": "https://arxiv.org/pdf/2607.15620v1",
  "html_url": "https://arxiv.org/html/2607.15620v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.15582",
  "slug": "a-model-based-decoupling-strategy-for-proprioception-and-contact-sensi",
  "title": "A Model-Based Decoupling Strategy for Proprioception and Contact Sensing in an Architected Soft Manipulator",
  "abstract": "Soft continuum robots require embedded sensing for proprioception and contact detection, yet integrating sensors into sparse, highly deformable architected structures remains challenging. We present a model-based strategy that decouples proprioceptive and contact signals from a common set of fluidic pressure sensors embedded in a soft architected segment. Each segment of the Innervated Trimmed Helicoid (ITH) contains six air channels routed in a localized zigzag pattern along the circumference. With only three principal kinematic degrees of freedom (axial compression, bending in x, bending in y), the six pressure readings form an overdetermined system. A piecewise constant curvature model maps pressures to shape, and Huber regression identifies outlier channels whose residuals indicate external contact. On a single ITH segment, this approach achieves proprioceptive shape estimation with a relative bending error of 0.11 +/- 0.02 and a contact detection rate of 97% across 178 trials. We integrate eight ITH segments into Air-Helix, a tendon-driven soft continuum manipulator, and present exploratory whole-arm demonstrations that include tactile teaching by demonstration, admittance-controlled force regulation, and tactile object reconstruction. The results suggest that localized fluidic innervation combined with model-based redundancy resolution is a practical path toward concurrent proprioception and contact sensing in architected soft robots.",
  "published": "2026-07-17",
  "updated": "2026-07-17",
  "year": "2026",
  "authors": [
   "Francesco Stella",
   "Annan Zhang",
   "Cosimo Della Santina",
   "Josie Hughes",
   "Daniela Rus"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Francesco Stella",
    "id": "2244219485",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Annan Zhang",
    "id": "113329770",
    "h_index": 10,
    "papers": 15
   },
   {
    "name": "C. D. Santina",
    "id": "35178897",
    "h_index": 25,
    "papers": 177
   },
   {
    "name": "Josie Hughes",
    "id": "40662586",
    "h_index": 25,
    "papers": 190
   },
   {
    "name": "Daniela Rus",
    "id": "2261287511",
    "h_index": 4,
    "papers": 13
   }
  ],
  "comment": "Accepted for publication in the proceedings of the 2026 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS)",
  "topics": [
   "tactile",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.15582v1",
  "pdf_url": "https://arxiv.org/pdf/2607.15582v1",
  "html_url": "https://arxiv.org/html/2607.15582v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2607.15579",
  "slug": "pace-persona-adaptation-through-conversational-elicitation-in-human-ro",
  "title": "PACE: Persona Adaptation through Conversational Elicitation in Human-Robot Interaction",
  "abstract": "Equipping humanoid robots with coherent and adaptable personas is crucial for fostering natural, engaging, and trustworthy human-robot interaction (HRI). However, existing approaches often rely on static, hard-coded identities that lack the flexibility to adapt to individual user contexts. In this paper, we present PACE (Persona Adaptation through Conversational Elicitation), a novel framework for the interactive generation and deployment of structured personas on the Ameca humanoid robot. Our system introduces an Interactive Persona Elicitation Pipeline, enabling the robot to dynamically synthesize a tailored, psychologically grounded identity through user Q&A. This elicitation process feeds into a persona prompt compilation phase, generating a structured persona prompt built upon multi-perspective dimensions. We detail the Embodied System Integration required to translate this structured specification into expressive, multimodal humanoid behaviors. Through a comprehensive empirical HRI evaluation, we assess the impact of dynamically generated personas on user trust, perceived anthropomorphism, persona consistency, personal relevance, and interaction quality compared to a generic baseline. These contributions establish a scalable pathway for deploying personalized, interactive, and reliable identities in embodied humanoid assistants. Video demo is available at: https://lipzh5.github.io/PACE/",
  "published": "2026-07-17",
  "updated": "2026-07-21",
  "year": "2026",
  "authors": [
   "Peizhen Li",
   "Longbing Cao",
   "Megani Rajendran",
   "Timothy Liu",
   "Aik Beng Ng",
   "Simon See"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.HC"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "PACE (Persona Adaptation through Conversational Elicitation), a novel framework for the interactive generation and deployment of structured personas on the Ameca humanoid robot, introduces an Interactive Persona Elicitation Pipeline, enabling the robot to dynamically synthesize a tailored, psychologically grounded identity through user Q&A.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Peizhen Li",
    "id": "2327378572",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Longbing Cao",
    "id": "2328683625",
    "h_index": 3,
    "papers": 14
   },
   {
    "name": "Megani Rajendran",
    "id": "2163710152",
    "h_index": 3,
    "papers": 13
   },
   {
    "name": "Timothy Liu",
    "id": "2109992450",
    "h_index": 4,
    "papers": 14
   },
   {
    "name": "A. Ng",
    "id": "1557327062",
    "h_index": 7,
    "papers": 41
   },
   {
    "name": "Simon See",
    "id": "2266630271",
    "h_index": 5,
    "papers": 29
   }
  ],
  "comment": "8 pages, 5 figures",
  "topics": [
   "humanoids",
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.15579v2",
  "pdf_url": "https://arxiv.org/pdf/2607.15579v2",
  "html_url": "https://arxiv.org/html/2607.15579v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.15542",
  "slug": "improvedvbgs-real-time-continual-variational-bayes-gaussian-splatting",
  "title": "ImprovedVBGS: Real-time Continual Variational Bayes Gaussian Splatting",
  "abstract": "On-the-fly reconstruction is a key requirement for many applications in robotics and autonomous navigation. Variational Bayes Gaussian Splatting (VBGS) enables continual learning without replay buffers using Coordinate Ascent Variational Inference (CAVI), but its per-frame iterations over all observed points make it too slow for real-time use with strict memory and latency requirements. We present ImprovedVBGS, an accelerated framework for on-the-fly continual reconstruction. This is achieved primarily through (i) spatially truncated variational inference, and (ii) improved reassignment that uses forwarding, truncation and eliminates wasteful dynamic recompilation. On the NeRF synthetic dataset, we reduce mean per-frame latency from ~84.0 s to ~0.050 s on an RTX 3070 Ti, a 1680x speed-up while maintaining reconstruction quality.",
  "published": "2026-07-17",
  "updated": "2026-07-17",
  "year": "2026",
  "authors": [
   "Damani Mguni-Coker"
  ],
  "author_count": 1,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "ImprovedVBGS is presented, an accelerated framework for on-the-fly continual reconstruction of Variational Bayes Gaussian Splatting through spatially truncated variational inference, and improved reassignment that uses forwarding, truncation and eliminates wasteful dynamic recompilation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Damani Mguni-Coker",
    "id": "2451045092",
    "h_index": 0,
    "papers": 1
   }
  ],
  "comment": "5 pages, 4 figure. Technical Report. This work introduces ImprovedVBGS, accelerated continual learning for 3D Gaussian Splatting based Reconstruction. Code available at [https://github.com/damanimc/ImprovedVBGS](https://github.com/damanimc/ImprovedVBGS)",
  "topics": [
   "spatial-3d",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.15542v1",
  "pdf_url": "https://arxiv.org/pdf/2607.15542v1",
  "html_url": "https://arxiv.org/html/2607.15542v1",
  "code_url": "https://github.com/damanimc/ImprovedVBGS",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2607.15483",
  "slug": "risk-aware-preference-learning-for-stochastic-outcomes",
  "title": "Risk-Aware Preference Learning for Stochastic Outcomes",
  "abstract": "Learning reward functions from human preferences is a widely used approach for aligning robot behavior with user expectations in human-robot interaction. Most existing approaches assume that humans evaluate uncertain outcomes using expected utility (EU), aggregating outcome utilities linearly with their probabilities. However, behavioral evidence shows that humans are systematically risk-sensitive, overweighting rare negative events and exhibiting loss aversion. We study the consequences of this mismatch in social robot navigation, where safety-critical outcomes (e.g., collisions) are rare but highly consequential. We compare EU with Cumulative Prospect Theory (CPT), a nonlinear model of human decision-making, within a Bradley-Terry preference learning framework. Our preliminary experiments show that when preferences are generated by risk-sensitive users, CPT-based learners recover reward functions with substantially lower regret compared to EU-based learners. Our results highlight the importance of modeling human risk sensitivity when learning rewards from preferences over stochastic robot outcomes.",
  "published": "2026-07-16",
  "updated": "2026-07-16",
  "year": "2026",
  "authors": [
   "Yi-Shiuan Tung",
   "Yuni Wu",
   "Wei Jiang",
   "Alessandro Roncone",
   "Bradley Hayes"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA 2026",
  "venue_source": "arxiv-comment",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "This work compares EU with Cumulative Prospect Theory (CPT), a nonlinear model of human decision-making, within a Bradley-Terry preference learning framework and shows that when preferences are generated by risk-sensitive users, CPT-based learners recover reward functions with substantially lower regret compared to EU-based learners.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yi-Shiuan Tung",
    "id": "1906867",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Yuning Wu",
    "id": "2108407304",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Wei Jiang",
    "id": "2350044938",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Alessandro Roncone",
    "id": "3077150",
    "h_index": 5,
    "papers": 30
   },
   {
    "name": "Bradley Hayes",
    "id": "2263552055",
    "h_index": 4,
    "papers": 19
   }
  ],
  "comment": "ICRA 2026 Workshop on Bridging the Gap between Robot Learning and Human-Robot Interaction",
  "topics": [
   "navigation",
   "hri",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.15483v1",
  "pdf_url": "https://arxiv.org/pdf/2607.15483v1",
  "html_url": "https://arxiv.org/html/2607.15483v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.8
 },
 {
  "id": "2607.15448",
  "slug": "vtap-gripper-synergizing-fingertip-sensing-and-a-visuo-tactile-active",
  "title": "VTAP Gripper: Synergizing Fingertip Sensing and a Visuo-Tactile Active Palm for Dexterous In-Hand Manipulation",
  "abstract": "This paper presents a tactile-reactive gripper that integrates a Visuo-Tactile Active Palm (VTAP) and compliant, reconfigurable fingers equipped with tactile array sensors. The design exploits structured finger-palm synergy and multi-modal perception to achieve both robust grasping and fine manipulation. The actuated bi-modal palm seamlessly combines long-range visual localization with contact-rich tactile feedback, substantially extending the system's manipulation capability. To bridge the embodiment gap between human hand motion and the heterogeneous three-finger structure, we further propose a staged, gesture-conditioned retargeting framework for dexterous teleoperation. Extensive experiments validate the system across a range of challenging tasks: reactive grasping of YCB and fragile objects, in-hand syringe reorientation and plunger actuation, singulation of clustered objects down to 3 mm in diameter, and vision-tactile peg-in-hole insertion. Results demonstrate that high manipulation performance can be achieved through coordinated finger-palm interaction and multi-modal sensing, without resorting to high degrees of freedom anthropomorphic designs. The VTAP gripper and its retargeting framework offer a practical reference architecture for dexterous gripper design, manipulation, and contact-rich data collection in support of learning-based approaches. Project webpage: https://yuhochau.github.io/vtap/.",
  "published": "2026-07-16",
  "updated": "2026-07-16",
  "year": "2026",
  "authors": [
   "Yuhao Zhou",
   "Sheeraz Athar",
   "Zhixian Hu",
   "Binghao Huang",
   "Yunzhu Li",
   "Juan Wachs",
   "Yu She"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A tactile-reactive gripper that integrates a Visuo-Tactile Active Palm and compliant, reconfigurable fingers equipped with tactile array sensors that offers a practical reference architecture for dexterous gripper design, manipulation, and contact-rich data collection in support of learning-based approaches.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yuhao Zhou",
    "id": "2314297484",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Sheeraz Athar",
    "id": "2238336847",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Zhixian Hu",
    "id": "2348410905",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Binghao Huang",
    "id": "2287019710",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Yunzhu Li",
    "id": "2374457025",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Juan P Wachs",
    "id": "2266149003",
    "h_index": 4,
    "papers": 14
   },
   {
    "name": "Yu She",
    "id": "2317112691",
    "h_index": 3,
    "papers": 12
   }
  ],
  "comment": "8 pages, 10 figures, accepted to the 2026 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS 2026). Project webpage: https://yuhochau.github.io/vtap/",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.15448v1",
  "pdf_url": "https://arxiv.org/pdf/2607.15448v1",
  "html_url": "https://arxiv.org/html/2607.15448v1",
  "code_url": "https://yuhochau.github.io/vtap/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.5
 },
 {
  "id": "2607.15422",
  "slug": "robust-silicone-pour-casting-and-sensor-embedding-procedures-for-soft",
  "title": "Robust Silicone Pour Casting and Sensor Embedding Procedures for Soft Robotic Actuators",
  "abstract": "Soft robots are well-suited for applications such as rehabilitation and surgery that require adaptable and safe interaction with their environment. However, the challenges of reproducible and scalable fabrication of soft robots limit their real-world deployment. Various fabrication methods have been introduced, but many are labor-intensive and prone to human error. Therefore, traditional two-part pour casting remains an attractive option. This paper presents procedures for robust, repeatable, and scalable fabrication of soft pneumatic actuators using two-part pour casting. The presented methods prevent internal cavity clogging and ensure air-tight sealing. Additionally, a robust sensor embedding procedure for thin-film flex sensors is presented, which allows for accurate and repeatable data acquisition. Finite Element Modeling (FEM) of the soft actuator is performed to analyze stress and deformation from internal pressure loadings. Pneumatic actuation experiments with PID pressure control are performed. Automated image processing is used to calibrate the embedded flex sensor to bending angle measurements. Staircase and sinusoidal profile actuation experiments validate the performance of the fabricated actuator. Angle response experiments for the staircase input show repeatable performance, and the sinusoidal input shows a small amount of hysteresis consistent with viscoelastic response to pneumatic actuation of soft actuators. Simulated and real-world bending angles show comparable response. These methods provide a repeatable and robust fabrication procedure, validated across two operators and 24 successful fabrications, along with benchmark simulations and experimental testing. These benchmarks will enable more widespread adoption of soft robotics.",
  "published": "2026-07-16",
  "updated": "2026-07-16",
  "year": "2026",
  "authors": [
   "Harshit Thakker",
   "Paul Dela Cruz",
   "Mostafa Mo. Massoud",
   "Jacqueline Libby"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "H. Thakker",
    "id": "2120061454",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "P. D. Cruz",
    "id": "101557912",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "M. Massoud",
    "id": "2197601565",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Jacqueline Libby",
    "id": "2326023415",
    "h_index": 2,
    "papers": 5
   }
  ],
  "comment": "8 pages, 17 figures, to be published in: IEEE RAS/EMBS 11th International Conference on Biomedical Robotics and Biomechatronics (BioRob 2026)",
  "topics": [
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.15422v1",
  "pdf_url": "https://arxiv.org/pdf/2607.15422v1",
  "html_url": "https://arxiv.org/html/2607.15422v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.15330",
  "slug": "xiaomi-robotics-1-scaling-vision-language-action-models-with-over-100k",
  "title": "Xiaomi-Robotics-1: Scaling Vision-Language-Action Models with over 100K Hours of Real-World Trajectories",
  "abstract": "We present Xiaomi-Robotics-1, a foundational vision-language-action (VLA) model capable of (1) following diverse language instructions to perform a wide range of mobile manipulation tasks in unseen environments out-of-the-box, and (2) efficiently adapting to novel downstream tasks with minimal fine-tuning data. We propose a two-stage training recipe consisting of pre-training and post-training. During pre-training, we imbue the model with broad and generalizable action-generation capabilities by training on over 100k hours of real-world manipulation trajectories collected via UMI devices. Crucially, we develop a scalable auto-labeling pipeline that annotates trajectory clips with natural languages describing scene state transitions, providing rich and precise conditioning for action learning. During post-training, we aim to align these capabilities with robot embodiments and imperative instructions that humans naturally use to prompt robots. Extensive experiments demonstrate strong scaling behavior. Xiaomi-Robotics-1 consistently improves with increased data scales and model sizes during pre-training. This scaling behavior directly transfers to post-training, where a stronger pre-training model yields better out-of-the-box real-robot performance in unseen environments. Furthermore, Xiaomi-Robotics-1 serves as a strong robot foundation policy that can be efficiently fine-tuned on complex, dexterous tasks with high data efficiency. Across multiple simulation benchmarks, Xiaomi-Robotics-1 outperforms state-of-the-art methods. Notably, it establishes a new state-of-the-art with a 57.4% success rate on RoboCasa365, surpassing the previous best of 46.6%. Furthermore, it achieves an average score of 20.07 on RoboDojo, significantly outperforming the prior state-of-the-art (13.07). Code and model checkpoints will be released. Project page: https://robotics.xiaomi.com/xiaomi-robotics-1.html",
  "published": "2026-07-16",
  "updated": "2026-07-22",
  "year": "2026",
  "authors": [
   " Xiaomi Robotics Team",
   "Jun Guo",
   "Piaopiao Jin",
   "Jason Li",
   "Peiyan Li",
   "Yingyan Li",
   "Futeng Liu",
   "Wanli Peng",
   "Optimus Qin",
   "Yifei Su",
   "Nan Sun",
   "Qiao Sun",
   "Runze Suo",
   "Heyun Wang",
   "Yunhong Wang",
   "Rujie Wu",
   "Caoyu Xia",
   "Lina Zhang",
   "Jack Zhao",
   "Guoliang Chen",
   "Wenlong Chen",
   "Xinze He",
   "Bin Li",
   "Qing Li",
   "Zhuorong Li",
   "Heng Qu",
   "Wenxuan Song",
   "Diyun Xiang",
   "Yifan Xie",
   "Peiran Xu",
   "Hangjun Ye",
   "Wen Ye",
   "Han Zhao",
   "Quanyun Zhou"
  ],
  "author_count": 34,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 8,
  "influential_citations": 1,
  "tldr": "Xiao-Robotics-1 serves as a strong robot foundation policy that can be efficiently fine-tuned on complex, dexterous tasks with high data efficiency and across multiple simulation benchmarks, Xiaomi-Robotics-1 outperforms state-of-the-art methods.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xiaomi Robotics Team Jun Guo",
    "id": "2453226596",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Piaopiao Jin",
    "id": "2378954331",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Jason Li",
    "id": "2382945637",
    "h_index": 2,
    "papers": 11
   },
   {
    "name": "Peiyan Li",
    "id": "2305635155",
    "h_index": 7,
    "papers": 26
   },
   {
    "name": "Yingyan Li",
    "id": "2306074629",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Futeng Liu",
    "id": "2411090440",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Wanli Peng",
    "id": "2449437929",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Optimus Qin",
    "id": "2451042246",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yifei Su",
    "id": "2332311229",
    "h_index": 7,
    "papers": 17
   },
   {
    "name": "Nan Sun",
    "id": "2322924729",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Qiao Sun",
    "id": "2107458720",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Runze Suo",
    "id": "2346981469",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Heyun Wang",
    "id": "2447881347",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Yunhong Wang",
    "id": "2281412623",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Rujie Wu",
    "id": "2179089110",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "Caoyu Xia",
    "id": "2449704713",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Lina Zhang",
    "id": "2449635810",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Jack Zhao",
    "id": "2449712522",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Guoliang Chen",
    "id": "2448943467",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Wenlong Chen",
    "id": "2395268324",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Xin He",
    "id": "2271984080",
    "h_index": 6,
    "papers": 19
   },
   {
    "name": "Bin Li",
    "id": "2446890741",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Qing Li",
    "id": "2282731856",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Zhuo Li",
    "id": "2451025455",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Hengxu Qu",
    "id": "2211044047",
    "h_index": 5,
    "papers": 15
   },
   {
    "name": "Wenxuan Song",
    "id": "2293142288",
    "h_index": 14,
    "papers": 46
   },
   {
    "name": "Diyun Xiang",
    "id": "2315503913",
    "h_index": 4,
    "papers": 15
   },
   {
    "name": "Yifan Xie",
    "id": "2393037213",
    "h_index": 1,
    "papers": 11
   },
   {
    "name": "Peiran Xu",
    "id": "2240803189",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Hangjun Ye",
    "id": "2384401186",
    "h_index": 7,
    "papers": 34
   },
   {
    "name": "Wen Ye",
    "id": "2256992266",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Han Zhao",
    "id": "2266256598",
    "h_index": 16,
    "papers": 36
   },
   {
    "name": "Quanyun Zhou",
    "id": "2355359215",
    "h_index": 1,
    "papers": 4
   }
  ],
  "comment": "Project page: https://robotics.xiaomi.com/xiaomi-robotics-1.html",
  "topics": [
   "vla",
   "dexterous-manipulation",
   "sim2real",
   "navigation",
   "foundation-pretraining",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.15330v2",
  "pdf_url": "https://arxiv.org/pdf/2607.15330v2",
  "html_url": "https://arxiv.org/html/2607.15330v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.95
 },
 {
  "id": "2607.15325",
  "slug": "interactive-3d-tangible-display-with-a-high-speed-stiffness-variable-j",
  "title": "Interactive 3D Tangible Display with a High-Speed Stiffness-Variable Jamming Module",
  "abstract": "Multisensory integration, particularly through visual and tactile feedback, plays a crucial role in enhancing audience engagement with artworks. Although recent research has increasingly explored tactile experiences in art, existing systems often lack real-time variable stiffness modulation and depend on bulky mechanical infrastructures. In this work, we propose a novel tangible display based on a magnetic jamming mechanism, enabling real-time, low-noise, and low-voltage stiffness modulation integrated into traditional sculptural artworks. Our system combines visual motion and dynamic tactile feedback within a compact standalone module, allowing audiences to interactively experience variations in the rigidity and form of features such as those found in the traditional Korean mask Hahoetal. This approach offers a new paradigm for interactive art, enabling more immersive, multisensory engagement through the fusion of cultural artifacts and modern technology. Our project page is available at https://cold-young.github.io/jamming_tangible/.",
  "published": "2026-07-16",
  "updated": "2026-07-16",
  "year": "2026",
  "authors": [
   "Chanyoung Ahn",
   "Jaesung Lee",
   "Donhyun Hwang"
  ],
  "author_count": 3,
  "categories": [
   "cs.HC",
   "cs.RO"
  ],
  "primary_category": "cs.HC",
  "venue": "ICRA 2025",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Chanyoung Ahn",
    "id": "2391070835",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Jaesung Lee",
    "id": "2450186380",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Donhyun Hwang",
    "id": "2451041297",
    "h_index": 0,
    "papers": 1
   }
  ],
  "comment": "8 pages, 7 figures. Exhibited at the ICRA 2025 Arts in Robotics",
  "topics": [
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.15325v1",
  "pdf_url": "https://arxiv.org/pdf/2607.15325v1",
  "html_url": "https://arxiv.org/html/2607.15325v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2607.15275",
  "slug": "robottt-context-scaling-for-robot-policies",
  "title": "RoboTTT: Context Scaling for Robot Policies",
  "abstract": "Recent robot foundation models operate with single-step or short-history visuomotor context. We introduce Test-Time-Training Robot Policies (RoboTTT), a robot model and training recipe that scale visuomotor context to 8K timesteps, three orders of magnitude beyond state-of-the-art policies, without growing inference latency. At this context length, we unlock new robot capabilities: one-shot in-context imitation from human video demonstrations, on-the-fly policy improvement, robustness to perturbations, and stronger performance on multi-stage, long-horizon tasks. We also observe, for the first time, steady gains in closed-loop performance as pretraining context length scales. At its core, RoboTTT integrates Test-Time Training into robot foundation models such as Vision-Language-Action policies, yielding a sequence model whose recurrent state consists of fast weights, parameters updated by gradient descent during both training and inference, compressing histories into weight space and retrieving contextual information for long-context conditioning. To scale training context length, the recipe combines sequence action forcing with truncated backpropagation through time. On challenging real-robot manipulation tasks, RoboTTT improves overall performance by 87% over the single-step context baseline and fully completes a five-minute, ten-stage assembly task, which no baseline ever does. RoboTTT trained with 8K-timestep context outperforms the same model pretrained with 1K timesteps by 62%, suggesting context length as a new scaling axis for robot foundation models. Videos are available at https://research.nvidia.com/labs/gear/robottt/",
  "published": "2026-07-16",
  "updated": "2026-07-16",
  "year": "2026",
  "authors": [
   "Yunfan Jiang",
   "Yevgen Chebotar",
   "Ruijie Zheng",
   "Fengyuan Hu",
   "Yunhao Ge",
   "Jimmy Wu",
   "Tianyuan Dai",
   "Scott Reed",
   "Li Fei-Fei",
   "Yuke Zhu",
   "Linxi \"Jim\" Fan"
  ],
  "author_count": 11,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 3,
  "influential_citations": 0,
  "tldr": "RoboTTT, a robot model and training recipe that scale visuomotor context to 8K timesteps, three orders of magnitude beyond state-of-the-art policies, without growing inference latency, unlocks new robot capabilities: one-shot in-context imitation from human video demonstrations, on-the-fly policy improvement, robustness to perturbations, and stronger performance on multi-stage, long-horizon tasks.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yunfan Jiang",
    "id": "2171112793",
    "h_index": 11,
    "papers": 16
   },
   {
    "name": "Yevgen Chebo-tar",
    "id": "2334885136",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Ruijie Zheng",
    "id": "2345931905",
    "h_index": 9,
    "papers": 15
   },
   {
    "name": "Fengyuan Hu",
    "id": "2352992614",
    "h_index": 8,
    "papers": 11
   },
   {
    "name": "Yunhao Ge",
    "id": "2299104362",
    "h_index": 11,
    "papers": 20
   },
   {
    "name": "Jimmy Wu",
    "id": "2390404352",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Tianyuan Dai",
    "id": "37304639",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Scott Reed",
    "id": "2344615904",
    "h_index": 8,
    "papers": 9
   },
   {
    "name": "Fei-Fei Li",
    "id": "2330589126",
    "h_index": 3,
    "papers": 15
   },
   {
    "name": "Yuke Zhu",
    "id": "2258068214",
    "h_index": 13,
    "papers": 21
   },
   {
    "name": "LinxiJimFan",
    "id": "2350861618",
    "h_index": 10,
    "papers": 15
   }
  ],
  "comment": "Project website: http://research.nvidia.com/labs/gear/robottt/",
  "topics": [
   "vla",
   "egocentric-data",
   "foundation-pretraining",
   "safety-eval"
  ],
  "orgs": [
   "NVIDIA"
  ],
  "abs_url": "https://arxiv.org/abs/2607.15275v1",
  "pdf_url": "https://arxiv.org/pdf/2607.15275v1",
  "html_url": "https://arxiv.org/html/2607.15275v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.1
 },
 {
  "id": "2607.15172",
  "slug": "ahead-anticipatory-hand-driven-teleoperation-via-human-intent-predicti",
  "title": "AHEAD: Anticipatory Hand-Driven Teleoperation via Human Intent Prediction",
  "abstract": "Direct hand-driven teleoperation maps an operator's hand motion to robot end-effector commands at every frame, enabling precise control, but it requires constant monitoring and correction during approach, grasp, and placement, which can be slow and fatiguing. For repetitive pick-and-place tasks, supervisory (goal-based) teleoperation simplifies this process: the operator specifies goals/waypoints, and the robot executes the motion using planning algorithms. Yet, this introduces latency, as the robot must wait for the next command before it can plan and act. \"How can we reduce robot reaction time while lowering operator workload?\" To tackle this question, we present AHEAD, a real-time VR teleoperation system that anticipates operator intent to enable proactive, hand-driven control. In a digital twin, the operator performs pick-and-place naturally, using hand motion to convey high-level commands rather than a continuous robot trajectory. AHEAD processes a short window of 3D hand and head signals together with scene context through an attention-based classifier to predict the intended grasp object and placement slot. A state machine converts intent predictions into stable robot goals, enabling early motion while remaining stable under noisy predictions and corrective hand movements. AHEAD's intent prediction module achieves Top1 accuracy: 76% for grasp objects and 76% for target slots. Moreover, our user study shows AHEAD reduces robot reaction latency by 0.6 s (object) and 1.4 s (slot) relative to baselines. Participants also reported lower operator load, indicating faster robot responses while maintaining low operator effort in practice.",
  "published": "2026-07-16",
  "updated": "2026-07-16",
  "year": "2026",
  "authors": [
   "Seok Joon Kim",
   "Junho Lee",
   "Federica Spinola",
   "Taein Kwon",
   "Mohsen Moghaddam"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "AHEAD, a real-time VR teleoperation system that anticipates operator intent to enable proactive, hand-driven control, and reduces robot reaction latency by 0.6 s and 1.4 s relative to baselines, respectively.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "S. Kim",
    "id": "2333046941",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Junho Lee",
    "id": "2449443858",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Federica Spinola",
    "id": "2134737584",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Taein Kwon",
    "id": "8197167",
    "h_index": 6,
    "papers": 17
   },
   {
    "name": "Mohsen Moghaddam",
    "id": "2333648599",
    "h_index": 1,
    "papers": 7
   }
  ],
  "comment": "Accepted to IROS2026, 8 pages, 6 figures",
  "topics": [
   "dexterous-manipulation",
   "data-teleop",
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.15172v1",
  "pdf_url": "https://arxiv.org/pdf/2607.15172v1",
  "html_url": "https://arxiv.org/html/2607.15172v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.15163",
  "slug": "scaling-behavior-foundation-model-for-humanoid-robots",
  "title": "Scaling Behavior Foundation Model for Humanoid Robots",
  "abstract": "Humanoid control requires natural whole-body coordination, precise real-time responses to control signals, and robust generalization across diverse environmental contexts, making it a cornerstone for generalist embodied agents. Behavior Foundation Models (BFMs) have recently emerged as a promising solution to address these challenges by leveraging large-scale behavioral data to achieve superior expressiveness, versatility and generalization. However, despite growing interest in scaling BFMs to further improve their capabilities, it remains unclear how key factors, including the learning paradigm, behavioral data and model architecture should be coordinated to enable effective scaling. In this work, we revisit the scaling recipe for BFMs and demonstrate that substantial performance gains can be achieved through the coordination of three core components: 1) the learning paradigm of motion tracking that reformulates diverse humanoid control problems as the reproduction of integrated whole-body behaviors in the global frame; 2) the strategic synergy between on-policy rollout quantity and reference motion diversity; and 3) the expressive and scalable model architecture termed Humanoid Transformer that facilitates the natural emergence of structured behavioral representations. Through extensive experiments in both simulation and real-world deployment, we demonstrate that our approach yields significant improvements in control fidelity and task generalization, reducing Mean Per-Keypoint Position Error (MPKPE) on the test set by over 10% in local mode and 82% in global mode compared with existing humanoid controllers. These results establish BFM as a principled and effective foundation for scalable and general-purpose humanoid control.",
  "published": "2026-07-16",
  "updated": "2026-07-16",
  "year": "2026",
  "authors": [
   "Weishuai Zeng",
   "Kangning Yin",
   "Xiaojie Niu",
   "Shunlin Lu",
   "Weixiang Zhong",
   "Jiahe Chen",
   "Feiyu Jia",
   "Xiao Chen",
   "Zirui Wang",
   "Furui Xu",
   "Ming Zhou",
   "Kailin Li",
   "Weinan Zhang",
   "He Wang",
   "Li Yi",
   "Dahua Lin",
   "Jiangmiao Pang",
   "Jingbo Wang"
  ],
  "author_count": 18,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "This work revisits the scaling recipe for BFMs and demonstrates that substantial performance gains can be achieved through the coordination of three core components: the learning paradigm of motion tracking that reformulates diverse humanoid control problems as the reproduction of integrated whole-body behaviors in the global frame.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Weishuai Zeng",
    "id": "2315923491",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Kangning Yin",
    "id": "2289841621",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Xiaojie Niu",
    "id": "2361881548",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Shunlin Lu",
    "id": "2351025318",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Weixiang Zhong",
    "id": "2445694003",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Jiahe Chen",
    "id": "2284724608",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Feiyu Jia",
    "id": "2345925355",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Xiao Chen",
    "id": "2276457168",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Zirui Wang",
    "id": "2257550085",
    "h_index": 21,
    "papers": 78
   },
   {
    "name": "Furui Xu",
    "id": "2392906627",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Ming Zhou",
    "id": "2152174952",
    "h_index": 9,
    "papers": 19
   },
   {
    "name": "Kailin Li",
    "id": "2023790905",
    "h_index": 9,
    "papers": 15
   },
   {
    "name": "Weinan Zhang",
    "id": "2344034124",
    "h_index": 5,
    "papers": 16
   },
   {
    "name": "He Wang",
    "id": "2336953381",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Li Yi",
    "id": "2279808630",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Dahua Lin",
    "id": "2237091231",
    "h_index": 16,
    "papers": 31
   },
   {
    "name": "Jiangmiao Pang",
    "id": "2405891293",
    "h_index": 2,
    "papers": 17
   },
   {
    "name": "Jingbo Wang",
    "id": "2363513787",
    "h_index": 5,
    "papers": 13
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.15163v1",
  "pdf_url": "https://arxiv.org/pdf/2607.15163v1",
  "html_url": "https://arxiv.org/html/2607.15163v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2607.15156",
  "slug": "assessing-physical-frailty-and-fall-risk-indicators-with-social-robots",
  "title": "Assessing Physical Frailty and Fall-Risk Indicators with Social Robots: An in situ Evaluation with Older Adults",
  "abstract": "Frailty assessments are crucial to evaluate the risk of adverse events and the health and social care needs of older adults, yet their administration remains resource-intensive and typically relies on coarse clinical outcomes, such as task completion times, which may overlook biomechanical indicators of functional decline. To address this, we present a robotic framework that guides older adults through standardised frailty and fall-risk tests while capturing clinical scores and additional frailty-related metrics, offering a deeper insight into a user's condition. The system uses a Behaviour Tree architecture that coordinates perception, decision-making, interaction, and measurement modules. Using vision-based skeleton tracking, the robot evaluates established clinical tests, including the Short Physical Performance Battery (SPPB) and the Timed Up and Go (TUG). The framework was co-designed with healthcare professionals and evaluated in situ during six months in a rehabilitation centre's research lab with N=81 older adults. Robot-derived measurements were compared against therapist assessments and clinical reference instruments, including a gait analysis walkway and an inertial measurement unit (IMU). Results showed excellent agreement for most test completion times and gait-related parameters ($ICC > 0.9$). And, substantial agreement for the overall SPPB score comparing the robot and the therapist ($k = 0.67$) and moderate agreement comparing the robot and the IMU ($k=0.55$). The findings highlight that social robots can provide reliable and objective frailty assessments in healthcare settings while enabling the collection of relevant mobility indicators beyond conventional outcomes.",
  "published": "2026-07-16",
  "updated": "2026-07-16",
  "year": "2026",
  "authors": [
   "Aniol Civit",
   "Antonio Andriella",
   "Alba Mart\u00ednez",
   "Joan Ars",
   "Aida Ribera",
   "Cristian Barru\u00e9",
   "Guillem Aleny\u00e0"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A robotic framework that guides older adults through standardised frailty and fall-risk tests while capturing clinical scores and additional frailty-related metrics, offering a deeper insight into a user's condition is presented.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Aniol Civit",
    "id": "2060204277",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Antonio Andriella",
    "id": "51115548",
    "h_index": 11,
    "papers": 60
   },
   {
    "name": "Alba Mart\u00ednez",
    "id": "2296737877",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "J. Ars",
    "id": "1742203411",
    "h_index": 9,
    "papers": 20
   },
   {
    "name": "Aida Ribera",
    "id": "2280739319",
    "h_index": 2,
    "papers": 11
   },
   {
    "name": "Cristian Barru\u00e9",
    "id": "2282159381",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Guillem Aleny\u00e0",
    "id": "2268206132",
    "h_index": 3,
    "papers": 29
   }
  ],
  "comment": "This work has been submitted to the IEEE for possible publication",
  "topics": [
   "hardware-codesign",
   "hri",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.15156v1",
  "pdf_url": "https://arxiv.org/pdf/2607.15156v1",
  "html_url": "https://arxiv.org/html/2607.15156v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.15129",
  "slug": "catch-throw-repeat-planning-for-human-robot-partner-juggling",
  "title": "Catch, Throw, Repeat: Planning for Human-Robot Partner Juggling",
  "abstract": "Dynamic object exchange between humans and robots remains a challenging problem due to uncertainty in perception, timing, and contact-rich interaction. Human-robot juggling represents a particularly demanding instance of this problem, requiring precise real-time coordination, predictive motion planning with feedback control, and robustness to variability in human motion. Enabling such skills is of interest for advancing physical human-robot interaction and shared autonomy. We present a real-time planning and control architecture for human-robot partner juggling that enables a robot to reliably catch and throw balls in synchronized multi-ball patterns with a human partner. The system integrates predictive ball tracking, adaptive online trajectory optimization using a multiple-shooting formulation, and a state-machine-based coordination logic to enable synchronized multi-ball human-robot partner juggling. In a user study with 8 participants of varying juggling skill from beginner to expert, we demonstrate that our system can achieve three-ball cascades shared between the robot and the human. All participants exceeded previously reported best-case results within a 10-minute test session, with one participant extending the previous record for shared three-ball cascade juggling fivefold to 20 consecutive robot catches, and another participant achieving a 100% success rate with 40 consecutive catches in a single-ball catch-and-return setting. Video documentation can be found at https://kai-ploeger.com/partner-juggling",
  "published": "2026-07-16",
  "updated": "2026-07-16",
  "year": "2026",
  "authors": [
   "Jonathan Rainer Lippert",
   "Kai Ploeger",
   "Abir Chowdhury",
   "Hermann M\u00fcller",
   "Jan Peters",
   "Alap Kshirsagar"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.HC",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A real-time planning and control architecture for human-robot partner juggling that enables a robot to reliably catch and throw balls in synchronized multi-ball patterns with a human partner.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jonathan Rainer Lippert",
    "id": "2309871471",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Kai Ploeger",
    "id": "51177013",
    "h_index": 4,
    "papers": 14
   },
   {
    "name": "Abir Chowdhury",
    "id": "1434553823",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "H. M\u00fcller",
    "id": "2238577689",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Jan Peters",
    "id": "2329405186",
    "h_index": 1,
    "papers": 11
   },
   {
    "name": "Alap Kshirsagar",
    "id": "2145217956",
    "h_index": 8,
    "papers": 39
   }
  ],
  "comment": "Accepted at the 2026 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS 2026)",
  "topics": [
   "tactile",
   "hri",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.15129v1",
  "pdf_url": "https://arxiv.org/pdf/2607.15129v1",
  "html_url": "https://arxiv.org/html/2607.15129v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2607.15065",
  "slug": "driftworld-fast-world-modeling-through-drifting",
  "title": "DriftWorld: Fast World Modeling through Drifting",
  "abstract": "Predictive world models enable robots to plan by imagining the outcomes of their actions, but their value for control hinges on generating many rollouts quickly. This creates a bottleneck for diffusion-based world models: multistep sampling makes each rollout expensive, limiting large-scale action search at inference time. We introduce DriftWorld, an action-conditioned world model based on drifting generative models. Rather than denoising iteratively at inference, DriftWorld learns an action-conditioned drift during training, allowing it to generate future frames from the current observation and a candidate action sequence in a single forward pass at 30+ fps, which is 17x faster on average than diffusion based baselines. We evaluate DriftWorld on standard vision-based robotic manipulation benchmarks, including Bridge-V2, RT-1, Language Table, Push-T, and Robomimic. By producing rollouts that are both accurate and fast, DriftWorld achieves state-of-the-art decision-making performance with far less inference time than diffusion-based world model baselines. Beyond online control, DriftWorld can also serve as an offline simulator for ranking real-world robot policies, with rollout-based scores correlating with ground truth at up to 0.99. These results show that drifting models are a strong fit for robot world modeling, where fast, high-quality imagination directly supports planning and policy evaluation.",
  "published": "2026-07-16",
  "updated": "2026-08-03",
  "year": "2026",
  "authors": [
   "Susie Lu",
   "Haonan Chen",
   "Weirui Ye",
   "Yilun Du"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Susie Lu",
    "id": "2375339161",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Haonan Chen",
    "id": "2309175413",
    "h_index": 6,
    "papers": 19
   },
   {
    "name": "Weirui Ye",
    "id": "83546634",
    "h_index": 12,
    "papers": 22
   },
   {
    "name": "Yilun Du",
    "id": "2383299857",
    "h_index": 2,
    "papers": 14
   }
  ],
  "comment": "Website at https://susie-lu.github.io/driftworld/",
  "topics": [
   "world-models",
   "sim2real"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.15065v2",
  "pdf_url": "https://arxiv.org/pdf/2607.15065v2",
  "html_url": "https://arxiv.org/html/2607.15065v2",
  "code_url": "https://susie-lu.github.io/driftworld/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2607.15058",
  "slug": "sufleca-scaling-up-feature-learning-for-cad-to-image-alignment",
  "title": "SUFLECA: Scaling Up Feature Learning for CAD-to-image Alignment",
  "abstract": "CAD-to-image alignment aims to estimate an object's 9D pose (rotation, translation, and anisotropic scale) from a single RGB image, enabling applications in robotics and augmented reality. Recent zero-shot methods use visual foundation models to match image regions to CAD models, yet typically their correspondences are appearance-driven and degrade under occlusion or sim-to-real domain shift. To address these limitations, we introduce SUFLECA (Scaling Up Feature LEarning for CAD Alignment), a weakly-supervised framework for zero-shot CAD alignment with two key contributions. First, SUFLECA scales up geometry-grounded feature learning from pretrained visual representations through Normalized Object Coordinates (NOCs) supervision on 674K images spanning 12 real and synthetic datasets, learning compact geometry-aware features that generalize across domains. Second, we propose a geometrically consistent matching algorithm that establishes reliable one-to-one CAD-to-image correspondences. Together, these contributions enable accurate, sub-second alignment per object instance without iterative pose refinement. On ScanNet25k, SUFLECA achieves 33.4%/42.3% category/instance accuracy, outperforming, with a smaller computational footprint, the strongest zero-shot baseline by 10.3/12.2 percentage points and, for the first time on this benchmark, even surpassing fully supervised methods. Code is available at: https://github.com/snt-arg/SUFLECA",
  "published": "2026-07-16",
  "updated": "2026-07-16",
  "year": "2026",
  "authors": [
   "Saad Ejaz",
   "Miguel Fernandez-Cortizas",
   "Javier Civera",
   "Holger Voos",
   "Jose Luis Sanchez-Lopez"
  ],
  "author_count": 5,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "SUFLECA (Scaling Up Feature LEarning for CAD Alignment), a weakly-supervised framework for zero-shot CAD alignment with two key contributions, and a geometrically consistent matching algorithm that establishes reliable one-to-one CAD-to-image correspondences.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Saad Ejaz",
    "id": "2320402847",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Miguel Fern\u00e1ndez-Cortizas",
    "id": "2151611669",
    "h_index": 8,
    "papers": 31
   },
   {
    "name": "Javier Civera",
    "id": "2273546248",
    "h_index": 5,
    "papers": 16
   },
   {
    "name": "Holger Voos",
    "id": "2060086740",
    "h_index": 12,
    "papers": 66
   },
   {
    "name": "J. L. S\u00e1nchez-L\u00f3pez",
    "id": "2249758142",
    "h_index": 7,
    "papers": 42
   }
  ],
  "comment": "",
  "topics": [
   "sim2real",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.15058v1",
  "pdf_url": "https://arxiv.org/pdf/2607.15058v1",
  "html_url": "https://arxiv.org/html/2607.15058v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.15016",
  "slug": "risk-aware-belief-control-barrier-functions-over-random-finite-sets",
  "title": "Risk-Aware Belief Control Barrier Functions over Random Finite Sets",
  "abstract": "Ensuring robot safety in unknown, dynamic environments is a fundamental requirement. It involves inferring the states of an unknown and time-varying number of moving objects from noisy, incomplete measurements. We address safe control under the induced multi-object state uncertainty with a risk-aware belief control barrier function (BCBF) framework. The uncertainty is captured by a random finite set (RFS) belief, estimated by a sequential Monte Carlo probability hypothesis density (SMC-PHD) filter that represents it with a set of particles. Building directly on these particles, we construct a nonsmooth BCBF, establish forward invariance of the safe set under continuous prediction, and derive an explicit condition under which discrete updates preserve safety. Simulation and real-world underwater experiments demonstrate the effectiveness and efficiency of the proposed approach.",
  "published": "2026-07-16",
  "updated": "2026-07-16",
  "year": "2026",
  "authors": [
   "Shaohang Han",
   "Gang Chen",
   "Yixi Cai",
   "Ignacio Torroba",
   "Ivan Stenius",
   "Patric Jensfelt",
   "Javier Alonso-Mora",
   "Jana Tumova"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work addresses safe control under the induced multi-object state uncertainty with a risk-aware belief control barrier function (BCBF) framework and establishes forward invariance of the safe set under continuous prediction, and derives an explicit condition under which discrete updates preserve safety.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Shaohang Han",
    "id": "2321715235",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Gang Chen",
    "id": "2298044561",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Yixi Cai",
    "id": "50204612",
    "h_index": 15,
    "papers": 37
   },
   {
    "name": "I. Torroba",
    "id": "134120611",
    "h_index": 9,
    "papers": 23
   },
   {
    "name": "Ivan Stenius",
    "id": "95137588",
    "h_index": 17,
    "papers": 55
   },
   {
    "name": "P. Jensfelt",
    "id": "1770066",
    "h_index": 55,
    "papers": 238
   },
   {
    "name": "Javier Alonso-Mora",
    "id": "2313388475",
    "h_index": 12,
    "papers": 42
   },
   {
    "name": "Jana Tumova",
    "id": "2326992384",
    "h_index": 2,
    "papers": 11
   }
  ],
  "comment": "",
  "topics": [
   "rl-control",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.15016v1",
  "pdf_url": "https://arxiv.org/pdf/2607.15016v1",
  "html_url": "https://arxiv.org/html/2607.15016v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.14997",
  "slug": "aeroact-action-centered-world-action-models-for-language-conditioned-q",
  "title": "AeroAct: Action-Centered World-Action Models for Language-Conditioned Quadrotor Flight",
  "abstract": "Language-conditioned quadrotor flight requires a policy to ground semantic goals, anticipate the visual consequences of ego-motion, and output control references that remain smooth and dynamically executable under rapidly changing first-person views. Existing aerial vision-language navigation and vision-language-action methods commonly use discrete actions, high-level waypoints, or instantaneous velocity commands, which provide limited supervision about how flight actions change future observations. We present AeroAct, an action-centered world-action model (WAM) for quadrotor navigation. To the best of our knowledge, AeroAct is the first WAM instantiated and demonstrated for real-world aerial flight. The model adapts a pretrained video diffusion Transformer to predict local trajectory-action chunks from egocentric visual history, proprioception, and language. Future first-person frames are used during training as dense consequence supervision, while deployment directly decodes actions without generating future video. To obtain aligned visual, state, language, and dynamically feasible action data, we build a DiffAero-based pipeline with complementary Isaac Lab and 3D Gaussian splatting renderers. We further introduce a low-cost handheld collection device that couples camera observations with motion estimates to recreate flight-like egocentric trajectories, and a self-guidance procedure that improves temporal consistency across overlapping trajectory chunks. Closed-loop simulation and real-world experiments show that temporal visual context improves target tracking and object-search performance, and that WAM-based policies can be executed on a physical quadrotor.",
  "published": "2026-07-16",
  "updated": "2026-07-16",
  "year": "2026",
  "authors": [
   "Xinhong Zhang",
   "Qiyuan Zhu",
   "Yubo Huang",
   "Haolin Chen",
   "Runqing Wang",
   "Yuhao Mo",
   "Zhongxin Chen",
   "Yu Hu",
   "Xinjiang Wang",
   "Jian Sun",
   "Gang Wang"
  ],
  "author_count": 11,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "AeroAct is the first WAM instantiated and demonstrated for real-world aerial flight, and closed-loop simulation and real-world experiments show that temporal visual context improves target tracking and object-search performance, and that WAM-based policies can be executed on a physical quadrotor.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xinhong Zhang",
    "id": "2108029654",
    "h_index": 4,
    "papers": 16
   },
   {
    "name": "Qiyuan Zhu",
    "id": "2450228090",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yubo Huang",
    "id": "2293777773",
    "h_index": 3,
    "papers": 16
   },
   {
    "name": "Haolin Chen",
    "id": "2449187121",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Runqing Wang",
    "id": "2345407885",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Yuhao Mo",
    "id": "2392258396",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Zhongxin Chen",
    "id": "2449462144",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yu Hu",
    "id": "2311440667",
    "h_index": 3,
    "papers": 15
   },
   {
    "name": "Xinjiang Wang",
    "id": "2457967098",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Jian Sun",
    "id": "2152148509",
    "h_index": 17,
    "papers": 84
   },
   {
    "name": "Gang Wang",
    "id": "2258688044",
    "h_index": 8,
    "papers": 53
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "egocentric-data",
   "spatial-3d",
   "navigation",
   "video-generation"
  ],
  "orgs": [
   "NVIDIA"
  ],
  "abs_url": "https://arxiv.org/abs/2607.14997v1",
  "pdf_url": "https://arxiv.org/pdf/2607.14997v1",
  "html_url": "https://arxiv.org/html/2607.14997v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2607.14853",
  "slug": "modeling-and-validation-of-quality-of-control-for-edge-offloaded-colla",
  "title": "Modeling and Validation of Quality of Control for Edge-Offloaded Collaborative Navigation",
  "abstract": "Collaborative control in complex environments is severely challenged by stochastic wireless delay and reliability variations, which can degrade navigation, tracking, and collision avoidance. These network-induced uncertainties complicate the maintenance of energy efficiency during collaborative tasks, and can potentially lead to over-provisioning of resources. In this paper, for a navigation setup with dynamic collision avoidance, we address this challenge by expanding the quality of control (QoC) framework from prior works to practical robotic models. Our approach (i) models end-to-end network effects on closed-loop performance, (ii) systematically explores the impact of various control parameters dictating robotic motion on network latency-reliability (iii) validates these models through experiments on a private 5G testbed across varying delay, reliability and control configurations. Our analysis indicates the optimal control-communication co-design operating regimes for practical robots and also compares the QoC performance of standard ROS~2 quality of service (QoS) policies under real-world conditions and showing how RELIABLE QoS offers 51.5% better QoC than BEST-EFFORT under certain experimental settings.",
  "published": "2026-07-16",
  "updated": "2026-07-16",
  "year": "2026",
  "authors": [
   "Neelabhro Roy",
   "Mikael Hammarling",
   "Victor Nan Fernandez-Ayala",
   "Gourav Prateek Sharma",
   "Mani H. Dhullipalla",
   "Dimos V. Dimarogonas",
   "James Gross"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "cs.MA",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This analysis indicates the optimal control-communication co-design operating regimes for practical robots and also compares the QoC performance of standard ROS~2 quality of service (QoS) policies under real-world conditions and shows how RELIABLE QoS offers 51.5% better QoC than BEST-EFFORT under certain experimental settings.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Neelabhro Roy",
    "id": "1579757578",
    "h_index": 4,
    "papers": 15
   },
   {
    "name": "Mikael Hammarling",
    "id": "2450184165",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Victor Nan Fernandez-Ayala",
    "id": "2221123414",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Gourav Prateek Sharma",
    "id": "2264381366",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Mani H. Dhullipalla",
    "id": "35725924",
    "h_index": 5,
    "papers": 20
   },
   {
    "name": "Dimos V. Dimarogonas",
    "id": "1722151",
    "h_index": 65,
    "papers": 593
   },
   {
    "name": "James Gross",
    "id": "145206802",
    "h_index": 5,
    "papers": 17
   }
  ],
  "comment": "Accpeted in IEEE VTC-Fall 2026",
  "topics": [
   "navigation",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.14853v1",
  "pdf_url": "https://arxiv.org/pdf/2607.14853v1",
  "html_url": "https://arxiv.org/html/2607.14853v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.14842",
  "slug": "kinefuse-kinematic-aware-haptic-fusion-for-in-hand-occluded-object-pos",
  "title": "KineFuse: Kinematic-Aware Haptic Fusion for In-Hand Occluded-Object Pose Tracking",
  "abstract": "Dexterous in-hand manipulation requires continuous 6D pose tracking, yet the manipulating fingers inevitably occlude the object from the camera. We study how to structure the sparse haptic signals already available on multi-fingered hands, including proprioception, proximal force/torque, and binary contact, to complement a pretrained visual pose tracker under occlusion. We propose a kinematic-aware finger-level encoder and systematically compare it against four alternative designs through three levels of evaluation: per-frame refinement, sequential open-loop tracking, and closed-loop manipulation. Our experiments reveal that (i) per-frame evaluation cannot distinguish encoder quality, while sequential tracking amplifies architectural differences by up to 15 times; (ii) the structured encoder learns task-specific cross-modal gating, using vision exclusively for translation and dedicating one attention head to haptics for rotation, without explicit supervision; and (iii) compact finger-level tokenization with 4 tokens outperforms both flat fusion and joint-level representations, which suppress vision through norm dominance. We validate that improved tracking yields higher success in a downstream reorientation task and provide qualitative real-world demonstrations. Our project page is available at https://cold-young.github.io/kine-fuse/.",
  "published": "2026-07-16",
  "updated": "2026-07-16",
  "year": "2026",
  "authors": [
   "Chanyoung Ahn",
   "Jaesung Lee",
   "Sungwoo Park",
   "Donghyun Hwang"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper studies how to structure the sparse haptic signals already available on multi-fingered hands, including proprioception, proximal force/torque, and binary contact, to complement a pretrained visual pose tracker under occlusion to propose a kinematic-aware finger-level encoder.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Chanyoung Ahn",
    "id": "2391070835",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Jaesung Lee",
    "id": "2450186380",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Sungwoo Park",
    "id": "2221104563",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Donghyun Hwang",
    "id": "2391070802",
    "h_index": 0,
    "papers": 3
   }
  ],
  "comment": "8 pages, 10 figures. Accepted for presentation at the 2026 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS 2026)",
  "topics": [
   "dexterous-manipulation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.14842v1",
  "pdf_url": "https://arxiv.org/pdf/2607.14842v1",
  "html_url": "https://arxiv.org/html/2607.14842v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2607.14826",
  "slug": "interventional-causal-circuits-for-safe-robot-action-testing-and-failu",
  "title": "Interventional Causal Circuits for Safe Robot Action Testing and Failure Recovery",
  "abstract": "Safe physical AI for robot actions are required not only likely to succeed but tested to be safe before execution. In practice, however, formal testing of motion parameters is computationally expensive, and the cost scales poorly with the dimensionality of the action space. When a proposed action is rejected by a tester, the naive response is to resample blindly until a passing candidate is found. This is wasteful, uninformative, and offers no convergence. We argue that rejection should instead trigger causal diagnosis: a principled identification of which action parameter caused the failure and what corrective value maximises the probability of passing testing under the interventional probability distribution. We propose a closed-loop framework that couples a Joint Probability Tree (JPT) with a Causal Circuit derived from a Marginal-Deterministic Variable Tree, enabling exact polytime computation without retraining, or additional data collection. The framework validates tractability of all interventional queries before the robot begins operating, and out-of-support candidates are detected and excluded from correction automatically. We perform experiments in a ROS2 simulation environment, and the framework demonstrates complementary roles across quality of distribution: under a high-quality JPT, the Causal Circuit reduces failed attempts by 10.3% and under a degraded JPT, it reduces total failed attempts by 37%. Every rejected plan produces a structured, interpretable causal report naming the primary cause variable, its observed value, and the recommended corrective region, supporting operator oversight and autonomous recovery without a separately trained failure model.",
  "published": "2026-07-16",
  "updated": "2026-07-16",
  "year": "2026",
  "authors": [
   "Naren Vasantakumaar",
   "Tom Schierenbeck",
   "Michael Beetz"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A closed-loop framework that couples a Joint Probability Tree with a Causal Circuit derived from a Marginal-Deterministic Variable Tree is proposed, enabling exact polytime computation without retraining, or additional data collection.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Naren Vasantakumaar",
    "id": "2405889949",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Tom Schierenbeck",
    "id": "2205660038",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Michael Beetz",
    "id": "2287786834",
    "h_index": 4,
    "papers": 11
   }
  ],
  "comment": "1st IJCAI Workshop on Safe Physical AI IJCAI 2026 workshop on Safe Physical AI",
  "topics": [
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.14826v1",
  "pdf_url": "https://arxiv.org/pdf/2607.14826v1",
  "html_url": "https://arxiv.org/html/2607.14826v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.14728",
  "slug": "vq-touch-a-data-efficient-tactile-generation-framework-across-sensors",
  "title": "VQ-Touch: A Data-Efficient Tactile Generation Framework Across Sensors and Scenarios",
  "abstract": "Tactile image generation significantly reduces the dependency on expensive and wear-prone sensors by synthesizing high-fidelity tactile data, offering an efficient solution for tactile information acquisition in robotic perception and human-machine interaction systems. However, existing methods depend on large-scale, diverse datasets from specific sensors and lack efficient data utilization and robust generalization capabilities, struggling in vision-limited environments. To address this, we introduce VQ-Touch, a tactile generation framework that supports both cross-sensor and multi-scenario applications. Specifically, to efficiently extract complex deformation and texture features from the data, we propose DM-VQGAN, an effective tactile representation learner. Furthermore, we introduce a discrete diffusion decoder with a unified conditioning interface, supporting multimodal generation tasks such as images and labels, and enhances the model's generalization capability through few-shot mixed training, thus achieving compatibility with current mainstream sensors and their variants. Experiments show that VQ-Touch surpasses state-of-the-art methods in multiple tasks.",
  "published": "2026-07-16",
  "updated": "2026-07-16",
  "year": "2026",
  "authors": [
   "Kailin Lyu",
   "Long Xiao",
   "Jianing Zeng",
   "Di Wu",
   "Lin Shu",
   "Jie Hao"
  ],
  "author_count": 6,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "To efficiently extract complex deformation and texture features from the data, this work proposes DM-VQGAN, an effective tactile representation learner, and introduces a discrete diffusion decoder with a unified conditioning interface that enhances the model's generalization capability through few-shot mixed training, thus achieving compatibility with current mainstream sensors and their variants.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Kailin Lyu",
    "id": "2391708903",
    "h_index": 2,
    "papers": 18
   },
   {
    "name": "Long Xiao",
    "id": "2370486657",
    "h_index": 2,
    "papers": 12
   },
   {
    "name": "Jianing Zeng",
    "id": "2332681987",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Di Wu",
    "id": "2391990030",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "L. Shu",
    "id": "2072175110",
    "h_index": 4,
    "papers": 26
   },
   {
    "name": "Jie Hao",
    "id": "2393010092",
    "h_index": 1,
    "papers": 8
   }
  ],
  "comment": "6 pages, 5 figures",
  "topics": [
   "tactile",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.14728v1",
  "pdf_url": "https://arxiv.org/pdf/2607.14728v1",
  "html_url": "https://arxiv.org/html/2607.14728v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2607.14708",
  "slug": "reinforcement-learning-for-the-full-strawberry-harvesting-process-obst",
  "title": "Reinforcement Learning for the Full Strawberry Harvesting Process: Obstacle Separation, Detachment, and Placement",
  "abstract": "Severe occlusions and deformable plant structures introduce complex contact dynamics that challenge robotic strawberry harvesting. A policy-driven reinforcement learning (RL) framework with heuristic phase coordination was developed, in which obstacle separation, fruit detachment, and placement were formulated as a sequential decision-making task. A shared interaction-aware policy generated Cartesian motions across all task phases, while lightweight heuristic logic coordinated task progression and gripper events. A shared structured observation space was used to represent target, obstacle, end-effector, and task-context information. A hierarchical architecture combined the high-level policy with low-level Cartesian impedance control for compliant interaction. To support zero-shot sim-to-real transfer, feasibility-first observation alignment and domain randomization were adopted. The policy achieved success rates of 89.7% in simulation and 82.0% in real-world experiments. As the occlusion level increased from 1 to 5, the average execution time increased from 12.99 s to 21.73 s, reflecting greater interaction complexity. These results demonstrated effective transfer of interaction-aware harvesting behaviors to a structurally different robotic platform.",
  "published": "2026-07-16",
  "updated": "2026-07-16",
  "year": "2026",
  "authors": [
   "Changyou Miao",
   "Teng Li",
   "Ya Xiong"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A policy-driven reinforcement learning framework with heuristic phase coordination was developed, in which obstacle separation, fruit detachment, and placement were formulated as a sequential decision-making task to demonstrate effective transfer of interaction-aware harvesting behaviors to a structurally different robotic platform.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Changyou Miao",
    "id": "2450183515",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Teng Li",
    "id": "2450151755",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Ya Xiong",
    "id": "2313624230",
    "h_index": 3,
    "papers": 11
   }
  ],
  "comment": "Accepted to IROS 2026",
  "topics": [
   "sim2real",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.14708v1",
  "pdf_url": "https://arxiv.org/pdf/2607.14708v1",
  "html_url": "https://arxiv.org/html/2607.14708v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2607.14643",
  "slug": "navcmpo-critic-guided-meanflow-policy-optimization-for-adaptive-naviga",
  "title": "NavCMPO: Critic-Guided MeanFlow Policy Optimization for Adaptive Navigation",
  "abstract": "End-to-end diffusion-based policies have demonstrated strong performance in mapless visual navigation, but their iterative denoising process introduces substantial inference latency, while behavior cloning limits performance to the quality of expert demonstrations. We present NavCMPO, a two-stage adaptive navigation framework that combines few-step MeanFlow trajectory generation, critic-guided refinement, and reinforcement learning fine-tuning. During pre-training, an obstacle proximity prediction task encourages the visual representation to capture obstacle-aware spatial information. To compensate for the degradation in obstacle avoidance caused by few-step generation, Critic-Guided Trajectory Refinement (CGTR) uses gradients from a critic trained with obstacle-point-cloud supervision to refine intermediate trajectories. During adaptation, the MeanFlow policy is fine-tuned using Proximal Policy Optimization with behavior-cloning regularization, while the critic is updated to accommodate embodiment-specific observation changes. Under a matched training budget on the InternVLA-N1 benchmark, NavCMPO achieves an average success rate of 74.7\\%, exceeding the retrained NavDP baseline by 6.4 percentage points, while reducing inference latency from 85\\,ms to 60\\,ms. Experiments on a Unitree Go2 further demonstrate effective sim-to-real transfer.",
  "published": "2026-07-16",
  "updated": "2026-07-16",
  "year": "2026",
  "authors": [
   "Junjie An",
   "Yi Wu",
   "Xiao Liu",
   "Yiqun Zhou",
   "Yuechen Wu",
   "Xiaoqing Guan",
   "You Wang",
   "Guang Li"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "NavCMPO is presented, a two-stage adaptive navigation framework that combines few-step MeanFlow trajectory generation, critic-guided refinement, and reinforcement learning fine-tuning, and reduces inference latency.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Junjie An",
    "id": "2189182906",
    "h_index": 0,
    "papers": 5
   },
   {
    "name": "Yi Wu",
    "id": "2319170668",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Xiao Liu",
    "id": "2312891150",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Yiqun Zhou",
    "id": "2450296571",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yuechen Wu",
    "id": "2447334638",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Xiaoqing Guan",
    "id": "1381737652",
    "h_index": 9,
    "papers": 36
   },
   {
    "name": "You Wang",
    "id": "2141040531",
    "h_index": 8,
    "papers": 36
   },
   {
    "name": "Guang Li",
    "id": "2151300097",
    "h_index": 5,
    "papers": 31
   }
  ],
  "comment": "Accepted for presentation at IROS 2026",
  "topics": [
   "sim2real",
   "imitation-diffusion",
   "rl-control",
   "navigation",
   "foundation-pretraining"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2607.14643v1",
  "pdf_url": "https://arxiv.org/pdf/2607.14643v1",
  "html_url": "https://arxiv.org/html/2607.14643v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.0
 },
 {
  "id": "2607.14635",
  "slug": "action-qformer-structured-representation-shaping-under-action-supervis",
  "title": "Action QFormer: Structured Representation Shaping under Action Supervision in Vision-Language-Action Models",
  "abstract": "Action supervision in vision-language-action (VLA) models is often treated as a downstream objective for learning action prediction. In this paper, we study it instead as a force that shapes inherited multimodal representations. We show that this shaping has a dual effect: it is necessary for forming action-compatible representations, but when action supervision is applied too directly to the inherited multimodal pathway, it can also destabilize representations that support language-side processing and object grounding. To address this tension, we introduce Action QFormer, a query-based action-facing interface that uses instruction-conditioned queries to reorganize inherited multimodal information into action-facing representations before downstream action generation. In zero-shot sim-to-real navigation, Action QFormer improves average closed-loop task success from 18.8% to 56.3%, raises fixed-instruction action-generation correctness from 22.5% to 75.5%, and nearly eliminates out-of-distribution instruction generations. Further analyses show that Action QFormer changes how action supervision shapes inherited multimodal representations, reducing broad upstream rewriting while preserving targeted and sometimes constructive action-supervised adaptation. These results suggest that improving VLA performance requires not only stronger pretrained backbones, but also better ways of selecting and organizing inherited multimodal information while controlling how it is shaped under action supervision.",
  "published": "2026-07-16",
  "updated": "2026-07-16",
  "year": "2026",
  "authors": [
   "Yufeng Ji",
   "Wenhao Tang",
   "Haoyi Niu",
   "Koushil Sreenath",
   "Yi Wu",
   "Zhongyu Li"
  ],
  "author_count": 6,
  "categories": [
   "cs.AI",
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.AI",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Action QFormer is introduced, a query-based action-facing interface that uses instruction-conditioned queries to reorganize inherited multimodal information into action-facing representations before downstream action generation, suggesting that improving VLA performance requires not only stronger pretrained backbones, but also better ways of selecting and organizing inherited multimodal information while controlling how it is shaped under action supervision.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yufeng Ji",
    "id": "2292928672",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Wenhao Tang",
    "id": "2323010636",
    "h_index": 5,
    "papers": 18
   },
   {
    "name": "Haoyi Niu",
    "id": "122919426",
    "h_index": 10,
    "papers": 25
   },
   {
    "name": "K. Sreenath",
    "id": "144116765",
    "h_index": 55,
    "papers": 230
   },
   {
    "name": "Yi Wu",
    "id": "2293552963",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Zhongyu Li",
    "id": "1491078398",
    "h_index": 24,
    "papers": 50
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "sim2real",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.14635v1",
  "pdf_url": "https://arxiv.org/pdf/2607.14635v1",
  "html_url": "https://arxiv.org/html/2607.14635v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.14609",
  "slug": "representation-aligned-tactile-grounding-for-contact-rich-robotic-mani",
  "title": "Representation-Aligned Tactile Grounding for Contact-Rich Robotic Manipulation",
  "abstract": "Tactile-enhanced vision-language-action (VLA) policies have been introduced for contact-rich manipulation, where critical interaction states are often hidden from vision. Future tactile prediction is a promising way to use touch because it turns tactile outcomes into supervision for action-induced contact dynamics. Yet VLA policies contain representations with different roles, from perceptual encoding to motor prediction, making it unclear where this supervision should be applied. We study this as a representation-alignment problem. Through a linear probe analysis, we find that future tactile states are most predictable from intermediate action-expert features, rather than from vision-language features or final action states. Motivated by this observation, we introduce a lightweight Latent Tactile Predictor (LTP), which predicts compact future tactile embeddings from the identified intermediate representation. By avoiding direct prediction of noisy raw tactile signals, LTP provides an action-outcome grounding signal that aligns intermediate action representations with future contact consequences. Experiments on real-world contact-rich manipulation tasks show that representation-aligned tactile grounding outperforms less aligned or multi-interface tactile prediction, highlighting the importance of where tactile supervision is applied.",
  "published": "2026-07-16",
  "updated": "2026-07-16",
  "year": "2026",
  "authors": [
   "Ruilin Chen",
   "Jingkai Jia",
   "Tong Yang",
   "Xinyu Zhou",
   "Qiao Sun",
   "Jiangwei Zhong",
   "Shizeng Zhang",
   "Nuo Chen",
   "Bailin He",
   "Wei Li",
   "Wenqiang Zhang"
  ],
  "author_count": 11,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A lightweight Latent Tactile Predictor (LTP) is introduced, which predicts compact future tactile embeddings from the identified intermediate representation, and provides an action-outcome grounding signal that aligns intermediate action representations with future contact consequences.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ruilin Chen",
    "id": "2449965909",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Jingkai Jia",
    "id": "2117241998",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Tong Yang",
    "id": "2263762687",
    "h_index": 7,
    "papers": 17
   },
   {
    "name": "Xinyu Zhou",
    "id": "2395637062",
    "h_index": 2,
    "papers": 13
   },
   {
    "name": "Qiao Sun",
    "id": "2349426707",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Jiangwei Zhong",
    "id": "2349071447",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Shizeng Zhang",
    "id": "2450394558",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Nuo Chen",
    "id": "2450267067",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Bailin He",
    "id": "2349085421",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Wei Li",
    "id": "2380649662",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Wenqiang Zhang",
    "id": "2450396401",
    "h_index": 0,
    "papers": 2
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.14609v1",
  "pdf_url": "https://arxiv.org/pdf/2607.14609v1",
  "html_url": "https://arxiv.org/html/2607.14609v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.14578",
  "slug": "beyond-implicit-force-evaluating-explicit-force-torque-proxies-in-acti",
  "title": "Beyond Implicit Force: Evaluating Explicit Force-Torque Proxies in Action Chunking with Transformers",
  "abstract": "Contact-rich manipulation requires policies to infer interaction state from signals that are often weakly observable through vision and kinematics alone. Action Chunking with Transformers (ACT) has shown strong performance in fine-grained manipulation, but many deployments collect demonstrations through leader-follower teleoperation, where tracking error between commanded leader motion and executed follower motion implicitly encodes contact, resistance, and constraint violation. This paper examines whether ACT's apparent force-awareness depends on this hidden interaction cue. We introduce an observation-centric ACT variant that predicts future follower joint states instead of leader commands, thereby removing the teleoperation-induced discrepancy signal while preserving the rest of the learning pipeline. We then evaluate whether simple joint-torque proxies, derived from onboard motor current or joint effort, can recover contact-aware behavior without external force/torque sensors. Across four real-world tasks spanning surface following, insertion, stiffness discrimination, and force-based stopping, removing the implicit cue leads to severe failures in force-critical phases. In contrast, torque-augmented policies recover robust contact behavior and improve the base ACT policy. These results demonstrate that, on real hardware, the implicit teleoperation cue is a recoverable source of force-awareness, where torque signals are available, a simple proxy matches, surpasses, or further enhances it.",
  "published": "2026-07-16",
  "updated": "2026-07-16",
  "year": "2026",
  "authors": [
   "King Hang Wong",
   "Lingqiao Liu",
   "Feras Dayoub"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper introduces an observation-centric ACT variant that predicts future follower joint states instead of leader commands, thereby removing the teleoperation-induced discrepancy signal while preserving the rest of the learning pipeline and demonstrating that the implicit teleoperation cue is a recoverable source of force-awareness.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "King Hang Wong",
    "id": "2450231665",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Lingqiao Liu",
    "id": "2264248423",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Feras Dayoub",
    "id": "1757942",
    "h_index": 30,
    "papers": 122
   }
  ],
  "comment": "Accepted to IROS 2026",
  "topics": [
   "tactile",
   "imitation-diffusion",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.14578v1",
  "pdf_url": "https://arxiv.org/pdf/2607.14578v1",
  "html_url": "https://arxiv.org/html/2607.14578v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2607.14539",
  "slug": "communication-efficient-relative-pose-estimation-with-vision-foundatio",
  "title": "Communication-Efficient Relative Pose Estimation with Vision Foundation Models for Ephemeral Collaborative Perception",
  "abstract": "Relative pose estimation is a fundamental capability for collaborative perception and coordination in multi-robot systems. However, robots encountering each other in real-world environments often operate in short interaction windows and must operate under limited communication bandwidth with intermittent or missing visual overlap caused by occlusions or limited fields of view. Existing approaches typically rely on global reference frames, assume sustained view overlap, or incur prohibitive communication costs, thereby limiting their applicability to ephemeral collaborative perception. To address these challenges, we introduce communication-efficient relative pose estimation (CERPE), a system-level framework that coordinates vision foundation models to jointly estimate ego-motion and inter-robot relative pose. CERPE reduces unnecessary raw-observation exchange by using continuously shared fixed-size descriptors to gate event-triggered raw-image requests independently of pose estimation. Non-overlapping encounters are handled by propagating inter-robot relative poses through metrically scaled ego-motion, thus maintaining relative pose estimates even in the absence of visual overlap. Experiments in simulation and real-world robots show that CERPE improves 6-DoF relative pose estimation over selected baselines in ephemeral collaborative perception.",
  "published": "2026-07-16",
  "updated": "2026-07-16",
  "year": "2026",
  "authors": [
   "Qihang Li",
   "Jo-Hao Huang",
   "Jiewen Liu",
   "Suyoung Kang",
   "Hao Zhang",
   "Peng Gao"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Communication-efficient relative pose estimation (CERPE) is introduced, a system-level framework that coordinates vision foundation models to jointly estimate ego-motion and inter-robot relative pose in multi-robot systems.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Qihang Li",
    "id": "2448639614",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Jo-Hao Huang",
    "id": "2450393551",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Jiewen Liu",
    "id": "2250926613",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Suyoung Kang",
    "id": "2200090414",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Hao Zhang",
    "id": "2294828693",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Peng Gao",
    "id": "2097573796",
    "h_index": 9,
    "papers": 22
   }
  ],
  "comment": "8 pages, 6 figures",
  "topics": [
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.14539v1",
  "pdf_url": "https://arxiv.org/pdf/2607.14539v1",
  "html_url": "https://arxiv.org/html/2607.14539v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.14488",
  "slug": "safe-execution-of-rl-policies-via-acceleration-based-cbf-qp-constraint",
  "title": "Safe Execution of RL Policies Via Acceleration-Based CBF-QP Constraint Enforcement for Real-World Robotic Deployments",
  "abstract": "Reinforcement Learning (RL) has demonstrated remarkable capabilities for solving complex robotic control problems, but its lack of safety guarantees severely limits deployment on hardware. In particular, as legged robots and manipulators often operate near safety-critical boundaries, out-of-distribution states can lead to failure upon deployment. To address this, we introduce Acc-CBF-QP, an acceleration-based Quadratic Program (QP) safety filter using Control Barrier Functions (CBFs) that constrains any RL policy onto a safe set at runtime without modifying training. The method applies to unconstrained and Safe-RL policies, and enforces joint position, velocity, torque, and collision constraints within a unified optimization framework. A key contribution is the formulation of RL+QP tasks that regulate deviation from the RL command when constraints would otherwise be violated. We introduce a TorqueTask, minimizing torque deviation, and a Forward Dynamics Task, minimizing induced acceleration deviation, thus providing principled control over safety-performance trade-offs. Experiments on a 7-DoF Kinova Gen3 manipulator and a 19-DoF Unitree H1 humanoid, both in simulation and on hardware, highlight substantial reductions in constraint violations. On the real H1 hardware, a Safe-RL policy alone yielded 10.04 violations/s, which were reduced by 92% to 0.80 violations/s when augmented with Acc-CBF-QP. On the Kinova Gen3, Acc-CBF-QP fully eliminated violations. Nominal task performance of the RL objective is preserved in violation-free regimes. Under aggressive velocity commands on H1, Acc-CBF-QP improves execution by preventing constraint-induced shutdowns, yielding longer survival times. The full pipeline is open-source.",
  "published": "2026-07-16",
  "updated": "2026-07-16",
  "year": "2026",
  "authors": [
   "Bastien Muraccioli",
   "Alice Cariou",
   "Pierre-Alexandre Leziart",
   "Mathieu Celerier",
   "Arnaud Demont",
   "Gentiane Venture",
   "Mehdi Benallegue"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Acc-CBF-QP is introduced, an acceleration-based Quadratic Program (QP) safety filter using Control Barrier Functions (CBFs) that constrains any RL policy onto a safe set at runtime without modifying training, thus providing principled control over safety-performance trade-offs.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Bastien Muraccioli",
    "id": "2153732465",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Alice Cariou",
    "id": "2442246043",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Pierre-Alexandre L\u00e9ziart",
    "id": "10164829",
    "h_index": 9,
    "papers": 15
   },
   {
    "name": "Mathieu C\u00e9l\u00e9rier",
    "id": "2349444547",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Arnaud Demont",
    "id": "2307466279",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "G. Venture",
    "id": "1775438",
    "h_index": 30,
    "papers": 324
   },
   {
    "name": "M. Benallegue",
    "id": "2097614",
    "h_index": 21,
    "papers": 83
   }
  ],
  "comment": "8 pages, 4 figures. Accepted for publication at IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS 2026)",
  "topics": [
   "humanoids",
   "rl-control",
   "safety-eval"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2607.14488v1",
  "pdf_url": "https://arxiv.org/pdf/2607.14488v1",
  "html_url": "https://arxiv.org/html/2607.14488v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.0
 },
 {
  "id": "2607.14487",
  "slug": "midas-hand-modular-low-impedance-direct-drive-anthropomorphic-sensing",
  "title": "MIDAS Hand: Modular low-Impedance Direct-drive Anthropomorphic Sensing Hand",
  "abstract": "Dexterous manipulation is limited not only by algorithms but by a shortage of accessible hand hardware that combines human-scale morphology, ease of manufacturing or maintenance, tactile sensing, and practical cost. Existing dexterous hands tend to optimize some of these properties at the expense of others. We present MIDAS Hand, a low-cost, open-source, human-scale dexterous hand with integrated tactile sensing for manipulation research. MIDAS Hand provides 16 total degrees of freedom (DoF) with 13 active DoF, directly driven actuation with measurably low backdrive torque, and 283 three-axis tactile taxels in a compact 700 g package with a bill of materials under 3,000 USD. Built from 3D-printed components, it assembles in under three hours while providing the strength, repeatability, and maintainability needed for repeated real-world experiments. Alongside the hardware, we release a full stack: design files, build documentation, control and tactile Python APIs, simulation models, and retargeting and teleoperation pipelines. We characterize MIDAS Hand through workspace and grasp-taxonomy analysis, payload and reliability tests, backdrivability measurements, and teleoperation demonstrations with tactile sensing, showing that it offers a balanced, reproducible platform for tactile dexterous manipulation and human-to-robot data collection. Project page: https://midas-hand.com",
  "published": "2026-07-16",
  "updated": "2026-07-16",
  "year": "2026",
  "authors": [
   "Alvin Zhu",
   "Mingzhang Zhu",
   "Beom Jun Kim",
   "Quanyou Wang",
   "Jose Victor S. H. Ramos",
   "Dennis Hong"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Alvin Zhu",
    "id": "2359447907",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Mingzhang Zhu",
    "id": "2350497013",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Beomdo Kim",
    "id": "2345408475",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Quanyou Wang",
    "id": "2359580930",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "J.H. Ramos",
    "id": "2361905795",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Dennis W. Hong",
    "id": "2376201986",
    "h_index": 1,
    "papers": 8
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "data-teleop",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.14487v1",
  "pdf_url": "https://arxiv.org/pdf/2607.14487v1",
  "html_url": "https://arxiv.org/html/2607.14487v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2607.14470",
  "slug": "g-2-sr-geometric-methods-for-fast-and-memory-efficient-gaussian-based",
  "title": "G$^2$SR: Geometric Methods for Fast and Memory-Efficient Gaussian-based Surface Reconstruction",
  "abstract": "Few-view surface reconstruction recovers the visible surfaces of a scene from a few posed RGB images, providing the 3D models that robots need to explore and interact online. On mobile platforms, the reconstruction must be fast and geometrically accurate while keeping a small memory footprint to ensure safe and efficient operation. 3D Gaussian Splatting (3DGS) offers a high-fidelity scene representation, but building it from a few views is ill-posed, as many distinct surfaces reproduce the same images, making traditional photometric methods prone to \"floater\" artifacts. End-to-end methods resolve the ambiguity by regressing splats with large, usually Transformer-based, networks that require heavy compute and memory while generalizing poorly to new scenes. We propose G2SR, which exploits a well-posed core of the task: given cross-view 2D splat correspondences, 3D splats follow analytically from multi-view geometry. G2SR employs a lightweight neural frontend to detect and track 2D Gaussian splats on the image plane and an analytic backend to triangulate each into a metric-scale 3D splat. On ScanNet, Replica, and DTU, G2SR matches or exceeds the geometric accuracy of state-of-the-art end-to-end methods while running at 69-89 reconstructions per second within 203 MB of GPU memory (5-107x less) for 2- and 3-view inputs at 384 x 512 resolution, offering a practical path to online Gaussian-based surface reconstruction.",
  "published": "2026-07-16",
  "updated": "2026-07-16",
  "year": "2026",
  "authors": [
   "Dasong Gao",
   "Vivienne Sze",
   "Sertac Karaman"
  ],
  "author_count": 3,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "G2SR matches or exceeds the geometric accuracy of state-of-the-art end-to-end methods while running at 69-89 reconstructions per second within 203 MB of GPU memory (5-107x less) for 2- and 3-view inputs at 384 x 512 resolution, offering a practical path to online Gaussian-based surface reconstruction.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Dasong Gao",
    "id": "1894982568",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Vivienne Sze",
    "id": "2321406965",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "S. Karaman",
    "id": "143612763",
    "h_index": 61,
    "papers": 270
   }
  ],
  "comment": "8 pages, 3 figures",
  "topics": [
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.14470v1",
  "pdf_url": "https://arxiv.org/pdf/2607.14470v1",
  "html_url": "https://arxiv.org/html/2607.14470v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.14341",
  "slug": "beyond-visual-grasping-benchmarking-complex-grasping-from-detection-to",
  "title": "Beyond Visual Grasping: Benchmarking Complex Grasping from Detection to Execution",
  "abstract": "Robust robotic grasping remains a fundamental challenge for complex real-world applications. Recent advances in large-scale models demonstrate promising capabilities for reasoning in robotic tasks. However, existing benchmarks for grasping primarily focus on isolated, visual-based grasp pose detection, failing to capture the complexity of grasping tasks that require multi-step reasoning and semantic understanding during execution. To address this gap, we propose GCA-Bench, a benchmark featuring challenging \\textit{grasping with complex action} scenarios that involve both scene-level reasoning and semantic constraints. GCA-Bench enables the evaluation of recent large foundation models under the same settings. To demonstrate the effectiveness of our new benchmark, we implement a diverse set of baselines, ranging from traditional grasp detection pipelines to end-to-end learning methods. Empirical studies achieve success rates below 70\\% on complex grasping scenarios, underscoring critical limitations. In addition, we propose new evaluation metrics, analyze critical failure models, and provide insights to guide the development of more robust and generalizable grasping strategies.",
  "published": "2026-07-15",
  "updated": "2026-07-15",
  "year": "2026",
  "authors": [
   "Hanyi Zhang",
   "Khang Nguyen",
   "Charith Munasinghe",
   "Basu Hela",
   "Tianyu Li",
   "Zihong Luo",
   "Hoan Nguyen",
   "Hans Wernher van de Venn",
   "Yalin Zheng",
   "Ravi Prakash",
   "Tung D. Ta",
   "Anh Nguyen",
   "Baoru Huang"
  ],
  "author_count": 13,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "GCA-Bench is proposed, a benchmark featuring challenging grasping with complex action scenarios that involve both scene-level reasoning and semantic constraints and enables the evaluation of recent large foundation models under the same settings.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hanyi Zhang",
    "id": "2346426387",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Khang Nguyen",
    "id": "2398778183",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Charith Munasinghe",
    "id": "2188873476",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Basu Hela",
    "id": "2367142348",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Tianyu Li",
    "id": "2450220433",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Zihong Luo",
    "id": "2283421778",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Hoan Nguyen",
    "id": "2267878838",
    "h_index": 4,
    "papers": 14
   },
   {
    "name": "H. W. V. D. Venn",
    "id": "98681611",
    "h_index": 9,
    "papers": 47
   },
   {
    "name": "Yalin Zheng",
    "id": "2271556582",
    "h_index": 6,
    "papers": 18
   },
   {
    "name": "Ravi Prakash",
    "id": "2297769213",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Tung D. Ta",
    "id": "2322942638",
    "h_index": 2,
    "papers": 16
   },
   {
    "name": "Anh Nguyen",
    "id": "2154623771",
    "h_index": 10,
    "papers": 27
   },
   {
    "name": "Baoru Huang",
    "id": "2243927673",
    "h_index": 10,
    "papers": 37
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "foundation-pretraining",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.14341v1",
  "pdf_url": "https://arxiv.org/pdf/2607.14341v1",
  "html_url": "https://arxiv.org/html/2607.14341v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.14236",
  "slug": "never-too-late-for-force-accelerating-vla-post-training-with-reactive",
  "title": "Never Too Late for Force: Accelerating VLA Post-Training with Reactive Force Injection",
  "abstract": "Pretrained vision-language-action (VLA) policies provide strong language-conditioned manipulation knowledge, but they remain largely vision-driven and can struggle once manipulation enters contact states where the scene is occluded, depth is ambiguous, or small force errors push execution off the offline demonstration distribution. We present LIFT (Late Reactive Injection of Force for VLA Post-Training), a force-aware post-training framework that adds contact reactivity to a pretrained VLA policy while preserving its general manipulation knowledge. LIFT grafts a reactive action expert beside the original action expert, initializes it from pretrained action weights, and injects recent 6D end-effector force through causal force memory and zero-initialized cross attention, enabling actions to be refreshed during execution. To address the policy-dependent distribution shift of contact feedback, LIFT further couples reactive force injection with an online DAgger loop that trains on a mixture of offline task-alignment data and human-corrected online rollouts. Across towel folding, book insertion, and Hanoi ring placement, LIFT learns faster and reaches higher performance than vision-only post-training, while ablations show that reactive force memory and online corrective data are both important for robust contact-rich manipulation. Our code and data will be publicly available.",
  "published": "2026-07-15",
  "updated": "2026-07-15",
  "year": "2026",
  "authors": [
   "Yi Wang",
   "Wendi Chen",
   "Zimo Wen",
   "Han Xue",
   "Xueqi Li",
   "Wenye Yu",
   "Zhijie Chen",
   "Hao Yang",
   "Jun Lv",
   "Chuan Wen",
   "Cewu Lu"
  ],
  "author_count": 11,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Across towel folding, book insertion, and Hanoi ring placement, LIFT learns faster and reaches higher performance than vision-only post-training, while ablations show that reactive force memory and online corrective data are both important for robust contact-rich manipulation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yi Wang",
    "id": "2363937133",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Wendi Chen",
    "id": "2326063487",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Zimo Wen",
    "id": "2392914761",
    "h_index": 1,
    "papers": 9
   },
   {
    "name": "Han Xue",
    "id": "2351107813",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Xueqi Li",
    "id": "2319814817",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Wenye Yu",
    "id": "2298210236",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Zhijie Chen",
    "id": "2316662510",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Hao Yang",
    "id": "2330419650",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Jun Lv",
    "id": "2054671126",
    "h_index": 11,
    "papers": 26
   },
   {
    "name": "Chuan Wen",
    "id": "2381956964",
    "h_index": 5,
    "papers": 20
   },
   {
    "name": "Cewu Lu",
    "id": "2301174899",
    "h_index": 7,
    "papers": 29
   }
  ],
  "comment": "Project page: https://lift-policy.github.io/",
  "topics": [
   "vla",
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.14236v1",
  "pdf_url": "https://arxiv.org/pdf/2607.14236v1",
  "html_url": "https://arxiv.org/html/2607.14236v1",
  "code_url": "https://lift-policy.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2607.14187",
  "slug": "rxbrain-embodied-cognition-foundation-model-with-joint-language-visual",
  "title": "RxBrain: Embodied Cognition Foundation Model with Joint Language-Visual Reasoning and Imagination",
  "abstract": "Embodied cognition requires agents to connect high-level task reasoning with the physical states to be achieved. We introduce Hy-Embodied-RxBrain, an embodied cognition foundation model with joint language-visual reasoning and imagination. Unlike vision-language models that emphasize scene understanding and textual decision making, or generative world models that mainly predict future visual states, RxBrain represents embodied plans in a single planning sequence where language and visual imagination play complementary roles. Language provides the abstract structure of a plan, including task decomposition, planning primitives, constraints, temporal order, and decision logic, while visual imagination grounds this structure through world state prediction and joint subgoal planning, associating each planning step with intermediate and final physical states. RxBrain adopts a unified multimodal Mixture-of-Transformers architecture that supports language, image, and video understanding and generation within one model. To train this capability, we build an automatic pipeline that converts embodied videos into joint text-visual planning supervision by decomposing videos into planning steps and aligning them with visual state transitions. We further introduce RxBrain-Bench to evaluate whether models can represent embodied plans through joint textual and visual components rather than separate understanding or generation. Experiments show that RxBrain maintains embodied understanding and generation abilities, and produces plans with coupled textual reasoning, world state prediction, and joint subgoal planning. We also extend RxBrain to continuous robot action generation, where it shows promising real-robot performance without large-scale action-data pretraining. These results provide an initial step toward foundation models for embodied cognition.",
  "published": "2026-07-15",
  "updated": "2026-07-15",
  "year": "2026",
  "authors": [
   "Haotian Liang",
   "Mingkang Chen",
   "Yufei Huang",
   "Yuchun Guo",
   "Xiaomeng Zhu",
   "Xiangli Shi",
   "Kaixuan Wang",
   "Yunxuan Mao",
   "Weijie Zhou",
   "Ling Chen",
   "Shirong Zeng",
   "Yueyu Long",
   "Yuchen Si",
   "Yajuan Zhu",
   "Xingyu Zhou",
   "Minghui Wang",
   "Wanjia He",
   "Xin Yang",
   "Lingzhu Xiang",
   "Zhiqing Liu",
   "Bohan Ma",
   "Xiran Huang",
   "Tianshuo Yang",
   "Zhiheng Liu",
   "Xuantang Xiong",
   "Zisheng Lu",
   "Ping Luo",
   "Yao Mu",
   "Han Hu",
   "Zhengyou Zhang"
  ],
  "author_count": 30,
  "categories": [
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.AI",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "Hy-Embodied-RxBrain, an embodied cognition foundation model with joint language-visual reasoning and imagination, is introduced and extended to continuous robot action generation, where it shows promising real-robot performance without large-scale action-data pretraining.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Haotian Liang",
    "id": "2365384725",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Mingkang Chen",
    "id": "2028618901",
    "h_index": 6,
    "papers": 19
   },
   {
    "name": "Yufei Huang",
    "id": "2362740559",
    "h_index": 4,
    "papers": 14
   },
   {
    "name": "Yuchun Guo",
    "id": "49813545",
    "h_index": 13,
    "papers": 28
   },
   {
    "name": "Xiaomeng Zhu",
    "id": "2379722842",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Xiangli Shi",
    "id": "2445896134",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Kaixuan Wang",
    "id": "2370941146",
    "h_index": 5,
    "papers": 16
   },
   {
    "name": "Yunxuan Mao",
    "id": "2376499882",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Weijie Zhou",
    "id": "2350037052",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Ling Chen",
    "id": "2315646281",
    "h_index": 3,
    "papers": 14
   },
   {
    "name": "Shirong Zeng",
    "id": "2279791694",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Yueyue Long",
    "id": "2147432961",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Yuchen Si",
    "id": "2450185831",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yajuan Zhu",
    "id": "2450219880",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Xingyu Zhou",
    "id": "2447057990",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Minghui Wang",
    "id": "2266139113",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Wanjiao He",
    "id": "2237411316",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Xin Yang",
    "id": "2367089762",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Lingzhu Xiang",
    "id": "2276931450",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Zhi-qing Liu",
    "id": "2449450244",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Bohan Ma",
    "id": "2450223866",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Xiran Huang",
    "id": "2450295208",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Tianshuo Yang",
    "id": "2300491960",
    "h_index": 9,
    "papers": 20
   },
   {
    "name": "Zhiheng Liu",
    "id": "2360881881",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Xuantang Xiong",
    "id": "2347707155",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Zisheng Lu",
    "id": "2442573262",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Ping Luo",
    "id": "2143481782",
    "h_index": 29,
    "papers": 55
   },
   {
    "name": "Yao Mu",
    "id": "1675357512",
    "h_index": 14,
    "papers": 23
   },
   {
    "name": "Han Hu",
    "id": "2359181758",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Zhengyou Zhang",
    "id": "2148905709",
    "h_index": 8,
    "papers": 19
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "foundation-pretraining",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.14187v1",
  "pdf_url": "https://arxiv.org/pdf/2607.14187v1",
  "html_url": "https://arxiv.org/html/2607.14187v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2607.14183",
  "slug": "open-aoe-an-open-egocentric-manipulation-dataset-and-toolchain-for-emb",
  "title": "Open-AoE: An Open Egocentric Manipulation Dataset and Toolchain for Embodied Learning",
  "abstract": "Egocentric videos of human manipulation provide scalable supervision for embodied intelligence, yet existing resources rarely combine low-cost continuous capture, manipulation-level structured annotations, and reusable tools for robot learning. We present Open-AoE, an open, community-oriented egocentric manipulation dataset and toolchain spanning the full pipeline from smartphone capture to model training. Its first release contains approximately 2,000 hours of manipulation video collected in natural environments by 500+ contributors using 400+ smartphones. The dataset provides text annotations, MANO-based hand poses, camera trajectories, and temporally localized atomic actions. Open-AoE further includes a data processing pipeline that transforms raw recordings into structured samples through temporal action segmentation, semantic annotation, hand reconstruction, and camera trajectory reconstruction. Meanwhile, we provide a separate downstream toolchain supports visualization, cross-embodiment retargeting, model-specific data conversion, and training recipes for VLA policies, WAMs, and World Models. By integrating scalable capture, structured processing, and downstream adaptation, Open-AoE reduces the barriers to both data contribution and reuse, providing practical open infrastructure for embodied model training, human-to-robot transfer, and world modeling.",
  "published": "2026-07-15",
  "updated": "2026-07-18",
  "year": "2026",
  "authors": [
   "Zishuo Li",
   "Bowen Yang",
   "Changtao Miao",
   "Kai Zhu",
   "Hao Chen",
   "Qingze Guan",
   "Zhengxing Wu",
   "Wanke Zhan",
   "Yang Sun",
   "Zhiyi Huang",
   "Zitong Shan",
   "Zhenchao Jin",
   "Jiadong Hong",
   "Taowen Wang",
   "Yushi Feng",
   "You Liu",
   "Yibo Wang",
   "Yifan Yang",
   "Zhaowen Zhou",
   "Man Luo",
   "Hao Cheng",
   "Bo Zhang",
   "Jianshu Li",
   "Jiansheng Cai",
   "Guocai Yao",
   "Jize Zhang",
   "Chenhao Lin",
   "Renjing Xu",
   "Lequan Yu",
   "Chao Shen",
   "Chunhua Shen",
   "Zhe Li"
  ],
  "author_count": 32,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "By integrating scalable capture, structured processing, and downstream adaptation, Open-AoE reduces the barriers to both data contribution and reuse, providing practical open infrastructure for embodied model training, human-to-robot transfer, and world modeling.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zishuo Li",
    "id": "2387443541",
    "h_index": 9,
    "papers": 48
   },
   {
    "name": "Bowen Yang",
    "id": "2337042873",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Changtao Miao",
    "id": "2313730242",
    "h_index": 4,
    "papers": 19
   },
   {
    "name": "Kai Zhu",
    "id": "2394240901",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Hao Chen",
    "id": "2363709527",
    "h_index": 6,
    "papers": 21
   },
   {
    "name": "Qingze Guan",
    "id": "2450183177",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Zhengxing Wu",
    "id": "2290608",
    "h_index": 34,
    "papers": 195
   },
   {
    "name": "Wanke Zhan",
    "id": "2450185411",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yang Sun",
    "id": "2401114303",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Zhiyi Huang",
    "id": "2449455143",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Zitong Shan",
    "id": "2212302468",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Zhenchao Jin",
    "id": "2152843665",
    "h_index": 13,
    "papers": 19
   },
   {
    "name": "Jiadong Hong",
    "id": "2450416520",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Taowen Wang",
    "id": "2311730488",
    "h_index": 4,
    "papers": 20
   },
   {
    "name": "Yushi Feng",
    "id": "2346441717",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "You Liu",
    "id": "2394027555",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Yibo Wang",
    "id": "2448920124",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yifan Yang",
    "id": "2313033754",
    "h_index": 4,
    "papers": 24
   },
   {
    "name": "Zhaowen Zhou",
    "id": "2237600989",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Man Luo",
    "id": "2314968675",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Hao Cheng",
    "id": "2446300967",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Bo Zhang",
    "id": "2354113552",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Jianshu Li",
    "id": "2183730672",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Jiansheng Cai",
    "id": "2295809954",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Guocai Yao",
    "id": "2376597393",
    "h_index": 6,
    "papers": 23
   },
   {
    "name": "Jize Zhang",
    "id": "2310746135",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Chenhao Lin",
    "id": "2281063360",
    "h_index": 8,
    "papers": 52
   },
   {
    "name": "Renjing Xu",
    "id": "2385455909",
    "h_index": 2,
    "papers": 21
   },
   {
    "name": "Lequan Yu",
    "id": "2259850862",
    "h_index": 7,
    "papers": 16
   },
   {
    "name": "Chao Shen",
    "id": "2276319867",
    "h_index": 9,
    "papers": 45
   },
   {
    "name": "Chunhua Shen",
    "id": "2257324242",
    "h_index": 19,
    "papers": 78
   },
   {
    "name": "Zhe Li",
    "id": "2385507969",
    "h_index": 7,
    "papers": 15
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "vla",
   "egocentric-data",
   "foundation-pretraining",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.14183v2",
  "pdf_url": "https://arxiv.org/pdf/2607.14183v2",
  "html_url": "https://arxiv.org/html/2607.14183v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2607.14182",
  "slug": "semantic-audio-driven-understanding-for-dynamic-humanoid-whole-body-co",
  "title": "Semantic Audio-driven Understanding for Dynamic Humanoid Whole Body Control",
  "abstract": "Recent advances in humanoid robotics and reinforcement learning have enabled the acquisition of highly expressive whole-body motion policies. However, most robotic performances remain based on pre-scripted sequences or externally triggered behaviors, limiting autonomy and responsiveness to dynamic environments. In this work, we introduce a novel multi-modal orchestration framework for semantic audio-driven humanoid control, enabling robots to autonomously select and execute appropriate motion skills in real time. The system processes continuous audio streams and routes them into music or speech branches. Music input is handled via audio fingerprinting and semantic embeddings to retrieve track identity and temporal alignment, allowing dynamic mapping between musical segments and motion policies. Speech input is grounded into a discrete library of imitation-learned skills, enabling direct human-robot interaction. Both modalities share a unified interface that schedules skill execution over a reinforcement learning control pipeline. We validate the approach in simulation and on a Unitree G1 humanoid, showing robust sim-to-real transfer and consistent audio-conditioned policy selection. Supplementary materials are available at the following site: https://lab-rococo-sapienza.github.io/semantic-WBC/",
  "published": "2026-07-15",
  "updated": "2026-07-15",
  "year": "2026",
  "authors": [
   "J. M. A. Marcelo",
   "M. Brienza",
   "E. Bugli",
   "L. Comito",
   "D. Nardi",
   "D. D. Bloisi",
   "V. Suriani"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work introduces a novel multi-modal orchestration framework for semantic audio-driven humanoid control, enabling robots to autonomously select and execute appropriate motion skills in real time.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "J. Marcelo",
    "id": "80949573",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "M. Brienza",
    "id": "2185586695",
    "h_index": 4,
    "papers": 20
   },
   {
    "name": "E. Bugli",
    "id": "2332354861",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "L. Comito",
    "id": "2352513790",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "D. Nardi",
    "id": "2064462333",
    "h_index": 6,
    "papers": 32
   },
   {
    "name": "D. Bloisi",
    "id": "2064400416",
    "h_index": 6,
    "papers": 33
   },
   {
    "name": "V. Suriani",
    "id": "22269673",
    "h_index": 7,
    "papers": 51
   }
  ],
  "comment": "Accepted at 29th Robocup International Symposium, held on July 6th, 2026 in Incheon, Republic of Korea",
  "topics": [
   "humanoids",
   "sim2real",
   "rl-control",
   "hri"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2607.14182v1",
  "pdf_url": "https://arxiv.org/pdf/2607.14182v1",
  "html_url": "https://arxiv.org/html/2607.14182v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2607.14047",
  "slug": "zero2skill-bootstrapping-robot-skills-through-autonomous-data-collecti",
  "title": "Zero2Skill: Bootstrapping Robot Skills through Autonomous Data Collection, Training, and Deployment",
  "abstract": "Autonomous data collection governs the volume and quality of real-world trajectories for manipulation policy learning. Existing pipelines reduce human effort via self-resetting, VLM verification, or language-guided correction, yet episode-scoped fixes must be reissued whenever the same failure recurs, so oversight cost grows with session length rather than with the number of distinct problems. We present Zero2Skill, a human-robot symbiotic agentic system in which corrections are retained and reused across rounds. The collection loop collects, verifies, and resets autonomously, pausing for a remote operator only when a phase exhausts an explicit retry budget. An LLM parser maps each natural-language utterance to a structured adjustment stored in Corrective Memory, so addressed failure modes typically need not be corrected again under the same conditions. On a real-robot desktop-clearing testbed, Zero2Skill matches teleoperation episode success while reducing human working time to 16%. Language corrections improve verifier-human agreement in all four evaluated settings and raise average single-attempt success from 12.5% to 47.5% (arm-selection: 20.0% to 50.0%). Policies fine-tuned on Zero2Skill data match teleoperation-trained policy success at a fraction of collection human cost.",
  "published": "2026-07-15",
  "updated": "2026-07-22",
  "year": "2026",
  "authors": [
   "Boyuan Wang",
   "Zhenyuan Zhang",
   "Zhiqin Yang",
   "Peijun Gu",
   "Shuya Wang",
   "Xiaofeng Wang",
   "Xianghui Ze",
   "Yifan Chang",
   "Guosheng Zhao",
   "Jiangnan Shao",
   "Guan Huang",
   "Hengyu Liu",
   "Yonggang Zhang",
   "Wei Xue",
   "Chunyuan Guan",
   "Chenglin Pu",
   "Yike Guo",
   "Xingang Wang",
   "Zheng Zhu"
  ],
  "author_count": 19,
  "categories": [
   "cs.RO",
   "cs.HC",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Zero2Skill is presented, a human-robot symbiotic agentic system in which corrections are retained and reused across rounds, and policies fine-tuned on Zero2Skill data match teleoperation-trained policy success at a fraction of collection human cost.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Boyuan Wang",
    "id": "2295591878",
    "h_index": 11,
    "papers": 23
   },
   {
    "name": "Zhenyuan Zhang",
    "id": "2371083016",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Zhiqin Yang",
    "id": "2257141336",
    "h_index": 10,
    "papers": 25
   },
   {
    "name": "Peijun Gu",
    "id": "2422385972",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Shuya Wang",
    "id": "2450178171",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Xiaofeng Wang",
    "id": "2349399535",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Xianghui Ze",
    "id": "2248275741",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Yifan Chang",
    "id": "2349397350",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Guosheng Zhao",
    "id": "2292092480",
    "h_index": 14,
    "papers": 27
   },
   {
    "name": "Jiangnan Shao",
    "id": "2150092260",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Guan Huang",
    "id": "2256954306",
    "h_index": 16,
    "papers": 42
   },
   {
    "name": "Hengyu Liu",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Yonggang Zhang",
    "id": "2360840966",
    "h_index": 3,
    "papers": 13
   },
   {
    "name": "Wei Xue",
    "id": "2315133973",
    "h_index": 5,
    "papers": 20
   },
   {
    "name": "Chunyuan Guan",
    "id": "2450000930",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Chenglin Pu",
    "id": "2450000108",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yike Guo",
    "id": "2403462291",
    "h_index": 3,
    "papers": 17
   },
   {
    "name": "Xingang Wang",
    "id": "2295655821",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Zheng Zhu",
    "id": "2109516240",
    "h_index": 17,
    "papers": 50
   }
  ],
  "comment": "WebPage: https://open-gigaai.github.io/Zero2Skill",
  "topics": [
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.14047v3",
  "pdf_url": "https://arxiv.org/pdf/2607.14047v3",
  "html_url": "https://arxiv.org/html/2607.14047v3",
  "code_url": "https://open-gigaai.github.io/Zero2Skill",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2607.14021",
  "slug": "industrial-dexterity-benchmark-a-hardware-software-benchmarking-platfo",
  "title": "Industrial Dexterity Benchmark: A Hardware-Software Benchmarking Platform for Industrial Dexterous Manipulation",
  "abstract": "Dexterous manipulation remains a critical bottleneck in industrial automation; tasks such as cable routing, connector insertion, and precision assembly still rely heavily on manual labor despite decades of robotics research. This work presents a progression from classical, modular robotics pipelines toward an end-to-end multimodal imitation-learning framework for industrial dexterous manipulation. As a part of this work, we introduce three key contributions: a set of Industrial Dexterity Benchmark (IDB) boards aimed to mimic datacenter cable management, automotive cable harnesses, and gearbox assembly tasks; a scalable imitation learning framework (DAG-ROS); and a multimodal diffusion-based policy framework (AG-iDP3) that creates models fusing RGB images, point clouds, joint positions, and wrist-frame wrench data. Focusing on the datacenter cable manipulation board, we evaluate the performance of a task involving cleaning a single cable over variations of an end-to-end AI policy using 48 trials per configuration. The best performing configuration, a multimodal expansion Diffusion Policy (DP), includes a multi-view RGB image source passed through an R3M encoder and reaches a 78% grasp and insert combined task success rate. This performance marks a significant improvement over the 36% observed from the single-camera RGB DP baseline. Each of the tested configurations requires only approximately 100 teleoperated demonstrations per task phase. These results indicate that the correct learned policy can outperform classical vision and control robotic methods in robustness, generalization, and deployment efficiency, justifying a shift toward scalable robotic automation for high up-time industrial environments.",
  "published": "2026-07-15",
  "updated": "2026-07-24",
  "year": "2026",
  "authors": [
   "Honglu He",
   "Jacob Laufer",
   "Zhiwu Zheng",
   "David Elkan-gonzalez",
   "Raman Goyal",
   "Xinyi Li",
   "Su Lu",
   "Mishek Musa",
   "Berke Saat",
   "Nicolas Tan",
   "Colm Prendergast"
  ],
  "author_count": 11,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Results indicate that the correct learned policy can outperform classical vision and control robotic methods in robustness, generalization, and deployment efficiency, justifying a shift toward scalable robotic automation for high up-time industrial environments.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Honglu He",
    "id": "2111909963",
    "h_index": 6,
    "papers": 18
   },
   {
    "name": "Jacob Laufer",
    "id": "2449999836",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Zhiwu Zheng",
    "id": "80081362",
    "h_index": 7,
    "papers": 18
   },
   {
    "name": "David Elkan-gonzalez",
    "id": "2450000013",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Raman Goyal",
    "id": "2290686022",
    "h_index": 2,
    "papers": 13
   },
   {
    "name": "Xinyi Li",
    "id": "2450231486",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Su Lu",
    "id": "2450176215",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Mishek Musa",
    "id": "1914359673",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Berke Saat",
    "id": "1393659694",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Nicolas Tan",
    "id": "2449999982",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Colm Prendergast",
    "id": "1404422097",
    "h_index": 3,
    "papers": 5
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "imitation-diffusion",
   "spatial-3d",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.14021v2",
  "pdf_url": "https://arxiv.org/pdf/2607.14021v2",
  "html_url": "https://arxiv.org/html/2607.14021v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.14005",
  "slug": "m-text-4-world-a-multi-view-multimodal-driving-world-model-for-interac",
  "title": "M$^\\text{4}$World: A Multi-view Multimodal Driving World Model for Interactive Object Manipulation and Minute-long Streaming",
  "abstract": "Driving-world generation has emerged as a core capability for scalable autonomous-driving simulation, yet existing methods remain limited in object-level controllability and long-horizon stability. We present M$^\\text{4}$World, a Multi-view and Multimodal generative driving world model that synthesizes future surround-view video streams and synchronized LiDAR scans while supporting interactive object Manipulation and stable Minute-long streaming. Fine-grained object manipulation is realized through a flexible conditioning interface that supports explicit control over both the spatial layout and visual appearance of individual objects. Stable minute-long streaming, on the other hand, is achieved through a multi-stage training framework that enables online causal generation in only four denoising steps while maintaining coherent world dynamics throughout extended rollouts. Building on these components, we introduce an efficient few-clip post-training as well as a suite of visual reference-conditioned generation models, preserving general generation ability while allowing rare-case customization for long-tail controllability. To assess controllability beyond realism, we further introduce an automated VLM-based judging pipeline that evaluates scene-level condition adherence, view-wise object controllability, and cross-view object consistency. Comprehensive experiments show that M$^\\text{4}$World consistently delivers high generation quality, precise controllability, and stable minute-long streaming. Together with downstream long-tail augmentation and scene editing, these results demonstrate the potential of M$^\\text{4}$World for controllable, scalable driving simulation.",
  "published": "2026-07-15",
  "updated": "2026-07-15",
  "year": "2026",
  "authors": [
   "Ke Cheng",
   "Hanqiao Ye",
   "Lei Shi",
   "Yahui Liu",
   "Yunhan Shen",
   "Jingtao Dong",
   "Zhenke Wang",
   "Wenxuan Ao",
   "Weixiang Xu",
   "Kaining Huang",
   "Shuhan Shen"
  ],
  "author_count": 11,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "M$^\\text{4}$World, a Multi-view and Multimodal generative driving world model that synthesizes future surround-view video streams and synchronized LiDAR scans while supporting interactive object Manipulation and stable Minute-long streaming is presented.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ke Cheng",
    "id": "1998966851",
    "h_index": 8,
    "papers": 19
   },
   {
    "name": "Hanqiao Ye",
    "id": "2111973101",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Lei Shi",
    "id": "2261687709",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Yahui Liu",
    "id": "2367201392",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Yunhan Shen",
    "id": "2450222855",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Jingtao Dong",
    "id": "2450224601",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Zhenke Wang",
    "id": "2451009296",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Wenxuan Ao",
    "id": "2181320302",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Weixiang Xu",
    "id": "2348984334",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Kai Huang",
    "id": "2448901359",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Shuhan Shen",
    "id": "2296220648",
    "h_index": 4,
    "papers": 15
   }
  ],
  "comment": "24 pages, 13 figures",
  "topics": [
   "world-models"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.14005v1",
  "pdf_url": "https://arxiv.org/pdf/2607.14005v1",
  "html_url": "https://arxiv.org/html/2607.14005v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.13960",
  "slug": "gigaworld-policy-0-5-a-faster-and-stronger-wam-empowered-by-autoresear",
  "title": "GigaWorld-Policy-0.5: A Faster and Stronger WAM Empowered by AutoResearch",
  "abstract": "World Action Models (WAMs) improve robot policy learning by jointly modeling actions and future visual observations, using future scene evolution as dense supervision for physically grounded action generation. However, a common design in existing WAMs is to explicitly generate future videos at inference time, incurring substantial computational overhead and hindering real-time closed-loop deployment. GigaWorld-Policy addresses this issue with an action-centered formulation, where future visual dynamics are used during training while action-only decoding is used at inference time. Building upon this framework, we present GigaWorld-Policy-0.5, an enhanced action-centered WAM designed for more efficient robot control. During pretraining, GigaWorld-Policy-0.5 adopts a mixed Action-Conditioned World Modeling (AC-WM) and WAM training strategy. This strengthens the coupling between visual dynamics and robot actions and improves the transferability of action representations for downstream policy learning. For efficient inference, GigaWorld-Policy-0.5 introduces a Mixture-of-Transformers architecture that separates visual dynamics modeling and action generation into specialized experts, reducing active computation during action-only inference and achieving 85 ms inference latency on a local RTX 4090 setup. In addition, we employ an agent-based AutoResearch pipeline to systematically search training configurations, enabling more efficient identification of optimal experimental setups while reducing the time and manual intervention required for hyperparameter tuning. Experiments and ablations show that GigaWorld-Policy-0.5 preserves the training benefits of future visual dynamics while improving inference efficiency for robot control.",
  "published": "2026-07-15",
  "updated": "2026-07-17",
  "year": "2026",
  "authors": [
   " GigaWorld Team",
   "Angen Ye",
   "Angyuan Ma",
   "Boyuan Wang",
   "Chaojun Ni",
   "Fangzheng Ye",
   "Guan Huang",
   "Guo Li",
   "Guosheng Zhao",
   "Haodong Yan",
   "Hengtao Li",
   "Jiwen Lu",
   "Kai Wang",
   "Mingming Yu",
   "Qitang Hu",
   "Qiuping Deng",
   "Songling Liu",
   "Xiaoyu Tian",
   "Xiaofeng Wang",
   "Xinyu Zhou",
   "Xiuwei Xu",
   "Xinze Chen",
   "Yang Wang",
   "Yejun Zeng",
   "Yifan Chang",
   "Yun Ye",
   "Zhenyu Wu",
   "Zhanqian Wu",
   "Zheng Zhu"
  ],
  "author_count": 29,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "GigaWorld-Policy-0.5 preserves the training benefits of future visual dynamics while improving inference efficiency for robot control, and introduces a Mixture-of-Transformers architecture that separates visual dynamics modeling and action generation into specialized experts.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "GigaWorld Team",
    "id": "2394171060",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Angen Ye",
    "id": "2373031539",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Angyuan Ma",
    "id": "2378923451",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Boyuan Wang",
    "id": "2295591878",
    "h_index": 11,
    "papers": 23
   },
   {
    "name": "Chaojun Ni",
    "id": "2326295492",
    "h_index": 12,
    "papers": 29
   },
   {
    "name": "Fangzheng Ye",
    "id": "2450000056",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Guan Huang",
    "id": "2256954306",
    "h_index": 16,
    "papers": 42
   },
   {
    "name": "Guo Li",
    "id": "2184570319",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Guosheng Zhao",
    "id": "2292092480",
    "h_index": 14,
    "papers": 27
   },
   {
    "name": "Haodong Yan",
    "id": "2321603038",
    "h_index": 5,
    "papers": 16
   },
   {
    "name": "Hengtao Li",
    "id": "2218230392",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "Jiwen Lu",
    "id": "2243332262",
    "h_index": 6,
    "papers": 16
   },
   {
    "name": "Kai Wang",
    "id": "2382828533",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Mingming Yu",
    "id": "2387333550",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Qitang Hu",
    "id": "2450003998",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Qiuping Deng",
    "id": "2391543086",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Songlin Liu",
    "id": "2258678044",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Xiaoyu Tian",
    "id": "2149326880",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Xiaofeng Wang",
    "id": "2349399535",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Xinyu Zhou",
    "id": "2395637062",
    "h_index": 2,
    "papers": 13
   },
   {
    "name": "Xiuwei Xu",
    "id": "2158440998",
    "h_index": 14,
    "papers": 42
   },
   {
    "name": "Xinze Chen",
    "id": "1391211885",
    "h_index": 17,
    "papers": 34
   },
   {
    "name": "Yang Wang",
    "id": "2373745208",
    "h_index": 6,
    "papers": 14
   },
   {
    "name": "Yejun Zeng",
    "id": "2388179036",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Yifan Chang",
    "id": "2349397350",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Yun Ye",
    "id": "2384823293",
    "h_index": 6,
    "papers": 16
   },
   {
    "name": "Zhenyu Wu",
    "id": "2328403813",
    "h_index": 9,
    "papers": 22
   },
   {
    "name": "Zhanqian Wu",
    "id": "2366154677",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Zheng Zhu",
    "id": "2109516240",
    "h_index": 17,
    "papers": 50
   }
  ],
  "comment": "project page: https://open-gigaai.github.io/giga-world-policy/",
  "topics": [
   "world-models",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.13960v3",
  "pdf_url": "https://arxiv.org/pdf/2607.13960v3",
  "html_url": "https://arxiv.org/html/2607.13960v3",
  "code_url": "https://open-gigaai.github.io/giga-world-policy/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.3
 },
 {
  "id": "2607.13938",
  "slug": "discriminative-barrier-functions-for-safe-adversarial-imitation-learni",
  "title": "Discriminative Barrier Functions for Safe Adversarial Imitation Learning from Observation",
  "abstract": "Inverse Reinforcement Learning (IRL) algorithms are powerful tools for learning from and generalizing expert demonstrations, but they often rely on unconstrained exploration, rendering them unsafe for real-world deployment. Meanwhile, Control Barrier Functions (CBFs) can guarantee the safety of control systems, but the analytical design of CBFs can be time-consuming and esoteric. In this work, we address these limitations jointly by constraining reward function candidacy during IRL to the space of CBFs, yielding a formulation that exhibits safe online control with continuous experiential improvement. Crucially, this framework enables the data-driven recovery of barrier functions directly from unlabeled expert observations. We demonstrate that the recovered barrier function is robust to unsafe states entirely absent from the expert data. Furthermore, we benchmark our method against standard IRL baselines in a simulated navigation environment, demonstrating improved safety performance. Finally, we investigate the trade-offs of planning-based versus policy-based IRL methods across both simulation and a real world obstacle avoidance task.",
  "published": "2026-07-15",
  "updated": "2026-07-15",
  "year": "2026",
  "authors": [
   "Anubhav Vishwakarma",
   "Bhaumik Mehta",
   "Caleb Hsu",
   "Byron Boots",
   "Karen Leung",
   "Tyler Han"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work constraining reward function candidacy during IRL to the space of CBFs yields a formulation that exhibits safe online control with continuous experiential improvement, and demonstrates that the recovered barrier function is robust to unsafe states entirely absent from the expert data.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Anubhav Vishwakarma",
    "id": "2140362109",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Bhaumik Mehta",
    "id": "2344832766",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Caleb Hsu",
    "id": "2450163444",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Byron Boots",
    "id": "2292033603",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Karen Leung",
    "id": "144579384",
    "h_index": 17,
    "papers": 34
   },
   {
    "name": "Tyler Han",
    "id": "2267489132",
    "h_index": 4,
    "papers": 9
   }
  ],
  "comment": "20 pages, 5 figures",
  "topics": [
   "imitation-diffusion",
   "rl-control",
   "navigation",
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.13938v1",
  "pdf_url": "https://arxiv.org/pdf/2607.13938v1",
  "html_url": "https://arxiv.org/html/2607.13938v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.13799",
  "slug": "vision-based-obstacle-separation-for-strawberry-harvesting-in-clusters",
  "title": "Vision-Based Obstacle Separation for Strawberry Harvesting in Clusters Using Hierarchical Reinforcement Learning",
  "abstract": "Selective harvesting in clustered strawberry environments is challenging because ripe fruits are often occluded by surrounding unripe fruits, making direct grasping unreliable. To address this problem, this paper proposes a hierarchical reinforcement learning framework, termed VGPA, which integrates a vision-guided decision mechanism and a Progressive Adaptive Exploration Strategy (PAES) for vision-based obstacle separation and harvesting. The task was decomposed into two sequential stages: obstacle separation and target grasping. At the high level, the vision-guided mechanism improved option selection and accelerated policy convergence. At the low level, PAES improved exploration efficiency and training stability during continuous control learning. In simulation experiments, the learned policy achieved a success rate of 96.7%. In addition, sim-to-real transfer experiments on a self-developed parallel robot showed that the proposed method achieved success rates ranging from 71.7% to 88.3%, outperforming direct picking while requiring only 1.22~s more average harvesting time. These results verified the effectiveness, generalization ability, and practical potential of the proposed method for robotic harvesting in complex clustered environments.",
  "published": "2026-07-15",
  "updated": "2026-07-15",
  "year": "2026",
  "authors": [
   "Teng Li",
   "Hanfei Shi",
   "Chunjiang Zhao",
   "Ya Xiong"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The proposed hierarchical reinforcement learning framework, termed VGPA, which integrates a vision-guided decision mechanism and a Progressive Adaptive Exploration Strategy for vision-based obstacle separation and harvesting outperforms direct picking while requiring only 1.22~s more average harvesting time.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Teng Li",
    "id": "2450151755",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Hanfei Shi",
    "id": "2450154644",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Chunjiang Zhao",
    "id": "2389548755",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Ya Xiong",
    "id": "2313624230",
    "h_index": 3,
    "papers": 11
   }
  ],
  "comment": "Accepted to IROS 2026",
  "topics": [
   "dexterous-manipulation",
   "sim2real",
   "rl-control",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.13799v1",
  "pdf_url": "https://arxiv.org/pdf/2607.13799v1",
  "html_url": "https://arxiv.org/html/2607.13799v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2607.13768",
  "slug": "the-nonsmooth-impact-direction-nsid-of-robotic-systems",
  "title": "The Nonsmooth Impact Direction (NSID) of Robotic Systems",
  "abstract": "Collisions of rigid-link robots and rigid environments are often modeled as instantaneous events. Under this idealization, the impact forces become impulsive and the system velocities nonsmooth. In this work, we systematically analyze pre- and post-impact velocities focusing on what we refer to as the nonsmooth impact direction (NSID). We show that it is a characteristic direction of a robotic impact and largely independent of contact properties. The results are directly applicable to large classes of backdrivable robotic systems with rigid links. We address particularities of systems with nonelastic and flexible joints, unconstrained as well as constrained systems. Further, we show that the approach direction w.r.t the NSID sets the direction of the impulsive force in frictional, inelastic impacts. The comprehensive theoretical analysis of this work supported by an experimental validation may serve as a foundation for future planning and control algorithms for various robotic impact applications. These can include humanoid locomotion on a slippery surface or repetitive hammering.",
  "published": "2026-07-15",
  "updated": "2026-07-15",
  "year": "2026",
  "authors": [
   "Annika Kirner",
   "Christian Ott"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "T-RO",
  "venue_source": "semantic-scholar",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "10.1109/tro.2026.3723945",
  "oa_pdf": "https://doi.org/10.1109/tro.2026.3723945",
  "s2_authors": [
   {
    "name": "Annika Kirner",
    "id": "2187482789",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Christian Ott",
    "id": "2308321145",
    "h_index": 1,
    "papers": 6
   }
  ],
  "comment": "18 pages, 17 figures, under review",
  "topics": [
   "humanoids"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.13768v1",
  "pdf_url": "https://arxiv.org/pdf/2607.13768v1",
  "html_url": "https://arxiv.org/html/2607.13768v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2607.13674",
  "slug": "wave-stereo-warp-aligned-volume-encoding-for-stereo-matching",
  "title": "WAVE-Stereo: Warp-Aligned Volume Encoding for Stereo Matching",
  "abstract": "Existing iterative stereo matching methods primarily adopt two types of correspondence representation: explicit matching search via correlation volumes and local residual refinement via warped features, yet the two remain separately modeled. We propose WAVE-Stereo, built on a core insight: correlation volumes and feature warping provide complementary matching cues. \\textbf{GeoWarp Correspondence Encoder (GWCE)} encodes matching search, residual alignment, and disparity prior in parallel at the ConvGRU input. To mitigate matching degradation in textureless regions, we propose \\textbf{Periodic Global Context Propagation (PGCP)}, which propagates global spatial information in a periodic manner. On five real-world benchmarks -- Middlebury, ETH3D, KITTI 2012, KITTI 2015, and Booster -- WAVE-Stereo achieves competitive zero-shot generalization accuracy without any external foundation model prior, achieving 3.18\\% D1-all on KITTI 2015, 4.42\\% Bad-2.0 on Booster, and 66ms real-time inference, striking a favorable balance between accuracy and efficiency. Our code is available at https://github.com/yamanoko-do/WAVE-Stereo.",
  "published": "2026-07-15",
  "updated": "2026-07-15",
  "year": "2026",
  "authors": [
   "Zehan Liu",
   "Yage He",
   "Xianwu Gong"
  ],
  "author_count": 3,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "WAVE-Stereo is proposed, built on a core insight: correlation volumes and feature warping provide complementary matching cues, and achieves competitive zero-shot generalization accuracy without any external foundation model prior.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zehan Liu",
    "id": "2372253877",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Yage He",
    "id": "2373550479",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Xianwu Gong",
    "id": "2193269",
    "h_index": 2,
    "papers": 14
   }
  ],
  "comment": "",
  "topics": [
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.13674v1",
  "pdf_url": "https://arxiv.org/pdf/2607.13674v1",
  "html_url": "https://arxiv.org/html/2607.13674v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.13653",
  "slug": "exploratory-communicative-and-deployable-vision-driven-embodied-agents",
  "title": "Exploratory, Communicative, and Deployable: Vision-Driven Embodied Agents for Open-World Mobile Manipulation",
  "abstract": "Real-world deployment of embodied agents requires active exploration, visual grounding, and interactive intent disambiguation. However, existing frameworks often rely on privileged simulator states or assume complete instructions, bypassing realistic deployment challenges. To bridge this gap, we present REAL, an agentic framework for open-world mobile manipulation. REAL establishes sim-to-real-consistent environment APIs without oracle perception and integrates a simulated user to enable human-in-the-loop interaction. Within this environment, we design diverse task compositions to drive data collection, supervised fine-tuning, and online reinforcement learning, systematically optimizing agent performance. To comprehensively evaluate this approach, we introduce REAL-Bench, a benchmark spanning 241 tasks across active exploration, visual distraction, articulated manipulation, and interactive disambiguation. Experimental results demonstrate that our trained agent outperforms leading commercial closed-source VLMs on interactive tasks with a 56.9% success rate. Further empirical analysis reveals that our hierarchical training pipeline successfully aligns the model's tool-use capabilities while maintaining robust open-vocabulary reasoning under extended exploration horizons. Finally, we deploy and evaluate our framework on a physical dual-arm mobile robot, where it achieves a 78.3% end-to-end success rate over 60 real-world episodes. These physical trials demonstrate robust zero-shot transferability to unseen household scenarios, validating that our sim-to-real-consistent design successfully bridges the reality gap for long-horizon mobile manipulation. Code is available at https://github.com/InternRobotics/REAL.",
  "published": "2026-07-15",
  "updated": "2026-07-27",
  "year": "2026",
  "authors": [
   "Boyu Mi",
   "Mengchen Ma",
   "Yifei Yao",
   "Xing Gao",
   "Junting Chen",
   "Yangzi Li",
   "Zihou Zhu",
   "Guohao Li",
   "Zhenfei Yin",
   "Tai Wang",
   "Yao Mu",
   "Jiangmiao Pang",
   "Hanqing Wang"
  ],
  "author_count": 13,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "ECCV 2026",
  "venue_source": "arxiv-comment",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "This work presents REAL, an agentic framework for open-world mobile manipulation, which establishes sim-to-real-consistent environment APIs without oracle perception and integrates a simulated user to enable human-in-the-loop interaction.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Boyu Mi",
    "id": "2257002702",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Mengchen Ma",
    "id": "2450001720",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yifei Yao",
    "id": "2321330037",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Xing Gao",
    "id": "2445502394",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Junting Chen",
    "id": "2284997004",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Yangzi Li",
    "id": "2450174664",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Zihou Zhu",
    "id": "2444936929",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Guohao Li",
    "id": "2328310852",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Zhenfei Yin",
    "id": "2348879787",
    "h_index": 9,
    "papers": 30
   },
   {
    "name": "Tai Wang",
    "id": "2359108866",
    "h_index": 11,
    "papers": 26
   },
   {
    "name": "Yao Mu",
    "id": "2348161293",
    "h_index": 2,
    "papers": 18
   },
   {
    "name": "Jiangmiao Pang",
    "id": "2377561990",
    "h_index": 10,
    "papers": 28
   },
   {
    "name": "Hanqing Wang",
    "id": "2311307527",
    "h_index": 11,
    "papers": 27
   }
  ],
  "comment": "Accepted to ECCV 2026. 57 pages. Code available at https://github.com/InternRobotics/REAL",
  "topics": [
   "sim2real",
   "rl-control",
   "navigation",
   "data-teleop",
   "hri"
  ],
  "orgs": [
   "Shanghai AI Lab"
  ],
  "abs_url": "https://arxiv.org/abs/2607.13653v2",
  "pdf_url": "https://arxiv.org/pdf/2607.13653v2",
  "html_url": "https://arxiv.org/html/2607.13653v2",
  "code_url": "https://github.com/InternRobotics/REAL",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.3
 },
 {
  "id": "2607.13622",
  "slug": "design-modeling-and-experimental-validation-of-a-miniature-hybrid-unde",
  "title": "Design, Modeling and Experimental Validation of a Miniature Hybrid Underwater Glider With Large-Range Foldable Deflectable Wings",
  "abstract": "Miniature hybrid underwater gliders have attracted increasing attention for long-endurance ocean observation and confined-space inspection. Large-range wing reconfiguration offers a promising yet largely unexplored approach for simultaneously enhancing maneuverability and shape adaptability in constrained underwater environments. However, such morphing introduces substantial challenges in mechanical integration, dynamic modeling, and hydrodynamic characterization. This paper presents FoDeGlider, a miniature hybrid underwater glider equipped with two independently actuated wings capable of large-range folding and deflection. To capture configuration-dependent variations in mass distribution, center-of-geometry location, and hydrodynamic loading, a multibody dynamics model is developed by treating wing configuration as a structural variable. A composite rigid body algorithm (CRBA)-based projection formulates the composite inertia, wrench transformations, and component-level hydrodynamics into a unified Fossen-form dynamic model applicable to arbitrary wing configurations. A sequential parameter-identification framework is further proposed to estimate fuselage and wing hydrodynamic coefficients, resulting in an open benchmark dataset for model identification and validation. Extensive experiments are conducted, the results of which demonstrate accurate dynamic modeling and parameter identification across diverse morphing configurations. Gate traversal experiments further validate FoDeGlider's ability to actively reconfigure its morphology during locomotion, enabling enhanced navigation in confined underwater environments.",
  "published": "2026-07-15",
  "updated": "2026-07-15",
  "year": "2026",
  "authors": [
   "Yongjian Zhu",
   "Yusen Tao",
   "Feitian Zhang"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yongjian Zhu",
    "id": "2157583735",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Yusen Tao",
    "id": "2449983732",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Feitian Zhang",
    "id": "145979419",
    "h_index": 15,
    "papers": 45
   }
  ],
  "comment": "11 pages, 8 figures, journal",
  "topics": [
   "humanoids",
   "navigation",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.13622v1",
  "pdf_url": "https://arxiv.org/pdf/2607.13622v1",
  "html_url": "https://arxiv.org/html/2607.13622v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.13595",
  "slug": "active-trust-management-for-successful-human-robot-teaming-moving-from",
  "title": "Active Trust Management for Successful Human-Robot Teaming: Moving from a Trust Repair to a Trust Satisficing Perspective",
  "abstract": "Integrating mobile robots into human teams promises significant capability improvements for tasks such as searching hazardous environments. Unlike existing teleoperated robots, future robot systems will increasingly be endowed with some level of artificial intelligence (AI), giving them a degree of autonomy in how they pursue mission goals. This autonomy could make a human-agent (robot) team more effective but also put inter-agent trust under strain if robots make a mistake, or (appear to) pursue task priorities that conflict with the team's best interest. During a mission, agents' trust states are anticipated to vary according to the situation as understood by each teammate (trustor). If component-level (agent) or system-level trust falls below sufficient levels for cooperative tasks to be completed, it could critically affect mission success . We argue that active trust management will be an important precondition for the success of human-robot teams (HRTs, a subcategory of human-agent teams with embodied agents), especially in dynamic, high-risk environments. We present a trust satisficing perspective which acknowledges and attempts to account for the fluctuating, multi-faceted, and context-dependent nature of trust and trust requirements even under normal operating conditions. Our outline of a trust management framework for human-robot teaming includes online measurement of proxy metrics for trust, closed-loop adaptation of robot behavior, and variable autonomy to give space for human responsibility in situations requiring value judgements. We refer to a recent experimental exploration of 'swift trust' and a novel behavioral trust metric for HRT, and we highlight issues for further investigation.",
  "published": "2026-07-15",
  "updated": "2026-07-15",
  "year": "2026",
  "authors": [
   "Nicola Webb",
   "Edmund R. Hunt"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "It is argued that active trust management will be an important precondition for the success of human-robot teams (HRTs, a subcategory of human-agent teams with embodied agents), especially in dynamic, high-risk environments.",
  "doi": "10.1108/S1534-085620260000021005",
  "oa_pdf": "https://arxiv.org/pdf/2607.13595",
  "s2_authors": [
   {
    "name": "Nicola Webb",
    "id": "2316432060",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Edmund R. Hunt",
    "id": "2238951540",
    "h_index": 3,
    "papers": 12
   }
  ],
  "comment": "",
  "topics": [
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.13595v1",
  "pdf_url": "https://arxiv.org/pdf/2607.13595v1",
  "html_url": "https://arxiv.org/html/2607.13595v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.13524",
  "slug": "colmar-cooperative-view-policy-learning-for-multi-agent-active-3d-reco",
  "title": "COLMAR: Cooperative View Policy Learning for Multi-Agent Active 3D Reconstruction",
  "abstract": "Active 3D reconstruction requires selecting informative viewpoints under limited sensing budgets. In multi-agent settings, coordination inefficiencies such as redundant observations and spatial clustering can significantly reduce reconstruction quality. We present COLMAR, a cooperative view policy learning framework for multi-agent active 3D reconstruction. COLMAR formulates viewpoint allocation as a shared policy optimization over map-centric observations and introduces a reconstruction-aware objective that promotes overlap-aware coverage, team-level discovery, and collision-safe exploration. Dense feedback derived from incremental reconstruction updates aligns exploration behavior with downstream geometric quality. The policy is trained using parameter-sharing Proximal Policy Optimization (PPO) with independent per-agent action selection at deployment, conditioned on a fused team map and without inter-agent message passing for decision making. Selected viewpoints are then reconstructed with 3D Gaussian Splatting (3DGS) for high-fidelity photometric evaluation. Experiments on GLEAM and Replica demonstrate consistent improvements over heuristic and non-cooperative baselines, achieving up to 54% higher reconstruction accuracy and 49% greater coverage under matched sensing budgets.",
  "published": "2026-07-15",
  "updated": "2026-07-15",
  "year": "2026",
  "authors": [
   "Phu Pham",
   "Damon Conover",
   "Aniket Bera"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Phu-Cuong Pham",
    "id": "2140639881",
    "h_index": 6,
    "papers": 16
   },
   {
    "name": "Damon Conover",
    "id": "2322792986",
    "h_index": 3,
    "papers": 15
   },
   {
    "name": "Aniket Bera",
    "id": "2286848568",
    "h_index": 5,
    "papers": 30
   }
  ],
  "comment": "8 pages, 5 figures, IROS 2026",
  "topics": [
   "rl-control",
   "spatial-3d",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.13524v1",
  "pdf_url": "https://arxiv.org/pdf/2607.13524v1",
  "html_url": "https://arxiv.org/html/2607.13524v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2607.13497",
  "slug": "layered-risk-mapping-for-autonomous-patient-transport-in-expeditionary",
  "title": "Layered Risk Mapping for Autonomous Patient Transport in Expeditionary Medical Facilities",
  "abstract": "In expeditionary medical facilities, routine patient transport imposes a compounding burden of personal protective equipment consumption, staff diversion, and elevated infection risk that becomes unsustainable under surge conditions. While autonomous wheelchairs could absorb this operational load, the safety-critical nature of patient transit within these highly unstructured and dynamic environments poses complex navigational challenges. To address this, we present a layered risk mapping framework that fuses four heterogeneous environmental hazards (terrain slope, static and dynamic obstacles, and semantic traversability) into a unified probabilistic cost surface via a Noisy-OR fusion model. In a paired Monte-Carlo evaluation, risk-informed fusion reduces collision rates from over 73% to under 32% and more than doubles obstacle clearance relative to a risk-unaware baseline. Additionaly, Noisy-OR achieves the highest clearance to obstacles and the lowest conditional peak risk across all tested hazard densities. We further validate the framework on a commercial powered wheelchair across three representative mission profiles in indoor and outdoor deployments, demonstrating that this architecture successfully meets the planning requirements of this previously unaddressed operational regime.",
  "published": "2026-07-15",
  "updated": "2026-07-15",
  "year": "2026",
  "authors": [
   "Lorena Maria Genua",
   "Sarvesh Prajapati",
   "Damla Leblebicioglu",
   "Ta\u015fk\u0131n Pad\u0131r"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A layered risk mapping framework that fuses four heterogeneous environmental hazards into a unified probabilistic cost surface via a Noisy-OR fusion model and achieves the highest clearance to obstacles and the lowest conditional peak risk across all tested hazard densities is presented.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "L. Genua",
    "id": "2273992508",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Sarvesh Prajapati",
    "id": "2287922847",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Damla Leblebicioglu",
    "id": "2143199297",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Tacskin Padir",
    "id": "1490886634",
    "h_index": 1,
    "papers": 2
   }
  ],
  "comment": "",
  "topics": [
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.13497v1",
  "pdf_url": "https://arxiv.org/pdf/2607.13497v1",
  "html_url": "https://arxiv.org/html/2607.13497v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.13472",
  "slug": "egohtr-egocentric-4d-demonstrations-of-human-terrain-traversal",
  "title": "EgoHTR: Egocentric 4D Demonstrations of Human Terrain Traversal",
  "abstract": "Deploying humanoid robots in unstructured terrain remains an open problem. While classic reinforcement learning struggles with the sheer complexity of real-world interactions, more promising methods leveraging human priors remain limited to models lacking contextual awareness. The restricted motion synthesis is a direct consequence of existing dataset pipelines failing to capture human-scene sequences in challenging environments. To bridge this gap between humanoid learning and scene reconstruction, we introduce the Egocentric Human-Terrain Reconstruction (EgoHTR) dataset. We develop and open-source a reconstruction pipeline capturing 55 scene-aligned 4D human motion sequences in diverse, complex environments using a multi-sensor setup of egocentric wearables and a portable 3D scanner. The resulting dataset comprises over 150k frames, which we evaluate against motion-capture ground truth, demonstrating state-of-the-art accuracy and establishing a rigorous benchmark for human motion analysis and synthesis. Further, we leverage this data to train perceptive locomotion policies, demonstrating hardware deployment on a Unitree G1 for reconstructed reference motions. Our pipeline enables community-driven dataset extensions and factors the problem to help researchers build foundational, context-aware robots that reliably traverse uneven terrain.",
  "published": "2026-07-15",
  "updated": "2026-07-15",
  "year": "2026",
  "authors": [
   "Alex Brandes",
   "Haig Conti Georges Sajelian",
   "Manthan Patel",
   "Dominik Hollidt",
   "Chenhao Li",
   "Matthias Heyrman",
   "Oliver Hausdoerfer",
   "Manuel Kaufmann",
   "Xi Wang",
   "Jonas Frey",
   "Angela P. Schoellig",
   "Christian Holz",
   "Marc Pollefeys",
   "Marco Hutter"
  ],
  "author_count": 14,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work develops and open-source a reconstruction pipeline capturing 55 scene-aligned 4D human motion sequences in diverse, complex environments using a multi-sensor setup of egocentric wearables and a portable 3D scanner, and introduces the Egocentric Human-Terrain Reconstruction (EgoHTR) dataset.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Alex Brandes",
    "id": "2449957278",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Haig Conti Georges Sajelian",
    "id": "2449955147",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Manthan Patel",
    "id": "2289075975",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Dominik Hollidt",
    "id": "2345011935",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Chenhao Li",
    "id": "2306055264",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Matthias Heyrman",
    "id": "2397376662",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Oliver Hausdoerfer",
    "id": "2442801977",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Manuel Kaufmann",
    "id": "35090707",
    "h_index": 13,
    "papers": 26
   },
   {
    "name": "Xi Wang",
    "id": "2352959391",
    "h_index": 2,
    "papers": 11
   },
   {
    "name": "Jonas Frey",
    "id": "2249531943",
    "h_index": 10,
    "papers": 29
   },
   {
    "name": "Angela P. Schoellig",
    "id": "2321572233",
    "h_index": 4,
    "papers": 20
   },
   {
    "name": "Christian Holz",
    "id": "2319602671",
    "h_index": 6,
    "papers": 19
   },
   {
    "name": "Marc Pollefeys",
    "id": "2385471198",
    "h_index": 3,
    "papers": 17
   },
   {
    "name": "Marco Hutter",
    "id": "2340685198",
    "h_index": 6,
    "papers": 20
   }
  ],
  "comment": "Project webpage: https://egohtr.github.io",
  "topics": [
   "humanoids",
   "egocentric-data",
   "rl-control"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2607.13472v1",
  "pdf_url": "https://arxiv.org/pdf/2607.13472v1",
  "html_url": "https://arxiv.org/html/2607.13472v1",
  "code_url": "https://egohtr.github.io",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2607.13461",
  "slug": "joint-on-and-off-policy-learning-for-vision-and-language-navigation",
  "title": "Joint On-and-Off Policy Learning for Vision-and-Language Navigation",
  "abstract": "Vision-and-Language Navigation (VLN) necessitates an embodied agent to navigate in the physical world by adhering to natural language instructions. Recent advancements in Vision-Language Models (VLM) have propelled the development of VLM-based VLN methods with two predominant paradigms: (1) imitation learning (IL) on expert demonstrations, followed by the Dataset Aggregation (DAgger) algorithm to bolster error recovery capabilities; (2) reinforcement learning (RL) driven by verifiable rewards to enhance reasoning and exploration. A notable gap is the absence of integration between these two distinct paradigms. This paper introduces JOP-VLN, a novel VLN framework that synergistically combines off-policy imitation learning and on-policy exploration within a three-stage training pipeline. Initially, IL is employed on expert demonstrations to acquire basic navigation skills. Subsequently, the DAgger algorithm is utilized to generate heuristic exploration trajectories, which are then used for imitation learning to improve error recovery capabilities. Finally, a joint on-and-off policy learning framework is implemented, featuring high-entropy trajectory sampling to enhance RL training efficiency and an error-correction-prioritized trajectory sorting strategy for effective error correction. Extensive experiments demonstrate the efficacy of JOP-VLN, achieving success rates of 69.9% and 68.0% on the VLN-CE R2R and RxR benchmarks, respectively, setting a new state-of-the-art on R2R. Project page: https://qingrongh.github.io/JOP-VLN.",
  "published": "2026-07-15",
  "updated": "2026-07-15",
  "year": "2026",
  "authors": [
   "Qingrong He",
   "Lin Zhao",
   "Kevin Zheng",
   "Liang Lin"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "JOP-VLN is introduced, a novel VLN framework that synergistically combines off-policy imitation learning and on-policy exploration within a three-stage training pipeline, featuring high-entropy trajectory sampling to enhance RL training efficiency and an error-correction-prioritized trajectory sorting strategy for effective error correction.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Qin He",
    "id": "2409992744",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Lingqing Zhao",
    "id": "2350015947",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Kevin Zheng",
    "id": "2308486856",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Liang Lin",
    "id": "2316519420",
    "h_index": 7,
    "papers": 18
   }
  ],
  "comment": "Accepted by IROS 2026",
  "topics": [
   "imitation-diffusion",
   "rl-control",
   "navigation",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.13461v1",
  "pdf_url": "https://arxiv.org/pdf/2607.13461v1",
  "html_url": "https://arxiv.org/html/2607.13461v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2607.13455",
  "slug": "reverse-to-advance-teleoperation-cost-effective-hard-policy-learning-f",
  "title": "Reverse to Advance: Teleoperation-Cost Effective Hard Policy Learning from Reversed Easy Tasks",
  "abstract": "High-quality teleoperation datasets are costly to collect, particularly for hard tasks. We observe that many tasks exhibit directional asymmetry: completing the forward hard task is difficult, whereas reversing it by relaxing or disrupting the environment is comparatively easy. This suggests that reversed easy-task trajectories can serve as a scalable supervision signal for the hard task, reducing the cost of manual demonstration collection. However, reversed data can be noisy, and directly training on it may yield suboptimal policies. To enable largely automated acquisition and effective use of reversed data, we propose a teleoperation-cost effective framework for hard policy learning via temporal reversal of easy tasks, consisting of three key components: a closed-loop data collection pipeline that alternates between hard-task and easy-task policies to autonomously reset the environment and generate diverse trajectories; a hierarchical data refinement pipeline that temporally inverts easy-task rollouts and filters low-quality motion using kinematic priors and a critic-guided advantage filter; and an iterative policy learning method that trains the hard-task policy using both initial reversed easy-task demonstrations and the filtered reversed data in a continuous online learning loop. By combining automated collection, hierarchical refinement, and iterative learning, our method enables scalable, reliable training of complex, high-precision manipulation tasks. Across two simulated benchmarks and real-robot experiments, we demonstrate that our method improves hard-task success rates with higher data efficiency and more stable training compared to reversal-based and reinforcement-learning baselines, without requiring extensive hard-task teleoperation.",
  "published": "2026-07-15",
  "updated": "2026-07-15",
  "year": "2026",
  "authors": [
   "Qiyuan Qiao",
   "Ge Yuan",
   "Can Wang",
   "Dong Xu"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes a teleoperation-cost effective framework for hard policy learning via temporal reversal of easy tasks, and demonstrates that the method improves hard-task success rates with higher data efficiency and more stable training compared to reversal-based and reinforcement-learning baselines, without requiring extensive hard-task teleoperation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Qiyuan Qiao",
    "id": "2367270387",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Genlan Yuan",
    "id": "90322228",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Canye Wang",
    "id": "2199807142",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Dong Xu",
    "id": "2238157352",
    "h_index": 9,
    "papers": 23
   }
  ],
  "comment": "17 pages, 17 figures",
  "topics": [
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.13455v1",
  "pdf_url": "https://arxiv.org/pdf/2607.13455v1",
  "html_url": "https://arxiv.org/html/2607.13455v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.13451",
  "slug": "learning-physics-guided-residual-dynamics-for-deformable-object-simula",
  "title": "Learning Physics-Guided Residual Dynamics for Deformable Object Simulation",
  "abstract": "Simulating deformable objects is essential for a wide range of robotic manipulation applications, yet accurately predicting their dynamics remains challenging. We propose Physics-Guided Residual Dynamics (PGRD), a hybrid simulation framework that combines the advantages of physics-based and learning-based approaches. Specifically, PGRD combines an optimizable spring-mass simulator as a backbone with a learned neural network that predicts residual corrections to the physics-based predictions. We adopt a velocity-based formulation to ensure stable simulation and a sliding-window transformer architecture to capture temporal dependencies. We show that PGRD produces more accurate results than both purely physics-based and learning-based methods on a set of diverse real-world deformable objects. We further demonstrate the utility of PGRD in two applications: manipulation planning via Model Predictive Control, including a language-conditioned setting with a generated goal image; and interactive simulation via action-conditioned video prediction by 3D Gaussian Splatting.",
  "published": "2026-07-15",
  "updated": "2026-07-15",
  "year": "2026",
  "authors": [
   "Shivansh Patel",
   "Kaifeng Zhang",
   "Sanjay Pokkali",
   "Svetlana Lazebnik",
   "Yunzhu Li"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "This work proposes Physics-Guided Residual Dynamics (PGRD), a hybrid simulation framework that combines the advantages of physics-based and learning-based approaches, and combines an optimizable spring-mass simulator as a backbone with a learned neural network that predicts residual corrections to the physics-based predictions.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Shivansh Patel",
    "id": "2403572295",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Kai Zhang",
    "id": "2274105131",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Sanjay Pokkali",
    "id": "2410288979",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Svetlana Lazebnik",
    "id": "2267723274",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Yunzhu Li",
    "id": "2310656658",
    "h_index": 4,
    "papers": 11
   }
  ],
  "comment": "Website: https://pgrd-robot.github.io/",
  "topics": [
   "world-models",
   "sim2real",
   "rl-control",
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.13451v1",
  "pdf_url": "https://arxiv.org/pdf/2607.13451v1",
  "html_url": "https://arxiv.org/html/2607.13451v1",
  "code_url": "https://pgrd-robot.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.3
 },
 {
  "id": "2607.13429",
  "slug": "generalizable-vla-finetuning-via-representation-anchoring-and-language",
  "title": "Generalizable VLA Finetuning via Representation Anchoring and Language-Action Alignment",
  "abstract": "Finetuning a pretrained vision-language model (VLM) on robot demonstrations via behavior cloning (BC) has become the standard recipe for vision-language-action (VLA) policies. However, BC finetuning progressively overwrites the pretrained representations that support visual and semantic generalization. Co-training on web image-text data, a common remedy, does not prevent this; it applies language and action losses to separate observations, leaving VLAs with language-action misalignment that standard manipulation benchmarks do not expose. We propose Anchor-Align, which augments BC with two objectives: Vision-Language Anchoring distills layer-wise representations from a frozen VLM copy to prevent this drift, while Language-Action Alignment converts each action target into a discrete motion-direction label and jointly trains language and action prediction on the same robot observation. On a physical xArm7 robot, across two widely used VLA architectures, Anchor-Align improves real-robot success on both (28% to 54% and 37% to 60%). At scale in simulation, we demonstrate consistent improvements on OOD perturbations, perceptual robustness, and long-horizon control across LIBERO-PRO, LIBERO-Plus, and CALVIN, respectively, suggesting that preserving pretrained representations and effective action learning are not fundamentally at odds. Project page: anchoralignvla.github.io",
  "published": "2026-07-15",
  "updated": "2026-07-15",
  "year": "2026",
  "authors": [
   "Dwip Dalal",
   "Shivansh Patel",
   "Chahit Jain",
   "Jeonghwan Kim",
   "Utkarsh Mishra",
   "Alex Baratian",
   "Hyeonjeong Ha",
   "Heng Ji",
   "Svetlana Lazebnik",
   "Unnat Jain"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "Anchor-Align is proposed, which augments BC with two objectives: Vision-Language Anchoring distills layer-wise representations from a frozen VLM copy to prevent this drift, and Language-Action Alignment converts each action target into a discrete motion-direction label and jointly trains language and action prediction on the same robot observation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Dwip Dalal",
    "id": "2213309144",
    "h_index": 6,
    "papers": 19
   },
   {
    "name": "Shivansh Patel",
    "id": "152264213",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Chahit Jain",
    "id": "2449954586",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Jeonghwan Kim",
    "id": "2116930037",
    "h_index": 8,
    "papers": 31
   },
   {
    "name": "Utkarsh Mishra",
    "id": "2385470372",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Alex Baratian",
    "id": "2449954387",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Hyeonjeong Ha",
    "id": "2347037972",
    "h_index": 5,
    "papers": 18
   },
   {
    "name": "Heng Ji",
    "id": "2375307612",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Svetlana Lazebnik",
    "id": "2267723274",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Unnat Jain",
    "id": "2387697720",
    "h_index": 2,
    "papers": 9
   }
  ],
  "comment": "Code: https://github.com/dwipddalal/Anchor-Align",
  "topics": [
   "vla",
   "imitation-diffusion",
   "video-generation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.13429v1",
  "pdf_url": "https://arxiv.org/pdf/2607.13429v1",
  "html_url": "https://arxiv.org/html/2607.13429v1",
  "code_url": "https://github.com/dwipddalal/Anchor-Align",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.3
 },
 {
  "id": "2607.13410",
  "slug": "ego-dynamics-augmented-world-model-for-autonomous-driving-with-zero-sh",
  "title": "Ego-Dynamics-Augmented World Model for Autonomous Driving with Zero-Shot Cross-Chassis Adaptation",
  "abstract": "World model (WM)-based reinforcement learning enables sample-efficient end-to-end autonomous driving learning by imagining long-horizon trajectories in latent space. However, most driving WMs operate on bird's-eye-view (BEV) representations that are inherently egocentric: the transition between consecutive frames entangles the ego vehicle's own motion with scene dynamics. As a result, the WM devotes significant capacity to recovering ego-motion from warped observations, at the cost of scene modeling fidelity and imagination accuracy. This work proposes DynaDreamer, a dynamics-augmented Dreamer-style reinforcement learning method to address this problem by augmenting the WM with an explicit ego-dynamics prior. A physics-informed ego-dynamics encoder-decoder extracts the ego-state history into a compact and identifiable context, which modulates a causal Transformer WM to condition both its prior and posterior latents. During imagination, the ego-dynamics predictor propagates this context forward to keep the ego-dynamics prior synchronized with the rollout. An information-theoretic analysis shows that conditioning on this context reduces both the predictive entropy of the observation transition and the prior--posterior Kullback--Leibler divergence, confining the WM's modeling burden to the scene dynamics beyond ego-motion. An additional benefit is zero-shot cross-chassis adaptation: the ego-dynamics context depends on identifiable chassis parameters, so that a vehicle with previously unseen dynamic characteristics can adapt the WM to the new chassis without retraining. Experiments demonstrate that DynaDreamer improves task success rates over the strongest baseline by 28% and 61% in urban and highway driving scenarios, respectively, with the advantage rising to 73% when extrapolating to unseen chassis.",
  "published": "2026-07-15",
  "updated": "2026-07-15",
  "year": "2026",
  "authors": [
   "Zhidong Wang",
   "Jingsong Liang",
   "Zirui Li",
   "Zhan Chen",
   "Han Yu",
   "Chen Lv"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "DynaDreamer is proposed, a dynamics-augmented Dreamer-style reinforcement learning method to address the problem of egocentric driving by augmenting the WM with an explicit ego-dynamics prior, and improves task success rates over the strongest baseline.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhidong Wang",
    "id": "2315060186",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Jingsong Liang",
    "id": "2284998910",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Zirui Li",
    "id": "48458251",
    "h_index": 22,
    "papers": 135
   },
   {
    "name": "Zhan Chen",
    "id": "2284198225",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Hankyeol Yu",
    "id": "2443605227",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Chen Lv",
    "id": "2289612016",
    "h_index": 3,
    "papers": 13
   }
  ],
  "comment": "13 pages, 13 figures",
  "topics": [
   "world-models",
   "egocentric-data",
   "rl-control",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.13410v1",
  "pdf_url": "https://arxiv.org/pdf/2607.13410v1",
  "html_url": "https://arxiv.org/html/2607.13410v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.13348",
  "slug": "safe-overtaking-for-autonomous-racing-using-hierarchical-optimization",
  "title": "Safe Overtaking for Autonomous Racing Using Hierarchical Optimization and Learning-Based Control",
  "abstract": "Autonomous racing overtaking requires balancing competitive performance with safety under nonlinear vehicle dynamics and real-time constraints. Model Predictive Control (MPC) combined with Control Barrier Functions (CBFs) provides a principled mechanism for certifying forward invariance of a safe set. However, commonly used fixed-decay discrete-time CBF formulations can become overly conservative in interactive racing scenarios, limiting overtaking performance and requiring manual tuning across track conditions. This paper proposes a hierarchical overtaking framework that explicitly separates maneuver-level decision making from safety-certified trajectory control, reducing conservatism while preserving safety. A high-level Mixed-Integer Quadratic Program (MIQP) resolves the combinatorial passing-side selection problem by selecting a feasible overtaking topology, while a nonlinear Frenet-frame MPC enforces vehicle dynamics and safety through embedded discrete-time CBF constraints. This decomposition isolates the combinatorial complexity of maneuver selection from the continuous trajectory optimization. To further mitigate the sensitivity of fixed-decay barrier constraints, a reinforcement learning policy adapts the discrete-time CBF decay parameter online, enabling context-dependent modulation of safety margins without directly controlling vehicle inputs. Simulation and scaled-hardware experiments show that no single fixed decay parameter achieves uniformly strong performance across tracks, whereas the adaptive strategy attains the highest aggregate success rate and consistently strong safety--performance trade-offs without per-track tuning, improving robustness to environment variation while maintaining safety constraint satisfaction in nominal operation.",
  "published": "2026-07-15",
  "updated": "2026-07-15",
  "year": "2026",
  "authors": [
   "Hassan Jardali",
   "Kai Yin",
   "Lantao Liu"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A hierarchical overtaking framework is proposed that explicitly separates maneuver-level decision making from safety-certified trajectory control, reducing conservatism while preserving safety, and improving robustness to environment variation while maintaining safety constraint satisfaction in nominal operation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hassan Jardali",
    "id": "2223985623",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Kai Yin",
    "id": "48214602",
    "h_index": 12,
    "papers": 40
   },
   {
    "name": "Lantao Liu",
    "id": "2355124346",
    "h_index": 2,
    "papers": 7
   }
  ],
  "comment": "",
  "topics": [
   "rl-control",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.13348v1",
  "pdf_url": "https://arxiv.org/pdf/2607.13348v1",
  "html_url": "https://arxiv.org/html/2607.13348v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.16312",
  "slug": "xperception-making-robotic-grasping-easier",
  "title": "xperception -- Making Robotic Grasping Easier",
  "abstract": "The transition toward high-mix low-volume manufacturing demands flexibility in robotic manipulation. However, conventional vision systems remain a bottleneck, requiring extensive data collection and model retraining whenever a new object is introduced to the production line. To overcome this rigidity, we present xperception, a zero-shot 6D pose estimation technology that eliminates the need for object-specific fine-tuning and laborious data annotation. By directly utilizing typical CAD models and integrating the rich semantic features of foundation models (e.g. DINOv2, GeDi), xperception achieves millimeter-accurate 6D pose estimation. xperception showed robustness against severe occlusions in industrial tasks like bin picking and is engineered for deployment on industrial edge hardware, such as NVIDIA Jetson Thor. Validated at a TRL of 6, the core methodology behind xperception is based on the FreeZe algorithm, which won the international BOP Challenge 2024, paving the way for scalable, plug-and-play robotic automation in unstructured high-mix low-volume manufacturing industries.",
  "published": "2026-07-14",
  "updated": "2026-07-14",
  "year": "2026",
  "authors": [
   "Matteo Bortolon",
   "Andrea Caraffa",
   "Alice Fasoli",
   "Fabio Poiesi"
  ],
  "author_count": 4,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "M. Bortolon",
    "id": "1637460719",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Andrea Caraffa",
    "id": "2213993852",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Alice Fasoli",
    "id": "2332088553",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Fabio Poiesi",
    "id": "2269470688",
    "h_index": 10,
    "papers": 27
   }
  ],
  "comment": "White paper published for Ital-IA 2026",
  "topics": [
   "dexterous-manipulation",
   "foundation-pretraining",
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [
   "NVIDIA"
  ],
  "abs_url": "https://arxiv.org/abs/2607.16312v1",
  "pdf_url": "https://arxiv.org/pdf/2607.16312v1",
  "html_url": "https://arxiv.org/html/2607.16312v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2607.16310",
  "slug": "real-time-semg-based-telecontrol-of-an-assistive-robotic-arm-using-a-1",
  "title": "Real-Time sEMG-Based Telecontrol of an Assistive Robotic Arm Using a 1D Convolutional Neural Network",
  "abstract": "Motor impairments affecting the upper limb significantly reduce autonomy in daily activities, particularly for tasks involving object manipulation. Assistive robotic arms offer a promising solution, provided they can be controlled in an intuitive, reliable, and responsive manner. Among human--machine interface approaches, surface electromyography (sEMG) enables non-invasive access to muscle activity and thus to the user's motor intentions. This work proposes a real-time sEMG-based interface for the teleoperation of an assistive robotic arm. The system relies on four-channel sEMG acquisition, signal preprocessing, segmentation into sliding windows, and classification using a one-dimensional convolutional neural network (CNN). Several real-time strategies are investigated, including threshold-based onset detection, a two-stage classification approach (rest vs movement followed by gesture recognition), and a single classifier handling both rest and five gestures. The complete pipeline is implemented and evaluated both in simulation and on a real robotic platform. The CNN-based approach achieves high classification performance, with a test accuracy above 90\\% and strong generalization on experimentally acquired signals. The system exhibits stable real-time behavior, with an average latency of approximately 0.32 s consistent with the chosen windowing strategy, and the robot can be controlled reliably using discrete gestures, producing coherent and smooth movements in both simulated and real environments. These findings demonstrate the feasibility of sEMG-based telecontrol for assistive robotics and highlight the importance of integrating signal processing, deep learning, and control strategies within a unified real-time framework. Future work may explore hybrid control approaches combining sEMG with additional sensing modalities to further improve robustness and usability.",
  "published": "2026-07-14",
  "updated": "2026-07-14",
  "year": "2026",
  "authors": [
   "Edgar Manacorda",
   "Mena Samir Kama Abouseffien",
   "Olivier Lecompte",
   "Amandine Gesta",
   "Abolfazl Mohebbi"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The findings demonstrate the feasibility of sEMG-based telecontrol for assistive robotics and highlight the importance of integrating signal processing, deep learning, and control strategies within a unified real-time framework.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Edgar Manacorda",
    "id": "2451285920",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Mena Samir Kama Abouseffien",
    "id": "2451286000",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Olivier Lecompte",
    "id": "2167025406",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Amandine Gesta",
    "id": "2165266934",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Abolfazl Mohebbi",
    "id": "2005330403",
    "h_index": 6,
    "papers": 21
   }
  ],
  "comment": "",
  "topics": [
   "data-teleop",
   "hri",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.16310v1",
  "pdf_url": "https://arxiv.org/pdf/2607.16310v1",
  "html_url": "https://arxiv.org/html/2607.16310v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.13319",
  "slug": "adapting-generalist-vehicle-models-for-high-speed-mpc-across-terrains",
  "title": "Adapting Generalist Vehicle Models for High-Speed MPC Across Terrains",
  "abstract": "High-speed off-road autonomy requires precise closed-loop control for a target vehicle while remaining robust across changing terrains. Recent forward kinodynamic (FKD) prediction foundation models suggest a promising path, starting from a generalist model and specializing it to the target platform. However, effective specialization remains challenging, as it often requires substantial real-world data, and models adapted to one setting can still overfit to specific terrains or driving regimes. We present OptCar (Optimized Car), a recipe for bridging the gap from generalist to specialist FKD models that preserves cross-terrain generalization while optimizing performance for a specific vehicle. $\\texttt{OptCar}$ introduces a history-conditioned dynamics adaptation module that encodes recent state-action observations into a dynamics context token, and then fine-tunes the generalist model using limited real-world data together with targeted synthetic rollouts from environment-specific system identification. In closed-loop model predictive control (MPC) experiments across three terrains and an out-of-distribution cart-pulling task, the largest gains appear at 6~m/s, the highest speed evaluated and the regime in which slip dominates tracking error. On vegetation and dirt, the most slip-diverse terrain, OptCar reduces 6~m/s trajectory tracking error by roughly 55% relative to a fine-tuned AnyCar baseline, and remains the most accurate even when an unseen cart payload changes the dynamics. With only 5 minutes of real data per terrain, OptCar is competitive on road with a specialist trained on 30 minutes of road data, and substantially outperforms it once the terrain changes.",
  "published": "2026-07-14",
  "updated": "2026-07-14",
  "year": "2026",
  "authors": [
   "Rwik Rana",
   "Jesse Quattrociocchi",
   "Christian Ellis",
   "Nathan Tsoi",
   "Garrett Warnell",
   "Joydeep Biswas"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "OptCar (Optimized Car), a recipe for bridging the gap from generalist to specialist FKD models that preserves cross-terrain generalization while optimizing performance for a specific vehicle, is presented.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Rwik Rana",
    "id": "2115455962",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Jesse Quattrociocchi",
    "id": "2146095986",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Christian Ellis",
    "id": "2366012257",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Nathan Tsoi",
    "id": "39282796",
    "h_index": 9,
    "papers": 33
   },
   {
    "name": "Garrett Warnell",
    "id": "1938253",
    "h_index": 30,
    "papers": 85
   },
   {
    "name": "Joydeep Biswas",
    "id": "2322098688",
    "h_index": 4,
    "papers": 12
   }
  ],
  "comment": "",
  "topics": [
   "rl-control",
   "foundation-pretraining",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.13319v1",
  "pdf_url": "https://arxiv.org/pdf/2607.13319v1",
  "html_url": "https://arxiv.org/html/2607.13319v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.13174",
  "slug": "towards-end-to-end-optimization-in-multimaterial-3d-printing",
  "title": "Towards end-to-end optimization in multimaterial 3D printing",
  "abstract": "Multimaterial 3D printing enables the fabrication of functionally graded components, but optimizing their spatial material distribution alongside structural topology remains a formidable challenge due to high-dimensional design spaces and complex constitutive modeling. This paper presents an end-to-end computational framework integrating sparsified physics-augmented neural networks with finite-element-based topology optimization. By extracting closed-form, composition-aware hyperelastic constitutive laws from experimental data, this approach facilitates exact symbolic differentiation via the adjoint state method implemented with FEniCSx, efficiently circumventing the bottlenecks of applying neural network constitutive models. This pipeline is deployed on soft robotic gripper applications, demonstrating continuous composition optimization for highly anisotropic contact responses, and the concurrent optimization of macroscopic topology and material distribution under non-failure stretch constraints. This methodology could replace laborious empirical prototyping, establishing interpretable machine-learning models as practical, robust design primitives for advanced multimaterial additive manufacturing.",
  "published": "2026-07-14",
  "updated": "2026-07-14",
  "year": "2026",
  "authors": [
   "Xue-Ling Luo",
   "Steven Yang",
   "Jingye Tan",
   "Robert F. Shepherd",
   "Noy Cohen",
   "Nikolaos Bouklas"
  ],
  "author_count": 6,
  "categories": [
   "physics.comp-ph",
   "cs.CE",
   "cs.RO"
  ],
  "primary_category": "physics.comp-ph",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "An end-to-end computational framework integrating sparsified physics-augmented neural networks with finite-element-based topology optimization, which could replace laborious empirical prototyping and establish interpretable machine-learning models as practical, robust design primitives for advanced multimaterial additive manufacturing.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xue-Ling Luo",
    "id": "2449714195",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Steven J. Yang",
    "id": "2447209097",
    "h_index": 7,
    "papers": 12
   },
   {
    "name": "J. Tan",
    "id": "2226487139",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "R. Shepherd",
    "id": "1781229",
    "h_index": 58,
    "papers": 129
   },
   {
    "name": "Noy Cohen",
    "id": "34973968",
    "h_index": 21,
    "papers": 79
   },
   {
    "name": "Nikolaos Bouklas",
    "id": "2295733986",
    "h_index": 7,
    "papers": 18
   }
  ],
  "comment": "",
  "topics": [
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.13174v1",
  "pdf_url": "https://arxiv.org/pdf/2607.13174v1",
  "html_url": "https://arxiv.org/html/2607.13174v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.13154",
  "slug": "worlds-in-one-demo-a-synthetic-data-engine-for-learning-open-world-mob",
  "title": "Worlds in One Demo: A Synthetic Data Engine for Learning Open-World Mobile Manipulation",
  "abstract": "Learning open-world mobile manipulation policies requires vast data to achieve spatial generalization, long-horizon robustness, and scene generalization. Current prevailing data collection paradigms, teleoperation and UMI, demand prohibitive human effort and cost at scale. To scale beyond the limits of manual data collection, we seek to maximize the value of each human demonstration by scalable data generation. To this end, we introduce WANDA: learning open-World mobile mANipulation from one demonstration via a synthetic DAta engine. WANDA first reconstructs background Gaussian splats and robot-object interaction trajectories from source RGBD observations, as a world substrate for later planning and rendering. It then rearranges contact-rich robot-object interaction segments into extensive spatial configurations, utilizing whole-body motion planning to chain them into new trajectories. To enhance long-horizon robustness, it applies Corrective State Expansion to increase the robot and object state diversity at different stages of mobile manipulation. To unlock cross-environment generalization, trajectories are synthesized on diverse generated 3D worlds from everyday photos. Furthermore, we synthesize photo-realistic observations by compositing rendered robot and object meshes with Gaussian splatting backgrounds. We evaluate our approach on extensive simulation and real-world tasks in various scenes. Experiments show that policies trained with WANDA achieve long-horizon robustness, broad spatial generalization and cross-environment generalization from one real demonstration. Moreover, WANDA naturally supports cross-embodiment data generation, validated by zero-shot deployment on another mobile manipulator with a distinct morphology.",
  "published": "2026-07-14",
  "updated": "2026-07-17",
  "year": "2026",
  "authors": [
   "Lingxiao Guo",
   "Huanyu Li",
   "Guanya Shi"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Experiments show that policies trained with WANDA achieve long-horizon robustness, broad spatial generalization and cross-environment generalization from one real demonstration, and WANDA naturally supports cross-embodiment data generation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Lingxiao Guo",
    "id": "2366005432",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Huanyu Li",
    "id": "2329166925",
    "h_index": 6,
    "papers": 20
   },
   {
    "name": "Guanya Shi",
    "id": "2384824402",
    "h_index": 9,
    "papers": 24
   }
  ],
  "comment": "Project website: https://wanda.lecar-lab.org/",
  "topics": [
   "humanoids",
   "egocentric-data",
   "tactile",
   "spatial-3d",
   "navigation",
   "foundation-pretraining",
   "data-teleop",
   "hardware-codesign",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.13154v2",
  "pdf_url": "https://arxiv.org/pdf/2607.13154v2",
  "html_url": "https://arxiv.org/html/2607.13154v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.13033",
  "slug": "densereward-dense-reward-learning-via-failure-synthesis-for-robotic-ma",
  "title": "DenseReward: Dense Reward Learning via Failure Synthesis for Robotic Manipulation",
  "abstract": "Reinforcement learning holds great promise for improving robot policies beyond the limits of imitation learning. However, its practical adoption remains bottlenecked by the lack of reliable vision-language reward models that provide dense and informative feedback. Two key challenges remain: acquiring diverse failure data at scale and obtaining fine-grained reward signals beyond sparse trajectory-level success labels. Collecting failure trajectories typically requires laborious human effort, while pseudo-failures constructed by relabeling successful demonstrations fail to capture the diverse physical failure modes that arise during robot execution. Meanwhile, existing reward models often predict sparse binary or trajectory-level rewards, which provide limited guidance for efficient policy optimization. We introduce DenseReward, a dense robotic reward model that addresses both challenges. To train DenseReward, we develop an automated failure data generation pipeline that synthesizes physically realistic failure trajectories in simulation without human labeling, covering diverse failure modes such as collisions, missed grasps, object drops, and recovery behaviors. DenseReward predicts dense frame-level reward scores from visual observations and language instructions, enabling fine-grained estimation of task progress throughout an episode. Experiments show that DenseReward outperforms general-purpose VLMs and existing robotic reward models in dense reward prediction across both simulated and real-world manipulation. We further demonstrate that DenseReward provides effective reward guidance for downstream model predictive control and reinforcement learning. We release the dataset, trained reward models, and evaluation suite to support the development of failure-aware dense reward modeling for robot learning.",
  "published": "2026-07-14",
  "updated": "2026-07-14",
  "year": "2026",
  "authors": [
   "Yu Fang",
   "Wanxi Dong",
   "Jiaqi Liu",
   "Yue Yang",
   "Mingxiao Huo",
   "Yao Mu",
   "Huaxiu Yao",
   "Li Erran Li",
   "Daniel Szafir",
   "Mingyu Ding"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Experiments show that DenseReward outperforms general-purpose VLMs and existing robotic reward models in dense reward prediction across both simulated and real-world manipulation, and provides effective reward guidance for downstream model predictive control and reinforcement learning.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yu Fang",
    "id": "2351241177",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Wanxi Dong",
    "id": "2376156355",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Jiaqi Liu",
    "id": "2384148523",
    "h_index": 7,
    "papers": 24
   },
   {
    "name": "Yue Yang",
    "id": "2292430956",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Mingxiao Huo",
    "id": "2253663026",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Yao Mu",
    "id": "2348606790",
    "h_index": 4,
    "papers": 18
   },
   {
    "name": "Huaxiu Yao",
    "id": "2321402465",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "L. Li",
    "id": "2156057522",
    "h_index": 9,
    "papers": 14
   },
   {
    "name": "D. Szafir",
    "id": "2259503818",
    "h_index": 6,
    "papers": 21
   },
   {
    "name": "Mingyu Ding",
    "id": "2346837065",
    "h_index": 4,
    "papers": 15
   }
  ],
  "comment": "Website: https://dense-reward.github.io/",
  "topics": [
   "dexterous-manipulation",
   "imitation-diffusion",
   "rl-control",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.13033v1",
  "pdf_url": "https://arxiv.org/pdf/2607.13033v1",
  "html_url": "https://arxiv.org/html/2607.13033v1",
  "code_url": "https://dense-reward.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2607.13028",
  "slug": "terrazero-procedural-driving-simulation-for-zero-demonstration-self-pl",
  "title": "TerraZero: Procedural Driving Simulation for Zero-Demonstration Self-Play at Scale",
  "abstract": "Training robust autonomous driving agents requires a simulator fast enough for reinforcement learning at scale, realistic enough to ground behavior in real-world map structure, and diverse enough to cover the safety-critical long tail that logged data rarely contains. We present TerraZero, a procedural driving simulator and self-play training stack that meets these goals. A configurable C engine runs simulation on the CPU and policy inference on the GPU over a zero-copy path, sustaining 1.3M agent-steps per second on a single server-grade GPU, far faster than existing object-level simulators, while keeping fidelity lighter single-agent systems omit: heterogeneous agents, multiple dynamics models, and full traffic-rule enforcement. TerraZero uses logged data only as a source of real-world map geometry, populating each map with randomized rule-based road users and signal controllers and randomizing agent dynamics, rewards, and sizes per episode, so one map yields an effectively unbounded set of scenarios. Every reported policy trains from scratch by reinforcement learning alone, with zero human demonstrations, no imitation, no logged trajectories, and no fallback planner at inference, on a compute-efficient self-play recipe scaled across GPUs. The policies generalize zero-shot across cities and datasets, including emergent left-hand-traffic driving without explicit supervision. As an ego policy, a single checkpoint is, to our knowledge, the first fully learned policy to top both val14 and the interactive long-tail InterPlan suite. On Waymo Open Sim Agents realism the same recipe outperforms other demonstration-free methods and is competitive with the strongest reference-anchored self-play method. One stack serves both roles: state-of-the-art demonstration-free driving policies across dynamics for cars and trucks, and sim agents that jointly control vehicles, pedestrians, and cyclists.",
  "published": "2026-07-14",
  "updated": "2026-08-04",
  "year": "2026",
  "authors": [
   "Zhouchonghao Wu",
   "Akshay Rangesh",
   "Weixin Li",
   "Wei-Jer Chang",
   "Zachary Lee",
   "Saeed Bonab",
   "Tim Wang",
   "Wei Zhan"
  ],
  "author_count": 8,
  "categories": [
   "cs.LG",
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "TerraZero is a procedural driving simulator and self-play training stack that meets the goals of reinforcement learning at scale, realistic enough to ground behavior in real-world map structure, and diverse enough to cover the safety-critical long tail that logged data rarely contains.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhouchonghao Wu",
    "id": "2350302302",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Akshay Rangesh",
    "id": "3394813",
    "h_index": 18,
    "papers": 33
   },
   {
    "name": "Weixin Li",
    "id": "2377235423",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Wei-Jer Chang",
    "id": "2116373708",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Zachary Lee",
    "id": "2449810680",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "S. Bonab",
    "id": "1412430895",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Tim Wang",
    "id": "2449936716",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Wei Zhan",
    "id": "2366010456",
    "h_index": 2,
    "papers": 10
   }
  ],
  "comment": "Technical Report from Applied Intuition Research",
  "topics": [
   "egocentric-data",
   "sim2real",
   "rl-control",
   "navigation",
   "safety-eval"
  ],
  "orgs": [
   "Applied Intuition"
  ],
  "abs_url": "https://arxiv.org/abs/2607.13028v2",
  "pdf_url": "https://arxiv.org/pdf/2607.13028v2",
  "html_url": "https://arxiv.org/html/2607.13028v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.8
 },
 {
  "id": "2607.12965",
  "slug": "mammoth-a-multi-modal-end-to-end-policy-for-off-road-mobility-robust-t",
  "title": "MAMMOTH: A Multi-Modal End-to-End Policy for Off-Road Mobility Robust to Missing Modality",
  "abstract": "Reliable autonomous navigation in unstructured off-road environments remains a critical unsolved challenge due to extreme terrain diversity, drastic illumination variations and acute sensor degradation. Recent developments have approached the problem as a traversability costmap estimation or visual navigation task. However, many exhibit heavy reliance on RGB modality, leading to poor performance in varied illumination such as glares, shadows or low ambient light. Achieving robust generalization in such conditions requires integrating modalities that provide supplementary scene information. Such multi-modal methods suffer from a rigid dependency on the presence of near-perfect sensor inputs, leaving them unable to robustly handle sensor degradation or individual modality failure. To address these limitations, we introduce MAMMOTH (MAsking Multi-Modal inputs for Off-road Traversability Heuristic-informed navigation), a unified end-to-end navigation policy for robust off-road visual-goal-conditioned navigation and undirected exploration. Specifically, MAMMOTH efficiently fuses multi-modal observations (RGB, Thermal, 3D Pointcloud and Ego Velocity) and is trained with a modality dropout scheme, enabling it to generalize to missing modalities at inference time. Furthermore, we employ a diffusion policy to learn the joint conditional probability distribution of physically-grounded trajectories and a intrinsic traversability heuristic. MAMMOTH utilizes this heuristic to prefer safer, smoother trajectories. We validate MAMMOTH through extensive real-world robot experiments in distinct off-road environments, including night-time operation. Our results demonstrate superior performance, with significant improvements in collision avoidance, terrain-aware planning and generalization to missing modalities. The code and dataset used for this work will be made publicly available.",
  "published": "2026-07-14",
  "updated": "2026-07-14",
  "year": "2026",
  "authors": [
   "Ahaan Kotian",
   "Shivani Subramanyan",
   "Suresh Sundaram"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work introduces MAMMOTH (MAsking Multi-Modal inputs for Off-road Traversability Heuristic-informed navigation), a unified end-to-end navigation policy for robust off-road visual-goal-conditioned navigation and undirected exploration and employs a diffusion policy to learn the joint conditional probability distribution of physically-grounded trajectories and a intrinsic traversability heuristic.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ahaan Kotian",
    "id": "2449925515",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Shivani Subramanyan",
    "id": "2449924041",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Suresh Sundaram",
    "id": "2333589699",
    "h_index": 2,
    "papers": 9
   }
  ],
  "comment": "Accepted to IROS 2026 Main Conference",
  "topics": [
   "imitation-diffusion",
   "navigation",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.12965v1",
  "pdf_url": "https://arxiv.org/pdf/2607.12965v1",
  "html_url": "https://arxiv.org/html/2607.12965v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2607.12801",
  "slug": "autonomous-tracking-and-terminal-guidance-of-moving-targets-for-fixed",
  "title": "Autonomous Tracking and Terminal Guidance of Moving Targets for Fixed-Wing UAVs",
  "abstract": "This study introduces a unified control framework for fixed-wing unmanned aerial vehicles (UAVs) fitted with a pan-tilt (PT) camera, intended to perform an end-to-end mission spanning from initial target detection to accurate terminal engagement. The proposed system employs a three-phase strategy: a vision-based target acquisition phase, an NMPC-based tracking phase, and a terminal guidance phase. During tracking, the framework uses an Unscented Kalman Filter (UKF) to fuse YOLO-based visual detections with inertial measurements, enabling robust target state estimation under unknown dynamics. To ensure reliable visual contact, we introduce a constraint-aware Nonlinear Model Predictive Control (NMPC) strategy that incorporates Control Barrier Functions (CBFs) to explicitly prevent UAV self-occlusion -- a common limitation in fixed-wing tracking. Upon satisfying terminal engagement conditions, the system seamlessly transitions control to a quaternion-based Biased Proportional Navigation Guidance (BPNG) law, enforcing precise impact angle constraints. High-fidelity simulations demonstrate that the framework achieves stable, robust tracking and accurate terminal interception while strictly respecting the vehicle's dynamic limits and camera field-of-view constraints.",
  "published": "2026-07-14",
  "updated": "2026-07-14",
  "year": "2026",
  "authors": [
   "Wei-Hao Liou",
   "Teng-Hu Cheng"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO",
   "cs.AI",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A constraint-aware Nonlinear Model Predictive Control strategy that incorporates Control Barrier Functions (CBFs) to explicitly prevent UAV self-occlusion is introduced to explicitly prevent UAV self-occlusion -- a common limitation in fixed-wing tracking.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Wei-Hao Liou",
    "id": "2449923112",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Teng-Hu Cheng",
    "id": "2244431164",
    "h_index": 1,
    "papers": 6
   }
  ],
  "comment": "",
  "topics": [
   "rl-control",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.12801v1",
  "pdf_url": "https://arxiv.org/pdf/2607.12801v1",
  "html_url": "https://arxiv.org/html/2607.12801v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.12702",
  "slug": "vision-based-dribbling-for-humanoid-soccer-via-privileged-representati",
  "title": "Vision-Based Dribbling for Humanoid Soccer via Privileged Representation Learning",
  "abstract": "Recent advances in humanoid robotics have highlighted the importance of deployable loco-manipulation skills. Dribbling a soccer ball while evading active opponents requires simultaneous balance, precise ball control, and awareness of a dynamic adversary under onboard sensing and real-time constraints. Existing approaches typically separate perception and motion, which can be effective in controlled settings but may fail under occlusions, fast ball movements, and complex opponent interactions, since perception is not directly optimized for control. We propose an integrated approach in which a temporal depth encoder is embedded into a reinforcement learning policy through a task-specific projection layer. We apply this framework to a simulated Booster T1 humanoid robot and show that it is possible to learn vision-based, opponent-aware dribbling directly from depth observations, without explicit state estimation or privileged scene information. The learned policy achieves 100% success in nominal target-driven dribbling and 96% success with a single static obstacle, while reaching 46% success against an actively moving ball-attacker opponent. These results demonstrate that the proposed framework supports robust vision-based dribbling in nominal and moderately dynamic settings, and provides a strong foundation for handling more challenging moving-adversary scenarios.",
  "published": "2026-07-14",
  "updated": "2026-07-14",
  "year": "2026",
  "authors": [
   "Flavio Maiorana",
   "Valerio Spagnoli",
   "Eugenio Bugli",
   "Flavio Volpi",
   "Daniele Affinita",
   "Vincenzo Suriani",
   "Daniele Nardi",
   "Luca Iocchi"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes an integrated approach in which a temporal depth encoder is embedded into a reinforcement learning policy through a task-specific projection layer, and shows that it is possible to learn vision-based, opponent-aware dribbling directly from depth observations, without explicit state estimation or privileged scene information.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "F. Maiorana",
    "id": "2295992564",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "V. Spagnoli",
    "id": "2281642050",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "E. Bugli",
    "id": "2332354861",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "F. Volpi",
    "id": "2281643131",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "D. Affinita",
    "id": "2281641246",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "V. Suriani",
    "id": "22269673",
    "h_index": 7,
    "papers": 51
   },
   {
    "name": "D. Nardi",
    "id": "2064462333",
    "h_index": 6,
    "papers": 32
   },
   {
    "name": "L. Iocchi",
    "id": "1712013",
    "h_index": 43,
    "papers": 321
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.12702v1",
  "pdf_url": "https://arxiv.org/pdf/2607.12702v1",
  "html_url": "https://arxiv.org/html/2607.12702v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.12604",
  "slug": "streamlining-stereo-differentiable-rendering-for-marker-free-real-time",
  "title": "Streamlining stereo differentiable rendering for marker-free real-time tracking of surgical robots",
  "abstract": "Purpose: Marker-based tracking of surgical robots is occlusion-prone in cluttered operating rooms. We evaluate stereo differentiable rendering for marker-free, real-time robot pose tracking, potentially improving safety, reducing setup time, and enabling multi-robot interaction. Methods: We extend the markerless pose estimation framework roboreg to online dynamic tracking via (i) sequential optimisation that propagates pose estimates across frames with motion-adaptive hyperparameter tuning, and (ii) CUDA stream parallelisation of segmentation and optimisation, combined with CUDA-graph accelerated segmentation. We evaluate on 38 unobstructed and 5 occluded displacement sequences with static start/end ground-truth calibrations and dynamic marker-based reference tracking. Results: We achieve real-time 1080p tracking at 30 fps (up from 14 fps for vanilla roboreg), matching the camera frame rate. Accuracy reaches 1.7 cm / 0.6 deg against static ground truth and 1.2 cm mean 3D error over 27,460 frames against the marker-based reference (1.53 cm over 1,242 occluded frames). Our method outperforms FoundationPose by 11% in dynamic estimation (63% under occlusion) and 250% in static estimation, with 6x faster inference. Conclusions: Stereo differentiable rendering enables real-time, high-resolution marker-free surgical robot tracking, on par with marker-based approaches and surpassing foundation-model baselines.",
  "published": "2026-07-14",
  "updated": "2026-07-14",
  "year": "2026",
  "authors": [
   "Yanghe Hao",
   "Martin Huber",
   "Christos Bergeles",
   "Tom Vercauteren"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Stereo-differentiable rendering-based pose estimation for marker-free real-time surgical robots tracking, mitigating occlusion-prone marker-based tracking in cluttered surgical environments, potentially improving safety, reducing setup times, and enabling intelligent multi-robot interaction is evaluated.",
  "doi": "10.1007/s11548-026-03730-z",
  "oa_pdf": "https://link.springer.com/content/pdf/10.1007/s11548-026-03730-z.pdf",
  "s2_authors": [
   {
    "name": "Yanghe Hao",
    "id": "2432164229",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "M. Huber",
    "id": "2267488881",
    "h_index": 3,
    "papers": 16
   },
   {
    "name": "Christos Bergeles",
    "id": "2241433151",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "T. Vercauteren",
    "id": "2290428866",
    "h_index": 4,
    "papers": 15
   }
  ],
  "comment": "",
  "topics": [
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.12604v1",
  "pdf_url": "https://arxiv.org/pdf/2607.12604v1",
  "html_url": "https://arxiv.org/html/2607.12604v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.12423",
  "slug": "model-based-diffusion-optimal-control-for-multi-robot-motion-planning",
  "title": "Model-Based Diffusion Optimal Control for Multi-Robot Motion Planning",
  "abstract": "Multi-Robot Motion Planning in continuous environments, where robots must generate dynamically feasible, collision-free trajectories, is challenging due to the combinatorial growth of the joint trajectory space and the difficulty of enforcing dynamic feasibility and hard safety constraints. Recent approaches recast trajectory planning as probabilistic inference, sampling from a posterior over trajectories using diffusion models whose score functions are learned from demonstration data. While showing promising performance, these approaches are limited: they often rely on sizable demonstration datasets and struggle to rigorously enforce dynamics and hard safety constraints during sampling. To this end, we introduce Model-Based Diffusion Optimal Control (MDOC), a model-based diffusion planner that efficiently produces dynamically feasible trajectories without relying on data. Crucially, we show that MDOC's safety mechanism -- combining known dynamics models with Control Barrier Function-constrained projections -- naturally scales to multi-robot planning settings through Conflict-Based Search. Across simulation experiments, this integrated method consistently outperforms representative baseline planners in sample efficiency, geometric smoothness, and success rate, while reducing computation time and producing collision-free trajectories.",
  "published": "2026-07-14",
  "updated": "2026-07-14",
  "year": "2026",
  "authors": [
   "Zhilin He",
   "Yorai Shaoul",
   "Jiaoyang Li"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 1,
  "tldr": "This work introduces Model-Based Diffusion Optimal Control (MDOC), a model-based diffusion planner that efficiently produces dynamically feasible trajectories without relying on data, and shows that MDOC's safety mechanism naturally scales to multi-robot planning settings through Conflict-Based Search.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhilin He",
    "id": "2449944847",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Yorai Shaoul",
    "id": "2027611706",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Jiaoyang Li",
    "id": "2294313888",
    "h_index": 7,
    "papers": 27
   }
  ],
  "comment": "",
  "topics": [
   "rl-control",
   "video-generation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.12423v1",
  "pdf_url": "https://arxiv.org/pdf/2607.12423v1",
  "html_url": "https://arxiv.org/html/2607.12423v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2607.12370",
  "slug": "stratmamba-strategic-and-reactive-stream-partitioning-for-path-efficie",
  "title": "StratMamba: Strategic and Reactive Stream Partitioning for Path-Efficient LiDAR-Based Obstacle Avoidance",
  "abstract": "This paper proposes StratMamba, a dual-stream Mamba-based temporal modeling architecture, to more efficiently capture long-horizon temporal dependencies required for robot navigation in complex and obstacle-rich environments. StratMamba leverages a combination of fast-decay and slow-decay memory architectures, where the fast-decay component processes high-frequency LiDAR data for reactive obstacle avoidance, while the slow-decay component maintains longer-horizon goal information for strategic planning. We perform extensive evaluations of different obstacle avoidance scenarios in IsaacLab and Gazebo, while also validating successful sim-to-real deployment on a Unitree GO1 quadruped robot navigating in the presence of static/dynamic obstacles. Comparisons with other temporal RL baselines, such as LSTM, Transformer, and Vanilla-Mamba, show that our StratMamba achieves exceptional temporal reasoning efficiency with a lower timeout rate, while maintaining the fastest navigation speed (576 median steps, 5.0% better than Vanilla-Mamba). It also achieves the highest path optimality (0.915 path efficiency) across all baselines. Real-world evaluation reveals that StratMamba maintains more robust performance across extended LiDAR ranges compared to vanilla Mamba and the Transformer, demonstrating that dual-stream partitioning effectively balances reactive safety with strategic navigation under challenging sensing conditions.",
  "published": "2026-07-14",
  "updated": "2026-07-14",
  "year": "2026",
  "authors": [
   "Hung-Chieh Wu",
   "Xiaopan Zhang",
   "Kasra Sinaei",
   "Ryan Abnavi",
   "Kasun Weerakoon",
   "Christopher Bradley",
   "Seyed Fakoorian",
   "Jiachen Li",
   "Donald Ebeigbe"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Real-world evaluation reveals that StratMamba maintains more robust performance across extended LiDAR ranges compared to vanilla Mamba and the Transformer, demonstrating that dual-stream partitioning effectively balances reactive safety with strategic navigation under challenging sensing conditions.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hung-Chieh Wu",
    "id": "2376521887",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Xiaopan Zhang",
    "id": "2312867599",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "K. Sinaei",
    "id": "2149075788",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Ryan Abnavi",
    "id": "2449775955",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Kasun Weerakoon",
    "id": "123689410",
    "h_index": 12,
    "papers": 38
   },
   {
    "name": "Christopher Bradley",
    "id": "10289493",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "S. Fakoorian",
    "id": "2317839201",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Jiachen Li",
    "id": "2449464303",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Donald Ebeigbe",
    "id": "3420796",
    "h_index": 6,
    "papers": 23
   }
  ],
  "comment": "Accepted to IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS 2026). 8 pages, 6 figures. Video: https://www.youtube.com/watch?v=Z0FfO_AVaSw",
  "topics": [
   "humanoids",
   "sim2real",
   "navigation",
   "safety-eval"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2607.12370v1",
  "pdf_url": "https://arxiv.org/pdf/2607.12370v1",
  "html_url": "https://arxiv.org/html/2607.12370v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.0
 },
 {
  "id": "2607.16299",
  "slug": "algorithmic-accuracy-as-a-motivational-driver-in-robot-mediated-learni",
  "title": "Algorithmic Accuracy as a Motivational Driver in Robot-Mediated Learning: A Comparative Study of Cross-Correlation and CNN-Based Sound Detection in an Interactive Quiz Game",
  "abstract": "In competitive learning activities, inaccurate robot decisions may reduce students' perceptions of fairness and competence, ultimately affecting their motivation. This paper investigates whether the accuracy of sound detection algorithms influences student motivation during a robot-mediated quiz game. A Pepper humanoid robot hosted an interactive buzzer-based quiz in which two sound detection approaches, a Convolutional Neural Network (CNN) and a Cross-Correlation algorithm, were evaluated using a controlled between-subjects experiment involving 40 university students. Participants were equally assigned to a CNN group (n = 20) and a Cross-Correlation group (n = 20). Both groups completed the same quiz under identical conditions, differing only in the sound detection algorithm used for first-responder identification. Student motivation was assessed using the Intrinsic Motivation Inventory (IMI), while algorithm performance was evaluated through real-time detection accuracy. The results indicate that the Cross-Correlation approach achieved more reliable sound detection under classroom conditions and produced significantly higher scores across all IMI subscales, demonstrating greater student interest, perceived competence, effort, perceived choice, and lower perceived pressure (after reverse coding). These findings provide empirical support for the proposed Algorithmic Precision-Motivation Relationship (APMR) model, demonstrating that algorithmic accuracy is not merely an engineering performance metric but an important factor influencing learner motivation in robot-assisted educational environments.",
  "published": "2026-07-13",
  "updated": "2026-07-13",
  "year": "2026",
  "authors": [
   "Rezaul Tutul",
   "Ilona Buchem",
   "Niels Pinkwart"
  ],
  "author_count": 3,
  "categories": [
   "cs.HC",
   "cs.RO"
  ],
  "primary_category": "cs.HC",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Investigating whether the accuracy of sound detection algorithms influences student motivation during a robot-mediated quiz game provides empirical support for the proposed Algorithmic Precision-Motivation Relationship (APMR) model, demonstrating that algorithmic accuracy is not merely an engineering performance metric but an important factor influencing learner motivation in robot-assisted educational environments.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "R. Tutul",
    "id": "2192495224",
    "h_index": 3,
    "papers": 13
   },
   {
    "name": "I. Buchem",
    "id": "2276637132",
    "h_index": 2,
    "papers": 14
   },
   {
    "name": "N. Pinkwart",
    "id": "2276639609",
    "h_index": 2,
    "papers": 8
   }
  ],
  "comment": "",
  "topics": [
   "humanoids"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.16299v1",
  "pdf_url": "https://arxiv.org/pdf/2607.16299v1",
  "html_url": "https://arxiv.org/html/2607.16299v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.12114",
  "slug": "gaitspan-growing-humanoid-locomotion-from-walking-to-running",
  "title": "GaitSpan: Growing Humanoid Locomotion from Walking to Running",
  "abstract": "A humanoid that can walk should not relearn locomotion from scratch to jog or run. Yet current approaches often obtain gait diversity by prescribing gait schedules, imitating motion clips, training experts to switch between or distilling skills into one policy. These strategies can produce impressive behaviors, but offer limited flexibility across continuous speed commands, terrains, and morphologies. We study skill growth with GaitSpan, a framework that expands a pretrained, basic walking policy into faster locomotion. It treats walking as a seed skill: reusable motor structure for balance, support, body coordination, and contact transition that can be regenerated at new rhythms, extended into longer/higher strides, and corrected by residual adaptation. This expansion has three aspects: 1) rhythm generation, which modulates the frozen walking policy with multiple internal clocks and learns command-conditioned combinations of the resulting canonical actions; 2) stride shaping, which rewards dynamic locomotion patterns appropriate for higher commanded speeds using a physically grounded objective inspired by spring-loaded inverted pendulum dynamics; and 3) residual adaptation, which captures motion details not accounted for by rhythm generation or stride shaping. GaitSpan is the first to deliver a single command-conditioned humanoid policy that spans walking, jogging, and running-like regimes covering a continuous speed range, transfers across morphologies, and deploys zero-shot on unseen sim-to-sim, and real-world terrains. Compared with baselines either trained with multi-experts or imitation from humans, it learns faster and achieves stronger gait performance.",
  "published": "2026-07-13",
  "updated": "2026-07-13",
  "year": "2026",
  "authors": [
   "Kwan-Yee Lin",
   "Zilin Wang",
   "Janelle J. Liu",
   "Stella X. Yu"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "GaitSpan is the first to deliver a single command-conditioned humanoid policy that spans walking, jogging, and running-like regimes covering a continuous speed range, transfers across morphologies, and deploys zero-shot on unseen sim-to-sim, and real-world terrains.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Kwan-Yee Lin",
    "id": "2363675973",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Zilin Wang",
    "id": "2259065671",
    "h_index": 6,
    "papers": 37
   },
   {
    "name": "Jane Liu",
    "id": "2135267088",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Stella X.Yu",
    "id": "2449777048",
    "h_index": 0,
    "papers": 1
   }
  ],
  "comment": "Project Page: https://gaitspan2026.github.io/",
  "topics": [
   "humanoids"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.12114v1",
  "pdf_url": "https://arxiv.org/pdf/2607.12114v1",
  "html_url": "https://arxiv.org/html/2607.12114v1",
  "code_url": "https://gaitspan2026.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2607.12105",
  "slug": "robust-in-hand-manipulation-via-priors-in-reinforcement-learning-and-m",
  "title": "Robust In-Hand Manipulation via Priors in Reinforcement Learning and Mechanical Design",
  "abstract": "In-hand manipulation without external sensing is challenging due to uncertainties from finger-object contacts and disturbances by gravity. While reinforcement learning has shown promise in learning complex finger gaiting, existing approaches do not prioritize maintaining well-conditioned grasps for sustained manipulation. We introduce two complementary physics priors for robust in-hand rolling: a global grasp-quality prior derived from classical grasp analysis and a local contact-geometry prior based on fingertip curvature. The grasp-quality prior is used as a dense reward-shaping term that encourages well-distributed contacts with improved worst-case wrench resistance. The contact-geometry prior is expressed in the fingertip geometry that mechanically shapes the contact interface toward task-aligned rolling while reducing off-axis drift. We evaluate the effect of these priors on learning in-hand rolling manipulation for a multifingered robotic hand manipulating three different objects at four palm orientations. Results show significant improvement in rotation efficiency, grasp stability, and disturbance rejection, suggesting that physics priors embedded in both learning and fingertip morphology improve task robustness and sim-to-real transfer. An overview video can be found at https://youtu.be/pdd1wHxQnJM?si=dM-U5kiiPTYsk3Pk.",
  "published": "2026-07-13",
  "updated": "2026-07-13",
  "year": "2026",
  "authors": [
   "Yifei Chen",
   "Shihan Lu",
   "Ed Colgate",
   "Kevin Lynch"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Two complementary physics priors for robust in-hand rolling manipulation are introduced: a global grasp-quality prior derived from classical grasp analysis and a local contact-geometry prior based on fingertip curvature.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yifei Chen",
    "id": "2312867762",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Shihan Lu",
    "id": "2143514741",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Ed Colgate",
    "id": "47814061",
    "h_index": 8,
    "papers": 17
   },
   {
    "name": "Kevin M. Lynch",
    "id": "2237594018",
    "h_index": 2,
    "papers": 5
   }
  ],
  "comment": "25 pages, 15 figures, 9 tables",
  "topics": [
   "dexterous-manipulation",
   "sim2real",
   "rl-control",
   "hardware-codesign",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.12105v1",
  "pdf_url": "https://arxiv.org/pdf/2607.12105v1",
  "html_url": "https://arxiv.org/html/2607.12105v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.12050",
  "slug": "eflux-elastic-multi-robot-formation-navigation-and-adaptation-with-age",
  "title": "EFLUX: Elastic Multi-Robot Formation Navigation and Adaptation with Agentic LLMs",
  "abstract": "Multi-robot teams operating in confined or cluttered environments must adapt both their formation geometry and group topology to navigate through complex obstacles. This adaptation requires two complementary behaviors: deformation, where the team continuously reshapes its geometry while remaining connected, and reconfiguration, where robots split into subgroups or merge back into a single formation. Existing methods often model these behaviors independently, connect them through handcrafted rules, or lack explicit geometric criteria for determining when each behavior should be invoked. However, challenging environments may require online changes in formation shape, connectivity, and effective team composition, making decoupled or rule-based approaches prone to suboptimal trajectories and deadlock. We propose EFLUX, a geometry-grounded LLM agentic framework for automatic and elastic multi-robot formation navigation. EFLUX extracts a structured scene representation and uses an LLM to reason jointly over both deformation actions, such as scaling and shearing, and reconfiguration actions, such as splitting and merging. These strategies are then translated into executable per-robot waypoints through a closed-loop generation, verification, and correction pipeline. Simulation and hardware experiments show that EFLUX enables safe, continuous, and elastic formation navigation in constrained environments, reducing deadlock and navigation failures compared with baselines while maintaining coherent multi-robot coordination.",
  "published": "2026-07-13",
  "updated": "2026-07-13",
  "year": "2026",
  "authors": [
   "Jinyuan Zhang",
   "Yuwei Wu",
   "Guangyao Shi",
   "Jonathan Diller",
   "Gaurav S. Sukhatme",
   "Vijay Kumar"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Simulation and hardware experiments show that EFLUX enables safe, continuous, and elastic formation navigation in constrained environments, reducing deadlock and navigation failures compared with baselines while maintaining coherent multi-robot coordination.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jinyuan Zhang",
    "id": "2449971239",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yuwei Wu",
    "id": "2277991230",
    "h_index": 6,
    "papers": 20
   },
   {
    "name": "Guangyao Shi",
    "id": "2292015106",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Jonathan Diller",
    "id": "147054873",
    "h_index": 5,
    "papers": 18
   },
   {
    "name": "Gaurav S. Sukhatme",
    "id": "2282955113",
    "h_index": 4,
    "papers": 14
   },
   {
    "name": "Vijay Kumar",
    "id": "2321487678",
    "h_index": 3,
    "papers": 8
   }
  ],
  "comment": "",
  "topics": [
   "spatial-3d",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.12050v1",
  "pdf_url": "https://arxiv.org/pdf/2607.12050v1",
  "html_url": "https://arxiv.org/html/2607.12050v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.11884",
  "slug": "mixture-of-frames-policy-multi-frame-action-denoising-for-bimanual-mob",
  "title": "Mixture of Frames Policy: Multi-Frame Action Denoising for Bimanual Mobile Manipulation",
  "abstract": "Robotic manipulation is inherently multi-frame: local actions may be simple in an end-effector frame, while transport, upright-object handling, and whole-body coordination are better represented in a base-aligned frame. However, modern diffusion-based visuomotor policies typically commit to a single predefined action frame, forcing one denoiser to model action distributions that are often unnecessarily complex in that frame. We propose Mixture of Frames Policy (MoF), a diffusion policy that performs synchronized action denoising across multiple coordinate frames. MoF maintains a single canonical diffusion state, re-expresses it in several task-relevant frames, applies frame-specialized denoisers, and fuses their noise predictions back in the canonical frame. To make this possible for intermediate noisy diffusion states, we introduce a column-based 6D rotation representation within an SE(3) action parameterization that supports exact, differentiable frame transformations without requiring noisy rotations to lie on the SO(3) manifold. Across nine simulated bimanual manipulation tasks, we show that the best action frame is task-dependent and that MoF improves over oracle frame selection and standard Mixture-of-Experts (MoE) baselines. We further evaluate MoF on two real-world bimanual mobile manipulation tasks, demonstrating that it outperforms all constituent single-frame baselines. Project homepage: https://mofpo.github.io",
  "published": "2026-07-13",
  "updated": "2026-07-13",
  "year": "2026",
  "authors": [
   "Dian Wang",
   "Jisang Park",
   "Xiaomeng Xu",
   "Han Zhang",
   "Shuran Song",
   "Jeannette Bohg"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "This work proposes Mixture of Frames Policy (MoF), a diffusion policy that performs synchronized action denoising across multiple coordinate frames and demonstrates that it improves over oracle frame selection and standard Mixture-of-Experts (MoE) baselines.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Dian Wang",
    "id": "2119264352",
    "h_index": 17,
    "papers": 36
   },
   {
    "name": "Jisang Park",
    "id": "2155095608",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Xiaomeng Xu",
    "id": "2286521452",
    "h_index": 7,
    "papers": 9
   },
   {
    "name": "Han Zhang",
    "id": "2367748183",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Shuran Song",
    "id": "2364257433",
    "h_index": 6,
    "papers": 17
   },
   {
    "name": "Jeannette Bohg",
    "id": "2323565347",
    "h_index": 6,
    "papers": 12
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "imitation-diffusion",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.11884v1",
  "pdf_url": "https://arxiv.org/pdf/2607.11884v1",
  "html_url": "https://arxiv.org/html/2607.11884v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2607.11874",
  "slug": "a-minimalist-retargeting-guided-reinforcement-learning-recipe-for-dext",
  "title": "A Minimalist Retargeting-Guided Reinforcement Learning Recipe for Dexterous Manipulation",
  "abstract": "Recent work in humanoid whole-body control has found success with a simple recipe: retarget human motion to robot kinematic references, then train policies via reinforcement learning (RL) to track them. But how does this recipe transfer to dexterous manipulation? The answer is not obvious, as manipulation involves complex, contact-rich dynamics and requires delicate regulation of contact modes and forces. We present REGRIND, a minimalist retargeting-guided RL pipeline that learns dexterous manipulation policies from a single human demonstration. REGRIND retargets human hand-object motion to a robot reference that preserves hand-object spatial and contact relationships, trains a residual RL policy in simulation to track object-centric keypoints along that reference, and transfers the resulting policy zero-shot to hardware with careful system identification. The resulting policies produce fluid, human-like behavior on two different multi-fingered hands across contact-rich tool-use tasks, including operating a pair of scissors and turning a screwdriver. Through systematic hardware experiments, we identify and analyze the key factors that govern sim-to-real transfer in dexterous manipulation, offering practical guidance for retargeting-based learning in contact-rich settings. Videos and code are available at https://yunhaifeng.com/REGRIND.",
  "published": "2026-07-13",
  "updated": "2026-07-13",
  "year": "2026",
  "authors": [
   "Yunhai Feng",
   "Natalie Leung",
   "Jiaxuan Wang",
   "Lujie Yang",
   "Haozhi Qi",
   "Preston Culbertson"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Through systematic hardware experiments, this work identifies and analyze the key factors that govern sim-to-real transfer in dexterous manipulation, offering practical guidance for retargeting-based learning in contact-rich settings.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yunhai Feng",
    "id": "2348093839",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Natalie Leung",
    "id": "2449708548",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Jiaxuan Wang",
    "id": "2449715169",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Lujie Yang",
    "id": "2383205600",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Haozhi Qi",
    "id": "2247951244",
    "h_index": 11,
    "papers": 18
   },
   {
    "name": "Preston Culbertson",
    "id": "2386019755",
    "h_index": 1,
    "papers": 7
   }
  ],
  "comment": "Website: https://yunhaifeng.com/REGRIND",
  "topics": [
   "dexterous-manipulation",
   "humanoids",
   "egocentric-data",
   "tactile",
   "sim2real",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.11874v1",
  "pdf_url": "https://arxiv.org/pdf/2607.11874v1",
  "html_url": "https://arxiv.org/html/2607.11874v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.11734",
  "slug": "neuralactuator-neural-actuation-modeling-for-robot-dynamics-and-extern",
  "title": "NeuralActuator: Neural Actuation Modeling for Robot Dynamics and External Force Perception",
  "abstract": "Differentiable simulators have advanced policy learning and model-based control across robotic tasks. Yet actuator dynamics remain underexplored and can be a major source of sim-to-real error, particularly on low-cost platforms, where the linear current-to-joint-torque approximation $\u03c4= K_t I$ becomes unreliable because of friction, hysteresis, backlash, and thermal effects. Accurate actuator models can also support force perception and integrated force/position control. We present NeuralActuator, which jointly predicts (i) a torque surrogate for trajectory propagation on low-cost servo platforms, (ii) external forces with a contact-probability gate for sensorless force perception, and (iii) a motor-condition score for a supervised joint, distinguishing normal from mechanically restricted operation. A twin-arm teleoperation system records robot states and actuator telemetry alongside external-force labels, yielding the Neural Actuation Dataset (NAD). The torque-surrogate head is trained through differentiable simulation from pose trajectories without ground-truth joint-torque measurements. A Transformer captures temporal dependencies while enabling real-time inference. We validate NeuralActuator on a 5-DoF OpenManipulator-X, a 6-DoF SO-101 from LeRobot, and a 7-DoF Franka Emika Panda, spanning three actuator families and costs from approximately \\$500 to more than \\$30{,}000. The low-cost platforms support physically plausible dynamics and force evaluation, while the offline Franka experiment provides a payload-force-estimation benchmark. We also demonstrate motor-condition estimation and improved behavior-cloning performance using NeuralActuator as a pretrained module. We release the dataset, code, and hardware configurations on the project page: https://frank-zy-dou.github.io/projects/NeuralActuator/index.html.",
  "published": "2026-07-13",
  "updated": "2026-07-17",
  "year": "2026",
  "authors": [
   "Zhiyang Dou",
   "John U. Onyemelukwe",
   "Hangxing Zhang",
   "Heng Zhang",
   "Minghao Guo",
   "Yunsheng Tian",
   "Michal Piotr Lipiec",
   "Joshua Jacob",
   "Chao Liu",
   "Peter Yichen Chen",
   "Yuri Ivanov",
   "Wojciech Matusik"
  ],
  "author_count": 12,
  "categories": [
   "cs.RO",
   "cs.CV",
   "cs.GR",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "RSS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work validates NeuralActuator on a 5-DoF OpenManipulator-X, a 6-DoF SO-101 from LeRobot, and a 7-DoF Franka Emika Panda, spanning three actuator families and costs from approximately \\$500 to more than \\$30{,}000.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhiyang Dou",
    "id": "2292386296",
    "h_index": 8,
    "papers": 24
   },
   {
    "name": "John U. Onyemelukwe",
    "id": "2449707184",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Hang Zhang",
    "id": "2446728746",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Heng Zhang",
    "id": "2294361958",
    "h_index": 0,
    "papers": 5
   },
   {
    "name": "Minghao Guo",
    "id": "2258740917",
    "h_index": 7,
    "papers": 35
   },
   {
    "name": "Yunsheng Tian",
    "id": "2249538776",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Michal Piotr Lipiec",
    "id": "2379667161",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Joshua Jacob",
    "id": "2225246289",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Chao Liu",
    "id": "2315736390",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "P. Y. Chen",
    "id": "2294384470",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Yuri Ivanov",
    "id": "2351947806",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Wojciech Matusik",
    "id": "2295306221",
    "h_index": 5,
    "papers": 15
   }
  ],
  "comment": "RSS 2026. Outstanding Systems Paper Award. Project Page: https://people.csail.mit.edu/frankzydou/projects/NeuralActuator/index.html Code: https://github.com/Frank-ZY-Dou/Dynamics-Modeling/tree/main/NeuralActuator",
  "topics": [
   "sim2real",
   "data-teleop",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.11734v2",
  "pdf_url": "https://arxiv.org/pdf/2607.11734v2",
  "html_url": "https://arxiv.org/html/2607.11734v2",
  "code_url": "https://github.com/Frank-ZY-Dou/Dynamics-Modeling",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.5
 },
 {
  "id": "2607.11690",
  "slug": "requirement-driven-design-of-whole-body-social-tactile-sensing-via-vir",
  "title": "Requirement-Driven Design of Whole-Body Social Tactile Sensing via Virtual Human-Robot Interaction",
  "abstract": "Tactile sensing for social-physical human-robot interaction (spHRI) is designed in a hardware-driven manner, where predefined sensor configurations constrain coverage, spatial resolution, and the range of recognizable gestures. We propose a requirement-driven framework that derives sensing requirements, specifically spatial resolution and placement, directly from interaction data. Using a VR-based platform with haptic feedback, we collected high-resolution whole-body contact distributions across multiple social scenarios, from which we identified nine recurring social touch gestures. Eight gestures were selected for controlled data collection with 18 participants, yielding an open-source dataset of 5,520 trials. Analysis of contact distributions and simulated tactile encodings provides quantitative baselines for skin coverage and sensor density on a humanoid robot platform. While demonstrated on a single robot platform, the methodology is designed to be transferable to other robot morphologies, potentially enabling morphology-specific sensing requirements to be derived prior to hardware fabrication.",
  "published": "2026-07-13",
  "updated": "2026-07-13",
  "year": "2026",
  "authors": [
   "Dakarai Crowder",
   "Ruohan Zhang",
   "Alexis E. Block",
   "Wenzhen Yuan"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.HC"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "While demonstrated on a single robot platform, the methodology is designed to be transferable to other robot morphologies, potentially enabling morphology-specific sensing requirements to be derived prior to hardware fabrication.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Dakarai Crowder",
    "id": "2164874906",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Ruohan Zhang",
    "id": "2248277538",
    "h_index": 7,
    "papers": 16
   },
   {
    "name": "Alexis E. Block",
    "id": "3395075",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Wenzhen Yuan",
    "id": "2342300498",
    "h_index": 2,
    "papers": 8
   }
  ],
  "comment": "8 pages, 6 figures, accepted to IROS 2026",
  "topics": [
   "humanoids",
   "tactile",
   "data-teleop",
   "hardware-codesign",
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.11690v1",
  "pdf_url": "https://arxiv.org/pdf/2607.11690v1",
  "html_url": "https://arxiv.org/html/2607.11690v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2607.11688",
  "slug": "automated-synthesis-of-facial-mechanisms-for-conversational-animatroni",
  "title": "Automated Synthesis of Facial Mechanisms for Conversational Animatronic Robots",
  "abstract": "Animatronic faces are a central component of socially interactive robots, enabling rich nonverbal communication through facial articulation. However, state-of-the-art animatronic faces are typically tailored systems: each new facial geometry requires extensive manual mechanical redesign, making large-scale personalization prohibitively slow and costly. In this work, we pursue automated and scalable mechanical face synthesis, aiming to rapidly generate a physically realizable facial mechanism for a wide range of facial geometries. We introduce a parametric, linkage-driven mechanical face template whose topology and actuator layout are explicitly parameterized to support systematic scaling and retargeting across diverse facial morphologies. Building on this template, we propose a hierarchical automatic design algorithm that takes a single 2D portrait as input, reconstructs a target 3D face, and synthesizes a collision-free, manufacturable internal mechanism. The algorithm combines anatomy-guided feasible motion volumes, Action Unit (AU)-derived trajectory-based expressiveness objectives, and a collision-driven outer-loop refinement strategy. Beyond hardware synthesis, we argue that future mechanical faces deployed at scale must engage in bidirectional, multi-turn conversation rather than functioning solely as speaking or listening heads. To this end, we develop a dual-identity conversational facial motion synthesis framework that jointly models speaking and listening behaviors from audio, producing temporally coherent 3D facial motion suitable for physical execution. We validate our system through extensive experiments, including (i) quantitative evaluation of automatic mechanism synthesis across diverse facial geometries, (ii) comparisons against manual mechanical design, (iii) benchmarks on conversational facial motion synthesis and real-time deployment, and (iv) perceptual user studies.",
  "published": "2026-07-13",
  "updated": "2026-07-13",
  "year": "2026",
  "authors": [
   "Zongzheng Zhang",
   "Zi Lin",
   "Jiawen Yang",
   "Ziqiao Peng",
   "Junyan Lao",
   "Lin Cheng",
   "Huazhe Xu",
   "Hang Zhao",
   "Hao Zhao"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "RSS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "It is argued that future mechanical faces deployed at scale must engage in bidirectional, multi-turn conversation rather than functioning solely as speaking or listening heads, and a dual-identity conversational facial motion synthesis framework is developed that jointly models speaking and listening behaviors from audio.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zongzheng Zhang",
    "id": "2294931371",
    "h_index": 7,
    "papers": 22
   },
   {
    "name": "Zixuan Lin",
    "id": "2447992504",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Jiawen Yang",
    "id": "2287390651",
    "h_index": 9,
    "papers": 26
   },
   {
    "name": "Ziqiao Peng",
    "id": "2372193932",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Junyan Lao",
    "id": "2211013073",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Lin Cheng",
    "id": "2407764945",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Huazhe Xu",
    "id": "2373743788",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Hang Zhao",
    "id": "2363967646",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Hao Zhao",
    "id": "2363967648",
    "h_index": 4,
    "papers": 9
   }
  ],
  "comment": "Accepted by RSS 2026. Project page: https://zzongzheng0918.github.io/automated-facial-mechanisms-synthesis/",
  "topics": [
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.11688v1",
  "pdf_url": "https://arxiv.org/pdf/2607.11688v1",
  "html_url": "https://arxiv.org/html/2607.11688v1",
  "code_url": "https://zzongzheng0918.github.io/automated-facial-mechanisms-synthesis/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.5
 },
 {
  "id": "2607.11643",
  "slug": "xiaomi-robotics-u0-unified-embodied-synthesis-with-world-foundation-mo",
  "title": "Xiaomi-Robotics-U0: Unified Embodied Synthesis with World Foundation Model",
  "abstract": "Recent foundation image and video generation models offer strong generalization and controllability, but their direct application to embodied scenarios is limited by requirements for multi-view consistency, geometric coherence, and robot embodiment constraints. Existing methods typically adapt foundation models with limited robot data, often sacrificing visual knowledge acquired during large-scale pre-training. We present Xiaomi-Robotics-U0, a 38-billion-parameter multimodal autoregressive model for unified embodied synthesis. It treats embodied generation as an extension of foundation image and video generation and jointly optimizes text-to-image generation, image editing, embodied scene generation, embodied transfer, and embodied video generation. This unified framework preserves the generalization of the pre-trained world foundation model while adapting it to embodied settings. Xiaomi-Robotics-U0 is the first model to support high-quality multi-view scene generation across multiple robot embodiments and to introduce structured, controllable embodied transfer for fine-grained editing while preserving multi-view consistency and interaction dynamics. It achieves state-of-the-art results on single-step and sequential generation tasks, outperforming GPT-Image-2.0 in human evaluations of embodied scene generation and transfer, ranking first on World Arena for embodied video generation, and improving the out-of-distribution success rate of pi_0.5 from 36.9% to 63.2% on challenging real-world manipulation tasks. These results show that foundation world models can serve both as embodied world models and scalable data engines for embodied intelligence. Code and checkpoints are available at https://robotics.xiaomi.com/xiaomi-robotics-u0.html.",
  "published": "2026-07-13",
  "updated": "2026-07-13",
  "year": "2026",
  "authors": [
   "Xinghang Li",
   "Jun Guo",
   "Qiwei Li",
   "Long Qian",
   "Hang Lai",
   "Yueze Wang",
   "Hongyu Yan",
   "Jiahang Cao",
   "Xi Chen",
   "Jingen Qu",
   "Jiaxi Song",
   "Nan Sun",
   "Hanye Zhao",
   "Futeng Liu",
   "Wanli Peng",
   "Heyun Wang",
   "Yunhong Wang",
   "Caoyu Xia",
   "Jack Zhao",
   "Diyun Xiang",
   "Hangjun Ye",
   "Heng Qu",
   "Huaping Liu",
   "Jason Li"
  ],
  "author_count": 24,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "Xiaomi-Robotics-U0 is the first model to support high-quality multi-view scene generation across multiple robot embodiments and to introduce structured, controllable embodied transfer for fine-grained editing while preserving multi-view consistency and interaction dynamics.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xinghang Li",
    "id": "2155447887",
    "h_index": 8,
    "papers": 26
   },
   {
    "name": "Jun Guo",
    "id": "2293357006",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Qiwei Li",
    "id": "2277552764",
    "h_index": 6,
    "papers": 24
   },
   {
    "name": "Long Qian",
    "id": "2159713431",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Hang Lai",
    "id": "2276781675",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Yueze Wang",
    "id": "2217456303",
    "h_index": 16,
    "papers": 22
   },
   {
    "name": "Hongyu Yan",
    "id": "2375738389",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Jiahang Cao",
    "id": "2348487240",
    "h_index": 3,
    "papers": 14
   },
   {
    "name": "Xi Chen",
    "id": "2447359905",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Jingen Qu",
    "id": "2345878245",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Jia Song",
    "id": "2442218978",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Nan Sun",
    "id": "2322924729",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Han Zhao",
    "id": "2266256598",
    "h_index": 16,
    "papers": 36
   },
   {
    "name": "Futeng Liu",
    "id": "2411090440",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Wanli Peng",
    "id": "2449437929",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Heyun Wang",
    "id": "2447881347",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Yunhong Wang",
    "id": "2281412623",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Caoyu Xia",
    "id": "2449704713",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Jack Zhao",
    "id": "2449712522",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Diyun Xiang",
    "id": "2315503913",
    "h_index": 4,
    "papers": 15
   },
   {
    "name": "Hangjun Ye",
    "id": "2367554550",
    "h_index": 8,
    "papers": 35
   },
   {
    "name": "Hengxu Qu",
    "id": "2211044047",
    "h_index": 5,
    "papers": 15
   },
   {
    "name": "Huaping Liu",
    "id": "2293552267",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Jason Li",
    "id": "2382945637",
    "h_index": 2,
    "papers": 11
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "foundation-pretraining",
   "video-generation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.11643v1",
  "pdf_url": "https://arxiv.org/pdf/2607.11643v1",
  "html_url": "https://arxiv.org/html/2607.11643v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2607.11638",
  "slug": "da-nav-direction-aware-city-scale-vision-language-navigation",
  "title": "DA-Nav: Direction-Aware City-Scale Vision-Language Navigation",
  "abstract": "City-scale outdoor navigation is currently hindered by the heavy reliance on dense maps or costly navigation supervision. In this work, we introduce a novel paradigm for leveraging directional instructions from commercial navigation tools (e.g., Google Maps). To bridge the gap between commercial instructions and executable navigation actions, while mitigating long-horizon error accumulation through robust trajectory recovery, we propose DA-Nav, a Direction-Aware vision-language Navigation framework that reformulates navigation as a discrete spatial grounding problem on the egocentric 2D image plane. To achieve trajectory recovery, DA-Nav employs a Chain-of-Thought (CoT) reasoning process encompassing deviation assessment, action prediction, and target grid selection. We further introduce ReDA, a dataset that provides direction-aware instructions and recovery trajectories to enhance spatial grounding and support CoT recovery reasoning. Extensive experiments in CARLA demonstrate that DA-Nav achieves a high success rate of 56.16% in unseen urban environments, outperforming existing State-of-The-Art (SoTA) methods while maintaining a substantially stronger recovery capability. Furthermore, without fine-tuning, DA-Nav seamlessly adapts to both quadruped and humanoid robots, enabling stable kilometer-scale closed-loop outdoor navigation in complex real world environments.",
  "published": "2026-07-13",
  "updated": "2026-07-14",
  "year": "2026",
  "authors": [
   "Ye Yuan",
   "Kehan Chen",
   "Xinqiang Yu",
   "Wentao Xu",
   "Heng Wang",
   "Libo Huang",
   "Chuanguang Yang",
   "Yan Huang",
   "Jiawei He",
   "Zhulin An"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "DA-Nav is proposed, a Direction-Aware vision-language Navigation framework that reformulates navigation as a discrete spatial grounding problem on the egocentric 2D image plane, outperforming existing State-of-The-Art (SoTA) methods while maintaining a substantially stronger recovery capability.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ye Yuan",
    "id": "2449700122",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Kehan Chen",
    "id": "2335489827",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Xinqiang Yu",
    "id": "2328936584",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Wentao Xu",
    "id": "2448146282",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Heng Wang",
    "id": "2449754249",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Libo Huang",
    "id": "2247874007",
    "h_index": 10,
    "papers": 50
   },
   {
    "name": "Chuanguang Yang",
    "id": "102756770",
    "h_index": 19,
    "papers": 79
   },
   {
    "name": "Yan Huang",
    "id": "2369163799",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Jiawei He",
    "id": "2153103015",
    "h_index": 21,
    "papers": 56
   },
   {
    "name": "Zhulin An",
    "id": "2127813",
    "h_index": 28,
    "papers": 115
   }
  ],
  "comment": "9 pages, 8 figures",
  "topics": [
   "humanoids",
   "egocentric-data",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.11638v2",
  "pdf_url": "https://arxiv.org/pdf/2607.11638v2",
  "html_url": "https://arxiv.org/html/2607.11638v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.11633",
  "slug": "breaking-the-15-barrier-a-real-world-data-driven-system-for-proactive",
  "title": "Breaking the 15% Barrier: A Real-World Data-Driven System for Proactive Social Robot Triggered by User Nonverbal Cues",
  "abstract": "Service robots in retail stores increasingly rely on cascaded speech pipelines (STT-LLM-TTS), yet many customer-robot interactions are initiated or guided by nonverbal behaviors such as approaching, waving, pointing, or showing items. This paper studies such cues in a real-world store deployment with a teleoperated humanoid robot and shows that a non-negligible portion of robot turns are triggered by nonverbal behaviors rather than spoken input, revealing a limitation of audio-only dialogue systems. In a 6-day in-the-wild deployment, 15.3\\% of robot utterances were initiated by users' nonverbal behaviors rather than spoken input. Based on an analysis of observed customer behaviors, we define a set of frequent, service-relevant nonverbal cues and develop a real-time multi-person, multi-label recognizer that runs online from video. We then propose a dialogue framework that conditions LLM-based utterance generation on recognized nonverbal cue tokens, and optionally leverages a vision-language model when items are shown, enabling proactive robot responses without hand-crafted rules. We evaluate the approach offline on nonverbal-triggered turns and demonstrate an online prototype that reacts to users' nonverbal cues in real time.",
  "published": "2026-07-13",
  "updated": "2026-07-13",
  "year": "2026",
  "authors": [
   "Yuga Yano",
   "Yuki Okafuji",
   "Ryo Miyoshi",
   "Sanae Yamashita",
   "Yoshiki Ohira"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "A dialogue framework is proposed that conditions LLM-based utterance generation on recognized nonverbal cue tokens, and optionally leverages a vision-language model when items are shown, enabling proactive robot responses without hand-crafted rules.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yuga Yano",
    "id": "2184338694",
    "h_index": 2,
    "papers": 16
   },
   {
    "name": "Yuki Okafuji",
    "id": "2266493740",
    "h_index": 4,
    "papers": 22
   },
   {
    "name": "Ryo Miyoshi",
    "id": "2373026175",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Sanae Yamashita",
    "id": "81698366",
    "h_index": 3,
    "papers": 22
   },
   {
    "name": "Yoshiki Ohira",
    "id": "2357375547",
    "h_index": 1,
    "papers": 6
   }
  ],
  "comment": "8 pages, accepted as a conference paper for IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS2026)",
  "topics": [
   "humanoids",
   "hri",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.11633v1",
  "pdf_url": "https://arxiv.org/pdf/2607.11633v1",
  "html_url": "https://arxiv.org/html/2607.11633v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2607.11627",
  "slug": "ibpa-real-time-free-form-manifold-mesh-reconstruction-via-incremental",
  "title": "IBPA: Real-time Free-form Manifold Mesh Reconstruction via Incremental Ball Pivoting with Integrated Hole Detection",
  "abstract": "Both Remotely Operated underwater Vehicles (ROVs) and Autonomous Underwater Vehicles (AUVs) are frequently deployed to acquire geometric bathymetric data. However, it is often discovered post-survey that the acquired data coverage is incomplete. Given the high operational cost associated with underwater deployments, it is essential to incrementally visualize surface coverage in real-time to support informed decision-making by both the operators of ROVs and the AUVs during data collection. In addition, traditional incremental surface reconstruction methods, such as Digital Terrain Models (DTMs), are inherently limited in expressiveness: they represent surfaces as height fields, allows only one elevation value per $(x, y)$ coordinate and thus cannot capture overhangs or vertical structures. To overcome these limitations, we adapt the original Ball Pivoting Algorithm (BPA) into an incremental, real-time, and free-form surface reconstruction method, referred to as Incremental BPA (IBPA). Our method incrementally constructs an orientable, manifold mesh from streaming point cloud data without imposing assumptions regarding point cloud overlap or spatial distribution. Furthermore, we introduce a hole detection mechanism that identifies and highlights incomplete mesh regions. Compared to existing approaches, our method supports more complex surface topologies without prior structural assumptions. The source code of our reference implementation is available: https://github.com/Mauhing/Incremental-BPA",
  "published": "2026-07-13",
  "updated": "2026-07-13",
  "year": "2026",
  "authors": [
   "Mauhing Yip",
   "Mohit Singh",
   "Kostas Alexis",
   "Christian Schellewald",
   "Annette Stahl"
  ],
  "author_count": 5,
  "categories": [
   "cs.GR",
   "cs.RO"
  ],
  "primary_category": "cs.GR",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The original Ball Pivoting Algorithm is adapted into an incremental, real-time, and free-form surface reconstruction method, referred to as Incremental BPA (IBPA), which incrementally constructs an orientable, manifold mesh from streaming point cloud data without imposing assumptions regarding point cloud overlap or spatial distribution.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Mauhing Yip",
    "id": "1478385138",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Mohit Singh",
    "id": "2261688957",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Kostas Alexis",
    "id": "2261492914",
    "h_index": 10,
    "papers": 39
   },
   {
    "name": "Christian Schellewald",
    "id": "2839806",
    "h_index": 12,
    "papers": 39
   },
   {
    "name": "A. Stahl",
    "id": "39487865",
    "h_index": 15,
    "papers": 95
   }
  ],
  "comment": "The source code will be made public after the pre-print paper is available online",
  "topics": [
   "spatial-3d",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.11627v1",
  "pdf_url": "https://arxiv.org/pdf/2607.11627v1",
  "html_url": "https://arxiv.org/html/2607.11627v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.11624",
  "slug": "skoop-symmetric-koopman-predictions-for-faster-and-more-generalizable",
  "title": "SKooP: Symmetric Koopman Predictions for Faster and More Generalizable Legged Robot Locomotion with Reinforcement Learning",
  "abstract": "Reinforcement learning (RL) algorithms classically suffer from poor sample efficiency. In robotics, a recent line of work has emerged addressing this problem by encoding physics priors in the learning process. However, most of these approaches are validated on well-defined, low-dimensional benchmark systems rather than high-dimensional robots with complex nonlinear dynamics. In this paper, we introduce \\textit{SKooP (Symmetric Koopman Predictions)}, an approach combining the advantages of morphological symmetries with those of a Koopman model learned via autoencoder to enhance policy learning. SKooP learns a Koopman model of the system dynamics alongside the policy. The resulting Koopman predictions are used as privileged observations for the critic, allowing the agent to learn based on smoother, more informative features. We also incorporate group symmetries into the actor, critic, encoder and decoder networks to produce a highly equivariant policy. The SKooP approach is validated via in-depth analysis of the learned Koopman models and symmetric policies to showcase how each of these influences the agent's performance. We also show that the learned policies are transferable to different simulation environments. Our results show that SKooP consistently reduces convergence time and increases the learned reward for multiple challenging bipedal locomotion tasks on a quadruped robot. Project page: https://evelyd.github.io/SymmetricKoopmanPredictions",
  "published": "2026-07-13",
  "updated": "2026-07-16",
  "year": "2026",
  "authors": [
   "Evelyn D'Elia",
   "Weishu Zhan",
   "Giulio Turrisi",
   "Giulio Romualdi",
   "Giuseppe L'Erario",
   "Raffaello Camoriano",
   "Wei Pan",
   "Daniele Pucci"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "SKooP (Symmetric Koopman Predictions) is introduced, an approach combining the advantages of morphological symmetries with those of a Koopman model learned via autoencoder to enhance policy learning.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Evelyn D'Elia",
    "id": "1405284446",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Weishu Zhan",
    "id": "2294005929",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Giulio Turrisi",
    "id": "2111996905",
    "h_index": 9,
    "papers": 26
   },
   {
    "name": "Giulio Romualdi",
    "id": "51451149",
    "h_index": 12,
    "papers": 32
   },
   {
    "name": "Giuseppe Lerario",
    "id": "1516283157",
    "h_index": 8,
    "papers": 29
   },
   {
    "name": "R. Camoriano",
    "id": "1824575",
    "h_index": 11,
    "papers": 40
   },
   {
    "name": "Wei Pan",
    "id": "2273515589",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Daniele Pucci",
    "id": "2287942224",
    "h_index": 6,
    "papers": 21
   }
  ],
  "comment": "This paper has been accepted for publication at the IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS), Pittsburgh, USA, 2026",
  "topics": [
   "humanoids",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.11624v3",
  "pdf_url": "https://arxiv.org/pdf/2607.11624v3",
  "html_url": "https://arxiv.org/html/2607.11624v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2607.11603",
  "slug": "warpmpc-large-batch-mpc-on-gpu-via-admm-with-unrolled-ldl-top-factoriz",
  "title": "WarpMPC: Large-Batch MPC on GPU via ADMM with Unrolled $LDL^\\top$ Factorization",
  "abstract": "This paper introduces numerical optimizations for maximizing throughput on GPU when solving large batches (10,000 to over 100,000) of sequential quadratic programming (SQP) iterations, where all problems have the same structure. The optimizations are implemented in a toolbox WarpMPC for model-predictive control (MPC) in JAX and Warp. Based on the insight that all MPC problem instances in a batch share the same sparsity in time, cost, and constraints, we propose unrolling sparse linear factorizations and solves, which dominate alternating direction method of multipliers (ADMM) solver runtime. We avoid memory access bottlenecks and wasting computations via optimized memory layout, padding-reducing segmentation of the unrolled factorization, and dependency level scheduled backsolves, additionally accelerating sensitivity computation. We achieve throughputs of 8,000 to 250,000 SQP iterations per second on nonlinear cartpole, quadrotor, and humanoid robot benchmarks, outperforming baselines by 3$\\times$ to 25$\\times$. We illustrate practical usefulness by synthesizing a dataset and training a neural network approximation of an MPC in under 4 minutes that stabilizes a nano quadrotor in hardware experiments.",
  "published": "2026-07-13",
  "updated": "2026-07-13",
  "year": "2026",
  "authors": [
   "Henrik Hose",
   "Se Hwan Jeon",
   "Charles Khazoom",
   "Sangbae Kim",
   "Sebastian Trimpe"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "eess.SY",
   "math.OC"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Numerical optimizations for maximizing throughput on GPU when solving large batches of sequential quadratic programming (SQP) iterations, where all problems have the same structure are introduced.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Henrik Hose",
    "id": "118266493",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Seungmin Jeon",
    "id": "2158860285",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Charles Khazoom",
    "id": "50812170",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Sangbae Kim",
    "id": "2283965697",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Sebastian Trimpe",
    "id": "2334864746",
    "h_index": 8,
    "papers": 28
   }
  ],
  "comment": "",
  "topics": [
   "humanoids"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.11603v1",
  "pdf_url": "https://arxiv.org/pdf/2607.11603v1",
  "html_url": "https://arxiv.org/html/2607.11603v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.11481",
  "slug": "towards-human-level-dexterous-teleoperation",
  "title": "Towards Human-level Dexterous Teleoperation",
  "abstract": "Humans routinely wield tools, swap grasps, and reposition objects within a single hand, seamlessly orchestrating contact transitions that span translation, reorientation, and finger gaiting. Endowing robot dexterous hands with this level of in-hand dexterity through teleoperation requires precise control of object motion via dynamic hand-object contact, yet current teleoperation systems remain far from this capability. To bridge this gap, we take a major step towards human-level dexterous teleoperation by introducing TeleDexter, a hand-object co-tracking controller that maps operator intent into learned, low-level contact execution. The controller is trained on consecutive co-tracking subgoals derived from human reference motions, utilizing a hybrid reward that couples sparse subgoal objectives with dense tracking rewards to enable learning across diverse interaction modalities rather than frame-wise trajectory imitation. The entire pipeline requires only single-stage RL and, with random action masking and domain randomization, transfers zero-shot to the real robot. We evaluate TeleDexter on seven challenging dexterous teleoperation tasks spanning object reorientation and long-horizon tool use across two dexterous hands, achieving a 75% average success rate where all baselines consistently fail. Furthermore, the collected demonstrations successfully train autonomous policies via behavioral cloning, marking a concrete step towards human-level dexterous teleoperation.",
  "published": "2026-07-13",
  "updated": "2026-07-13",
  "year": "2026",
  "authors": [
   "Puhao Li",
   "Zeyuan Chen",
   "Yingying Wu",
   "Pengkun Wei",
   "Yuyang Li",
   "Tianyu Wang",
   "Jiaxiao Shi",
   "Mingrui Yu",
   "Baoxiong Jia",
   "Song-chun Zhu",
   "Tengyu Liu",
   "Siyuan Huang"
  ],
  "author_count": 12,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "TeleDexter is introduced, a hand-object co-tracking controller that maps operator intent into learned, low-level contact execution that successfully train autonomous policies via behavioral cloning, marking a concrete step towards human-level dexterous teleoperation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Puhao Li",
    "id": "2145015272",
    "h_index": 13,
    "papers": 18
   },
   {
    "name": "Zeyuan Chen",
    "id": "2362661308",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Yingying Wu",
    "id": "2367899887",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Pengkun Wei",
    "id": "2233339178",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Yuyang Li",
    "id": "2261448933",
    "h_index": 8,
    "papers": 11
   },
   {
    "name": "Tianyue Wang",
    "id": "2447600606",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Jiaxiao Shi",
    "id": "2449700750",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Mingrui Yu",
    "id": "2047407832",
    "h_index": 9,
    "papers": 32
   },
   {
    "name": "Baoxiong Jia",
    "id": "26663607",
    "h_index": 27,
    "papers": 61
   },
   {
    "name": "Song-Chun Zhu",
    "id": "2347682849",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Tengyu Liu",
    "id": "2110032600",
    "h_index": 21,
    "papers": 31
   },
   {
    "name": "Siyuan Huang",
    "id": "2264375840",
    "h_index": 13,
    "papers": 26
   }
  ],
  "comment": "Project Website: https://bigai-dex.github.io/blog/teledexter/",
  "topics": [
   "dexterous-manipulation",
   "sim2real",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.11481v1",
  "pdf_url": "https://arxiv.org/pdf/2607.11481v1",
  "html_url": "https://arxiv.org/html/2607.11481v1",
  "code_url": "https://bigai-dex.github.io/blog/teledexter/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2607.11270",
  "slug": "towards-predictive-aligned-and-scalable-robot-learning",
  "title": "Towards Predictive, Aligned, and Scalable Robot Learning",
  "abstract": "Learning, at its core, extends beyond memorization to the ability to reason and solve novel problems by navigating a space of possibilities. We introduce Lumo-2, a latent world-action model that generates actions by reasoning over world dynamics in latent space. The learned latent world dynamics capture physically grounded visual transitions, naturally encoding future possibilities and providing a unified substrate for cross-modal alignment. This formulation enables predictive reasoning akin to world modelling while remaining lightweight and focused on physical dynamics relevant to control. Central to our approach is the hypothesis that action generation quality is governed by the geometry of the latent space. We observe that standard reconstruction-based action tokenization objectives induce representations biased toward low-level signal fidelity, leading to misalignment between reconstruction quality and downstream control performance. To address this limitation, we propose a multi-stage modality pre-alignment strategy in which action representations are progressively aligned with latent world dynamics, vision, and language. This process enforces cross-modal consistency, promotes abstraction, and induces a structured latent space for predictive reasoning. We provide a systematic empirical study of latent world modelling and modality alignment, analyzing their roles in scaling laws and out-of-distribution generalization. Results show that Lumo-2 consistently outperforms strong vision-language-action (VLA) and world-action model (WAM) baselines, with gains on challenging real-world tasks requiring temporal reasoning, physical understanding, or high control complexity, including long-horizon and dexterous manipulation. These findings suggest that structured multimodal alignment and predictive reasoning are fundamental principles for advancing embodied intelligence.",
  "published": "2026-07-13",
  "updated": "2026-07-13",
  "year": "2026",
  "authors": [
   "Peijun Tang",
   "Shangjin Xie",
   "Baifu Huang",
   "Binyan Sun",
   "Haotian Yang",
   "Kuncheng Luo",
   "Weiqi Jin",
   "Shilin Fang",
   "Jianan Wang"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Lumo-2 is introduced, a latent world-action model that generates actions by reasoning over world dynamics in latent space that consistently outperforms strong vision-language-action and world-action model baselines, with gains on challenging real-world tasks requiring temporal reasoning, physical understanding, or high control complexity.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Peijun Tang",
    "id": "2359636324",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Shang-Ping Xie",
    "id": "2244801731",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Baifu Huang",
    "id": "2398007608",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Bin Sun",
    "id": "2407601378",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Hao Yang",
    "id": "2290848696",
    "h_index": 6,
    "papers": 19
   },
   {
    "name": "Kuncheng Luo",
    "id": "2397488098",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Weiqi Jin",
    "id": "2398972686",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Shilin Fang",
    "id": "2375724303",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Jianan Wang",
    "id": "2396383245",
    "h_index": 4,
    "papers": 15
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "vla",
   "dexterous-manipulation",
   "foundation-pretraining",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.11270v1",
  "pdf_url": "https://arxiv.org/pdf/2607.11270v1",
  "html_url": "https://arxiv.org/html/2607.11270v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.11184",
  "slug": "geogs-slam-online-monocular-reconstruction-using-gaussian-splatting-wi",
  "title": "GeoGS-SLAM: Online Monocular Reconstruction Using Gaussian Splatting with Geometric Priors",
  "abstract": "SLAM methods based on 3D Gaussian Splatting (3DGS) have demonstrated impressive tracking and mapping performance, but typically require additional geometric information from external depth sensors. Meanwhile, recent SLAM systems that leverage geometric priors from pre-trained feed-forward models enable real-time dense reconstruction, yet often discard original RGB information during optimization, thus degrading overall reconstruction quality. We present GeoGS-SLAM, an online monocular dense reconstruction system that combines the 3DGS-based map representation with learned geometric priors. Given uncalibrated RGB input, we first employ a feed-forward visual geometry model to predict camera and scene priors. The Gaussian scene map is then expanded by directly sampling Gaussian primitives from both RGB input and geometric priors. Camera poses and the scene map are jointly optimized through a coarse-to-fine strategy that minimizes both photometric and geometric losses. To ensure global consistency, we further incorporate online loop closure detection and pose graph optimization. Extensive experiments across indoor and outdoor benchmarks demonstrate that GeoGS-SLAM achieves superior rendering quality and tracking accuracy compared to state-of-the-art methods while maintaining online real-time performance. Project page: https://rlgao.github.io/geogs_slam.",
  "published": "2026-07-13",
  "updated": "2026-07-13",
  "year": "2026",
  "authors": [
   "Ruilan Gao",
   "Letian Jin",
   "Yu Zhang"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "GeoGS-SLAM is presented, an online monocular dense reconstruction system that combines the 3DGS-based map representation with learned geometric priors and achieves superior rendering quality and tracking accuracy compared to state-of-the-art methods while maintaining online real-time performance.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ruilan Gao",
    "id": "2238103480",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Letian Jin",
    "id": "2449757300",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yu Zhang",
    "id": "2315644627",
    "h_index": 2,
    "papers": 5
   }
  ],
  "comment": "",
  "topics": [
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.11184v1",
  "pdf_url": "https://arxiv.org/pdf/2607.11184v1",
  "html_url": "https://arxiv.org/html/2607.11184v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.11167",
  "slug": "pix2act-image-space-manipulation-policies-with-equivariant-augmentatio",
  "title": "Pix2Act: Image-Space Manipulation Policies with Equivariant Augmentation",
  "abstract": "Representing manipulation actions as 2D trajectories in the camera plane provides a compact and interpretable basis for learning complex 3D manipulation policies. However, it also creates challenges from out-of-frame trajectories and limited precision. We propose Pix2Act, an imitation learning method that addresses these challenges by generating continuous image-space keypoint trajectories in each camera plane and losslessly recovering end-effector poses via triangulation. This reformulates high-dimensional 3D control as a simpler, more learnable 2D prediction problem. Crucially, it aligns observations and actions in the same coordinate space, enabling equivariant transformations to jointly rotate individual camera images together with their image-space actions. We analyze the symmetry properties of this augmentation and design a network architecture that can fuse multiple camera views while respecting their per-view rotations. As a result, Pix2Act implicitly enlarges the support of the data distribution and learns invariant action structures across transformations, yielding improved generalization and overall performance. Across diverse simulated and real-world manipulation tasks, Pix2Act outperforms state-of-the-art baselines and remains robust under camera perturbations.",
  "published": "2026-07-13",
  "updated": "2026-07-13",
  "year": "2026",
  "authors": [
   "Haojie Huang",
   "Linfeng Zhao",
   "Haotian Liu",
   "Zhang Ye",
   "Si-Yuan Huang",
   "Mingxi Jia",
   "Boce Hu",
   "Fangzhou Lin",
   "Yu Qi",
   "Dian Wang",
   "Robin Walters",
   "Robert Platt"
  ],
  "author_count": 12,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "P Pix2Act is proposed, an imitation learning method that addresses high-dimensional 3D control as a simpler, more learnable 2D prediction problem by generating continuous image-space keypoint trajectories in each camera plane and losslessly recovering end-effector poses via triangulation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hao-zhe Huang",
    "id": "2143569284",
    "h_index": 8,
    "papers": 23
   },
   {
    "name": "Linfeng Zhao",
    "id": "2308044351",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Haotian Liu",
    "id": "2307380968",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Zhangchen Ye",
    "id": "2402503135",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Si-Yuan Huang",
    "id": "2449699889",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Ming Jia",
    "id": "2148250127",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Boce Hu",
    "id": "2312110011",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Fangzhou Lin",
    "id": "2331651062",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Yu Qi",
    "id": "2311499051",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Dian Wang",
    "id": "2119264352",
    "h_index": 17,
    "papers": 36
   },
   {
    "name": "Robin Walters",
    "id": "2287354248",
    "h_index": 7,
    "papers": 22
   },
   {
    "name": "Robert Platt",
    "id": "2280136750",
    "h_index": 8,
    "papers": 25
   }
  ],
  "comment": "Project Website: https://haojhuang.github.io/pix2act_page/",
  "topics": [
   "imitation-diffusion"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.11167v1",
  "pdf_url": "https://arxiv.org/pdf/2607.11167v1",
  "html_url": "https://arxiv.org/html/2607.11167v1",
  "code_url": "https://haojhuang.github.io/pix2act_page/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.3
 },
 {
  "id": "2607.11128",
  "slug": "comparison-based-ordinal-learning-for-proactive-driving-risk-assessmen",
  "title": "Comparison-Based Ordinal Learning for Proactive Driving Risk Assessment",
  "abstract": "Real-time driving risk assessment provides an essential basis for proactive safety by identifying and quantifying the danger of ongoing road interactions before adverse outcomes occur. However, due to the scarcity of collision data and frame-level risk labels, existing driving risk assessment methods often rely on surrogate objectives, which may imperfectly align with true collision risk and not faithfully reflect the relative danger of driving interaction. This paper proposes a comparison-based ordinal risk learning framework that learns collision-relevant risk scores from pairwise supervision in driving data, directly modeling relative risk ordering without requiring numerical frame-level risk labels. We derive pairwise comparisons from three sources of event-structured driving data for such ordinal risk learning: temporal progression within safety-critical sequences, event-level contrast between dangerous and normal interactions, and physics-based counterfactual perturbations. On this basis, instantiations with three risk-scoring function parameterizations are implemented, including directly learning risk scores from comparison data, and aligning existing single or multiple surrogate-based risk models. The proposed framework is evaluated on the 100-Car and SHRP2 naturalistic driving datasets using a proactive collision warning task. Results show that the proposed framework improves high-recall risk discrimination, warning precision, and warning lead time over representative surrogate-based baselines across both in-distribution and out-of-distribution evaluations. These results suggest that the proposed framework can contribute to proactive safety research by providing more reliable risk assessment for automated driving systems and safety-critical driving interactions.",
  "published": "2026-07-13",
  "updated": "2026-07-13",
  "year": "2026",
  "authors": [
   "Zhuoren Li",
   "Yi Zhong",
   "Weiqi Zhang",
   "Xinrui Zhang",
   "Lu Xiong",
   "Chongfeng Wei",
   "Bo Leng"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A comparison-based ordinal risk learning framework that learns collision-relevant risk scores from pairwise supervision in driving data, directly modeling relative risk ordering without requiring numerical frame-level risk labels is proposed.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhuoren Li",
    "id": "1570118579",
    "h_index": 7,
    "papers": 45
   },
   {
    "name": "Yifu Zhong",
    "id": "2448200038",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Weiqi Zhang",
    "id": "2257082330",
    "h_index": 0,
    "papers": 6
   },
   {
    "name": "Xinrui Zhang",
    "id": "2311568440",
    "h_index": 4,
    "papers": 17
   },
   {
    "name": "Lu Xiong",
    "id": "2284081713",
    "h_index": 5,
    "papers": 16
   },
   {
    "name": "Chongfeng Wei",
    "id": "2223269355",
    "h_index": 8,
    "papers": 24
   },
   {
    "name": "B. Leng",
    "id": "12947744",
    "h_index": 15,
    "papers": 101
   }
  ],
  "comment": "15 pages, 5 figures",
  "topics": [
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.11128v1",
  "pdf_url": "https://arxiv.org/pdf/2607.11128v1",
  "html_url": "https://arxiv.org/html/2607.11128v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.11119",
  "slug": "via-visual-interface-agent-for-robot-control",
  "title": "VIA: Visual Interface Agent for Robot Control",
  "abstract": "Robot manipulation is a complex task that requires visual understanding, physical reasoning, planning, and closed-loop control. General-purpose foundation models (FMs) have grown remarkably capable of some of these, especially vision and reasoning. To leverage this for generalist robot policies, current methods typically involve converting existing FMs into vision-language-action (VLA) models by fine-tuning on robot data to output low-level actions. However, VLAs are often orders of magnitude smaller than frontier FMs given the limited data and compute available for fine-tuning, which in turn limits their general capability. Inspired by the growing ability of FMs to operate software through visual interfaces, we ask whether that same competence suffices to control a robot. We present VIA (Visual Interface Agent for robot control), a framework that recasts robot control as an agentic task: an off-the-shelf FM-powered agent drives a manipulator through a browser-based 3D interface by taking screenshots, issuing intuitive commands, observing the outcome, and adjusting. The agent receives no robot-specific fine-tuning and no access to privileged state information: it perceives visual input and acts through a small set of general tools. VIA inherits the agent's general reasoning, closed-loop error recovery, and ability to plan and re-plan from what it observes. It solves a diverse suite of tabletop manipulation tasks zero-shot with both Claude Code and Codex. With the strongest model (Fable 5) it achieves 96.7% success on three LIBERO-Goal tasks and 100% on a long-horizon rainbow assembly task. Performance improves with the scale and strength of the underlying model. These results suggest that frontier agents already possess skills that transfer directly to robot control given the right interface: your coding or computer-use agent is, in a sense, secretly a robot-control agent.",
  "published": "2026-07-13",
  "updated": "2026-07-13",
  "year": "2026",
  "authors": [
   "Hengyuan Hu",
   "Priya Sundaresan",
   "Jensen Gao",
   "Dorsa Sadigh"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work presents VIA (Visual Interface Agent for robot control), a framework that recasts robot control as an agentic task: an off-the-shelf FM-powered agent drives a manipulator through a browser-based 3D interface by taking screenshots, issuing intuitive commands, observing the outcome, and adjusting.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hengyuan Hu",
    "id": "2265518772",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Priya Sundaresan",
    "id": "123235030",
    "h_index": 19,
    "papers": 31
   },
   {
    "name": "Jensen Gao",
    "id": "2238154243",
    "h_index": 8,
    "papers": 10
   },
   {
    "name": "Dorsa Sadigh",
    "id": "1779671",
    "h_index": 69,
    "papers": 225
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.11119v1",
  "pdf_url": "https://arxiv.org/pdf/2607.11119v1",
  "html_url": "https://arxiv.org/html/2607.11119v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.11041",
  "slug": "pake-learning-whole-body-loco-manipulation-with-partial-kinematic-embe",
  "title": "PAKE: Learning Whole-Body Loco-Manipulation with Partial Kinematic Embeddings",
  "abstract": "Loco-manipulation has recently shown promising capabilities; however, achieving high-precision control, managing the high-dimensional action space induced by many degrees of freedom (DoFs), and fully exploiting the inherent redundancy of whole-body systems remain challenging. In this paper, we propose a novel whole-body control framework that effectively addresses these challenges by decomposing the complex loco-manipulation problem into partial reference motion generation and low-level imitation control. We introduce a new Kinematic Normalizing Flow (KNF) model, trained on a large-scale kinematic dataset, that generates diverse yet feasible partial reference motions. A high-level controller is then trained to navigate the KNF's latent space to exploit redundant solutions, while a low-level controller ensures physically feasible and accurate motion execution. We validate our approach on the quadrupedal robot equipped with a six-DoF robotic arm. In simulation, experimental results show that our approach significantly outperforms state-of-the-art methods in terms of tracking accuracy and feasible workspace coverage. For hardware deployment, we evaluate the system over 24 episodes across 8 different mobile loco-manipulation tasks. The system achieves end-effector pose-tracking errors of 4.5 cm and 0.14 rad, while maintaining accurate locomotion tracking with linear and angular velocity errors of 0.1 m/s and 0.01 rad/s, respectively, outperforming competitive baselines. Our method represents a practical and powerful solution for accurate and generalized whole-body loco-manipulation in high-DoF robotic systems, with promising potential for diverse downstream robotic tasks.",
  "published": "2026-07-13",
  "updated": "2026-07-13",
  "year": "2026",
  "authors": [
   "Zhengmao He",
   "Moonkyu Jung",
   "Hyeongjun Kim",
   "Jiseong Lee",
   "Hui Zhang",
   "Jemin Hwangbo",
   "Jie Song"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper introduces a new Kinematic Normalizing Flow (KNF) model, trained on a large-scale kinematic dataset, that generates diverse yet feasible partial reference motions that effectively addresses whole-body control challenges by decomposing the complex loco-manipulation problem into partial reference motion generation and low-level imitation control.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhengmao He",
    "id": "2265617102",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Moonkyu Jung",
    "id": "2233287834",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Hyeongjun Kim",
    "id": "2249537064",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Jiseong Lee",
    "id": "9174317",
    "h_index": 4,
    "papers": 15
   },
   {
    "name": "Hui Zhang",
    "id": "2238389387",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Jemin Hwangbo",
    "id": "1707297",
    "h_index": 25,
    "papers": 48
   },
   {
    "name": "Jie Song",
    "id": "2319387162",
    "h_index": 4,
    "papers": 7
   }
  ],
  "comment": "",
  "topics": [
   "humanoids"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.11041v1",
  "pdf_url": "https://arxiv.org/pdf/2607.11041v1",
  "html_url": "https://arxiv.org/html/2607.11041v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.11027",
  "slug": "segdiff-segmented-trajectory-diffusion-for-consistent-and-adaptive-rob",
  "title": "SegDiff: Segmented Trajectory Diffusion for Consistent and Adaptive Robot Manipulation",
  "abstract": "Imitation learning enables robots to acquire manipulation skills from demonstrations by mapping observations to actions. Existing approaches predict either short-horizon continuous action sequences or discrete keyposes. However, continuous prediction methods suffer from compounding errors due to short prediction horizons and struggle with multi-modal action distributions, whereas keypose-based methods necessitate an external planner, constraining real-time applicability. To address these challenges, we introduce SegDiff, a closed-loop visuomotor policy that integrates the strengths of both paradigms. SegDiff decomposes demonstrations into motion segments between keyposes and learns to predict the continuous trajectory from the current state to the next keypose, enabling long-horizon prediction with real-time refinement. Furthermore, we leverage the capability of diffusion models and DDIM inversion to propose a Dynamic Temporal Ensembling mechanism, which allows the policy to efficiently respond to dynamic environments and mitigate discontinuities caused by inconsistent multi-modal sampling. SegDiff demonstrates significant performance gains over existing approaches across various simulated and real-world scenarios, indicating its strong ability to reason over extended temporal dependencies while maintaining real-time adaptability and control stability.",
  "published": "2026-07-13",
  "updated": "2026-07-13",
  "year": "2026",
  "authors": [
   "Haidong Cao",
   "Wenjun Cao",
   "Quanhao Li",
   "Sicheng Xie",
   "Zhiying Du",
   "Jiaqi Leng",
   "Zuxuan Wu",
   "Yu-Gang Jiang"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "SegDiff, a closed-loop visuomotor policy that decomposes demonstrations into motion segments between keyposes and learns to predict the continuous trajectory from the current state to the next keypose, enabling long-horizon prediction with real-time refinement.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Haidong Cao",
    "id": "2347189486",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Wenjun Cao",
    "id": "2449716827",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Quanhao Li",
    "id": "2146088185",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Sicheng Xie",
    "id": "2269466881",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Zhiying Du",
    "id": "2390621033",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Jiaqi Leng",
    "id": "2215640415",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Zuxuan Wu",
    "id": "2341646178",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Yu-Gang Jiang",
    "id": "1717861",
    "h_index": 47,
    "papers": 141
   }
  ],
  "comment": "",
  "topics": [
   "imitation-diffusion",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.11027v1",
  "pdf_url": "https://arxiv.org/pdf/2607.11027v1",
  "html_url": "https://arxiv.org/html/2607.11027v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.11018",
  "slug": "whole-body-semantic-to-actuation-grounding-of-elephant-inspired-soft-t",
  "title": "Whole-Body Semantic-to-Actuation Grounding of Elephant-Inspired Soft-Trunk Motion via Lightweight Flow Matching",
  "abstract": "For close-contact human-robot interaction (HRI), trunk-like continuum manipulators provide a physical channel for diverse whole-body expression, but grounding open-vocabulary responses into such robots is difficult: end-effector motion underspecifies body shape, whereas direct whole-body commands are high-dimensional and hard to keep feasible. We propose a whole-body semantic-to-actuation grounding framework for elephant-inspired soft-trunk HRI based on lightweight flow matching. The framework converts responses from a multimodal large language model into bounded, morphology-aligned intent-intensity tuples, parameterizes tendon-actuation trajectories with compact Catmull-Rom spline controls, and uses a rectified-flow generator to sample feasible whole-body trunk motions. Experiments show that the proposed framework improves held-out grounding correctness from 25.0% to 77.2% over a raw-response dense-regression baseline. Compared with a denoising-diffusion baseline, it improves correctness from 71.9% to 77.2% and reduces inference time from 7.86 ms to 4.87 ms while preserving motion diversity. A 100-participant physical HRI study further shows that adding the generated soft-trunk motion channel increases the positive overall-satisfaction rating from 46% to 82% over the audiovisual-only baseline.",
  "published": "2026-07-13",
  "updated": "2026-07-13",
  "year": "2026",
  "authors": [
   "Tingcong Liu",
   "Tongshun Chen",
   "Siyi Ma",
   "Yuhao Wang",
   "Aye Phyu Phyu Aung",
   "Ibrahim Alsarraj",
   "J. Senthilnath",
   "Bo An",
   "Ke Wu"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A whole-body semantic-to-actuation grounding framework for elephant-inspired soft-trunk HRI based on lightweight flow matching is proposed and shows that adding the generated soft-trunk motion channel increases the positive overall-satisfaction rating from 46% to 82% over the audiovisual-only baseline.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Tingcong Liu",
    "id": "2362387251",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Tongshun Chen",
    "id": "2380082277",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Siyi Ma",
    "id": "2448646006",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Yuhao Wang",
    "id": "2281682178",
    "h_index": 1,
    "papers": 10
   },
   {
    "name": "A. Phyu",
    "id": "2753785",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Phyu Aung",
    "id": "2284716001",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Ibrahim Alsarraj",
    "id": "2353625812",
    "h_index": 1,
    "papers": 9
   },
   {
    "name": "J. Senthilnath",
    "id": "2279918120",
    "h_index": 2,
    "papers": 18
   },
   {
    "name": "Bo An",
    "id": "2449705325",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Ke Wu",
    "id": "2390529706",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Mllm Vla",
    "id": "2449711488",
    "h_index": 0,
    "papers": 1
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "hardware-codesign",
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.11018v1",
  "pdf_url": "https://arxiv.org/pdf/2607.11018v1",
  "html_url": "https://arxiv.org/html/2607.11018v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.10975",
  "slug": "real-time-rulebook-aware-nonlinear-mpc-for-autonomous-driving-with-pri",
  "title": "Real-Time Rulebook-Aware Nonlinear MPC for Autonomous Driving with Priority-Biased Tiered Slacks",
  "abstract": "Autonomous-vehicle motion planners must resolve conflicts among safety, regulation, comfort, and efficiency in real time while exposing those decisions for audit. We present W-SQP, a weighted tiered-slack nonlinear model predictive controller (NMPC) that compiles nine driving-rule families into a four-tier shared-slack nonlinear program solved online with CasADi and IPOPT; the name denotes the weighted quadratic slack penalty, not a sequential-quadratic-programming solver. Strongly separated tier penalties bias residual violations toward lower-priority rules while leaving actuation bounds hard. The controller replans from its executed state at $10$\\,Hz and records per-rule residuals on every cycle. A $90$\\,ms solver-time limit returns an anytime iterate that is projected through the vehicle dynamics before execution; median and maximum observed wall-clock solve times were $28$ and $104$\\,ms. We evaluate W-SQP in closed loop on 150 Waymo Open Motion Dataset scenarios in Waymax against reactive and proposal-and-select baselines, and introduce a log-independent protocol that separates safety and regulatory compliance from resemblance to the recorded human trajectory. Under this protocol, W-SQP shows no systematic group-level deficit relative to expert replay on the log-independent safety and regulatory rules, with several localized regressions in the hardest, highest-divergence scenarios. The results characterize W-SQP as an auditable, priority-biased, anytime-capable NMPC prototype rather than a hard-real-time or formally safe controller.",
  "published": "2026-07-13",
  "updated": "2026-07-13",
  "year": "2026",
  "authors": [
   "Hadi Hajieghrary",
   "Benedikt Walter",
   "Chaitanya Shinde",
   "Paul Schmitt",
   "Miguel Hurtado"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "math.OC"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "W-SQP is presented, a weighted tiered-slack nonlinear model predictive controller that compiles nine driving-rule families into a four-tier shared-slack nonlinear program solved online with CasADi and IPOPT and introduced a log-independent protocol that separates safety and regulatory compliance from resemblance to the recorded human trajectory.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hadi Hajieghrary",
    "id": "1814561",
    "h_index": 9,
    "papers": 26
   },
   {
    "name": "Benedikt Walter",
    "id": "2352945630",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Chaitanya Shinde",
    "id": "2312430530",
    "h_index": 2,
    "papers": 12
   },
   {
    "name": "Paul Schmitt",
    "id": "2056025419",
    "h_index": 6,
    "papers": 18
   },
   {
    "name": "M. Hurtado",
    "id": "145988317",
    "h_index": 3,
    "papers": 17
   }
  ],
  "comment": "The manuscript is submitted to the Journal of Control Engineering Practice, A journal of The International Federation of Automatic Control (IFAC)",
  "topics": [
   "rl-control",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.10975v1",
  "pdf_url": "https://arxiv.org/pdf/2607.10975v1",
  "html_url": "https://arxiv.org/html/2607.10975v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2607.10892",
  "slug": "a-single-diffusion-policy-controller-for-multi-task-block-pushing-with",
  "title": "A Single Diffusion-Policy Controller for Multi-Task Block Pushing with Zero-Shot Sim-to-Real Transfer",
  "abstract": "Diffusion policies have shown promising empirical performance in representing and learning complex maneuvers for robots using behavior cloning (BC). In this paper, we explore training diffusion policies from scratch using reinforcement learning (RL) for multi-task robotic manipulation. Specifically, we aim to train a single diffusion policy for block-pushing tasks with multiple shapes. The proposed framework features a simple policy loss function, which is a reweighted evidence lower bound used in BC-based diffusion policy training and can seamlessly serve as the policy learning module in RL algorithms. To address the exploration challenges arising from the absence of demonstrations, we incorporate reverse curriculum generation and objective-centric representations. Combined with the expressiveness of diffusion policies, our design supports learning of multi-task block-pushing policies in our sparse-reward simulation setting. We further evaluate whether the trained diffusion policy transfers in zero-shot to real-world tasks under varying environmental conditions including goal positions, block shapes, block weights and surface friction, providing evidence that this pipeline can transfer to our real-world block-pushing setup under the tested variations.",
  "published": "2026-07-12",
  "updated": "2026-07-12",
  "year": "2026",
  "authors": [
   "Haitong Ma",
   "Haldun Balim",
   "Yang Hu",
   "Bo Dai",
   "Na Li"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper aims to train a single diffusion policy for block-pushing tasks with multiple shapes using reinforcement learning for multi-task robotic manipulation and incorporates reverse curriculum generation and objective-centric representations to address the exploration challenges arising from the absence of demonstrations.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Haitong Ma",
    "id": "2295874222",
    "h_index": 5,
    "papers": 22
   },
   {
    "name": "Haldun Balim",
    "id": "40928122",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Yang Hu",
    "id": "2296611544",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Bo Dai",
    "id": "2295666916",
    "h_index": 4,
    "papers": 17
   },
   {
    "name": "Na Li",
    "id": "2300488405",
    "h_index": 2,
    "papers": 7
   }
  ],
  "comment": "8 pages, 7 figures",
  "topics": [
   "sim2real",
   "imitation-diffusion",
   "rl-control",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.10892v1",
  "pdf_url": "https://arxiv.org/pdf/2607.10892v1",
  "html_url": "https://arxiv.org/html/2607.10892v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.10842",
  "slug": "d-safempc-diffusion-driven-safe-model-predictive-control-with-discrete",
  "title": "D-SafeMPC: Diffusion-Driven Safe Model Predictive Control with Discrete-Time Control Barrier Functions",
  "abstract": "A key limitation on the use of diffusion models in robotic planning is their inability to inherently enforce safety or dynamical constraints, which often results in physically infeasible or unsafe outputs. Hybrid approaches that employ model predictive control (MPC) to address this problem can be unstable, as poor trajectory initializations from the diffusion model prevent the MPC from converging to a safe and feasible solution. To overcome these challenges, we propose D-SafeMPC, which enhances the interaction between diffusion and control. Our method guides the reverse diffusion process with control barrier functions (CBFs) and control Lyapunov functions (CLFs) and employs an iterative-projection scheme where an MPC refines the trajectory at each denoising step. This steers sampling toward safe, goal-directed regions and provides reliable MPC warm starts. In simulations on a Franka manipulator across four scenarios (one static-obstacle and three dynamic-obstacle settings) and in a sim-to-real experiment on a physical Franka robot, D-SafeMPC improves safety, task success rates, and planning efficiency over state-of-the-art baselines. To facilitate reproducibility, our source code and experimental configurations are available in a repository at https://github.com/erdiphd/D-SafeMPC",
  "published": "2026-07-12",
  "updated": "2026-07-12",
  "year": "2026",
  "authors": [
   "Erdi Sayar",
   "Ersin Da\u015f",
   "Joel W. Burdick",
   "Alois Knoll",
   "Erdal Kayacan"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "D-SafeMPC improves safety, task success rates, and planning efficiency over state-of-the-art baselines, and employs an iterative-projection scheme where an MPC refines the trajectory at each denoising step.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Erdi Sayar",
    "id": "81496006",
    "h_index": 4,
    "papers": 18
   },
   {
    "name": "Ersin Dacs",
    "id": "2449708763",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "J. W. Burdick",
    "id": "2303258085",
    "h_index": 5,
    "papers": 23
   },
   {
    "name": "A. Knoll",
    "id": "2199865233",
    "h_index": 12,
    "papers": 81
   },
   {
    "name": "Erdal Kayacan",
    "id": "144332507",
    "h_index": 42,
    "papers": 217
   }
  ],
  "comment": "",
  "topics": [
   "sim2real",
   "rl-control",
   "video-generation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.10842v1",
  "pdf_url": "https://arxiv.org/pdf/2607.10842v1",
  "html_url": "https://arxiv.org/html/2607.10842v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.10815",
  "slug": "learning-roller-skating-motions-of-humanoid-robots-based-on-adversaria",
  "title": "Learning Roller-Skating Motions of Humanoid Robots Based on Adversarial Motion Priors",
  "abstract": "Humanoid roller-skating is difficult because the robot must coordinate whole-body balance, rolling contacts, and velocity-dependent posture regulation. This paper presents an adversarial motion prior based reinforcement learning framework for two humanoid roller-skating gaits: Pump Glide skating and Push Glide skating. The two gait datasets are collected independently through motion capture and retargeted to the humanoid robot separately. The retargeted data are then smoothed and resampled into reference motion states for AMP training. The two gaits are learned by independent AMP training pipelines with separate reference datasets, separate policies, and independent reward architectures. Simulation experiments are designed to evaluate gait quality, velocity tracking, turning, and gait-specific reward ablations.",
  "published": "2026-07-12",
  "updated": "2026-07-12",
  "year": "2026",
  "authors": [
   "Yunkang Cheng",
   "Yutong Wu",
   "Menghan Li",
   "Shihe Zhou",
   "Mingguo Zhao"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "An adversarial motion prior based reinforcement learning framework for two humanoid roller-skating gaits: Pump Glide skating and Push Glide skating with separate reference datasets, separate policies, and independent reward architectures is presented.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yunkang Cheng",
    "id": "2449701083",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yutong Wu",
    "id": "2448361468",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Menghan Li",
    "id": "2449145295",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Shihe Zhou",
    "id": "2449755760",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Mingguo Zhao",
    "id": "2239286448",
    "h_index": 4,
    "papers": 14
   }
  ],
  "comment": "12 pages. Submitted preprint version. Accepted for oral presentation at CLAWAR 2026",
  "topics": [
   "humanoids",
   "egocentric-data",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.10815v1",
  "pdf_url": "https://arxiv.org/pdf/2607.10815v1",
  "html_url": "https://arxiv.org/html/2607.10815v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.10762",
  "slug": "tolid-bridging-the-architecture-gap-in-vision-foundation-model-to-lida",
  "title": "TOLiD: Bridging the Architecture Gap in Vision Foundation Model to LiDAR Pretraining via Token Lifting for Distillation",
  "abstract": "Cross-modal distillation from Vision Foundation Models (VFMs) to LiDAR backbones has recently emerged as a self-supervised pretraining strategy that reduces reliance on dense point-wise annotation for 3D scene understanding. However, existing distillation pipelines typically treat the VFM as a frozen feature source and train a heterogeneous 3D backbone to match fixed image embeddings, forcing the student to bridge both the modality gap and the cross-architecture gap between dense ViT token representations and sparse 3D encoders. We propose TOLiD, a self-supervised pretraining method for LiDAR representation learning that addresses this gap by coupling a LiDAR backbone with a student Vision Transformer (ViT) initialized from a frozen VFM teacher and applying supervision over compatible patch-token representations. TOLiD converts the set of point features within each image patch frustum into a token using Frustum Pooling followed by Frustum Attention, and performs token-level distillation with visibility masking. For LiDAR-only deployment, we lift token features back to per-point representations using masked bilinear sampling to avoid patches that have limited LiDAR points. We extensively evaluate TOLiD on five heterogeneous LiDAR datasets and four cross-sensor adaptation pairs, demonstrating improved transfer with frozen backbones and lightweight heads.",
  "published": "2026-07-12",
  "updated": "2026-07-12",
  "year": "2026",
  "authors": [
   "Sutharsan Mahendran",
   "Darshana Priyasad",
   "Kaushik Roy",
   "Tharindu Fernando",
   "Sridha Sridharan",
   "Clinton Fookes",
   "Peyman Moghadam"
  ],
  "author_count": 7,
  "categories": [
   "cs.CV",
   "cs.LG",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "TOLiD is proposed, a self-supervised pretraining method for LiDAR representation learning that addresses the gap between dense ViT token representations and sparse 3D encoders by coupling a LiDAR backbone with a student Vision Transformer initialized from a frozen VFM teacher and applying supervision over compatible patch-token representations.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Sutharsan Mahendran",
    "id": "2449705277",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Darshana Priyasad",
    "id": "51123172",
    "h_index": 6,
    "papers": 16
   },
   {
    "name": "Kaushik Roy",
    "id": "2268120688",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Tharindu Fernando",
    "id": "34735743",
    "h_index": 26,
    "papers": 91
   },
   {
    "name": "S. Sridharan",
    "id": "1729760",
    "h_index": 65,
    "papers": 676
   },
   {
    "name": "C. Fookes",
    "id": "3140440",
    "h_index": 59,
    "papers": 471
   },
   {
    "name": "P. Moghadam",
    "id": "2242950949",
    "h_index": 7,
    "papers": 32
   }
  ],
  "comment": "Accepted to The IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS) 2026",
  "topics": [
   "foundation-pretraining",
   "data-teleop",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.10762v1",
  "pdf_url": "https://arxiv.org/pdf/2607.10762v1",
  "html_url": "https://arxiv.org/html/2607.10762v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2607.10706",
  "slug": "action-map-policy-learning-3d-closed-loop-manipulation-via-pixel-class",
  "title": "Action Map Policy: Learning 3D Closed-loop Manipulation via Pixel Classification",
  "abstract": "The action space poses a major challenge in robot learning, since it is often high-dimensional, can span long time horizons, and frequently admits multi-modal optimal solutions. A good choice of action representation and loss function can help to address these concerns, but there are often trade offs. We propose Action Map Policy (AMP), which casts 3D closed-loop manipulation policy learning as a classification problem in image space. While classification has been an effective formulation in generative language models, applying it to robot action learning is difficult because naively discretizing high-dimensional continuous actions explodes the token vocabulary. Our key idea is to project 3D actions onto the camera image planes and treat each pixel location as a discrete class, thus controlling dimensionality while retaining multi-modality. This method supports millimeter-level precision for high-dimensional actions without requiring a prohibitively large vocabulary, while preserving fine-grained pixel-wise visual signals. Furthermore, it can predict the entire action chunk in a single forward pass, avoiding complex noise scheduling and iterative denoising while achieving substantially faster inference than diffusion policies. Experiments on various manipulation tasks show that AMP outperforms strong baselines, achieving higher success rates, faster inference, and enhanced spatial reasoning.",
  "published": "2026-07-12",
  "updated": "2026-07-12",
  "year": "2026",
  "authors": [
   "Haojie Huang",
   "Zhang Ye",
   "Linfeng Zhao",
   "Boce Hu",
   "Mingxi Jia",
   "Yu Qi",
   "Ahmed Agha",
   "Dian Wang",
   "Robert Platt",
   "Robin Walters"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes Action Map Policy (AMP), which casts 3D closed-loop manipulation policy learning as a classification problem in image space and treats each pixel location as a discrete class, thus controlling dimensionality while retaining multi-modality.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hao-zhe Huang",
    "id": "2143569284",
    "h_index": 8,
    "papers": 23
   },
   {
    "name": "Zhangchen Ye",
    "id": "2402503135",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Linfeng Zhao",
    "id": "2308044351",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Boce Hu",
    "id": "2312110011",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Ming Jia",
    "id": "2148250127",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Yu Qi",
    "id": "2311499051",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Ahmed Agha",
    "id": "49558195",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Dian Wang",
    "id": "2119264352",
    "h_index": 17,
    "papers": 36
   },
   {
    "name": "Robert Platt",
    "id": "2280136750",
    "h_index": 8,
    "papers": 25
   },
   {
    "name": "Robin Walters",
    "id": "2287354248",
    "h_index": 7,
    "papers": 22
   }
  ],
  "comment": "Project Website: https://haojhuang.github.io/amp_page/",
  "topics": [
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.10706v1",
  "pdf_url": "https://arxiv.org/pdf/2607.10706v1",
  "html_url": "https://arxiv.org/html/2607.10706v1",
  "code_url": "https://haojhuang.github.io/amp_page/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2607.10655",
  "slug": "artificial-foveated-perception-for-mitigating-shortcut-learning-in-rob",
  "title": "Artificial Foveated Perception for Mitigating Shortcut Learning in Robotic Foundation Models",
  "abstract": "Robotic foundation models have recently made substantial progress in multi-task capability, cross-embodiment transfer, and language-conditioned control. Yet robust deployment across diverse real-world settings remains difficult, in part because policies often fail to distinguish causally relevant visual structure from spurious scene-level correlations. We identify this failure mode as shortcut learning: the tendency to exploit predictive but non-causal correlations in the training distribution rather than the task-relevant visual evidence that determines successful action. Although shortcut learning has been extensively studied in computer vision and broader machine learning, its role in robotic foundation models remains comparatively underexplored. We propose Artificial Foveated Perception (AFP), a lightweight, policy-agnostic module that takes the same vision and language inputs as Vision-Language-Action and World Action Model pipelines and predicts task-conditioned masks over relevant objects, the robot, and other action-critical regions. We use these masks primarily as an auxiliary grounding signal during fine-tuning, aligning policy attention with task-relevant regions while leaving the core architecture unchanged. After fine-tuning, the policy executes on the original observation stream without requiring AFP in the control loop. We evaluate AFP across state-of-the-art robotic foundation models and show that foveated perception reduces fine-tuning time, suppresses overfitting, and improves generalization under environmental perturbations. Ablations over mask quality and grounding-loss design further show that these gains arise from directing policy learning toward task-relevant visual evidence. These results suggest that task-conditioned foveated perception is a practical mechanism for making robotic foundation models more robust, data-efficient, and scalable.",
  "published": "2026-07-12",
  "updated": "2026-07-12",
  "year": "2026",
  "authors": [
   "Xiatao Sun",
   "Yuan Zhuang",
   "Mateo Sanchez Lopez Negrete",
   "Matei-Victor Coldea",
   "Chen Liang",
   "Haoyang Zhang",
   "Che Liu",
   "Ziyao Zeng",
   "Shawn Li",
   "Qian Wang",
   "Fei Miao",
   "Daniel Rakita"
  ],
  "author_count": 12,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Artificial Foveated Perception is proposed, a lightweight, policy-agnostic module that takes the same vision and language inputs as Vision-Language-Action and World Action Model pipelines and predicts task-conditioned masks over relevant objects, the robot, and other action-critical regions and reduces fine-tuning time, suppresses overfitting, and improves generalization under environmental perturbations.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xiatao Sun",
    "id": "2440642902",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Zhuang Yuan",
    "id": "2324898338",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Mateo Sanchez Lopez Negrete",
    "id": "2449707115",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Matei-Victor Coldea",
    "id": "2449711629",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Chen Liang",
    "id": "2357823244",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Haoyang Zhang",
    "id": "1932916846",
    "h_index": 8,
    "papers": 14
   },
   {
    "name": "Che Liu",
    "id": "2390436735",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Ziyao Zeng",
    "id": "2282099512",
    "h_index": 6,
    "papers": 21
   },
   {
    "name": "Shawn Li",
    "id": "2330535399",
    "h_index": 1,
    "papers": 9
   },
   {
    "name": "Qian Wang",
    "id": "2305814350",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "Fei Miao",
    "id": "2307003840",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Daniel Rakita",
    "id": "2356784657",
    "h_index": 3,
    "papers": 8
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "foundation-pretraining",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.10655v1",
  "pdf_url": "https://arxiv.org/pdf/2607.10655v1",
  "html_url": "https://arxiv.org/html/2607.10655v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.10630",
  "slug": "world-models-as-adversaries-multi-agent-self-play-fine-tuning-for-robu",
  "title": "World Models as Adversaries: Multi-Agent Self-Play Fine-Tuning for Robust Motion Planning",
  "abstract": "Robust motion planning in dense traffic requires autonomous vehicles to interact in rare and safety-critical scenarios that are underrepresented in naturalistic driving data. Although adversarial training offers a feasible solution, existing methods often rely on external scenario generators, heuristic perturbations, or simulator-heavy rollouts, which makes them difficult to integrate with modern autoregressive planners. Here, we cast adversarially robust planner learning as a constrained min-max game and propose Adversarial World Modeling (AWM), a theoretically grounded multi-agent self-play fine-tuning framework. Since solving the exact game is intractable, AWM introduces a principled decoupled solver. In the inner minimization, the planner's predictive world model is converted into a role-conditioned adversary that learns sparse, scene-adaptive attack coalitions via counterfactual credit assignment. In the outer maximization, the ego planner optimizes a regret-aware robust best response against the frozen AWM, utilizing tail-risk weighting and reference-anchored trust regions to improve hard-case recovery while preserving nominal driving behavior. Experiments on the nuPlan and InterPlan benchmarks demonstrate that our method generates transferable adversarial interactions and yields a robust planner that achieves competitive closed-loop performance in both nominal and highly interactive long-tail scenarios. Theoretical analysis justifies the decoupled solver and the main optimization components.",
  "published": "2026-07-12",
  "updated": "2026-07-12",
  "year": "2026",
  "authors": [
   "Tong Nie",
   "Yuewen Mei",
   "Junlin He",
   "Yihong Tang",
   "Jian Sun",
   "Wei Ma"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Adversarial World Modeling (AWM) is proposed, a theoretically grounded multi-agent self-play fine-tuning framework that generates transferable adversarial interactions and yields a robust planner that achieves competitive closed-loop performance in both nominal and highly interactive long-tail scenarios.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Tong Nie",
    "id": "2385483186",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Yuewen Mei",
    "id": "2269471021",
    "h_index": 7,
    "papers": 15
   },
   {
    "name": "Junlin He",
    "id": "2316783851",
    "h_index": 6,
    "papers": 22
   },
   {
    "name": "Yihong Tang",
    "id": "2363225341",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Jiangming Sun",
    "id": "2028643500",
    "h_index": 10,
    "papers": 19
   },
   {
    "name": "Wei Ma",
    "id": "2277421553",
    "h_index": 7,
    "papers": 19
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "sim2real",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.10630v1",
  "pdf_url": "https://arxiv.org/pdf/2607.10630v1",
  "html_url": "https://arxiv.org/html/2607.10630v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.10625",
  "slug": "dual-process-atomic-skill-learning-decoupling-semantic-reasoning-and-r",
  "title": "Dual-Process Atomic Skill Learning: Decoupling Semantic Reasoning and Real-Time Control",
  "abstract": "Language-conditioned Imitation Learning (IL) is essential for enabling robots to perform complex tasks following natural language instructions. However, generalizing to multi-step compositional tasks remains a significant challenge. While hierarchical approaches attempt to address this by decomposing tasks into atomic skills, existing methods often suffer from training instability and codebook collapse due to the tight coupling between high-level skill reasoning and low-level action generation in joint training paradigms. Inspired by the Dual-Process Theory of cognition, we propose Dual-Process Atomic Skill Learning (DASL), a novel asynchronous hierarchical imitation learning framework that decouples slow semantic reasoning from fast, real-time motion control. DASL comprises a Slow-Frequency Policy that predicts interpretable, discrete skills via Vector Quantization, and a High-Frequency Policy that leverages a latent diffusion model and a Decision Transformer to generate precise actions conditioned on these latent skills. By asynchronously coordinating these modules and utilizing diffusion to structure the latent space, our framework mitigates the skill codebook interference problem common in joint training paradigms. Evaluations across simulation benchmarks and experiment demonstrate that DASL significantly outperforms state-of-the-art baselines, excelling in skill acquisition and compositional generalization to unseen instructions. GitHub page: https://github.com/Hatakekaka/DASL",
  "published": "2026-07-12",
  "updated": "2026-07-12",
  "year": "2026",
  "authors": [
   "Jun Chen",
   "Erdent Bao",
   "Wenlong Dong",
   "Jierui Liu",
   "Qi Cai",
   "Hao Wan",
   "Shaopeng Li",
   "Weijun Qin",
   "Jing Liang",
   "Huiping Zhuang"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes Dual-Process Atomic Skill Learning (DASL), a novel asynchronous hierarchical imitation learning framework that decouples slow semantic reasoning from fast, real-time motion control and mitigates the skill codebook interference problem common in joint training paradigms.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jun Chen",
    "id": "2303640667",
    "h_index": 7,
    "papers": 15
   },
   {
    "name": "Erdent Bao",
    "id": "2449708883",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Wenlong Dong",
    "id": "2296790559",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Jierui Liu",
    "id": "2449725485",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Qi Cai",
    "id": "2449710171",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Hao Wan",
    "id": "2333099804",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Shaopeng Li",
    "id": "2449722896",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Weijun Qin",
    "id": "2449710814",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Jingchuan Liang",
    "id": "2440912992",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Huiping Zhuang",
    "id": "2367195460",
    "h_index": 2,
    "papers": 9
   }
  ],
  "comment": "28 pages,20 figures,21 tables",
  "topics": [
   "sim2real",
   "imitation-diffusion",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.10625v1",
  "pdf_url": "https://arxiv.org/pdf/2607.10625v1",
  "html_url": "https://arxiv.org/html/2607.10625v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.10565",
  "slug": "bucketkd-a-safety-aware-bucket-based-knowledge-distillation-framework",
  "title": "BucketKD: A Safety-Aware Bucket-Based Knowledge Distillation Framework for End-to-End Motion Planning",
  "abstract": "End-to-end motion planning has emerged as a promising paradigm in autonomous driving, directly mapping raw sensor data to control commands via deep neural networks. Despite its advantages, its large model size hinders deployment in resource-constrained platforms. In this paper, we present BucketKD, a bucket-based knowledge distillation framework that yields compact and safety-aware end-to-end planners. Compared to the state-of-the-art approach, which relies on simplified planning state representations, BucketKD discretizes critical environmental variables into adaptive buckets that capture richer scene semantics while preserving efficiency. In addition, we design a safety-aware waypoint attention mechanism that evaluates each waypoint's risk level by accounting for both obstacle proximity and relative motion through a time-to-collision (TTC) formulation widely used in transportation research. This enables the student model to better retain safety-critical behaviors during distillation. Extensive experiments in CARLA using the Bench2Drive dataset show that BucketKD significantly outperforms the state-of-the-art in both planning accuracy and safety while maintaining strong compression ratios.",
  "published": "2026-07-12",
  "updated": "2026-07-12",
  "year": "2026",
  "authors": [
   "Md Nahidul Islam",
   "Mohd Hasan Ali",
   "Dipankar Dasgupta",
   "Myounggyu Won"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "BucketKD is presented, a bucket-based knowledge distillation framework that yields compact and safety-aware end-to-end planners that significantly outperforms the state-of-the-art in both planning accuracy and safety while maintaining strong compression ratios.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Md Nahidul Islam",
    "id": "2449700743",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Mohd. Hasan Ali",
    "id": "2298432968",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Dipankar Dasgupta",
    "id": "2284079099",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Myounggyu Won",
    "id": "2284078178",
    "h_index": 1,
    "papers": 6
   }
  ],
  "comment": "",
  "topics": [
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.10565v1",
  "pdf_url": "https://arxiv.org/pdf/2607.10565v1",
  "html_url": "https://arxiv.org/html/2607.10565v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.10438",
  "slug": "large-language-model-enhanced-differentiable-trajectory-planning-for-i",
  "title": "Large Language Model Enhanced Differentiable Trajectory Planning for IoT-Enabled Autonomous Driving",
  "abstract": "Autonomous driving planning is a key component of IoT-enabled intelligent transportation systems, requiring vehicles to generate safe, efficient, and executable trajectories in complex urban environments from multi-source contextual information. While imitation learning (IL) has shown promise on large-scale datasets, IL-based planners still suffer from limited coverage of complex long-tail interactions, weak consistency with downstream constrained refinement, and insufficient use of high level scene semantics under real time constraints. To address these issues, this paper proposes a large language model (LLM) enhanced differentiable trajectory planning framework for IoT-enabled autonomous driving. Specifically, we introduce a surrounding agent centric data augmentation strategy to reorganize sur rounding agent trajectories as additional planning supervision, thereby improving the training distribution without collecting additional raw data. We further design a complexity-aware asyn chronous LLM-based semantic enhancement module to extract scene-related high-level semantic features with controlled online overhead. In addition, a differentiable optimization module is incorporated to refine generated trajectories with explicit residual penalties while backpropagating optimization gradients to the upstream planner. Experiments show that the proposed method achieves the best overall scores of 83.63 and 78.29 on the nuPlan closed-loop nonreactive and reactive Hard20 benchmarks, respectively, and CARLA-ROS tests further verify its online deployment and real time closed-loop execution capability.",
  "published": "2026-07-11",
  "updated": "2026-07-11",
  "year": "2026",
  "authors": [
   "Shihao Zhang",
   "Jing Yang",
   "Ziyu Song",
   "Zheng Lin",
   "Sunil Prajapat",
   "Zhaochen Xia",
   "Hemant Ghayvat",
   "Haitao Ding",
   "Lip Yee Por",
   "Ashok Kumar Das"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO",
   "cs.NI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "A large language model (LLM) enhanced differentiable trajectory planning framework for IoT-enabled autonomous driving is proposed and a surrounding agent centric data augmentation strategy is introduced to reorganize sur rounding agent trajectories as additional planning supervision, thereby improving the training distribution without collecting additional raw data.",
  "doi": "10.1109/jiot.2026.3711819",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Shihao Zhang",
    "id": "2335445854",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Jing Yang",
    "id": "2258721359",
    "h_index": 9,
    "papers": 43
   },
   {
    "name": "Ziyu Song",
    "id": "2240715185",
    "h_index": 6,
    "papers": 25
   },
   {
    "name": "Zheng Lin",
    "id": "2284062432",
    "h_index": 15,
    "papers": 18
   },
   {
    "name": "Sunil Prajapat",
    "id": "2212875403",
    "h_index": 17,
    "papers": 89
   },
   {
    "name": "Zhaochen Xia",
    "id": "2441077753",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "H. Ghayvat",
    "id": "3424424",
    "h_index": 25,
    "papers": 89
   },
   {
    "name": "Haitao Ding",
    "id": "2240709094",
    "h_index": 8,
    "papers": 30
   },
   {
    "name": "L. Y. Por",
    "id": "2925662",
    "h_index": 31,
    "papers": 167
   },
   {
    "name": "Ashok Kumar Das",
    "id": "2240511396",
    "h_index": 21,
    "papers": 132
   }
  ],
  "comment": "13 pages, 5 figures",
  "topics": [
   "imitation-diffusion",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.10438v1",
  "pdf_url": "https://arxiv.org/pdf/2607.10438v1",
  "html_url": "https://arxiv.org/html/2607.10438v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2607.10383",
  "slug": "abot-n1-toward-a-general-visual-language-navigation-foundation-model",
  "title": "ABot-N1: Toward a General Visual Language Navigation Foundation Model",
  "abstract": "Visual Language Navigation foundation models aim to unify deep reasoning for grounded spatial decisions with broad versatility for diverse embodied tasks. Current approaches typically achieve this integration via monolithic policies that map observations directly to actions, yet they often suffer from coordinate drift and poor handling of long-tail semantics. Furthermore, these black-box mappings lack interpretability, hindering the simultaneous achievement of generality, robustness, and transparency. We present ABot-N1, a step toward a general Visual Language Navigation foundation model, that addresses these challenges by decoupling cognition from control via a slow-fast architecture guided by dual visual-language signals. More specifically, a slow vision-language reasoner performs explicit Chain-of-Thought reasoning while producing a pixel goal. This compact set of image-space anchor points serves as a universal interface for diverse tasks, including point-goal, object-goal, poi-goal, instruction-following, and person-following. Subsequently, a fast action expert leverages both the textual cues and the pixel guidance to generate continuous waypoints at the native control frequency. By bridging high-level intents and low-level control through pixel-grounded anchors paired with explicit linguistic traces, our approach ensures robust, generalizable, and interpretable navigation across simulation and real-world benchmarks. ABot-N1 establishes new state-of-the-art records, delivering massive gains specifically in urban-scale navigation: boosting POI arrival by 35.0% (to 77.3%) and achieving 95.4%/92.9% SR in complex indoor and outdoor scenes. It also maintains superior robustness across object-reaching, person-following, and instruction-following tasks. New Point-Goal/POI-Goal benchmarks are released as open source to advance the field of urban-scale navigation.",
  "published": "2026-07-11",
  "updated": "2026-07-17",
  "year": "2026",
  "authors": [
   "Ruiyan Gong",
   "Yingnan Guo",
   "Junjun Hu",
   "Jintao Kong",
   "Xiaoxu Leng",
   "Tianlun Li",
   "Weize Li",
   "Fei Liu",
   "Zhicheng Liu",
   "Jia Lu",
   "Minghua Luo",
   "Chenlin Ming",
   "Yanfen Shen",
   "Jiyue Tao",
   "Zhengbo Wang",
   "Mingyang Yin",
   "Minqi Gu",
   "Zihao Guan",
   "Wei Guo",
   "Guoqing Liu",
   "Huachong Pang",
   "Menglin Yang",
   "Zeqian Ye",
   "Xiaoxiao Geng",
   "Zhining Gu",
   "Honglin Han",
   "Di Jing",
   "Hongyu Pan",
   "Mingchao Sun",
   "Kuan Yang",
   "Jianfang Zhang",
   "Yanghong Chen",
   "Ye He",
   "Wei Mei",
   "Jiahao Shi",
   "Xiangpo Yang",
   "Yanqing Zhu",
   "Yang Cai",
   "Jingjing Ma",
   "Shihui Su",
   "Zixiao Tang",
   "Linbo Zheng",
   "Zedong Chu",
   "Xiaolong Wu",
   "Wenbin Tang",
   "Mu Xu"
  ],
  "author_count": 46,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "ABot-N1 establishes new state-of-the-art records, delivering massive gains specifically in urban-scale navigation: boosting POI arrival by 35.0% (to 77.3%) and achieving 95.4%/92.9% SR in complex indoor and outdoor scenes.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ruiyan Gong",
    "id": "2380827364",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Yingnan Guo",
    "id": "2380384885",
    "h_index": 2,
    "papers": 14
   },
   {
    "name": "Junjun Hu",
    "id": "2303462130",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Jintao Kong",
    "id": "2427168213",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "X. Leng",
    "id": "144031761",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Tianlun Li",
    "id": "2118910472",
    "h_index": 2,
    "papers": 12
   },
   {
    "name": "Weize Li",
    "id": "2376357439",
    "h_index": 2,
    "papers": 14
   },
   {
    "name": "Fei Liu",
    "id": "2395949565",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Zhichen Liu",
    "id": "2296273061",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Jia Lu",
    "id": "2396292240",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Minghua Luo",
    "id": "2384079347",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Chenlin Ming",
    "id": "2243338765",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Yanfen Shen",
    "id": "2394804148",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Jiyue Tao",
    "id": "2302585496",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Zhengbo Wang",
    "id": "2383056766",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Mingyang Yin",
    "id": "41075506",
    "h_index": 8,
    "papers": 17
   },
   {
    "name": "Minqi Gu",
    "id": "2448444135",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Zi-An Guan",
    "id": "2333649933",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Wei Guo",
    "id": "2271589639",
    "h_index": 15,
    "papers": 42
   },
   {
    "name": "Guoqing Liu",
    "id": "2274751895",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Huachong Pang",
    "id": "2449640721",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Meng-Yao Yang",
    "id": "2337354985",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Zeqian Ye",
    "id": "2449650155",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Xiaoxiao Geng",
    "id": "2448211903",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Zhining Gu",
    "id": "2395846668",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Honglin Han",
    "id": "2313939339",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "D. Jing",
    "id": "2240218619",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Hongyu Pan",
    "id": "2303398536",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Mingchao Sun",
    "id": "2386126343",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Kuan Yang",
    "id": "2394373958",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Jianfang Zhang",
    "id": "2107968948",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Yang Chen",
    "id": "2448410969",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Ye He",
    "id": "2342014906",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Wei Mei",
    "id": "2267381983",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Jiahao Shi",
    "id": "2449183590",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Xiangpo Yang",
    "id": "2410866537",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Yanqing Zhu",
    "id": "2410224299",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Yang Cai",
    "id": "2383107401",
    "h_index": 5,
    "papers": 16
   },
   {
    "name": "Jingjing Ma",
    "id": "2445395129",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Shihu Su",
    "id": "2437718563",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Zixiao Tang",
    "id": "2448619712",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Lin-Jing Zheng",
    "id": "2438706306",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Zedong Chu",
    "id": "2333428585",
    "h_index": 7,
    "papers": 14
   },
   {
    "name": "Xiaolong Wu",
    "id": "2382838519",
    "h_index": 6,
    "papers": 20
   },
   {
    "name": "Wenbin Tang",
    "id": "2387892194",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Mu Xu",
    "id": "2382940270",
    "h_index": 6,
    "papers": 22
   }
  ],
  "comment": "",
  "topics": [
   "navigation",
   "foundation-pretraining",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.10383v3",
  "pdf_url": "https://arxiv.org/pdf/2607.10383v3",
  "html_url": "https://arxiv.org/html/2607.10383v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.10369",
  "slug": "vine-taming-generative-control-policies-for-reinforcement-learning",
  "title": "VINE: Taming Generative Control Policies for Reinforcement Learning",
  "abstract": "Flow-matching policies have emerged as an effective policy parameterization for robot learning. They iteratively generate actions from noise, enabling highly expressive modeling of complex and multimodal action distributions. However, prior works observed that scaling these policies with value-gradient reinforcement learning (RL) often leads to training instability. Existing methods attribute this instability to iterative generation and therefore avoid end-to-end value-gradient optimization by sacrificing iterative generation, high expressiveness, or value-gradient optimization. Contrary to prior belief, we show the instability does not stem from iterative generation itself, but from the vanilla sampling strategy originally designed for behavior cloning, which becomes brittle under value-gradient RL. Motivated by this insight, we propose VINE, an RL-oriented sampling method that enables stable end-to-end value-gradient optimization for flow-matching policies. Instead of following a single flow trajectory, VINE reconstructs a new interpolation state at every denoising step, creating a stable differentiable path for value-gradient propagation while remaining compatible with the original flow-matching denoising process. As a result, VINE preserves the expressiveness and iterative generation of flow-matching without sacrificing end-to-end value-gradient optimization. Despite performing end-to-end backpropagation through all ten denoising steps, VINE achieves stable policy improvement and consistently outperforms state-of-the-art RL methods on the OGBench offline RL benchmark and real-world robotic manipulation task. Videos are available on our website: https://agibottech.github.io/vine.",
  "published": "2026-07-11",
  "updated": "2026-07-11",
  "year": "2026",
  "authors": [
   "Rushuai Yang",
   "Zhuo Han",
   "Houlin Li",
   "Hecheng Wang",
   "Zhichao Wu",
   "Rui Zhang",
   "Zhaowei Zhang",
   "Zihong Chen",
   "Xiaohan Yan",
   "Chiming Liu",
   "Yi Chen",
   "Wei Shan",
   "Maoqing Yao"
  ],
  "author_count": 13,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "VINE is proposed, an RL-oriented sampling method that enables stable end-to-end value-gradient optimization for flow-matching policies and achieves stable policy improvement and consistently outperforms state-of-the-art RL methods on the OGBench offline RL benchmark and real-world robotic manipulation task.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Rushuai Yang",
    "id": "2216653262",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Zhuo Han",
    "id": "2449754912",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Houlin Li",
    "id": "2444282785",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Hecheng Wang",
    "id": "2272429260",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Zhichao Wu",
    "id": "2278216011",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Rui Zhang",
    "id": "2239646543",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Zhaowei Zhang",
    "id": "2449701740",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Zihong Chen",
    "id": "2449753942",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Xiao Yan",
    "id": "2275997184",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Chiming Liu",
    "id": "2349426158",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Yi Chen",
    "id": "2382653683",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Wei Shan",
    "id": "2382133576",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Maoqing Yao",
    "id": "2395512682",
    "h_index": 2,
    "papers": 3
   }
  ],
  "comment": "",
  "topics": [
   "imitation-diffusion",
   "rl-control"
  ],
  "orgs": [
   "AgiBot"
  ],
  "abs_url": "https://arxiv.org/abs/2607.10369v1",
  "pdf_url": "https://arxiv.org/pdf/2607.10369v1",
  "html_url": "https://arxiv.org/html/2607.10369v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2607.10350",
  "slug": "abot-agentos-a-general-robotic-agent-os-with-lifelong-multi-modal-memo",
  "title": "ABot-AgentOS: A General Robotic Agent OS with Lifelong Multi-modal Memory",
  "abstract": "Recent VLM and VLA systems have improved robotic perception and action prediction, yet long-horizon embodied agents still require a general runtime layer for reasoning, memory, tool use, verification, and cross-embodiment execution. We present ABot-AgentOS, a general robotic Agent Operating System that sits above low-level controllers and provides a deliberative agent layer for scene-conditioned planning, context-isolated skill execution, multi-stage verification, multi-modal memory, and edge-cloud collaboration. To evaluate such systems, we introduce EmbodiedWorldBench, an executable benchmark with 16 indoor, outdoor, and hybrid scenes, four difficulty levels, and over 200 tasks involving navigation, object search, NPC dialogue, dynamic events, and trace-grounded scoring. ABot-AgentOS further introduces Universal Multi-modal Graph Memory, a persistent source-grounded substrate that converts dialogue, visual observations, spatial context, temporal relations, and task traces into typed nodes and edges. A failure-driven self-evolution loop converts diagnosed memory failures into gated runtime evo-assets that are promoted only to later evaluation splits, preventing current-split ground-truth leakage while enabling continual improvement. On an initial EmbodiedWorldBench subset, ABot-AgentOS improves over a single-controller baseline in both task success and goal completion. Across memory benchmarks, ABot-AgentOS Static achieves 87.5 on LoCoMo, 59.9 on OpenEQA EM-EQA, 88.6 on Mem-Gallery, and 76.5 Acc@All on NExT-QA; self-evolution further improves LoCoMo to 88.7, OpenEQA to 60.4, and Mem-Gallery to 89.0. These results suggest that a general Agent OS layer can improve long-horizon embodied execution while providing persistent, auditable memory for continual interaction.",
  "published": "2026-07-11",
  "updated": "2026-07-17",
  "year": "2026",
  "authors": [
   "Jiayi Tian",
   "Shiao Liu",
   "Yuting Xu",
   "Jia Lu",
   "Zihao Guan",
   "Honglin Han",
   "Di Yang",
   "Minqi Gu",
   "Yifei Qian",
   "Tianlin Zhang",
   "Yanqing Zhu",
   "Zeqian Ye",
   "Menglin Yang",
   "Fei Wang",
   "Xu Hu",
   "Xiuxian Li",
   "Wei Zhang",
   "Shihui Su",
   "Yiyan Ji",
   "Jingbo Wang",
   "Ziteng Feng",
   "Jiaheng Liu",
   "Zhaoxiang Zhang",
   "Xiaolong Wu",
   "Zixiao Tang",
   "Zhining Gu",
   "Yang Cai",
   "Linbo Zheng",
   "Jingjing Ma",
   "Mingyang Yin",
   "Zedong Chu",
   "Wenbin Tang",
   "Mu Xu"
  ],
  "author_count": 33,
  "categories": [
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.AI",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "ABot-AgentOS is presented, a general robotic Agent Operating System that sits above low-level controllers and provides a deliberative agent layer for scene-conditioned planning, context-isolated skill execution, multi-stage verification, multi-modal memory, and edge-cloud collaboration.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jiayi Tian",
    "id": "2344623736",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Shiao Liu",
    "id": "2449698907",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yuting Xu",
    "id": "2110175285",
    "h_index": 8,
    "papers": 14
   },
   {
    "name": "Jia Lu",
    "id": "2448344120",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Zi-An Guan",
    "id": "2333649933",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Honglin Han",
    "id": "2313939339",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Dianzhe Yang",
    "id": "2307887987",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Minqi Gu",
    "id": "2448444135",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Yifei Qian",
    "id": "2445495605",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Tianlin Zhang",
    "id": "2383230595",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Yanqing Zhu",
    "id": "2410224299",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Zeqian Ye",
    "id": "2449650155",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Meng-Yao Yang",
    "id": "2337354985",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Feifei Wang",
    "id": "2446888576",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Xu Hu",
    "id": "2282086127",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Xiuxian Li",
    "id": "2449769797",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "W. Zhang",
    "id": "2448640266",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Shihu Su",
    "id": "2437718563",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Yiyan Ji",
    "id": "2391923508",
    "h_index": 2,
    "papers": 11
   },
   {
    "name": "Jingbo Wang",
    "id": "2449443869",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Ziteng Feng",
    "id": "2294644664",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Jiaheng Liu",
    "id": "2346051109",
    "h_index": 9,
    "papers": 38
   },
   {
    "name": "Zhaoxiang Zhang",
    "id": "2374145316",
    "h_index": 2,
    "papers": 12
   },
   {
    "name": "Xiaolong Wu",
    "id": "2382838519",
    "h_index": 6,
    "papers": 20
   },
   {
    "name": "Mingyang Yin",
    "id": "41075506",
    "h_index": 8,
    "papers": 17
   },
   {
    "name": "Zedong Chu",
    "id": "2333428585",
    "h_index": 7,
    "papers": 14
   },
   {
    "name": "Mu Xu",
    "id": "2382940270",
    "h_index": 6,
    "papers": 22
   }
  ],
  "comment": "Code: https://github.com/amap-cvlab/ABot-AgentOS Project page: https://amap-cvlab.github.io/ABot-AgentOS",
  "topics": [
   "vla",
   "navigation",
   "foundation-pretraining",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.10350v3",
  "pdf_url": "https://arxiv.org/pdf/2607.10350v3",
  "html_url": "https://arxiv.org/html/2607.10350v3",
  "code_url": "https://github.com/amap-cvlab/ABot-AgentOS",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2607.10288",
  "slug": "pier-flow-physics-informed-efficient-rectified-flow-for-real-time-mobi",
  "title": "PIER-Flow: Physics-Informed Efficient Rectified Flow for Real-Time Mobile Robot Navigation",
  "abstract": "Autonomous navigation in dense and highly dynamic environments requires both physically feasible control and low-latency replanning. Optimization-based methods such as Model Predictive Control (MPC) explicitly handle robot kinematics and safety constraints, but repeated nonlinear optimization can limit real-time responsiveness. Deterministic behavior-cloning policies enable efficient inference but may fail to represent multimodal avoidance behaviors, whereas diffusion policies capture multimodality at the cost of time-consuming iterative denoising. We propose PIER-Flow (Physics-Informed Efficient Rectified Flow), a lightweight navigation policy for mobile robots. By distilling an MPC expert into a continuous-time Ordinary Differential Equation (ODE), PIER-Flow achieves single-step action generation through parallel latent sampling and lightweight feasibility selection. We introduce a physics-informed training objective to enforce kinematic consistency, paired with an asynchronous action chunking architecture for robust sim-to-real deployment. Extensive simulations demonstrate that PIER-Flow achieves a 98.85\\% success rate and zero collisions, with an average inference of $\\sim$1.29 ms, which accelerates planning by 37.2$\\times$ compared to MPC and over 800$\\times$ against standard diffusion models. Crucially, real-world deployment on a resource-constrained edge computer further achieves an approximately stable inference latency of $\\sim$5.3 ms, avoiding the latency spikes and freezing events observed with planning baselines.",
  "published": "2026-07-11",
  "updated": "2026-07-11",
  "year": "2026",
  "authors": [
   "Shibo Li",
   "Zhongcheng Wang",
   "Jiahe Cao",
   "Jianhua Yang",
   "Ke Wu"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "PIER-Flow (Physics-Informed Efficient Rectified Flow), a lightweight navigation policy for mobile robots, is proposed, which introduces a physics-informed training objective to enforce kinematic consistency and is paired with an asynchronous action chunking architecture for robust sim-to-real deployment.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Shibo Li",
    "id": "2118056567",
    "h_index": 7,
    "papers": 14
   },
   {
    "name": "Zhong-Yan Wang",
    "id": "2446301260",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Jiahe Cao",
    "id": "2314330896",
    "h_index": 0,
    "papers": 5
   },
   {
    "name": "Jianhua Yang",
    "id": "2109746434",
    "h_index": 4,
    "papers": 21
   },
   {
    "name": "Ke Wu",
    "id": "2446901109",
    "h_index": 0,
    "papers": 2
   }
  ],
  "comment": "",
  "topics": [
   "sim2real",
   "imitation-diffusion",
   "rl-control",
   "navigation",
   "data-teleop",
   "video-generation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.10288v1",
  "pdf_url": "https://arxiv.org/pdf/2607.10288v1",
  "html_url": "https://arxiv.org/html/2607.10288v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.10206",
  "slug": "source-lifted-flow-matching-for-intervenable-multimodal-imitation",
  "title": "Source-Lifted Flow Matching for Intervenable Multimodal Imitation",
  "abstract": "Flow-matching policies are promising for imitation learning because they model complex multimodal action distributions. However, their stochasticity is largely passive: repeated sampling may yield diverse behaviors, but users cannot directly choose among valid continuations from the same state. We propose Source-Lifted Flow Matching (SL-FM), a source-intervenable flow-matching policy that exposes such a handle while keeping the velocity field shared and latent-free. The handle selects only the source endpoint of the conditional flow, not a mode-specific field, preserving the standard formulation while avoiding decomposition into separate mode-conditioned dynamics. The core mechanism is \\textbf{Orthogonal Source Lifting}, designed to prevent path-crossing ambiguity. Instead of partitioning target actions by mode, SL-FM lifts handle-specific sources into auxiliary orthogonal coordinates and keeps targets in the original action subspace. This preserves the demonstrated action distribution while allowing one shared field to carry different branches without merging at crossings. To keep handles usable across states, we learn a state-dependent source mixture end to end and use a responsibility floor, giving each handle weak supervision and mitigating dead modes. Experiments on crossing-flow diagnostics and robot-control benchmarks show that SL-FM converts passive source randomness into an actionable intervention variable. It removes crossing-induced composite trajectories, changes future routes in 91.1\\% of matched-prefix interventions, and achieves strong free-deployment performance, with improvements in several benchmark settings. Overall, source geometry provides actionable multimodal control without conditioning the velocity field on the selected mode.",
  "published": "2026-07-11",
  "updated": "2026-07-11",
  "year": "2026",
  "authors": [
   "He Zhang",
   "Ying Sun",
   "Pengteng Li",
   "Ziyang Chen",
   "Yiren Zhao",
   "Ziyang Rao",
   "Weiyu Guo",
   "Yandong Guo",
   "Hui Xiong"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Source-Lifted Flow Matching (SL-FM), a source-intervenable flow-matching policy that exposes such a handle while keeping the velocity field shared and latent-free, and converts passive source randomness into an actionable intervention variable.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "He Zhang",
    "id": "2257389052",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Ying Sun",
    "id": "2257321422",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Pengteng Li",
    "id": "2342359936",
    "h_index": 3,
    "papers": 14
   },
   {
    "name": "Ziyang Chen",
    "id": "2347655949",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Yiren Zhao",
    "id": "2109919449",
    "h_index": 21,
    "papers": 88
   },
   {
    "name": "Ziyang Rao",
    "id": "2343504880",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Weiyu Guo",
    "id": "2257320765",
    "h_index": 7,
    "papers": 27
   },
   {
    "name": "Yandong Guo",
    "id": "2284207445",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Hui Xiong",
    "id": "2346984163",
    "h_index": 3,
    "papers": 13
   }
  ],
  "comment": "",
  "topics": [
   "imitation-diffusion"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.10206v1",
  "pdf_url": "https://arxiv.org/pdf/2607.10206v1",
  "html_url": "https://arxiv.org/html/2607.10206v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.10132",
  "slug": "tac-loco-unified-whole-body-control-for-quadrupedal-tactile-informed-l",
  "title": "TAC-LOCO: Unified Whole-Body Control for Quadrupedal TACtile-Informed LOCO-Manipulation",
  "abstract": "Dynamic loco-manipulation requires legged robots to coordinate whole-body motion while maintaining stable physical interaction with grasped objects under uncertain external forces. While tactile sensing has been widely studied for robotic manipulation, its role in dynamic whole-body control remains largely unexplored. Existing works without tactile feedback commonly grasp firmly rather than regulate the grasp according to the interaction. We propose TAC-LOCO, a tactile-augmented unified reinforcement learning framework that encodes tactile array observations from compliant grippers into a compact latent representation and joins it with proprioception for unified control of the legs, arm, and gripper. With effective grasp stability reward design, the policy learns to simultaneously track body velocity and end-effector trajectories, moderate grasp force, and prevent object slip under both gradual load changes and sudden release events. We deploy the policy zero-shot on a Unitree Go2 with an Interbotix WidowX 250 arm and tactile gripper, demonstrating dynamic tactile-informed loco-manipulation under varying external interactions, achieving a 47% reduction in grasping force and an object drop rate of less than 1%.",
  "published": "2026-07-11",
  "updated": "2026-07-15",
  "year": "2026",
  "authors": [
   "Muqun Hu",
   "Yuhao Zhou",
   "Kabir Ray Malik",
   "Chi Lin",
   "Won Suk Lee",
   "Yu She",
   "Yan Gu"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "TAC-LOCO is proposed, a tactile-augmented unified reinforcement learning framework that encodes tactile array observations from compliant grippers into a compact latent representation and joins it with proprioception for unified control of the legs, arm, and gripper.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Muqun Hu",
    "id": "2264743925",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Yuhao Zhou",
    "id": "2314297484",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Kabir Ray Malik",
    "id": "2449643018",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Chi Lin",
    "id": "2444934863",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "W. S. Lee",
    "id": "144832957",
    "h_index": 44,
    "papers": 179
   },
   {
    "name": "Yu She",
    "id": "2238342225",
    "h_index": 7,
    "papers": 24
   },
   {
    "name": "Yan Gu",
    "id": "2409888461",
    "h_index": 2,
    "papers": 5
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "humanoids",
   "tactile",
   "rl-control"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2607.10132v2",
  "pdf_url": "https://arxiv.org/pdf/2607.10132v2",
  "html_url": "https://arxiv.org/html/2607.10132v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2607.13067",
  "slug": "a-3dgs-driven-dynamic-viewpoint-and-vibrotactile-framework-for-subsea",
  "title": "A 3DGS-Driven Dynamic Viewpoint and Vibrotactile Framework for Subsea Teleoperation Validated via fNIRS",
  "abstract": "Teleoperating remotely operated vehicles (ROVs) in flooded, cluttered infrastructure is fundamentally limited by narrow 2D egocentric views and subsea communication latency. We present a multimodal teleoperation architecture built on a ROS-Unity framework that decouples proactive spatial planning from reactive boundary avoidance. The system replaces static camera feeds with a Dynamic Adaptive Viewpoint System (DAVS), which uses continuous optimization and real-time 3D Gaussian Splatting (3DGS) to synthesize an occlusion-free exocentric viewpoint from onboard state estimation. To further reduce sensory workload, a torso-mounted vibrotactile suit maps local obstacle clearance to intuitive haptic proximity cues. The architecture was evaluated in a controlled human-subject study (N = 30) using a BlueROV2 navigating a complex simulated underwater facility. A 3 x 4 repeated-measures design compared three interaction modalities (Egocentric, Haptic, Exocentric) under four communication delays (0.0-1.0 s). Performance was quantified using behavioral measures and functional near-infrared spectroscopy (fNIRS) to assess task-evoked prefrontal activation. Results show that reactive haptic feedback improves path adherence under minimal delay, whereas the 3DGS-driven exocentric visualization provides superior resilience under severe latency (0.5-1.0 s), significantly outperforming the other modalities. fNIRS further revealed a cognitive disengagement effect: increasing latency during conventional egocentric teleoperation overloaded working memory and reduced prefrontal activation, whereas the proactive spatial context provided by DAVS sustained executive control. These findings demonstrate that spatially grounded, multimodal assistance can substantially improve operator performance and cognitive endurance during latency-degraded underwater teleoperation.",
  "published": "2026-07-10",
  "updated": "2026-07-10",
  "year": "2026",
  "authors": [
   "Fang Xu",
   "Tianyu Zhou",
   "Ruitong Tian",
   "Md Jahidul Islam",
   "Jing Du"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "It is demonstrated that spatially grounded, multimodal assistance can substantially improve operator performance and cognitive endurance during latency-degraded underwater teleoperation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Fang Xu",
    "id": "2156159672",
    "h_index": 9,
    "papers": 26
   },
   {
    "name": "Tianyu Zhou",
    "id": "2114111522",
    "h_index": 16,
    "papers": 52
   },
   {
    "name": "Ruitong Tian",
    "id": "2296072834",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Md Jahidul Islam",
    "id": "2297737955",
    "h_index": 5,
    "papers": 20
   },
   {
    "name": "Jing Du",
    "id": "2116518997",
    "h_index": 15,
    "papers": 35
   }
  ],
  "comment": "9 pages, 6 figures",
  "topics": [
   "egocentric-data",
   "tactile",
   "spatial-3d",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.13067v1",
  "pdf_url": "https://arxiv.org/pdf/2607.13067v1",
  "html_url": "https://arxiv.org/html/2607.13067v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.10014",
  "slug": "runtime-safety-filtering-for-learned-small-uas-separation-policies-und",
  "title": "Runtime Safety Filtering for Learned Small UAS Separation Policies under GNSS Degradation",
  "abstract": "Learning-based separation assurance for small Unmanned Aircraft Systems (sUAS) achieves near-zero collision rates in simulation, but assumes accurate position and velocity information from Global Navigation Satellite Systems (GNSS). This assumption fails in urban environments, where multipath propagation, signal blockage, and intentional interference degrade navigation integrity. This raises a fundamental architectural question for deploying learned separation policies under GNSS degradation: should runtime safety mechanisms filter the policy's actions or its observations? This work evaluates both approaches for multi-agent sUAS separation under adversarial GNSS degradation. Both architectures first estimate a worst-case traffic state consistent with bounded observation uncertainty, then diverge: action filtering constrains policy outputs via discrete-time control barrier functions evaluated at the worst-case state, while observation filtering presents the worst-case state directly to the policy as corrected input. Experimental results show that action filtering provides negligible safety improvement, while observation filtering reduces near mid-air collisions by 90% and remains robust to the barrier function's tradeoff between separation distance and closing rate. These results suggest that, for policies with learned safety behaviors, preserving the policy's decision authority outperforms overriding its actions with hand-designed constraints.",
  "published": "2026-07-10",
  "updated": "2026-07-10",
  "year": "2026",
  "authors": [
   "Alex Zongo",
   "Peng Wei"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO",
   "cs.LG",
   "cs.MA",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Experimental results show that action filtering provides negligible safety improvement, while observation filtering reduces near mid-air collisions by 90% and remains robust to the barrier function's tradeoff between separation distance and closing rate, and suggest that preserving the policy's decision authority outperforms overriding its actions with hand-designed constraints.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Alex Zongo",
    "id": "2426798166",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Peng Wei",
    "id": "2261431339",
    "h_index": 3,
    "papers": 10
   }
  ],
  "comment": "Accepted for publication at the 2026 IEEE/AIAA Digital Avionics Systems Conference (DASC). 9 pages, 8 figures",
  "topics": [
   "rl-control",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.10014v1",
  "pdf_url": "https://arxiv.org/pdf/2607.10014v1",
  "html_url": "https://arxiv.org/html/2607.10014v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.09962",
  "slug": "task-planning-for-mobile-manipulation-in-retail-stores-using-foundatio",
  "title": "Task Planning for Mobile Manipulation in Retail Stores using Foundation Models with Iterative Re-planning",
  "abstract": "Automation in industries such as retail, warehousing and logistics presents opportunities for greater throughput, cost reduction and mitigation of disruptions from labour shortages. Previously, such efforts have focused on back-room operations involving packing and sorting in relatively structured environments. With advances in robotic mobile manipulation hardware and foundation models, automation can now be applied to more variable and human-centric environments such as retail store shelves. In this work, we present a task-planning approach using Large Language Models (LLMs) and Vision-Language Models (VLMs) to address the restocking problem in retail scenarios such as supermarkets. We demonstrate this system on a custom omnidirectional mobile manipulation platform, with user-driven prompts and a feedback-based iterative re-planning approach for error correction. The end-to-end system is validated in a PyBullet simulation environment for pick-and-place tasks.",
  "published": "2026-07-10",
  "updated": "2026-07-10",
  "year": "2026",
  "authors": [
   "Vismay Vakharia",
   "Sanjana Garai",
   "Rolif Lima",
   "Nijil George",
   "Vighnesh Vatsal",
   "Kaushik Das"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work presents a task-planning approach using Large Language Models (LLMs) and Vision-Language Models (VLMs) to address the restocking problem in retail scenarios such as supermarkets.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Vismay Vakharia",
    "id": "2148521314",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Sanjana Garai",
    "id": "2333354022",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Rolif Lima",
    "id": "30545360",
    "h_index": 6,
    "papers": 28
   },
   {
    "name": "Nijil George",
    "id": "92698974",
    "h_index": 4,
    "papers": 14
   },
   {
    "name": "Vighnesh Vatsal",
    "id": "2995657",
    "h_index": 6,
    "papers": 29
   },
   {
    "name": "Kaushik Das",
    "id": "2341340906",
    "h_index": 2,
    "papers": 12
   }
  ],
  "comment": "",
  "topics": [
   "navigation",
   "foundation-pretraining",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.09962v1",
  "pdf_url": "https://arxiv.org/pdf/2607.09962v1",
  "html_url": "https://arxiv.org/html/2607.09962v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.09959",
  "slug": "seamlis-visibility-aware-safety-for-perception-limited-multi-robot-exp",
  "title": "SEAMLiS: Visibility-Aware Safety for Perception-Limited Multi-Robot Exploration",
  "abstract": "Autonomous exploration in unknown environments is typically driven by informative frontiers, viewpoints, or trajectories, while local safety controllers avoid obstacles represented in the current map. Under finite sensing range and limited field of view, this separation can be unsafe: an exploration stack may plan optimistically through unobserved space and steer the sensor toward information gain rather than along the direction of motion, causing hidden obstacles to be detected too late for bounded-actuation avoidance. This paper presents SEAMLiS (Safe Exploration for Autonomous Multi-Robot Systems Under Limited Sensing), a modular execution-layer safety framework for decentralized multi-robot exploration. SEAMLiS preserves the upstream exploration stack, including the goal allocator and local planner, and enforces safety at the execution layer through perception-aware attitude and positional filters. A gatekeeper-based attitude filter switches between a visibility-promoting yaw policy and a velocity-tracking backup policy to preserve visibility of the critical known-free/unknown boundary with sufficient braking margin. A Control Barrier Function (CBF)-based positional filter then avoids known obstacles, newly detected obstacles, and other robots. We provide sufficient collision-avoidance conditions and validate the framework in randomized simulation, Isaac Sim, and Crazyflie hardware experiments. Results show collision-free exploration across tested single- and multi-robot settings while retaining much of the efficiency of visibility-promoting yaw control.",
  "published": "2026-07-10",
  "updated": "2026-07-10",
  "year": "2026",
  "authors": [
   "Taekyung Kim",
   "Rahul H Kumar",
   "Aswin D. Menon",
   "Tzu-Hsiang Lin",
   "Dimitra Panagou"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "SEAMLiS preserves the upstream exploration stack, including the goal allocator and local planner, and enforces safety at the execution layer through perception-aware attitude and positional filters, and shows collision-free exploration across tested single- and multi-robot settings while retaining much of the efficiency of visibility-promoting yaw control.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Taekyung Kim",
    "id": "2305753219",
    "h_index": 5,
    "papers": 17
   },
   {
    "name": "Rahul Kumar",
    "id": "2305567935",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Aswin D. Menon",
    "id": "2409242943",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Tzu-Hsiang Lin",
    "id": "2289663624",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Dimitra Panagou",
    "id": "2762989",
    "h_index": 32,
    "papers": 221
   }
  ],
  "comment": "Project page: https://www.taekyung.me/seamlis",
  "topics": [
   "rl-control",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.09959v1",
  "pdf_url": "https://arxiv.org/pdf/2607.09959v1",
  "html_url": "https://arxiv.org/html/2607.09959v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.09866",
  "slug": "robo-valuerl-reliable-value-estimation-for-offline-to-online-reinforce",
  "title": "Robo-ValueRL: Reliable Value Estimation for Offline-to-Online Reinforcement Learning",
  "abstract": "Offline-to-online reinforcement learning is promising for generalizable robotic manipulation, yet its full-stack complexity obscures reproduction and diagnosis. Within such systems, value estimation plays a central role in prioritizing heterogeneous data for policy improvement. Despite its importance, the central question remains underexplored: how value-function reliability shapes policy optimization in offline-to-online reinforcement learning. To answer this question, we propose Robo-ValueRL, a unified framework that enables reliable value estimation and systematically traces its downstream effects on policy pretraining and online improvement. Concretely, Robo-ValueRL learns a history-conditioned value estimator and evaluates its reliability through global-progress and local-preference metrics. These resulting value estimates are propagated into quality-conditioned consistency-policy pretraining and a residual adaptation module on online rollouts, providing a unified testbed for analyzing how value reliability shapes downstream policy performance. Across 240 hours of offline demonstrations and over 3,000 online rollout trajectories, our extensive experiments show that downstream performance is strongly associated with value reliability. Reliable value functions provide better action-quality estimates, allowing value-guided offline RL to scale more effectively than quality-agnostic behavior cloning, and stabilize online improvement by prioritizing high-quality rollout data. Integrating reliable value guidance through offline pretraining with online improvement, our system achieves 86% success on millimeter-level precise chip insertion and 84% on generalizable block disassembly. We hope these findings highlight the importance of value-guided data utilization for effective policy improvement from heterogeneous robotic experience.",
  "published": "2026-07-10",
  "updated": "2026-07-10",
  "year": "2026",
  "authors": [
   "Wenke Xia",
   "Pei Ren",
   "Wenbo Yu",
   "Yizhuo Zhang",
   "Jifan Li",
   "Yixue Zhang",
   "Yinuo Zhao",
   "Qingyang Gao",
   "Jianlong Fu",
   "Jian Tang",
   "Ji-Rong Wen",
   "Zhengping Che",
   "Di Hu"
  ],
  "author_count": 13,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "This work proposes Robo-ValueRL, a unified framework that enables reliable value estimation and systematically traces its downstream effects on policy pretraining and online improvement, and achieves 86% success on millimeter-level precise chip insertion and 84% on generalizable block disassembly.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Wenke Xia",
    "id": "2201319923",
    "h_index": 8,
    "papers": 16
   },
   {
    "name": "Pei Ren",
    "id": "2365229417",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Wenbo Yu",
    "id": "2374260069",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Yizhuo Zhang",
    "id": "2378472142",
    "h_index": 1,
    "papers": 9
   },
   {
    "name": "Jifan Li",
    "id": "2449712735",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yixue Zhang",
    "id": "2390530244",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Yinuo Zhao",
    "id": "1720832487",
    "h_index": 10,
    "papers": 18
   },
   {
    "name": "Qingyang Gao",
    "id": "2449651411",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Jianlong Fu",
    "id": "2396435303",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Jian Tang",
    "id": "2152779004",
    "h_index": 12,
    "papers": 33
   },
   {
    "name": "Ji-Rong Wen",
    "id": "2265728957",
    "h_index": 6,
    "papers": 22
   },
   {
    "name": "Zhengping Che",
    "id": "1939695",
    "h_index": 26,
    "papers": 88
   },
   {
    "name": "Di Hu",
    "id": "2265546488",
    "h_index": 5,
    "papers": 10
   }
  ],
  "comment": "Please refer to our website: https://gewu-lab.github.io/Robo-ValueRL/",
  "topics": [
   "imitation-diffusion",
   "rl-control",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.09866v1",
  "pdf_url": "https://arxiv.org/pdf/2607.09866v1",
  "html_url": "https://arxiv.org/html/2607.09866v1",
  "code_url": "https://gewu-lab.github.io/Robo-ValueRL/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.3
 },
 {
  "id": "2607.09818",
  "slug": "ts-mask-vla-2d-temporal-spatial-masking-for-vision-language-action-mod",
  "title": "TS-Mask VLA: 2D Temporal-Spatial Masking for Vision-Language-Action Model with Effective Bridging",
  "abstract": "Vision-language-action (VLA) models aim to understand natural-language instructions and visual observations, and to generate and execute corresponding actions as embodied agents. Recently, autoregressive token-based action generation has driven the development of many representative VLA models. However, this paradigm often reduces action generation to next-token prediction, thereby lacking explicit modeling of the spatiotemporal structure of action sequences and the disentanglement between vision-language representations and actions, which can limit performance in long-horizon and complex scenarios. In this paper, we propose TS-Mask VLA, a vision-language-action framework for robot manipulation. TS-Mask VLA is built upon two key designs: (1) a Discrete Diffusion Action Expert equipped with a Bridge Attention conditioning bridge, which enables multi-layer conditioning from the VLM and facilitates more accurate and stable action generation; and (2) a temporal-spatial 2D masking strategy for discrete action tokens that strengthens the model's understanding of cross-time dependencies and inter-dimensional coupling, leading to more structurally consistent action sequences. We conduct extensive experiments on simulation benchmarks and real-world tasks. On LIBERO, TS-Mask VLA achieves a 95.7 percent average success rate with only 0.5B parameters, outperforming significantly larger models. On CALVIN, it attains the best average sequence length of 4.19 and strong long-horizon performance. Comprehensive analyses and ablations further validate the effectiveness of our design.",
  "published": "2026-07-10",
  "updated": "2026-07-10",
  "year": "2026",
  "authors": [
   "Shengzhuo Yang",
   "Ronghao Yu",
   "Chuanjie Lv",
   "Linpeng Peng",
   "Hang Yu",
   "Jie Ren",
   "Jiajun Lv",
   "Yong Liu"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "TS-Mask VLA is built upon two key designs: a Discrete Diffusion Action Expert equipped with a Bridge Attention conditioning bridge, which enables multi-layer conditioning from the VLM and facilitates more accurate and stable action generation; and a temporal-spatial 2D masking strategy for discrete action tokens that strengthens the model's understanding of cross-time dependencies and inter-dimensional coupling, leading to more structurally consistent action sequences.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Shengzhuo Yang",
    "id": "2449938920",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Ronghao Yu",
    "id": "2323542702",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Chuanjie Lv",
    "id": "2449640097",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Linpeng Peng",
    "id": "89066848",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Hang Yu",
    "id": "2245270655",
    "h_index": 6,
    "papers": 16
   },
   {
    "name": "Jie Ren",
    "id": "2449700713",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Jiajun Lv",
    "id": "2054668538",
    "h_index": 12,
    "papers": 32
   },
   {
    "name": "Yong Liu",
    "id": "2317960641",
    "h_index": 5,
    "papers": 14
   }
  ],
  "comment": "9 pages, 5 figures, accepted to the 2026 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS 2026)",
  "topics": [
   "vla",
   "sim2real"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.09818v1",
  "pdf_url": "https://arxiv.org/pdf/2607.09818v1",
  "html_url": "https://arxiv.org/html/2607.09818v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2607.09815",
  "slug": "rasr-range-aware-scale-recovery-for-metric-uav-navigation",
  "title": "RASR: Range-Aware Scale Recovery for Metric UAV Navigation",
  "abstract": "A central challenge in image-goal UAV navigation under Global Navigation Satellite System (GNSS) denial is estimating metric distance and heading between current and goal views. Dense pairwise geometry models capture relative scene structure, but without a calibrated metric scale, they cannot directly provide reliable distance estimates for navigation. Although global scale calibration corrects the dominant scale bias, the remaining errors vary systematically with distance. In this paper, Range-Aware Scale Recovery (RASR) is proposed, which complements global scale calibration with range-aware residual correction. RASR encodes pairwise geometry extracted by a frozen Matching And Stereo 3D Reconstruction (MASt3R) backbone as a compact descriptor and separates the scale-recovery core from task-specific command calibration. On the official online evaluation of the UAVs in Multimedia 2026 PairUAV challenge, RASR achieved a total error of 0.003189, achieving a lower total error than global scale calibration alone. The results demonstrate that range-aware residual correction improves metric distance estimation beyond global scale calibration. Code and materials are available at https://github.com/lht-research/rasr-pairuav.",
  "published": "2026-07-10",
  "updated": "2026-07-16",
  "year": "2026",
  "authors": [
   "Hongtao Liang",
   "Xinyu Shao",
   "Chenxu Wang",
   "Yiyao Wan",
   "Jiahuan Ji",
   "Fangwei Ye",
   "Fuhui Zhou",
   "Qihui Wu"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "RASR encodes pairwise geometry extracted by a frozen Matching And Stereo 3D Reconstruction backbone as a compact descriptor and separates the scale-recovery core from task-specific command calibration, which improves metric distance estimation beyond global scale calibration.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hongtao Liang",
    "id": "2312343381",
    "h_index": 2,
    "papers": 14
   },
   {
    "name": "Xinyu Shao",
    "id": "2449642043",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Chenxu Wang",
    "id": "2449699108",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yiyao Wan",
    "id": "2015963118",
    "h_index": 5,
    "papers": 17
   },
   {
    "name": "Jiahuan Ji",
    "id": "2340563444",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Fangwei Ye",
    "id": "2292452892",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Fuhui Zhou",
    "id": "2114902865",
    "h_index": 28,
    "papers": 241
   },
   {
    "name": "Qihui Wu",
    "id": "2259827557",
    "h_index": 28,
    "papers": 225
   }
  ],
  "comment": "5 pages, 4 figures. Technical report for the UAVM 2026 PairUAV Challenge",
  "topics": [
   "spatial-3d",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.09815v2",
  "pdf_url": "https://arxiv.org/pdf/2607.09815v2",
  "html_url": "https://arxiv.org/html/2607.09815v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.09590",
  "slug": "pac-act-post-training-actor-critic-for-action-chunking-transformers",
  "title": "PAC-ACT: Post-training Actor-Critic for Action Chunking Transformers",
  "abstract": "Precision industrial contact manipulation requires reliable robot policies under pose perturbations and contact-force constraints. Vision-language-action models offer broad generalization but often introduce high inference latency and GPU-memory cost, while vision-action chunking policies are more suitable for real-time industrial control. However, these policies are usually trained by behavior cloning and suffer from distribution shift in contact-rich tasks. This paper proposes PAC-ACT, a reinforcement-learning post-training framework for pretrained Action Chunking Transformer policies. PAC-ACT reformulates policy optimization at the chunk level, constructs an ACT-transferred actor-critic architecture, and introduces a hybrid behavior-prior constraint to preserve the pretrained action distribution during online fine-tuning. Experiments on industrial precision-contact benchmarks show that PAC-ACT improves task success, contact stability, and force safety while retaining low latency and low GPU-memory usage. On the Contour task, PAC-ACT significantly reduces peak contact force and decreases the proportion of force readings above 60 N by 46 times. Sparse-reward ablations further show that the proposed behavior-prior constraint enables effective exploration under randomized initial poses.",
  "published": "2026-07-10",
  "updated": "2026-07-10",
  "year": "2026",
  "authors": [
   "Yujie Pang",
   "Zudong Li"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "PAC-ACT is proposed, a reinforcement-learning post-training framework for pretrained Action Chunking Transformer policies that improves task success, contact stability, and force safety while retaining low latency and low GPU-memory usage and introduces a hybrid behavior-prior constraint.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yujie Pang",
    "id": "2449506463",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Zudong Li",
    "id": "2185943831",
    "h_index": 1,
    "papers": 4
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "tactile",
   "imitation-diffusion",
   "rl-control",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.09590v1",
  "pdf_url": "https://arxiv.org/pdf/2607.09590v1",
  "html_url": "https://arxiv.org/html/2607.09590v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.09557",
  "slug": "coral-auv-cfd-oriented-reinforcement-learning-for-autonomous-underwate",
  "title": "CORAL-AUV: CFD Oriented Reinforcement Learning for Autonomous Underwater Vehicles",
  "abstract": "Fine grain control and positioning of autonomous underwater vehicles (AUVs) is critical for sampling, maintenance, and survey applications. Traditional control methods for AUVs are labor intensive and are not robust to changes in the vehicle configuration or environmental conditions. Reinforcement learning (RL) promises rapid controller development while handling a range of deployment parameters via domain randomization (DR). However, DR is still limited by the capacity of the underlying simulation to model real physics. In particular, drag physics are difficult to model and are a large contributor to sim-to-real gaps. Meanwhile, computational fluid dynamics (CFD) provides high fidelity drag models but is challenging to leverage within reinforcement learning frameworks due to its computational overhead. Thus, in this paper we exploit the idea of training surrogate approximations of CFD models of a given vehicle, enabling fast inference within RL pipelines. We are the first to successfully deploy a zero-shot RL policy on a 6-DOF AUV in which policy training is performed on surrogate drag models (SDMs) trained on CFD data. We find 31% lower energy usage compared to a controller using simplified physics while traversing between waypoints 11% faster with 19% less error. Our SDM based RL controller better predicts zero-shot transfer and is more robust across reward shaping design choices. When using DR to complete a task with perturbed parameters, we find that the CFD policy is the only controller that successfully transfers. The policies are evaluated in a controlled tank environment and in the field providing extensive testing of the policies' capabilities.",
  "published": "2026-07-10",
  "updated": "2026-07-10",
  "year": "2026",
  "authors": [
   "Steven Roche",
   "Milo Van Mooy",
   "Nathan McGuire",
   "Levi Cai",
   "Jonathan P. How",
   "Yogesh Girdhar"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper is the first to successfully deploy a zero-shot RL policy on a 6-DOF AUV in which policy training is performed on surrogate drag models (SDMs) trained on CFD data, enabling fast inference within RL pipelines.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Steven Roche",
    "id": "2399420976",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Milo Van Mooy",
    "id": "2449506591",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Nathan McGuire",
    "id": "2154619909",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Levi Cai",
    "id": "2301264924",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Jonathan P. How",
    "id": "2245011516",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Y. Girdhar",
    "id": "2263172754",
    "h_index": 5,
    "papers": 20
   }
  ],
  "comment": "16 pages, 13 figures",
  "topics": [
   "sim2real",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.09557v1",
  "pdf_url": "https://arxiv.org/pdf/2607.09557v1",
  "html_url": "https://arxiv.org/html/2607.09557v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.09365",
  "slug": "physv2a-reachability-gated-and-semantic-mask-constrained-feasibility-c",
  "title": "PhysV2A: Reachability-Gated and Semantic-Mask-Constrained Feasibility Completion for Video-to-Robot Manipulation",
  "abstract": "Video-based manipulation provides object-centric motion priors from human demonstrations, generated videos, or RGB-D observations, but such priors are typically embodiment-agnostic and cannot be directly executed by a specific robot. This paper presents \\textbf{PhysV2A}, a reachability-gated and semantic-mask-constrained feasibility-completion framework for converting video-derived 6D object motion into robot-executable manipulation trajectories. The key idea is to treat grasp feasibility as trajectory-conditioned rather than local: each RGB-D-generated 6-DoF grasp candidate is rigidly coupled with the recovered object motion to form a grasp-conditioned TCP trajectory hypothesis. PhysV2A then performs hierarchical reachability-gated selection, where infeasible grasp--trajectory pairs are rejected by robot-centric kinematic checks and surviving candidates are ranked by downstream execution suitability. For the selected reachable trajectory, a VLM-assisted and rule-validated S-Mask identifies task-critical and relaxable Cartesian components, enabling semantic-mask-constrained manipulability refinement through redundancy-first optimization and bounded Cartesian relaxation. Real-robot experiments on four tabletop manipulation tasks show that PhysV2A improves task success over representative video-prior and IK-only baselines, reduces kinematic-feasibility failures, and produces better-conditioned trajectories with bounded semantic deviations.",
  "published": "2026-07-10",
  "updated": "2026-07-10",
  "year": "2026",
  "authors": [
   "Haohui Huang",
   "Junda Duan",
   "Tao Teng",
   "Chenguang Yang"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Real-robot experiments on four tabletop manipulation tasks show that PhysV2A improves task success over representative video-prior and IK-only baselines, reduces kinematic-feasibility failures, and produces better-conditioned trajectories with bounded semantic deviations.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Haohui Huang",
    "id": "2282385733",
    "h_index": 2,
    "papers": 15
   },
   {
    "name": "Junda Duan",
    "id": "2422300696",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Tao Teng",
    "id": "2315508495",
    "h_index": 4,
    "papers": 17
   },
   {
    "name": "Chenguang Yang",
    "id": "2203800278",
    "h_index": 6,
    "papers": 15
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "egocentric-data"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.09365v1",
  "pdf_url": "https://arxiv.org/pdf/2607.09365v1",
  "html_url": "https://arxiv.org/html/2607.09365v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.09234",
  "slug": "implicit-behavior-coordination-from-unlabeled-sub-task-demonstrations",
  "title": "Implicit-Behavior Coordination from Unlabeled Sub-Task Demonstrations for Rearrangement Tasks",
  "abstract": "Long-horizon robotic rearrangement tasks are often treated as skill sequencing problems, requiring predefined skills, skill labels, or boundaries, and task-specific switching logic. Although effective, such explicit skill abstractions can become difficult to scale as the number of behaviors and the task horizon increase. We instead formulate rearrangement as implicit-behavior coordination from unlabeled sub-task demonstrations, where skill-like behaviors are learned directly from mixed behavior data and coordinated through value-guided action selection. Experiments in Habitat rearrangement tasks support this formulation in three ways. First, our method outperforms task-specific imitation baselines on more complex rearrangement tasks and approaches an oracle-planner baseline with behavior-cloned skills, while using no oracle task plan or skill-labeled full-task demonstrations. Second, ablations show that reliable critic-guided candidate selection is essential for coordinating multi-modal behaviors. Third, scaling experiments show that the method handles larger behavior repertoires and maintains stronger performance than task-specific imitation baselines as chained targets extend the horizon. These results suggest that explicit skill abstraction is not a prerequisite for long-horizon rearrangement, and that implicit-behavior coordination offers a promising data-driven alternative to explicit skill-based pipelines.",
  "published": "2026-07-10",
  "updated": "2026-07-10",
  "year": "2026",
  "authors": [
   "Ahmed Shokry",
   "Usama Ahmed Siddiquie",
   "Sicong Pan",
   "Maren Bennewitz"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The results suggest that explicit skill abstraction is not a prerequisite for long-horizon rearrangement, and that implicit-behavior coordination offers a promising data-driven alternative to explicit skill-based pipelines.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "A. Shokry",
    "id": "48084429",
    "h_index": 15,
    "papers": 80
   },
   {
    "name": "Usama Ahmed Siddiquie",
    "id": "2385460475",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Sicong Pan",
    "id": "10456004",
    "h_index": 9,
    "papers": 26
   },
   {
    "name": "Maren Bennewitz",
    "id": "2295575242",
    "h_index": 2,
    "papers": 17
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.09234v1",
  "pdf_url": "https://arxiv.org/pdf/2607.09234v1",
  "html_url": "https://arxiv.org/html/2607.09234v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.09190",
  "slug": "tactidex-a-real-world-tactile-guided-benchmark-for-human-like-dexterou",
  "title": "TactiDex: A Real-World Tactile-Guided Benchmark for Human-Like Dexterous Manipulation",
  "abstract": "Tactile feedback is fundamental to Hand-Object Interaction (HOI), governing contact formation, force regulation, and stable manipulation, making it essential for achieving true human-like dexterous manipulation. Yet, current human-to-robot dexterous transfer pipelines primarily rely on kinematic trajectories, resulting in motion imitation without physically grounded interaction. To address this, we introduce TactiDex, a real-world tactile-guided benchmark specifically designed to move dexterous manipulation beyond kinematic mimicry toward contact-level human-likeness. TactiDex provides a comprehensive dataset that elegantly aligns whole-hand tactile signals with multi-granularity kinematic and object states, coupled with standardized evaluation metrics. Building upon this data paradigm, we propose a tactile-driven transfer framework that effectively translates human demonstrations into physically plausible robotic execution. We introduce TactiSkill, a framework built upon a novel tri-component tactile reward that innovatively uses tactile signals as structured supervision. This reward unifies guidance, human-like alignment, and contact constraints into a single objective. Through comprehensive experiments on both single and bimanual tasks, we demonstrate that TactiSkill achieves superior performance in manipulation success and physical realism. This work lays a crucial foundation for advancing tactile-aware dexterous manipulation. Our project page at https://tactidex.github.io/.",
  "published": "2026-07-10",
  "updated": "2026-07-10",
  "year": "2026",
  "authors": [
   "Suting Ni",
   "Hanbing Zhang",
   "Zhenyu Wei",
   "Guo Chen",
   "Chixuan Zhang",
   "Ye Shi",
   "Jingya Wang"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 2,
  "influential_citations": 0,
  "tldr": "TactiDex is introduced, a real-world tactile-guided benchmark specifically designed to move dexterous manipulation beyond kinematic mimicry toward contact-level human-likeness and TactiSkill, a framework built upon a novel tri-component tactile reward that innovatively uses tactile signals as structured supervision.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Suting Ni",
    "id": "2338898182",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Hanbing Zhang",
    "id": "2448367979",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Zhenyu Wei",
    "id": "2394124879",
    "h_index": 4,
    "papers": 16
   },
   {
    "name": "Guoyin Chen",
    "id": "2440937631",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Chi Zhang",
    "id": "2269711204",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Ye Shi",
    "id": "2192580886",
    "h_index": 17,
    "papers": 50
   },
   {
    "name": "Jingya Wang",
    "id": "2273021500",
    "h_index": 11,
    "papers": 42
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "egocentric-data",
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.09190v1",
  "pdf_url": "https://arxiv.org/pdf/2607.09190v1",
  "html_url": "https://arxiv.org/html/2607.09190v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.48
 },
 {
  "id": "2607.09060",
  "slug": "dec-marvel-decentralized-multi-agent-exploration-without-communication",
  "title": "Dec-MARVEL: Decentralized Multi-Agent Exploration without Communication under Budget Constraints",
  "abstract": "Multi-UAV exploration is often constrained by unreliable communication, limited field-of-view sensing (e.g., lightweight onboard camera), and finite travel budgets that require each robot to reserve enough budget to return to its base. We present Dec-MARVEL, a decentralized budget-aware exploration framework for communication-free teams with directional sensing. Rather than exchanging maps, goals, or messages, each robot coordinates through its incidental observations: any teammate trajectory within its field of view serves as a coordination signal. A graph-attention actor fuses local frontier geometry, teammate motion, and budget features to select return-feasible waypoint-heading actions. The actor is trained with phase-conditioned critics, a training-only task-oriented privileged critic, and a mixture-based budget curriculum. Across 900 held-out trials spanning three team sizes (2, 4, 8 robots) and three travel budgets (720, 800, 1024 meters) against four baselines, Dec-MARVEL achieves the highest or tied-highest exploration rate and lowest sensing overlap across all nine team-size budget configurations. Under our tightest 720m budget, it reaches 53%, 94%, and 100% success for 2, 4, and 8 robots, versus 37%, 83%, and 99% for the strongest baseline. Physical-robot experiments demonstrate successful sim-to-real transfer and real-world deployment of Dec-MARVEL.",
  "published": "2026-07-10",
  "updated": "2026-07-13",
  "year": "2026",
  "authors": [
   "Janghyun Cho",
   "Jimmy Chiun",
   "Guillaume Sartoretti",
   "Changjoo Nam"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Dec-MARVEL is presented, a decentralized budget-aware exploration framework for communication-free teams with directional sensing that achieves the highest or tied-highest exploration rate and lowest sensing overlap across all nine team-size budget configurations.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Janghyun Cho",
    "id": "2449591275",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Jimmy Chiun",
    "id": "2284773169",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "G. Sartoretti",
    "id": "2292917033",
    "h_index": 11,
    "papers": 67
   },
   {
    "name": "Changjoo Nam",
    "id": "37951975",
    "h_index": 18,
    "papers": 49
   }
  ],
  "comment": "8 pages, 5 figures",
  "topics": [
   "sim2real",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.09060v2",
  "pdf_url": "https://arxiv.org/pdf/2607.09060v2",
  "html_url": "https://arxiv.org/html/2607.09060v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.08948",
  "slug": "splatctrl-perception-action-coupling-via-gaussian-scene-representation",
  "title": "SplatCtrl: Perception-Action Coupling via Gaussian Scene Representations and Reactive Robot Control",
  "abstract": "Robotic manipulators excel in structured environments but face substantial challenges in unstructured and dynamic settings. This paper presents SplatCtrl, a unified framework for real-time scene reconstruction and reactive robot motion generation to enable collision-free robotic arm control in previously unseen and continuously changing environments. Building on 3D Gaussian Splatting (3D-GS), we introduce a hybrid voxel-based filtering and dynamic Gaussian relocation strategy that supports efficient scene reconstruction from RGB-D streams while accommodating environmental changes. For safe and reactive control, we further propose a method for deriving continuous signed distance functions from isotropic Gaussians, providing stable and differentiable collision probability estimates that bridge classical distance fields with the modern implicit representation. These continuous distance metrics are incorporated into control barrier functions, resulting in a unified perception-action coupling framework that supports smooth and reliable real-time motion generation in response to scene changes. Experimental validation in simulation, on physical robot, and within shared human-robot workspace demonstrates the framework's effectiveness, achieving integrated scene reconstruction and reactive control in uncertain, and dynamic environments.",
  "published": "2026-07-09",
  "updated": "2026-07-09",
  "year": "2026",
  "authors": [
   "Siddarth Jain",
   "Ho Jin Choi"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "SplatCtrl is presented, a unified framework for real-time scene reconstruction and reactive robot motion generation to enable collision-free robotic arm control in previously unseen and continuously changing environments.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Siddarth Jain",
    "id": "1741915",
    "h_index": 9,
    "papers": 32
   },
   {
    "name": "H. Choi",
    "id": "2258533091",
    "h_index": 3,
    "papers": 6
   }
  ],
  "comment": "Published in 2026 International Conference on Robotics and Automation (ICRA). 8 pages, 8 figures",
  "topics": [
   "rl-control",
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.08948v1",
  "pdf_url": "https://arxiv.org/pdf/2607.08948v1",
  "html_url": "https://arxiv.org/html/2607.08948v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2607.08877",
  "slug": "flowdagger-human-in-the-loop-adaptation-of-generative-robot-policies-i",
  "title": "FlowDAgger: Human-in-the-Loop Adaptation of Generative Robot Policies in Latent Space",
  "abstract": "Pretrained generative robot policies based on flow matching and diffusion have achieved impressive results across a wide range of manipulation tasks. Yet real-world deployments routinely expose failure modes outside the pretraining distribution. Closing these gaps typically requires large-scale data collection or online reinforcement learning on physical hardware, which is impractical for rapid and safe adaptation. We present FlowDAgger, a sample- and compute-efficient method for adapting frozen generative robot policies from human interventions in latent space. Our key idea is action inversion: each human expert action is mapped to the noise that would have produced it under the frozen base policy, using reverse-time integration followed by local refinement. The resulting inverted noise provides supervision for a lightweight latent policy that steers the base model at deployment time, enabling rapid skill acquisition while preserving its behavioral priors. We evaluate FlowDAgger in simulation and on real-world bimanual and single-arm manipulation, adapting both action-head VLAs and world-action models from a handful of interventions. FlowDAgger outperforms supervised fine-tuning and latent-space RL baselines and preserves pretrained skills on held-out tasks, offering a practical path for adapting robot foundation models in the real world. Website: https://microsoft.github.io/FlowDAgger",
  "published": "2026-07-09",
  "updated": "2026-07-09",
  "year": "2026",
  "authors": [
   "Michael Murray",
   "Daphne Chen",
   "Simran Bagaria",
   "Dean Fortier",
   "Tess Hellebrekers",
   "Galen Mullins",
   "Harshavardhan Gajarla",
   "Oier Mees",
   "Maya Cakmak",
   "Andrey Kolobov"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The key idea is action inversion: each human expert action is mapped to the noise that would have produced it under the frozen base policy, using reverse-time integration followed by local refinement, which provides supervision for a lightweight latent policy that steers the base model at deployment time, enabling rapid skill acquisition while preserving its behavioral priors.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Michael Murray",
    "id": "2114300655",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Daphne Chen",
    "id": "2315702161",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Simran Bagaria",
    "id": "2334364751",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Dean Fortier",
    "id": "2400417876",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "T. Hellebrekers",
    "id": "2576308",
    "h_index": 18,
    "papers": 31
   },
   {
    "name": "Galen Mullins",
    "id": "2400414920",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "H. Gajarla",
    "id": "2405582304",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Oier Mees",
    "id": "7264115",
    "h_index": 28,
    "papers": 45
   },
   {
    "name": "Maya Cakmak",
    "id": "2282542087",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "A. Kolobov",
    "id": "2335445895",
    "h_index": 4,
    "papers": 8
   }
  ],
  "comment": "",
  "topics": [
   "rl-control",
   "foundation-pretraining",
   "data-teleop",
   "hri",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.08877v1",
  "pdf_url": "https://arxiv.org/pdf/2607.08877v1",
  "html_url": "https://arxiv.org/html/2607.08877v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.08857",
  "slug": "agenticfocus-object-preserving-mixed-reality-synthesis-from-human-fpv",
  "title": "AgenticFocus: Object-Preserving Mixed Reality Synthesis from Human FPV Video for Dexterous Humanoid Learning",
  "abstract": "Human egocentric video is a scalable supervision source for humanoid policy learning, but current pipelines struggle with hand-object occlusion, oversimplified motion, or specialized capture hardware. We introduce AgenticFocus, a Mixed Reality synthesis pipeline that converts ordinary first-person-view human videos into robot-trainable demonstrations by restoring occluded object geometry, reconstructing full-hand motion, and retargeting it to a humanoid embodiment through camera-relative alignment and layered compositing. The resulting dataset pairs focused visual observations with synchronized robot actions and states. AgenticFocus achieves lower trajectory error and smoother wrist motion than cross-embodiment baselines, with SPARC scores of -5.18 versus -5.56 and -6.05.",
  "published": "2026-07-09",
  "updated": "2026-07-13",
  "year": "2026",
  "authors": [
   "Iaroslav Kolomiets",
   "Miguel Altamirano Cabrera",
   "Artem Lykov",
   "Jeffrin Sam",
   "Dmitrii Iarchuk",
   "Yara Mahmoud",
   "Daniia Zinniatullina",
   "Mikhail Konenkov",
   "Dzmitry Tsetserukou"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "AgenticFocus, a Mixed Reality synthesis pipeline that converts ordinary first-person-view human videos into robot-trainable demonstrations by restoring occluded object geometry, reconstructing full-hand motion, and retargeting it to a humanoid embodiment through camera-relative alignment and layered compositing is introduced.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ia M Kolomiets",
    "id": "37491424",
    "h_index": 0,
    "papers": 6
   },
   {
    "name": "Miguel Altamirano Cabrera",
    "id": "144548970",
    "h_index": 10,
    "papers": 62
   },
   {
    "name": "Artem Lykov",
    "id": "2189477580",
    "h_index": 13,
    "papers": 33
   },
   {
    "name": "Jeffrin Sam",
    "id": "2360360752",
    "h_index": 2,
    "papers": 15
   },
   {
    "name": "Dmitrii Iarchuk",
    "id": "2339774165",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Yara Mahmoud",
    "id": "2351604209",
    "h_index": 2,
    "papers": 13
   },
   {
    "name": "Daniia Zinniatullina",
    "id": "2425366427",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Mikhail Konenkov",
    "id": "2226258666",
    "h_index": 8,
    "papers": 14
   },
   {
    "name": "D. Tsetserukou",
    "id": "48470616",
    "h_index": 27,
    "papers": 298
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "humanoids",
   "egocentric-data",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.08857v2",
  "pdf_url": "https://arxiv.org/pdf/2607.08857v2",
  "html_url": "https://arxiv.org/html/2607.08857v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.08751",
  "slug": "dexverse-a-modular-benchmark-for-multi-task-multi-embodiment-dexterous",
  "title": "DexVerse: A Modular Benchmark for Multi-Task, Multi-Embodiment Dexterous Manipulation",
  "abstract": "Building general-purpose dexterous manipulation policies requires benchmarks that go beyond isolated tasks to systematically evaluate policies across diverse interaction modes, sensory conditions, and robot embodiments. However, existing benchmarks remain limited in task and data diversity, embodiment coverage, or controllable visual variation, hindering studies of cross-task and cross-embodiment generalization. We present DexVerse, a large-scale and modular benchmark for dexterous manipulation. DexVerse includes 100 tasks spanning a broad range of manipulation skills, including object grasping and relocation, articulated-object interaction, functional tool use, bimanual coordination, non-prehensile control, contact-rich behaviors, multi-goal execution, and long-horizon multi-stage task completion. It supports 3 robot arms and 6 dexterous hands, and is extensible to new tasks, assets, and embodiments. To evaluate visuomotor generalization, DexVerse provides configurable visual variations in textures, background, lighting, and camera viewpoints. We further provide a VR-based teleoperation interface and 3,180 demonstrations with synchronized proprioceptive, RGB, depth, point-cloud, and state observations. We benchmark representative methods, including Diffusion Policy, DP3, OpenVLA, and $\u03c0_{0.5}$, across 19 tasks. Results reveal substantial challenges in task generalization and visuomotor robustness, establishing DexVerse as a promising testbed for general-purpose dexterous manipulation. Project page: https://ycyao216.github.io/DexVerse.site",
  "published": "2026-07-09",
  "updated": "2026-07-09",
  "year": "2026",
  "authors": [
   "Yunchao Yao",
   "Zhuxiu Xu",
   "Tianqi Zhang",
   "Zixian Liu",
   "Sikai Li",
   "Zhenyu Wei",
   "Feng Chen",
   "Dihong Huang",
   "Kechang Wan",
   "Chenyang Ma",
   "Shuqi Zhao",
   "Shenghua Gao",
   "Masayoshi Tomizuka",
   "Yi Ma",
   "Mingyu Ding"
  ],
  "author_count": 15,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "Results reveal substantial challenges in task generalization and visuomotor robustness, establishing DexVerse as a promising testbed for general-purpose dexterous manipulation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yunchao Yao",
    "id": "2352910959",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Zhuxiu Xu",
    "id": "2308983686",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Tianqi Zhang",
    "id": "2375099211",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Zixi Liu",
    "id": "2362147041",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Sikai Li",
    "id": "2283135687",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Zhenyu Wei",
    "id": "2394124879",
    "h_index": 4,
    "papers": 16
   },
   {
    "name": "Feng Chen",
    "id": "2377276163",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Di Huang",
    "id": "2292899023",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Kechang Wan",
    "id": "2275613019",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Chenyang Ma",
    "id": "2292350179",
    "h_index": 5,
    "papers": 19
   },
   {
    "name": "Shuqi Zhao",
    "id": "2288286963",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Shenghua Gao",
    "id": "2285702784",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Masayoshi Tomizuka",
    "id": "2261974717",
    "h_index": 7,
    "papers": 27
   },
   {
    "name": "Yi Ma",
    "id": "2255716385",
    "h_index": 7,
    "papers": 14
   },
   {
    "name": "Mingyu Ding",
    "id": "2346837065",
    "h_index": 4,
    "papers": 15
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "imitation-diffusion",
   "foundation-pretraining",
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.08751v1",
  "pdf_url": "https://arxiv.org/pdf/2607.08751v1",
  "html_url": "https://arxiv.org/html/2607.08751v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2607.08741",
  "slug": "ardy-autoregressive-diffusion-with-hybrid-representation-for-interacti",
  "title": "ARDY: Autoregressive Diffusion with Hybrid Representation for Interactive Human Motion Generation",
  "abstract": "Generating realistic 3D human motions in real-time within interactive applications is key for animation, simulation, and humanoid robotics. While recent offline motion generation approaches offer precise control via text and kinematic constraints, they lack the inference speed required for interactive settings. Conversely, existing online methods enable real-time synthesis but often sacrifice controllability or struggle with complex text semantics and long-horizon goals due to limited context windows. In this work, we introduce ARDY, a streaming generation framework that bridges this gap by enabling high-fidelity motion generation controllable via online text prompts and flexible kinematic constraints. ARDY employs a hybrid representation that combines explicit root features with a latent body embedding, balancing precise trajectory control with efficient generative learning. We propose a two-stage autoregressive transformer denoiser that features variable history context and supports conditioning on flexible, long-horizon kinematic constraints. By training on a large-scale motion capture dataset and being directly conditioned on text labels and kinematic constraints sampled from ground truth poses, ARDY natively learns controllable generation that supports online prompting and flexible long-horizon goals. Extensive evaluations on the HumanML3D benchmark and the large-scale, high-fidelity Bones Rigplay dataset demonstrate ARDY's high motion quality and constraint adherence, validating the efficacy of our key architectural decisions. Finally, we demonstrate the method's practical versatility through an interactive demo featuring dynamic text control, diverse keyframe pose constraints, path following, and interactive locomotion control via mouse and keyboard. Supplementary video results, code, and model releases can be found at https://research.nvidia.com/labs/sil/projects/ardy/.",
  "published": "2026-07-09",
  "updated": "2026-07-09",
  "year": "2026",
  "authors": [
   "Kaifeng Zhao",
   "Mathis Petrovich",
   "Haotian Zhang",
   "Tingwu Wang",
   "Siyu Tang",
   "Davis Rempe"
  ],
  "author_count": 6,
  "categories": [
   "cs.GR",
   "cs.CV",
   "cs.LG",
   "cs.RO"
  ],
  "primary_category": "cs.GR",
  "venue": "SIGGRAPH 2026",
  "venue_source": "arxiv-comment",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "ARDY is introduced, a streaming generation framework that bridges this gap by enabling high-fidelity motion generation controllable via online text prompts and flexible kinematic constraints and balancing precise trajectory control with efficient generative learning.",
  "doi": "10.1145/3811284",
  "oa_pdf": "https://doi.org/10.1145/3811284",
  "s2_authors": [
   {
    "name": "Kaifeng Zhao",
    "id": "2074108295",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Mathis Petrovich",
    "id": "1562113276",
    "h_index": 9,
    "papers": 15
   },
   {
    "name": "Haotian Zhang",
    "id": "2315848763",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Tingwu Wang",
    "id": "2392415176",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Siyu Tang",
    "id": "2312064418",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Davis Rempe",
    "id": "2279753371",
    "h_index": 6,
    "papers": 12
   }
  ],
  "comment": "ACM Transactions on Graphics (SIGGRAPH 2026)",
  "topics": [
   "humanoids",
   "egocentric-data"
  ],
  "orgs": [
   "NVIDIA"
  ],
  "abs_url": "https://arxiv.org/abs/2607.08741v1",
  "pdf_url": "https://arxiv.org/pdf/2607.08741v1",
  "html_url": "https://arxiv.org/html/2607.08741v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.3
 },
 {
  "id": "2607.08639",
  "slug": "native-video-action-pretraining-for-generalizable-robot-control",
  "title": "Native Video-Action Pretraining for Generalizable Robot Control",
  "abstract": "The advent of video-action models offers a promising path for robot control. Nevertheless, we argue that repurposing video generative models designed for digital content creation is inherently inadequate for physical environments. To bridge this gap, we present LingBot-VA 2.0, a video-action foundation model built from the ground up for embodiment. Four core design principles showcase its evolution from LingBot-VA. (1) Departing from traditional reconstruction-focused VAEs, we introduce a semantic visual-action tokenizer, which aligns visual representations with both semantics and actions, improving instruction following and action precision in subsequent policy learning. (2) Given the strictly causal nature of temporal dynamics, we adopt a causal pretraining paradigm, training from scratch to circumvent the catastrophic forgetting that frequently occurs when adapting bidirectional architectures. (3) To meet the demands of high-frequency inference, our model employs a sparse MoE backbone, expanding model capacity without compromising efficiency. (4) Real-time closed-loop control is realized through an enhanced asynchronous inference scheme, which predicts future latents in parallel with action execution while re-grounding each rollout on the latest observation via learned forward dynamics. Real-world deployment validates LingBot-VA 2.0 as a robust foundation model, as evidenced by its few-shot generalization across complex manipulation tasks.",
  "published": "2026-07-09",
  "updated": "2026-07-16",
  "year": "2026",
  "authors": [
   "Qihang Zhang",
   "Lin Li",
   "Luyao Zhang",
   "Shuai Yang",
   "Yiming Luo",
   "Shuaiting Li",
   "Ruilin Wang",
   "Junke Wang",
   "Jiahao Shao",
   "Gangwei Xu",
   "Jiaming Zhou",
   "Yishu Shen",
   "Yudong Jin",
   "Fangyi Xu",
   "Shuailei Ma",
   "Jiaqi Liao",
   "Guanxing Lu",
   "Zifan Shi",
   "Yongkun Wen",
   "Yujie Zhao",
   "Weixuan Tang",
   "Xinyang Wang",
   "Chaojian Li",
   "Jiapeng Zhu",
   "Ka Leong Cheng",
   "Nan Xue",
   "Xing Zhu",
   "Yujun Shen",
   "Yinghao Xu"
  ],
  "author_count": 29,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 7,
  "influential_citations": 1,
  "tldr": "LingBot-VA 2.0 is presented, a video-action foundation model built from the ground up for embodiment, which introduces a semantic visual-action tokenizer, which aligns visual representations with both semantics and actions, improving instruction following and action precision in subsequent policy learning.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Qihang Zhang",
    "id": "2112192853",
    "h_index": 10,
    "papers": 20
   },
   {
    "name": "Lin Li",
    "id": "2340512661",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Luyao Zhang",
    "id": "2224109770",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Shuai Yang",
    "id": "2302532869",
    "h_index": 9,
    "papers": 16
   },
   {
    "name": "Yiming Luo",
    "id": "2357052308",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Shuai Li",
    "id": "2448023008",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Ruilin Wang",
    "id": "2226504527",
    "h_index": 5,
    "papers": 39
   },
   {
    "name": "Junke Wang",
    "id": "2124919221",
    "h_index": 15,
    "papers": 32
   },
   {
    "name": "Jiahao Shao",
    "id": "2292307890",
    "h_index": 5,
    "papers": 18
   },
   {
    "name": "Gangwei Xu",
    "id": "2158317969",
    "h_index": 12,
    "papers": 32
   },
   {
    "name": "Jiaming Zhou",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Yishu Shen",
    "id": "2448916432",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yudong Jin",
    "id": "2336180963",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Fang Xu",
    "id": "2447445892",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Shuailei Ma",
    "id": "2293403497",
    "h_index": 9,
    "papers": 17
   },
   {
    "name": "Jiaqi Liao",
    "id": "2315613899",
    "h_index": 7,
    "papers": 16
   },
   {
    "name": "Guanxing Lu",
    "id": "2273936081",
    "h_index": 8,
    "papers": 16
   },
   {
    "name": "Zifan Shi",
    "id": "2125701522",
    "h_index": 12,
    "papers": 25
   },
   {
    "name": "Yongkun Wen",
    "id": "2114785666",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Yujie Zhao",
    "id": "2400145455",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Weixuan Tang",
    "id": "2320291038",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Xinyang Wang",
    "id": "2448705668",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Chaojian Li",
    "id": "28987646",
    "h_index": 19,
    "papers": 53
   },
   {
    "name": "Jiapeng Zhu",
    "id": "47054925",
    "h_index": 13,
    "papers": 32
   },
   {
    "name": "K. Cheng",
    "id": "2297462874",
    "h_index": 10,
    "papers": 23
   },
   {
    "name": "Nan Xue",
    "id": "2292027056",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "Xing Zhu",
    "id": "2406915796",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Yujun Shen",
    "id": "2392945842",
    "h_index": 6,
    "papers": 21
   },
   {
    "name": "Yinghao Xu",
    "id": "121983635",
    "h_index": 35,
    "papers": 73
   }
  ],
  "comment": "",
  "topics": [
   "foundation-pretraining",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.08639v2",
  "pdf_url": "https://arxiv.org/pdf/2607.08639v2",
  "html_url": "https://arxiv.org/html/2607.08639v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.9
 },
 {
  "id": "2607.08620",
  "slug": "a-new-human-likeness-and-comfort-index-for-robot-movements-along-presc",
  "title": "A New Human-Likeness and Comfort Index for Robot Movements Along Prescribed Paths",
  "abstract": "As human-robot interaction rapidly spreads in numerous fields, the subject of robot acceptance gains increasing importance. Visual similarity to the human body, as occurs for humanoids, is generally not enough to ensure acceptance in physical interaction, as acceptance directly links to comfort and ergonomics, which are measured in terms of the quality of the robot movement perceived by the human. This paper discusses the connection between comfort and similarity of the robot movement to the human one. By considering the kinematic characterization of human movement, this paper focuses on the time laws of such movements, wherein the end-effector path is prescribed. Based on the lognormality principle for modeling human movements, a human-likeness index is defined and used to provide an a priori characterization of trajectories. Such an index can be used to evaluate the performance of trajectory generation algorithms in producing human-like movements before they are actually executed. For validation purposes, 68 subjects are required to judge their comfort. The results of three experimental campaigns involving a physical interaction with a robot demonstrate a globally consistent trend between the preference in terms of perceived comfort and the distribution of the suggested human-likeness index.",
  "published": "2026-07-09",
  "updated": "2026-07-31",
  "year": "2026",
  "authors": [
   "Rosanna Coccaro",
   "Enrico Ferrentino",
   "Antonio Parziale",
   "Angelo Marcelli",
   "Pasquale Chiacchio"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "The results of three experimental campaigns involving a physical interaction with a robot demonstrate a globally consistent trend between the preference in terms of perceived comfort and the distribution of the suggested human-likeness index.",
  "doi": "10.1109/TCYB.2026.3707010",
  "oa_pdf": "https://doi.org/10.1109/tcyb.2026.3707010",
  "s2_authors": [
   {
    "name": "Rosanna Coccaro",
    "id": "2308667650",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Enrico Ferrentino",
    "id": "26974091",
    "h_index": 9,
    "papers": 35
   },
   {
    "name": "Antonio Parziale",
    "id": "144715993",
    "h_index": 13,
    "papers": 55
   },
   {
    "name": "Angelo Marcelli",
    "id": "2136508708",
    "h_index": 5,
    "papers": 30
   },
   {
    "name": "Pasquale Chiacchio",
    "id": "2261555049",
    "h_index": 3,
    "papers": 15
   }
  ],
  "comment": "13 pages, 5 figures. Accepted version, published at 10.1109/TCYB.2026.3707010, 2026 IEEE Transactions on Cybernetics",
  "topics": [
   "humanoids",
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.08620v3",
  "pdf_url": "https://arxiv.org/pdf/2607.08620v3",
  "html_url": "https://arxiv.org/html/2607.08620v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2607.08354",
  "slug": "skillplug-unsupervised-skill-mining-for-few-shot-adaptation-in-robotic",
  "title": "SkillPlug: Unsupervised Skill Mining for Few-Shot Adaptation in Robotic Manipulation",
  "abstract": "Learning transferable visuomotor imitation policies that generalize across diverse manipulation tasks and adapt rapidly to new tasks from only a handful of demonstrations remains challenging. Most modern policies are trained end-to-end to map observations directly to low-level actions, offering little explicit structure for reusing and recombining behaviors across tasks and making transfer data-inefficient under limited supervision. We propose SkillPlug, a plug-in framework that augments an existing visuomotor policy with a skill-conditioning module and mines a shared, transferable skill library from raw multi-task demonstrations. SkillPlug learns skills via self-supervised objectives that promote compact, reusable, and non-redundant behavior-level primitives, forming a task-shared prior for compositional control. After skill mining, we keep the learned skills fixed and specialize to unseen tasks by fine-tuning only lightweight router and action head, enabling efficient adaptation without full end-to-end retraining. We evaluate SkillPlug on two simulation benchmarks and on a real robot, and observe that the mined transferable skills consistently improve both multi-task performance and few-shot adaptation. Overall, SkillPlug offers a scalable way to mine reusable skills that improve data-efficient generalization in robotic manipulation.",
  "published": "2026-07-09",
  "updated": "2026-07-09",
  "year": "2026",
  "authors": [
   "Zi-han Ding",
   "Ziwei Wang"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "RA-L",
  "venue_source": "semantic-scholar",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "SkillPlug is proposed, a plug-in framework that augments an existing visuomotor policy with a skill-conditioning module and mines a shared, transferable skill library from raw multi-task demonstrations and offers a scalable way to mine reusable skills that improve data-efficient generalization in robotic manipulation.",
  "doi": "10.1109/LRA.2026.3703998",
  "oa_pdf": "https://arxiv.org/pdf/2607.08354",
  "s2_authors": [
   {
    "name": "Zi-han Ding",
    "id": "2410900441",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Ziwei Wang",
    "id": "2145033831",
    "h_index": 8,
    "papers": 11
   }
  ],
  "comment": "8 pages, 8 figures, published to RA-L",
  "topics": [
   "sim2real"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.08354v1",
  "pdf_url": "https://arxiv.org/pdf/2607.08354v1",
  "html_url": "https://arxiv.org/html/2607.08354v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.8
 },
 {
  "id": "2607.08341",
  "slug": "anydexrt-calibration-free-dexterous-hand-retargeting-with-few-shot-hum",
  "title": "AnyDexRT: Calibration-Free Dexterous Hand Retargeting with Few-Shot Human Guidance",
  "abstract": "Teleoperation is a key interface for controlling dexterous robotic hands and collecting demonstrations for imitation learning. Its effectiveness largely depends on kinematic retargeting, which maps operator hand motions to feasible and intuitive robot hand motions. Existing methods often require hand-crafted objectives, precise calibration, or global shape matching between human and robot hand spaces, making them sensitive to hand-specific tuning and less reliable across different dexterous hands. We propose AnyDexRT, a calibration-free retargeting method for intuitive dexterous teleoperation across human-like dexterous hands. AnyDexRT combines self-supervised fingertip correspondence learning with few-shot human guidance to anchor the mapping in task-relevant regions, and further refines pinch-related poses using a contact classifier. Experiments on diverse dexterous hands and real-world teleoperation tasks show that AnyDexRT improves retargeting quality, reduces manual tuning, and provides more intuitive and efficient control than prior retargeting methods. Project website: https://chenxi-wang.github.io/projects/anydexrt",
  "published": "2026-07-09",
  "updated": "2026-07-09",
  "year": "2026",
  "authors": [
   "Chenxi Wang",
   "Ying Feng",
   "Hongjie Fang",
   "Shangning Xia",
   "Lixin Yang",
   "Chuan Wen",
   "Cewu Lu"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "AnyDexRT is proposed, a calibration-free retargeting method for intuitive dexterous teleoperation across human-like dexterous hands that improves retargeting quality, reduces manual tuning, and provides more intuitive and efficient control than prior retargeting methods.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Chenxi Wang",
    "id": "2109436791",
    "h_index": 11,
    "papers": 20
   },
   {
    "name": "Ying Feng",
    "id": "2305733734",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Hongjie Fang",
    "id": "2152115958",
    "h_index": 16,
    "papers": 31
   },
   {
    "name": "Shangning Xia",
    "id": "2326972726",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Lixin Yang",
    "id": "2111809756",
    "h_index": 16,
    "papers": 41
   },
   {
    "name": "Chuan Wen",
    "id": "2381956964",
    "h_index": 5,
    "papers": 20
   },
   {
    "name": "Cewu Lu",
    "id": "2301174899",
    "h_index": 7,
    "papers": 29
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "imitation-diffusion",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.08341v1",
  "pdf_url": "https://arxiv.org/pdf/2607.08341v1",
  "html_url": "https://arxiv.org/html/2607.08341v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2607.08127",
  "slug": "understanding-and-mitigating-the-video-action-generalization-gap-via-t",
  "title": "Understanding and Mitigating the Video-Action Generalization Gap via Temporal Ratio",
  "abstract": "Generative video foundation models exhibit strong compositional priors, yet world-action models (WAMs) and video-action models (VAMs) often lose these priors after finetuning on robotic action data. We refer to this discrepancy as the video-action generalization gap. In this paper, we systematically investigate this gap by evaluating a comprehensive design space of VAMs, demonstrating that standard design choices yield no emergent explanation pattern. To explain this behavior, we introduce the Temporal Ratio (TR), an attention-based measure of how strongly the action head relies on future latent rollouts relative to the anchored current frame. TR has two key properties: first, a model's structural reliance on future-predictive latents, measured via TR, acts as a predictor of its compositional generalization capacity; second, it natively fluctuates based on task phase, shifting attention to future frames during planning and reverting to the present frame for precise manipulation. Finally, based on these findings, we propose an inference-time adaptive guidance method, which exploits this intrinsic feature attention pattern to dynamically amplify compositional video conditioning signals precisely when the policy relies on future rollouts. Evaluated on the LIBERO benchmark and real-world tasks, our approach mitigates the OOD-ID compositional generalization gap. More details: https://umishra.me/temporal-ratio/",
  "published": "2026-07-09",
  "updated": "2026-07-09",
  "year": "2026",
  "authors": [
   "Utkarsh A. Mishra",
   "Yongxin Chen",
   "Danfei Xu",
   "Yang Liu",
   "Xi Chen",
   "Jiayuan Mao"
  ],
  "author_count": 6,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "An inference-time adaptive guidance method is proposed, which exploits this intrinsic feature attention pattern to dynamically amplify compositional video conditioning signals precisely when the policy relies on future rollouts, and mitigates the OOD-ID compositional generalization gap.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "U. Mishra",
    "id": "2322505995",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Yongxin Chen",
    "id": "2243868887",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Danfei Xu",
    "id": "2322756258",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Yang Liu",
    "id": "2449159951",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Xi Chen",
    "id": "2384643239",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Jiayuan Mao",
    "id": "2323437497",
    "h_index": 3,
    "papers": 8
   }
  ],
  "comment": "26 pages, 9 figures",
  "topics": [
   "foundation-pretraining",
   "data-teleop",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.08127v1",
  "pdf_url": "https://arxiv.org/pdf/2607.08127v1",
  "html_url": "https://arxiv.org/html/2607.08127v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2607.08072",
  "slug": "post-training-in-end-to-end-autonomous-driving",
  "title": "Post-Training in End-to-End Autonomous Driving",
  "abstract": "End-to-end models that map multimodal inputs directly to future trajectories/maneuvers have emerged as an increasingly prominent research paradigm in autonomous driving. This class of models includes both Vision-Language-Action models and trajectory-generative planners. Unlike classic machine learning applications, autonomous vehicles operate in safety-critical and interaction-intensive environments where traditional open-loop imitation of expert demonstrations is not sufficient to ensure reliability. In particular, small execution errors can accumulate over time, while recovery behaviors are scarce in training data. In addition, long-horizon objectives such as safety and driving comfort are not captured by pointwise labels either. These limitations have motivated a shift toward post-training techniques, which further refine driving policies beyond pure imitation. This survey presents a unified view of post-training for autonomous driving by defining its scope and organizing the existing literature into four major families based on the form of supervision they use. For each family, we discuss its capabilities, limitations, and open challenges. We aim to facilitate a systematic understanding of this emerging area and stimulate future research on reliable and efficient post-training for autonomous driving.A collection of related papers is available at https://github.com/RYNing/Awesome-Post-Training-In-Autonomous-Driving-Papers.",
  "published": "2026-07-09",
  "updated": "2026-07-13",
  "year": "2026",
  "authors": [
   "Ruining Yang",
   "Muxing Wang",
   "Yixiao Chen",
   "Tongfei Guo",
   "Yi Xu",
   "Can Cui",
   "Zichong Yang",
   "Yitian Zhang",
   "Ziran Wang",
   "Yun Fu",
   "Lili Su"
  ],
  "author_count": 11,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "A unified view of post-training for autonomous driving is presented by defining its scope and organizing the existing literature into four major families based on the form of supervision they use, which aim to facilitate a systematic understanding of this emerging area and stimulate future research on reliable and efficient post-training for autonomous driving.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ruining Yang",
    "id": "2323040183",
    "h_index": 5,
    "papers": 15
   },
   {
    "name": "Muxing Wang",
    "id": "2448670157",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yixiao Chen",
    "id": "2381401226",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Tongfei Guo",
    "id": "2323121278",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Yi Xu",
    "id": "2257074262",
    "h_index": 7,
    "papers": 20
   },
   {
    "name": "Can Cui",
    "id": "2242949023",
    "h_index": 12,
    "papers": 32
   },
   {
    "name": "Zichong Yang",
    "id": "2267506629",
    "h_index": 7,
    "papers": 17
   },
   {
    "name": "Yitian Zhang",
    "id": "2291322254",
    "h_index": 7,
    "papers": 37
   },
   {
    "name": "Ziran Wang",
    "id": "4141749",
    "h_index": 34,
    "papers": 159
   },
   {
    "name": "Yun Fu",
    "id": "2257139684",
    "h_index": 6,
    "papers": 19
   },
   {
    "name": "Lili Su",
    "id": "2323437390",
    "h_index": 2,
    "papers": 6
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.08072v2",
  "pdf_url": "https://arxiv.org/pdf/2607.08072v2",
  "html_url": "https://arxiv.org/html/2607.08072v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2607.08436",
  "slug": "egowam-world-action-models-beyond-pixels-with-in-the-wild-egocentric-h",
  "title": "EgoWAM: World Action Models Beyond Pixels with In-the-Wild Egocentric Human Data",
  "abstract": "Egocentric human data offers scalable supervision for robot manipulation. However, behavior cloning entangles transferable content like objects, scenes, and task semantics, with non-transferable factors like human morphology, head motion, and behavioral style. We study whether World Action Models (WAMs) provide a better training signal by requiring policies to predict not only actions, but also how the scene evolves. The central question is what world representation best enables human-to-robot transfer. We hypothesize that an effective world target should abstract appearance, capture agent-invariant physical effects, and separate camera motion from environment change. We introduce EgoWAM, a controlled human-robot co-training framework that fixes the policy backbone, action head, and data mixture while varying only the world prediction target, comparing Pixel, DINO, and 3D motion flow. Across three real-world bimanual tasks, WAM co-training scales more effectively with in-the-wild egocentric human data than behavior cloning. Pixel-based prediction transfers weakly, while DINO and 3D flow yield substantial gains: DINO improves out-of-distribution object and scene generalization by up to 4x, and 3D flow improves in-domain performance by 20-30%. More details: https://gatech-rl2.github.io/egowam.github.io",
  "published": "2026-07-08",
  "updated": "2026-07-08",
  "year": "2026",
  "authors": [
   "Baoyu Li",
   "Xinchen Yin",
   "Mengying Lin",
   "Yixin Zhang",
   "Danfei Xu"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "EgoWAM is introduced, a controlled human-robot co-training framework that fixes the policy backbone, action head, and data mixture while varying only the world prediction target, comparing Pixel, DINO, and 3D motion flow.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Baoyu Li",
    "id": "2294776955",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Xi Yin",
    "id": "2315784082",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Mengying Lin",
    "id": "2292061484",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Yixin Zhang",
    "id": "2449184241",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Danfei Xu",
    "id": "2264393671",
    "h_index": 4,
    "papers": 15
   }
  ],
  "comment": "",
  "topics": [
   "egocentric-data",
   "imitation-diffusion",
   "hardware-codesign",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.08436v1",
  "pdf_url": "https://arxiv.org/pdf/2607.08436v1",
  "html_url": "https://arxiv.org/html/2607.08436v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2607.07972",
  "slug": "in-vivo-feasibility-study-of-humanoid-robots-in-surgery",
  "title": "In vivo feasibility study of humanoid robots in surgery",
  "abstract": "Recent advances in actuation, control and learning have rapidly pushed humanoid robots from a distant vision towards near-term real-world deployment. Healthcare is a particularly pressing domain, in which staffing shortages and increasing care demand are widening the gap between clinical workload and available skilled labour. Although current automation has largely focused on digital and logistical tasks, much hospital work remains embodied, requiring mobility, manipulation and safe interaction in human-designed environments. Humanoid form factors offer unique potential, particularly for assisting with surgical tasks. Traditionally, robotic systems for surgery are purpose-built platforms such as Intuitive Surgical's da Vinci Surgical System, and it remains unclear how close current humanoid systems are to meeting the precision, control and safety requirements of minimally invasive surgery. Here we present a systematic evaluation of contemporary humanoid technology for laparoscopic surgical tasks. We develop a humanoid-based laparoscopic teleoperation framework using general-purpose instruments and assess its abilities through benchtop characterization, dry-laboratory user studies spanning diverse surgical experience levels and in vivo porcine studies. Across these evaluations, we quantify technical feasibility, task performance and clinical readiness relative to established surgical platforms. Together, our study provides an evidence-based assessment of current humanoid abilities and limitations for surgical applications, highlighting both their promise and key technical challenges that must be addressed before clinical deployment.",
  "published": "2026-07-08",
  "updated": "2026-07-08",
  "year": "2026",
  "authors": [
   "Zekai Liang",
   "Nikita Thareja",
   "Peihan Zhang",
   "Calvin Joyce",
   "Soofiyan Atar",
   "Florian Richter",
   "Garth Jacobsen",
   "Shanglei Liu",
   "Ryan Broderick",
   "Michael Yip"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "Nature",
  "venue_source": "semantic-scholar",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "This study develops a humanoid-based laparoscopic teleoperation framework using general-purpose instruments and assess its abilities through benchtop characterization, dry-laboratory user studies spanning diverse surgical experience levels and in vivo porcine studies, highlighting both their promise and key technical challenges that must be addressed before clinical deployment.",
  "doi": "10.1038/s41586-026-10796-x",
  "oa_pdf": "https://arxiv.org/pdf/2607.07972",
  "s2_authors": [
   {
    "name": "Zekai Liang",
    "id": "2321422084",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "N. Thareja",
    "id": "93485059",
    "h_index": 3,
    "papers": 14
   },
   {
    "name": "Peihan Zhang",
    "id": "2351069369",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Calvin Joyce",
    "id": "2202692760",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Simranjeet Singh",
    "id": "2154835245",
    "h_index": 3,
    "papers": 17
   },
   {
    "name": "Florian Richter",
    "id": "2322445199",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "G. Jacobsen",
    "id": "34880884",
    "h_index": 35,
    "papers": 165
   },
   {
    "name": "Shanglei Liu",
    "id": "5508382",
    "h_index": 9,
    "papers": 42
   },
   {
    "name": "R. Broderick",
    "id": "5374261",
    "h_index": 20,
    "papers": 99
   },
   {
    "name": "Michael C. Yip",
    "id": "2322445999",
    "h_index": 1,
    "papers": 5
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.07972v1",
  "pdf_url": "https://arxiv.org/pdf/2607.07972v1",
  "html_url": "https://arxiv.org/html/2607.07972v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.8
 },
 {
  "id": "2607.07968",
  "slug": "soft-robotic-exogloves-for-dexterous-mobility-towards-personalized-reh",
  "title": "Soft Robotic Exogloves for Dexterous Mobility -- Towards Personalized Rehabilitation",
  "abstract": "Soft robotic exogloves can provide hand rehabilitation and assistance. Fitting these gloves often relies on standardized measurements not tailored to the individual, limiting their effectiveness, especially for fine articulation necessary for dexterous manipulation. We present the design, fabrication, modeling, and testing of a personalized pneumatically-actuated soft robotic exoglove. The glove was fit to a user's hand with topological scans and fabricated with silicone mold casting. Finite element analysis (FEA) was performed to evaluate actuator bending and forces from physical human-robot interaction (pHRI) between an actuator and a simplified personalized biomechanical finger model. Pneumatic pressure control experiments were conducted to flex the user's finger with static and dynamic references. Fabrication results show that topological scans enable precise tailoring to hand anatomy. Simulations showed that anatomical personalization enables analysis of pHRI contact forces, and results indicate sufficient joint mobilization with non-ideal compression on the proximal phalanx. Pneumatic testing indicates that pressure control allows accurate and targeted mobility of the metacarpophalangeal (MCP) and proximal interphalangeal (PIP) joints with intrinsic stiffness. Testing of multiple designs showed that relaxing the strain-limiting layer improves actuator-to-finger joint alignment during actuation. This work presents personalization to the human hand in structural conformability, joint topology, modeling of pHRI contact, and time-dependent actuation-deformation profiles. This lays a groundwork for informing exoglove design optimization to enable assistance in dexterous manipulation and neuromuscular rehabilitation of fine motor skills.",
  "published": "2026-07-08",
  "updated": "2026-07-08",
  "year": "2026",
  "authors": [
   "Paul Dela Cruz",
   "Mostafa Mo. Massoud",
   "Jacqueline Libby"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work presents personalization to the human hand in structural conformability, joint topology, modeling of pHRI contact, and time-dependent actuation-deformation profiles to lay a groundwork for informing exoglove design optimization to enable assistance in dexterous manipulation and neuromuscular rehabilitation of fine motor skills.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "P. D. Cruz",
    "id": "101557912",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "M. Massoud",
    "id": "2197601565",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Jacqueline Libby",
    "id": "2326023415",
    "h_index": 2,
    "papers": 5
   }
  ],
  "comment": "8 pages, 14 figures. To be published in The IEEE RAS/EMBS 11th International Conference on Biomedical Robotics and Biomechatronics (BioRob 2026)",
  "topics": [
   "dexterous-manipulation",
   "hardware-codesign",
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.07968v1",
  "pdf_url": "https://arxiv.org/pdf/2607.07968v1",
  "html_url": "https://arxiv.org/html/2607.07968v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.07958",
  "slug": "towards-soft-robotic-exogloves-for-musculoskeletal-manipulation-to-red",
  "title": "Towards Soft Robotic Exogloves for Musculoskeletal Manipulation to Reduce Pain and Spasticity",
  "abstract": "Hand spasticity and resulting pain affect 12 million people worldwide, including stroke survivors, arthritis patients, and those with other muscle and nerve deficiencies. Soft robotic exogloves are being introduced to help patients enhance mobility or manage pain; however, there are no current solutions that address both pain and mobility. We present preliminary development of a soft robotic exoglove that both aids in mobility and administers massage-like compression to relax spastic muscles. The glove consists of soft pneumatic actuators that are personalized to an individual's hand topology and kinematics, allowing for optimal conformability and targeted mobility. Novel soft actuators were designed, analyzed, fabricated, assembled into an exoglove, and experimentally tested. Actuators were 3D modeled and analyzed with finite element modeling under pressures of 100 and 200 kPa. Geometries were optimized to minimize stress before fabrication and testing. A dorsal finger actuator was successfully customized to a participant's hand topology, providing full conformal contact and maximal force distribution. A ventral finger actuator was successfully fabricated that can be drastically compressed in size to fit into the tight space of a hyperflexed spastic finger. A palmar actuator was successfully printed with stereolithography, showing potential for 3D-printed soft actuators with more complex geometries. The glove was assembled and successfully worn by a pilot user to validate initial findings in comfort and effectiveness.",
  "published": "2026-07-08",
  "updated": "2026-07-08",
  "year": "2026",
  "authors": [
   "Antonia Salluce",
   "Maeryn Erdheim",
   "Gailen Davis",
   "Lauren H. Sullivan",
   "Max-William Kanz",
   "Jacqueline Libby"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Preliminary development of a soft robotic exoglove that both aids in mobility and administers massage-like compression to relax spastic muscles is presented.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "A. Salluce",
    "id": "121002817",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Maeryn Erdheim",
    "id": "2448665697",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Gailen Davis",
    "id": "2448667557",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Lauren H. Sullivan",
    "id": "2448665679",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Max-William Kanz",
    "id": "2448665666",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Jacqueline Libby",
    "id": "2069358187",
    "h_index": 6,
    "papers": 12
   }
  ],
  "comment": "8 pages, 15 figures. To be published in the IEEE RAS/EMBS 11th International Conference on Biomedical Robotics and Biomechatronics (BioRob 2026)",
  "topics": [
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.07958v1",
  "pdf_url": "https://arxiv.org/pdf/2607.07958v1",
  "html_url": "https://arxiv.org/html/2607.07958v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.07885",
  "slug": "time-to-collision-based-dynamic-obstacle-avoidance-using-pretrained-vi",
  "title": "Time-to-Collision Based Dynamic Obstacle Avoidance Using Pretrained Vision Models for Robots in Unstructured Environments",
  "abstract": "Dynamic obstacle avoidance in unstructured outdoor environments remains a critical challenge for autonomous mobile robots, particularly when large-scale robot-specific training data and simulation-based policies are impractical. We present a data-efficient, interpretable method for vision-based dynamic obstacle avoidance that operates entirely on real-world data, avoiding the sim-to-real transfer problem inherent in simulation-trained policies. Our approach leverages UniDepth, a large pretrained monocular depth estimation model, to produce dense depth maps from RGB video without requiring stereo cameras or LiDAR at inference time. Dynamic obstacle avoidance is achieved by extending the SuperPoint and SuperGlue feature correspondence pipeline to track keypoints across long frame sequences, projecting their 2D pixel-space positions into 3D using camera intrinsics and predicted depth, running bundle adjustment initialized from these 3D keypoints, and computing per-keypoint time-to-collision (TTC). A 2D motion primitive in the ground plane is then selected to move the robot away from the closest point of approach of the minimum-TTC keypoint. Evaluated on real-world data from the M3ED dataset, our pipeline achieves a precision of 0.49 and a recall of 0.38 in identifying frames with a ground truth TTC below 1 second, and correctly generates the evasive motion direction in 84\\% of true positive detections. Crucially, it detects at least one frame with TTC less than 1 second for 20 out of 22 unique physical obstacles present in our test sequences. Unlike end-to-end learned methods that demand thousands of hours of robot-specific training data, our approach eliminates model training entirely, requiring only 74 seconds of data for hyperparameter tuning. This demonstrates exceptional data efficiency while preserving interpretable and generalizable behavior across diverse obstacle types.",
  "published": "2026-07-08",
  "updated": "2026-07-08",
  "year": "2026",
  "authors": [
   "Erik Jagnandan",
   "Mulugeta Haile",
   "Gregory Barber",
   "Pratik Chaudhari"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "eess.IV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work presents a data-efficient, interpretable method for vision-based dynamic obstacle avoidance that operates entirely on real-world data, avoiding the sim-to-real transfer problem inherent in simulation-trained policies.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Erik Jagnandan",
    "id": "2448666051",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "M. Haile",
    "id": "39004674",
    "h_index": 19,
    "papers": 78
   },
   {
    "name": "Gregory Barber",
    "id": "153374577",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Pratik Chaudhari",
    "id": "2258718499",
    "h_index": 7,
    "papers": 21
   }
  ],
  "comment": "9 pages, 8 figures",
  "topics": [
   "sim2real",
   "spatial-3d",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.07885v1",
  "pdf_url": "https://arxiv.org/pdf/2607.07885v1",
  "html_url": "https://arxiv.org/html/2607.07885v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.07844",
  "slug": "shift-drift-a-zero-shot-benchmark-for-generalizable-and-robust-autonom",
  "title": "Shift & Drift: A Zero-Shot Benchmark for Generalizable and Robust Autonomous Driving Motion Planning",
  "abstract": "While closed-loop motion planners trained on large-scale, object-level datasets, e.g., nuPlan, demonstrate strong in-distribution (ID) performance, their generalization to novel urban topologies and recovery mechanisms following execution perturbations remain under-explored. To address this, we present Shift & Drift, a novel dual-track benchmark designed to rigorously stress-test motion planners across two critical axes of distribution shift: (1) The Semantic Shift Track leverages a novel conversion pipeline that transforms the aerial, DeepScenario Open 3D dataset into the nuPlan simulation framework. This enables zero-shot evaluation of planners trained on North American and Singaporean data against 1,182 scenarios spanning four German cities and the US city of San Francisco featuring dense pedestrian-cyclist interactions. (2) The State-Distribution Drift Track injects stochastic perturbations into the ego vehicle's dynamics to quantify robustness against compounding execution errors. Based on this, we systematically evaluate the failure modes of diverse planning paradigms under semantic and state-distribution shifts. While imitation learning methods achieve high scores in ID benchmarks, they exhibit significant failures under semantic shift, particularly in pedestrian-dense environments, and suffer from persistent drift when subjected to temporally correlated actuation noise. In contrast, the evaluated reinforcement-learning-based planner demonstrates more graceful degradation, maintaining higher safety and progress metrics across both tracks. Our findings reveal an empirical trade-off between imitation fidelity and closed-loop resilience, providing the community with a rigorous benchmark to evaluate progress toward reliable deployment.",
  "published": "2026-07-08",
  "updated": "2026-07-08",
  "year": "2026",
  "authors": [
   "Alessandro Canevaro",
   "Hang Yu",
   "Julian Schmidt",
   "Peizheng Li",
   "Silvan Lindner",
   "Wilhelm Stork",
   "Georg Martius",
   "Julian Jordan"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "shift&Drift, a novel dual-track benchmark designed to rigorously stress-test motion planners across two critical axes of distribution shift, reveals an empirical trade-off between imitation fidelity and closed-loop resilience, providing the community with a rigorous benchmark to evaluate progress toward reliable deployment.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Alessandro Canevaro",
    "id": "2329100136",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Hang Yu",
    "id": "2329317029",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Julian Schmidt",
    "id": "2329139274",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Peizheng Li",
    "id": "2220434051",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Silvan Lindner",
    "id": "151212410",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Wilhelm Stork",
    "id": "2351293744",
    "h_index": 1,
    "papers": 12
   },
   {
    "name": "Georg Martius",
    "id": "2269471620",
    "h_index": 6,
    "papers": 19
   },
   {
    "name": "Julian Jordan",
    "id": "2329099511",
    "h_index": 2,
    "papers": 6
   }
  ],
  "comment": "Accepted at 2026 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS 2026)",
  "topics": [
   "imitation-diffusion",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.07844v1",
  "pdf_url": "https://arxiv.org/pdf/2607.07844v1",
  "html_url": "https://arxiv.org/html/2607.07844v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2607.07830",
  "slug": "physics-guided-biomechanical-gait-adaptation-for-humanoid-locomotion-o",
  "title": "Physics-Guided Biomechanical Gait Adaptation for Humanoid Locomotion on Extreme Sloped Terrains",
  "abstract": "Model-free reinforcement learning has enabled impressive humanoid locomotion; however, control on steep slopes remains largely unexplored. Unlike flat or discrete terrains, sloped terrains impose a persistent gravitational bias that demands simultaneous stability and posture control. Consequently, under generic reward formulations, policies can converge to slow, conservative low-center-of-mass (CoM) crouched gaits. In this work, we propose a novel two-stage physics-guided framework, dubbed HumoSlope, dedicated to robust humanoid locomotion on diverse sloped terrains. Specifically, Stage I establishes a terrain-consistent balance prior by introducing a slope-adaptive Zero Moment Point (ZMP) regularizer evaluated directly on the local inclined support plane rather than a world-horizontal reference. To prevent the resulting policy from defaulting to a crouched posture, Stage II introduces the Biomechanical Slope Gait Adapter (BSGA). Utilizing extracted macroscopic terrain descriptors as privileged, training-only signals, BSGA dynamically gates soft reward priors to modulate CoM height and lower-limb coordination based on the estimated slope geometry -- encouraging hip-dominant uphill propulsion and knee-oriented downhill braking. Crucially, the deployed actor remains entirely proprioceptive, requiring no online exteroceptive sensing. Extensive Sim-to-Real experiments demonstrate that our framework effectively mitigates posture degeneration and enables blind, continuous traversal of outdoor grass slopes up to 62.7% ($32.1^\\circ$), validating a physics-guided approach to challenging slope terrain adaptation.",
  "published": "2026-07-08",
  "updated": "2026-07-08",
  "year": "2026",
  "authors": [
   "Xuanyu Chen",
   "Mohan Liu",
   "Dengchen Mei",
   "Zhihao Gu",
   "Haitian Zhang",
   "Kaimin Mao",
   "Haiyue Zhu",
   "Shijun Yan",
   "Lin Wang"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xuanyu Chen",
    "id": "2322988535",
    "h_index": 0,
    "papers": 6
   },
   {
    "name": "Mohan Liu",
    "id": "2449133858",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Dengchen Mei",
    "id": "2448665920",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Zhihao Gu",
    "id": "2260941337",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Haitian Zhang",
    "id": "2448901630",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Kaimin Mao",
    "id": "2339473033",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Haiyue Zhu",
    "id": "2316882538",
    "h_index": 3,
    "papers": 14
   },
   {
    "name": "Shijun Yan",
    "id": "2448901239",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Lin Wang",
    "id": "2379501038",
    "h_index": 0,
    "papers": 6
   }
  ],
  "comment": "12 pages,6 figures",
  "topics": [
   "humanoids",
   "sim2real",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.07830v1",
  "pdf_url": "https://arxiv.org/pdf/2607.07830v1",
  "html_url": "https://arxiv.org/html/2607.07830v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.07601",
  "slug": "carla-gs-decoupling-representation-reasoning-and-physics-simulation-fo",
  "title": "CARLA-GS: Decoupling Representation, Reasoning, and Physics Simulation for Autonomous Driving Corner-Case Synthesis",
  "abstract": "Safety evaluation for autonomous driving is dominated by rare, safety-critical interactions, motivating simulators that can deliberately synthesize corner cases with photorealistic observations. Corner-case generation is inherently a multi-source problem spanning visual representation, scene reasoning, and vehicle trajectory generation and control. Prior knowledge- and model-based approaches typically focus on scene or trajectory components in isolation, while diffusion-based methods attempt end-to-end generation but still struggle to ensure spatiotemporal consistency and physical realism. To unify these aspects within a single framework, we propose CARLA-GS, a modular corner-case synthesis pipeline that decouples visual representation, semantic reasoning, and physics-based execution while maintaining tight cross-module coupling. Starting from real driving data, we reconstruct an editable gaussian scene with additional geometry-consistent constraints. A multi-agent LLM then performs scene-level reasoning to identify risky interactions and generate intent-level waypoint trajectories, while the low-level motion control is delegated to CARLA, where a PID controller ensures kinematic and dynamic feasibility. The simulated vehicle states are finally re-projected into the gaussian scene for ego-centric rendering. This design enables high-level semantic reasoning, low-level physically executable motion, and photorealistic corner-case generation within a unified pipeline. Experiments on the Waymo Open Dataset show, both quantitatively and qualitatively, that our framework enables controllable corner-case generation and produces photorealistic, spatiotemporally consistent videos aligned with semantic intent and physically feasible motion.",
  "published": "2026-07-08",
  "updated": "2026-07-08",
  "year": "2026",
  "authors": [
   "Kaicong Huang",
   "Meng Ma",
   "Ruimin Ke"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "This work proposes CARLA-GS, a modular corner-case synthesis pipeline that decouples visual representation, semantic reasoning, and physics-based execution while maintaining tight cross-module coupling, and experiments show that this framework enables controllable corner-case generation and produces photorealistic, spatiotemporally consistent videos aligned with semantic intent and physically feasible motion.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Kaicong Huang",
    "id": "2343609841",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Meng Ma",
    "id": "2334360519",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Ruimin Ke",
    "id": "2306522809",
    "h_index": 4,
    "papers": 23
   }
  ],
  "comment": "",
  "topics": [
   "sim2real",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.07601v1",
  "pdf_url": "https://arxiv.org/pdf/2607.07601v1",
  "html_url": "https://arxiv.org/html/2607.07601v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2607.07491",
  "slug": "smooth-operator-a-real-time-sampling-based-algorithm-for-kinematic-han",
  "title": "Smooth Operator: A Real-Time Sampling-Based Algorithm for Kinematic Hand Retargeting",
  "abstract": "Advances in learning-based robotic manipulation, such as Vision-Language-Action (VLA) models and Video Action Models (VAMs), heavily rely on high-quality teleoperation data. Their capabilities are strictly upper-bounded by the quality of the underlying human demonstrations. Current gradient-based retargeting algorithms often converge to different local minima, resulting in jitter that affects data quality and teleoperation experience. To address this, we introduce the Sampling-Based Retargeter (SBR), a novel gradient-free retargeting method drawn from the rich literature of sampling-based control and explicitly designed for low-jitter, real-time kinematic retargeting. We evaluate SBR both in simulation and through a rigorous real-world user study involving 18 participants performing 3 complex manipulation tasks. Compared to gradient-based baselines, SBR achieved the highest overall task success rate (54.1%) while significantly reducing operator cognitive fatigue, recording the lowest NASA-TLX workload score (36.4 out of 100). Ultimately, we establish SBR as a highly effective, intuitive retargeter for dexterous manipulation, providing the community with a rigorous benchmarking methodology to guide future retargeting research.",
  "published": "2026-07-08",
  "updated": "2026-07-13",
  "year": "2026",
  "authors": [
   "Robert Jomar Malate",
   "Erik Bauer",
   "Norica Bacuieti",
   "Stefanos Charalambous",
   "Elvis Nava",
   "Robert K. Katzschmann",
   "Benedek Forrai"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The Sampling-Based Retargeter (SBR) is introduced, a novel gradient-free retargeting method drawn from the rich literature of sampling-based control and explicitly designed for low-jitter, real-time kinematic retargeting.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "R. J. Malate",
    "id": "2354260083",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Erik Bauer",
    "id": "2157863513",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Norica Bacuieti",
    "id": "2165124046",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "S. Charalambous",
    "id": "91880520",
    "h_index": 19,
    "papers": 160
   },
   {
    "name": "Elvis Nava",
    "id": "2129786387",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "Robert K. Katzschmann",
    "id": "50191333",
    "h_index": 26,
    "papers": 87
   },
   {
    "name": "Benedek Forrai",
    "id": "2163581775",
    "h_index": 5,
    "papers": 7
   }
  ],
  "comment": "Minor cosmetic updates to figures",
  "topics": [
   "vla",
   "dexterous-manipulation",
   "egocentric-data",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.07491v2",
  "pdf_url": "https://arxiv.org/pdf/2607.07491v2",
  "html_url": "https://arxiv.org/html/2607.07491v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.07452",
  "slug": "geogs-slam-geometry-only-gaussian-splatting-for-dense-monocular-slam",
  "title": "GeoGS-SLAM: Geometry-Only Gaussian Splatting for Dense Monocular SLAM",
  "abstract": "Dense visual SLAM is a fundamental problem in robotics. Recent advances in 3DGS have demonstrated its potential for dense SLAM. Existing 3DGS frameworks focus on both appearance and geometry modeling. However, scene geometry is typically more critical for SLAM than novel view synthesis because downstream robotic tasks, such as navigation and obstacle avoidance, rely primarily on accurate spatial geometry rather than photorealistic rendering. This observation raises a natural question: Is it feasible for 3DGS to perform 3D reconstruction without scene appearance modeling? Motivated by this, we propose Geometry-only Gaussian Splatting (GeoGS), which directly reconstructs scene geometry, and further present GeoGS-SLAM, a dense visual SLAM system built upon this representation. Specifically, GeoGS retains only spatial parameters to reduce the number of per-primitive parameters by over 80%. In contrast to existing 3DGS methods, GeoGS focuses solely on geometric reconstruction, which significantly reduces the number of Gaussian primitives, accelerates geometric convergence, and enhances robustness to illumination variations. In addition, we present an effective training framework that optimizes the Gaussian primitives via single-view and multi-view geometric and photometric supervision, and speeds up geometry convergence with a local-plane driven initialization that better aligns primitives with local structures. Furthermore, we introduce a map update strategy for loop closure that globally transforms the Gaussian map to align it with the corrected pose estimates, thereby preventing map tearing caused by inconsistent per-viewpoint pose corrections in existing methods. Extensive experiments on synthetic and real-world benchmarks demonstrate that our method outperforms SOTA methods in terms of online mapping efficiency and geometric reconstruction quality.",
  "published": "2026-07-08",
  "updated": "2026-07-08",
  "year": "2026",
  "authors": [
   "Lipu Zhou",
   "Yaoyun Kang",
   "Junxiang Pang",
   "Shengkai Sun",
   "Tingting Bao",
   "Kehan Wang"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Geometry-only Gaussian Splatting (GeoGS), which directly reconstructs scene geometry, and GeoGS-SLAM, a dense visual SLAM system built upon this representation, which outperforms SOTA methods in terms of online mapping efficiency and geometric reconstruction quality.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Lipu Zhou",
    "id": "2365852681",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Yaoyun Kang",
    "id": "2448634976",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Junxiang Pang",
    "id": "2448445231",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Shengkai Sun",
    "id": "2448884796",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Tingting Bao",
    "id": "2448444841",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Kehan Wang",
    "id": "2448636273",
    "h_index": 0,
    "papers": 1
   }
  ],
  "comment": "",
  "topics": [
   "spatial-3d",
   "navigation",
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.07452v1",
  "pdf_url": "https://arxiv.org/pdf/2607.07452v1",
  "html_url": "https://arxiv.org/html/2607.07452v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.07430",
  "slug": "immersive-social-interaction-with-vr-and-llm-assisted-humanoids",
  "title": "Immersive Social Interaction with VR and LLM-Assisted Humanoids",
  "abstract": "Humanoid robots can extend human presence to remote, constrained, or hazardous environments, but existing teleoperation interfaces often require physically demanding motion tracking or cognitively demanding low-level control. This paper presents an immersive teleoperation framework that integrates voice-controlled locomotion, VR-based manipulation, and bidirectional social interaction for whole-body humanoid control. Using Apple Vision Pro, the operator receives egocentric visual feedback, issues natural-language locomotion commands, and teleoperates the robot's arms and dexterous hands through wrist and finger tracking. An LLM-assisted voice-control module converts spoken instructions into high-level locomotion commands, while the manipulation module retargets human hand motions to the robot through inverse kinematics and PD control. The system also records multimodal data, including egocentric RGB observations, voice/text commands, joint states, hand motions, and eye-gaze signals, supporting future imitation learning and autonomy. We evaluate the framework on a Unitree H1 humanoid equipped with dexterous hands in manipulation and social interaction tasks. Results show that novice users can successfully operate the system after brief familiarization, achieving 80\\% success in object manipulation and 70\\% success in a social cube-passing task. These results demonstrate the potential of immersive, language-assisted teleoperation as an accessible interface for humanoid interaction, remote assistance, and multimodal data collection.",
  "published": "2026-07-08",
  "updated": "2026-07-08",
  "year": "2026",
  "authors": [
   "Niraj Pudasaini",
   "Geeta Chandra Raju Bethala",
   "Pranav Doma",
   "Anthony Tzes",
   "Yi Fang"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "Humanoids",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "An immersive teleoperation framework that integrates voice-controlled locomotion, VR-based manipulation, and bidirectional social interaction for whole-body humanoid control and demonstrates the potential of immersive, language-assisted teleoperation as an accessible interface for humanoid interaction, remote assistance, and multimodal data collection.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Niraj Pudasaini",
    "id": "2092477521",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Geeta Chandra Raju Bethala",
    "id": "2320308161",
    "h_index": 2,
    "papers": 12
   },
   {
    "name": "Pranav Doma",
    "id": "2333424708",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Anthony Tzes",
    "id": "2282297404",
    "h_index": 6,
    "papers": 58
   },
   {
    "name": "Yi Fang",
    "id": "2264345449",
    "h_index": 6,
    "papers": 36
   }
  ],
  "comment": "IEEE-RAS International Conference on Humanoid Robots - Workshop: Designing Interactive Humanoids",
  "topics": [
   "dexterous-manipulation",
   "humanoids",
   "egocentric-data",
   "imitation-diffusion",
   "data-teleop"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2607.07430v1",
  "pdf_url": "https://arxiv.org/pdf/2607.07430v1",
  "html_url": "https://arxiv.org/html/2607.07430v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.0
 },
 {
  "id": "2607.07420",
  "slug": "initiation-safety-a-missing-dimension-in-generalist-robot-safety",
  "title": "Initiation Safety: A Missing Dimension in Generalist-Robot Safety",
  "abstract": "Safety for generalist robots is usually discussed in terms of motion or dialogue. We argue a third question is missing: should the robot take its first hard-to-undo social action at all, such as a greeting, an uninvited grasp, or stepping into someone's space? We call this initiation authorization. Current frameworks rarely treat it as a separate safety layer. Today's stacks often skip this step: a high engagement score or a confident VLA rollout is treated as permission to act. But seeing a person is not the same as having their consent to be addressed. We frame initiation authorization within generalist-robot safety and contrast it with post-plan VLA guardrails, implementing PAS (probe-authorize-speak) on a doorway humanoid, comparing it with direct-init on logged traces, and proposing a three-condition user study, with open questions on metrics, governance, and where initiation ends and foundation-model generation begins.",
  "published": "2026-07-08",
  "updated": "2026-07-15",
  "year": "2026",
  "authors": [
   "Zhijin Meng",
   "Francisco Cruz"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO",
   "cs.HC"
  ],
  "primary_category": "cs.RO",
  "venue": "RSS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work frames initiation authorization within generalist-robot safety and contrast it with post-plan VLA guardrails, implementing PAS (probe-authorize-speak) on a doorway humanoid, comparing it with direct-init on logged traces, and proposing a three-condition user study.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhijin Meng",
    "id": "2322470714",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Francisco Cruz",
    "id": "2350618707",
    "h_index": 2,
    "papers": 6
   }
  ],
  "comment": "4 pages, 2 figures. Accepted to RSS 2026 Workshop on Rethinking Safety for Generalist Robots",
  "topics": [
   "vla",
   "dexterous-manipulation",
   "humanoids",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.07420v2",
  "pdf_url": "https://arxiv.org/pdf/2607.07420v2",
  "html_url": "https://arxiv.org/html/2607.07420v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2607.07390",
  "slug": "communicative-efficiency-of-single-vs-multi-axis-robot-neck-motion",
  "title": "Communicative Efficiency of Single vs. Multi-Axis Robot Neck Motion",
  "abstract": "Nonverbal communication through head and neck movement is fundamental to human social signalling, yet how robotic neck morphology translates motion into communicative information remains poorly understood. We present an information-theoretic framework characterising robot neck movement as a communication channel, quantifying information transmitted and energy expended across varied configurations. Using a robotic neck platform, we recorded 84 video stimuli spanning three rotational degrees of freedom (DoF), varying amplitude, acceleration, and frequency, measuring Shannon entropy of pixel-change signals alongside energy consumption. A perceptual study validated communicative interpretations of each motion. While humans typically engage one axis per gesture, robots are unconstrained by biological architecture, motivating tests up to 3 DoF. Yet communicative information peaks at two DoF and decreases at three despite rising energy cost, a phenomenon we term the morphological information bottleneck. Motion parameter effects were parameter-dependent, some additive, others non-linear. We introduce the Motor Information Space, a framework mapping entropy against energy to expose communicative efficiency across morphologies, in which the optimal configuration achieves 5.26 bits at competitive energy cost. Perception data further confirm multi-axis movements reduce clarity. These findings challenge the assumption that anatomical completeness improves robotic expressiveness, establishing a quantitative basis for morphological design in robots, especially humanoids.",
  "published": "2026-07-08",
  "updated": "2026-07-08",
  "year": "2026",
  "authors": [
   "Chapa Sirithunge",
   "Haewon Jeong",
   "Qinghua Guan",
   "Fumiya Iida",
   "Josie Hughes"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.IT"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The Motor Information Space is introduced, a framework mapping entropy against energy to expose communicative efficiency across morphologies, in which the optimal configuration achieves 5.26 bits at competitive energy cost.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Chapa Sirithunge",
    "id": "52031585",
    "h_index": 6,
    "papers": 23
   },
   {
    "name": "Haewon Jeong",
    "id": "2290932095",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Qinghua Guan",
    "id": "2259706840",
    "h_index": 7,
    "papers": 18
   },
   {
    "name": "Fumiya Iida",
    "id": "2335095469",
    "h_index": 2,
    "papers": 21
   },
   {
    "name": "Josie Hughes",
    "id": "40662586",
    "h_index": 25,
    "papers": 190
   }
  ],
  "comment": "Under review",
  "topics": [
   "humanoids",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.07390v1",
  "pdf_url": "https://arxiv.org/pdf/2607.07390v1",
  "html_url": "https://arxiv.org/html/2607.07390v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.07370",
  "slug": "behavior-foundations-for-quadruped-robots-abot-c0-technical-report",
  "title": "Behavior Foundations for Quadruped Robots: ABot-C0 Technical Report",
  "abstract": "The motion controller is one of the most fundamental modules in embodied intelligence systems. Driven by large-scale human motion-capture data and the motion-tracking paradigm, humanoid control has achieved remarkable progress in recent years. However, migrating this recipe to the quadrupedal setting is far less straightforward: animal motion data is scarcer and harder to capture at scale than human data, and cross-embodiment retargeting remains fragile. We present ABot-C0, a generalist motion-control system for quadruped robots that establishes three complementary behavior foundations: a scalable multi-source motion-data pipeline, robust policy learning across motion tracking, locomotion, and scene interaction, and a unified deployment stack for reliable real-world operation. Fundamentally, we construct a data pyramid through conditional video-generation synthesis, annotated motion capture, teleoperation, and human design, producing 16,074 physically feasible motion clips as the data foundation for diverse motion-learning demands. With large-scale motion data, a Flow-Matching generalist policy demonstrates, for the first time, a scaling law for quadruped motion tracking: performance improves consistently as training scales up, with zero-shot capability to track unseen motions. We then go a step further toward robust all-terrain locomotion by adopting a three-stage privileged-to-perceptive framework with temporal LiDAR memory and terrain-predictive supervision. Collectively, these components form a motion generalist that coordinates multi-policy execution, smooth behavior transitions, energy-efficient control, and safety mechanisms for real-world deployment. Extensive experiments on urban-terrain autonomous navigation and companion-style multimodal interaction demonstrate that quadruped robots can move beyond functional demos toward product-level behavioral intelligence.",
  "published": "2026-07-08",
  "updated": "2026-07-09",
  "year": "2026",
  "authors": [
   "Xufeng Zhao",
   "Fuzhi Yang",
   "Jianhui Chen",
   "Li Gao",
   "Zhang Meng",
   "Jie Gao",
   "Yao Zheng",
   "Congyang Zhao",
   "Tianxiong Lv",
   "Menglin Yang",
   "Minqi Gu",
   "Yaru Zhao",
   "Wenyu Liu",
   "Honglin Han",
   "Shihui Su",
   "Zixiao Tang",
   "Liu Liu",
   "Mu Xu",
   "Yang Cai",
   "Wenbin Tang"
  ],
  "author_count": 20,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.HC",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "ABot-C0 is presented, a generalist motion-control system for quadruped robots that establishes three complementary behavior foundations: a scalable multi-source motion-data pipeline, robust policy learning across motion tracking, locomotion, and scene interaction, and a unified deployment stack for reliable real-world operation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xufeng Zhao",
    "id": "2145744284",
    "h_index": 11,
    "papers": 20
   },
   {
    "name": "Fuzhi Yang",
    "id": "2269697047",
    "h_index": 2,
    "papers": 15
   },
   {
    "name": "Jianhui Chen",
    "id": "2108513900",
    "h_index": 18,
    "papers": 32
   },
   {
    "name": "Li Gao",
    "id": "2373716829",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Zhang Meng",
    "id": "2448449058",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Jie Gao",
    "id": "2448586732",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Yaozhi Zheng",
    "id": "2367565779",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Congyang Zhao",
    "id": "2446294396",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "T. Lv",
    "id": "122471663",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Meng Yang",
    "id": "2446380135",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Minqi Gu",
    "id": "2448444135",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Yaru Zhao",
    "id": "2265141969",
    "h_index": 2,
    "papers": 19
   },
   {
    "name": "Wenyu Liu",
    "id": "2257432695",
    "h_index": 15,
    "papers": 33
   },
   {
    "name": "Honglin Han",
    "id": "2313939339",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Shihu Su",
    "id": "2437718563",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Zixiao Tang",
    "id": "2448619712",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Liu Liu",
    "id": "2261888536",
    "h_index": 2,
    "papers": 11
   },
   {
    "name": "Mu Xu",
    "id": "2382940270",
    "h_index": 6,
    "papers": 22
   },
   {
    "name": "Yang Cai",
    "id": "2383107401",
    "h_index": 5,
    "papers": 16
   },
   {
    "name": "Wenbin Tang",
    "id": "2387892194",
    "h_index": 3,
    "papers": 7
   }
  ],
  "comment": "Abot-C0 project page will be released soon",
  "topics": [
   "humanoids",
   "egocentric-data",
   "navigation",
   "foundation-pretraining",
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.07370v2",
  "pdf_url": "https://arxiv.org/pdf/2607.07370v2",
  "html_url": "https://arxiv.org/html/2607.07370v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.07287",
  "slug": "touchworld-a-predictive-and-reactive-tactile-foundation-model-for-dext",
  "title": "TouchWorld: A Predictive and Reactive Tactile Foundation Model for Dexterous Manipulation",
  "abstract": "Dexterous manipulation in everyday environments requires both anticipation and reaction: a robot must predict how contact should evolve while rapidly correcting local errors caused by slip, misalignment, unstable grasping, or force mismatch. Vision and language provide semantic and geometric guidance, but they cannot reliably reveal hidden contact states such as force, slip, and contact stability. Although tactile sensing exposes these physical cues, most existing policies treat touch as a low-frequency observation stream within a monolithic action model, coupling slow task reasoning, action generation, and fast contact feedback in a single loop. We introduce TouchWorld, a predictive-and-reactive tactile foundation model for dexterous manipulation. TouchWorld uses a hierarchical policy that separates vision-language subtask planning, tactile world-model prediction, visuo-tactile goal-conditioned action generation, and high-frequency tactile residual refinement. A High-Level Planning Layer produces executable subtasks and predicts tactile subgoals; a Visuo-Tactile Goal-Conditioned Policy generates nominal action chunks; and a Tactile-Conditioned Refinement Policy performs online residual correction using recent tactile and proprioceptive feedback. By using touch as both a predictive contact reference and a fast feedback signal, TouchWorld preserves the semantic generalization of vision-language-action policies while improving local contact adaptation. Across six long-horizon and contact-rich dexterous manipulation tasks, TouchWorld achieves 65.0% success in the clean setting and 53.7% success under human perturbations, outperforming the strongest baseline by 15.7 and 18.5 percentage points, respectively.",
  "published": "2026-07-08",
  "updated": "2026-07-09",
  "year": "2026",
  "authors": [
   "Jianyi Zhou",
   "Feiyang Hong",
   "Yunhao Li",
   "Yicheng Zhao",
   "Yongjue Cen",
   "Zirui Liu",
   "Jiakang Huang",
   "Zirui Chen",
   "Ruiyang Zhang",
   "Weizhuo Zhu",
   "Xuhua Song",
   "Shuo Yang"
  ],
  "author_count": 12,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 2,
  "influential_citations": 0,
  "tldr": "By using touch as both a predictive contact reference and a fast feedback signal, TouchWorld preserves the semantic generalization of vision-language-action policies while improving local contact adaptation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jianyi Zhou",
    "id": "2408442564",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Feiyang Hong",
    "id": "2448443717",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yunhao Li",
    "id": "2448657541",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yicheng Zhao",
    "id": "2448639714",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yongjue Cen",
    "id": "2448443687",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Zirui Liu",
    "id": "2326322742",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Jiakang Huang",
    "id": "2358186554",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Zirui Chen",
    "id": "2448913937",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Ruiyang Zhang",
    "id": "2447343670",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Weizhuo Zhu",
    "id": "2448581706",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Xuhua Song",
    "id": "2448650432",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Shuo Yang",
    "id": "2344197064",
    "h_index": 3,
    "papers": 9
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "vla",
   "dexterous-manipulation",
   "tactile",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.07287v2",
  "pdf_url": "https://arxiv.org/pdf/2607.07287v2",
  "html_url": "https://arxiv.org/html/2607.07287v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.48
 },
 {
  "id": "2607.07196",
  "slug": "validate-the-dream-before-you-trust-its-verdict-admissibility-for-worl",
  "title": "Validate the Dream Before You Trust Its Verdict: Admissibility for World-Model Simulators",
  "abstract": "Across robotics, World Models (WMs) are increasingly used to evaluate action policies by simulating the consequences of actions in an imagined world, and returning a success or safety verdict. Yet a verdict is only as trustworthy as the WM that produced it, and the WM itself needs to be certified. In video-generation WMs, fidelity metrics such as Fr\u00e9chet Video Distance (FVD) reward visual realism, but ignore whether the world responds correctly to the policy's actions, including those unseen in training. Classical simulation-based validation assumes a trusted simulator evaluating an untrusted policy, whereas generative WMs are themselves unverified learned artifacts. Hence, we argue that any WM used as a test oracle must first be accredited before its verdicts can serve as evidence. Building on credibility practices from safety-critical simulation, including Verification, Validation & Accreditation (VV&A), Safety of the Intended Functionality (SOTIF), and scenario-based testing standards, we define an admissibility ladder (L0-L4) that a WM must climb before its closed-loop verdicts are accepted as assurance evidence. Our framework is embodiment-agnostic, and is instantiated in autonomous driving (AD), where assurance methods for traditional simulation are most mature. Applied to two driving WMs, the lower rungs reveal a reversal: the model that ranks higher on visual generation quality (L0) ranks lower on action-following (L1-L2), so visual fidelity does not predict the action-robustness a closed-loop verdict depends on.",
  "published": "2026-07-08",
  "updated": "2026-07-08",
  "year": "2026",
  "authors": [
   "Christian Oefinger",
   "Finn Rasmus Sch\u00e4fer",
   "Korbinian Moller",
   "Mattia Piccinini",
   "Johannes Betz"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG",
   "cs.SE"
  ],
  "primary_category": "cs.RO",
  "venue": "RSS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work argues that any WM used as a test oracle must first be accredited before its verdicts can serve as evidence, and defines an admissibility ladder (L0-L4) that a WM must climb before its closed-loop verdicts are accepted as assurance evidence.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Christian Oefinger",
    "id": "2425366203",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "F. Schafer",
    "id": "2423034922",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Korbinian Moller",
    "id": "2268252626",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Mattia Piccinini",
    "id": "2049096210",
    "h_index": 11,
    "papers": 47
   },
   {
    "name": "Johannes Betz",
    "id": "2300176530",
    "h_index": 4,
    "papers": 16
   }
  ],
  "comment": "Accepted at RSS 2026 Workshop on Robot World Models",
  "topics": [
   "world-models",
   "sim2real",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.07196v1",
  "pdf_url": "https://arxiv.org/pdf/2607.07196v1",
  "html_url": "https://arxiv.org/html/2607.07196v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2607.07139",
  "slug": "disturbance-aware-motion-planning-for-over-actuated-underwater-vehicle",
  "title": "Disturbance-aware Motion Planning for Over-actuated Underwater Vehicles Exploiting Actuation Redundancy for High-fidelity 3D Reconstruction",
  "abstract": "Underwater robots often operate near delicate targets where high-power thrusters resuspend sediments and induce turbulence, degrading image quality at the sensor input. Conventional controllers optimize vehicle-centric objectives, such as tracking and stability, without accounting for the impact of actuation on sensing. We address this actuation-to-perception coupling by exploiting redundancy in over-actuated platforms. For an eight-thruster ROV, multiple thrust allocations can yield the same motion; we search this null space to minimize predicted disturbance in a task-relevant target region while enforcing motion constraints. Our method uses a control-oriented thruster-wake proxy derived from actuator-disk theory with directional attenuation and validated by PIV ($R^2 = 0.99$ near the wake axis; $R^2 > 0.82$ in the primary wake region), together with a real-time redundancy-resolving allocator running at 10 Hz (45 ms/solve). Across 440 trials, the approach reduces target-region particle velocity by 67% ($p < 0.001$), improves 3D reconstruction RMSE by 55% versus a disturbance-unaware baseline ($1.9 \\pm 0.4$ mm vs. $4.3 \\pm 1.8$ mm), and achieves a 98.5% reconstruction success rate. The framework supports autonomous scanning, which is quantitatively evaluated, and operator-assisted inspection, which is demonstrated in the supplementary materials.",
  "published": "2026-07-08",
  "updated": "2026-07-08",
  "year": "2026",
  "authors": [
   "Yuer Gao",
   "Tongqing Xu",
   "Qingyang Liu",
   "Yi Cai"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yuer Gao",
    "id": "2326549389",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Tongqing Xu",
    "id": "2326952155",
    "h_index": 0,
    "papers": 6
   },
   {
    "name": "Qingyang Liu",
    "id": "2349226228",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Yi Cai",
    "id": "2258335050",
    "h_index": 6,
    "papers": 19
   }
  ],
  "comment": "",
  "topics": [
   "spatial-3d",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.07139v1",
  "pdf_url": "https://arxiv.org/pdf/2607.07139v1",
  "html_url": "https://arxiv.org/html/2607.07139v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.07129",
  "slug": "compositional-motion-generation-from-demonstration-with-object-centric",
  "title": "Compositional Motion Generation from Demonstration with Object-Centric Neural Fields",
  "abstract": "Compositionality, by organizing complex behavior as combinations of simpler elements, enables robot learning that is scalable and data efficient. Leveraging this principle, we propose a generative learning-from-demonstration framework that enables compositional modeling of robotic behavior by connecting perception and motion through shared object-level representations. We render scenes from object-centric neural representations that integrate canonical neural fields with latent-conditioned deformations, capturing positional and geometric variations in a smooth, consistent, and interpretable way. For motion generation, a temporal mixture-of-experts (MoE) employs a gating mechanism to combine object-conditioned movement primitives over time, producing complete trajectories. This spatial-temporal compositionality maintains the data efficiency of movement primitives while grounding motion in visual structure, enabling systematic generalization across diverse scene configurations. In simulation, long-horizon manipulation tasks are successfully completed using the proposed model, which requires significantly less training data than other image-based baselines. Real-world experiments further demonstrate the method's robustness to noise, its ability to generalize at the category level through language-based segmentation models, and its capacity to operate directly on 3D scene representations.",
  "published": "2026-07-08",
  "updated": "2026-07-08",
  "year": "2026",
  "authors": [
   "Ahmet Ercan Tekden",
   "Yasemin Bekiroglu"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "RA-L",
  "venue_source": "semantic-scholar",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes a generative learning-from-demonstration framework that enables compositional modeling of robotic behavior by connecting perception and motion through shared object-level representations, and renders scenes from object-centric neural representations that integrate canonical neural fields with latent-conditioned deformations.",
  "doi": "10.1109/LRA.2026.3713713",
  "oa_pdf": "https://doi.org/10.1109/lra.2026.3713713",
  "s2_authors": [
   {
    "name": "A. Tekden",
    "id": "150925638",
    "h_index": 7,
    "papers": 18
   },
   {
    "name": "Yasemin Bekiroglu",
    "id": "1742732",
    "h_index": 19,
    "papers": 59
   }
  ],
  "comment": "Accepted by IEEE Robotics and Automation Letters (RAL)",
  "topics": [
   "spatial-3d",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.07129v1",
  "pdf_url": "https://arxiv.org/pdf/2607.07129v1",
  "html_url": "https://arxiv.org/html/2607.07129v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2607.07101",
  "slug": "geoprop-grounding-robot-state-in-vision-for-generalist-manipulation",
  "title": "GeoProp: Grounding Robot State in Vision for Generalist Manipulation",
  "abstract": "Proprioception is fundamental to robotic manipulation, yet standard fusion methods often treat it as an isolated vector lacking explicit alignment with visual tokens. Without a direct correspondence between 3D kinematics and 2D feature maps, manipulation policies struggle to ground the robot's state within the scene, frequently underperforming even vision-only baselines. To address this, we introduce GeoProp, a lightweight, plug-and-play adapter that aligns proprioception with vision through explicit geometric grounding and spatial feature sampling. GeoProp projects the robot state onto the image plane to sample localized visual features, constructing a grounded state token. It then injects state-derived spatial priors into the corresponding visual features via FiLM modulation. To capture motion intent, GeoProp further samples features at a short-horizon predicted coordinate derived from recent kinematics, providing look-ahead visual context. Across 67 tasks, GeoProp improves Diffusion Policy by 8.7% on 63 simulation tasks and pi_0 by 4.0% on the RoboTwin subset, and yields a 10.6% average gain across both policy families in the real world, while adding only 2-3% to the parameter count. These results demonstrate that GeoProp is a simple yet high-impact inductive bias for generalist embodied policies. Project page: https://alibaba-damo-academy.github.io/GeoProp/.",
  "published": "2026-07-08",
  "updated": "2026-07-08",
  "year": "2026",
  "authors": [
   "Guoyang Zhao",
   "Quanhao Qian",
   "Gongjie Zhang",
   "Wenhao Li",
   "Jiuniu Wang",
   "Xiaowei Lu",
   "Deli Zhao",
   "Ran Xu"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "GeoProp is a lightweight, plug-and-play adapter that aligns proprioception with vision through explicit geometric grounding and spatial feature sampling, and demonstrates that GeoProp is a simple yet high-impact inductive bias for generalist embodied policies.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Guoyang Zhao",
    "id": "2382081267",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Quanhao Qian",
    "id": "2372232181",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Gongjie Zhang",
    "id": "2372725300",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Wenhao Li",
    "id": "2350520914",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Jiuniu Wang",
    "id": "2372246114",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Xiao Lu",
    "id": "2447889009",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Deli Zhao",
    "id": "2373574966",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Ran Xu",
    "id": "2376110719",
    "h_index": 4,
    "papers": 12
   }
  ],
  "comment": "21 pages, 8 figures, 11 tables. Project page: https://alibaba-damo-academy.github.io/GeoProp/",
  "topics": [
   "imitation-diffusion"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.07101v1",
  "pdf_url": "https://arxiv.org/pdf/2607.07101v1",
  "html_url": "https://arxiv.org/html/2607.07101v1",
  "code_url": "https://alibaba-damo-academy.github.io/GeoProp/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2607.07076",
  "slug": "prigo-test-time-primitive-guidance-to-diffusion-and-flow-policies-for",
  "title": "PriGo: Test-Time Primitive Guidance to Diffusion and Flow Policies for Adaptive Robotic Manipulation",
  "abstract": "Imitation learning has enabled remarkable progress in robotic manipulation, especially with diffusion and flow-based policies that generate complex visuomotor behaviors directly from demonstrations. Yet, despite their strong performance, these policies often fail to generalize across tasks and environments. A key reason is that existing policies tend to imitate superficial action correlations rather than the underlying intent. Inspired by the compositional structure of human behaviors, we propose PriGo, a primitive-guided test-time adaptive framework for robust robotic manipulation. PriGo introduces PANet, a lightweight primitive prediction module that infers primitive distributions directly from observations. We further propose a differentiable primitive guidance mechanism that refines generated actions during inference, steering trajectories toward semantically consistent behaviors. Unlike prior primitive-conditioned approaches, PriGo operates entirely at test time and can be seamlessly integrated into pretrained diffusion and flow policies without retraining. Extensive experiments on LIBERO, CALVIN, SIMPLER, and real-world robotic tasks demonstrate that PriGo consistently improves robustness, long-horizon execution, and generalization ability across both diffusion and flow-based policies.",
  "published": "2026-07-08",
  "updated": "2026-07-08",
  "year": "2026",
  "authors": [
   "Zezeng Li",
   "Enda Xiang",
   "Thuy Tran",
   "Di Huang",
   "Momath Thiam",
   "Liming Chen"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Inspired by the compositional structure of human behaviors, PriGo is proposed, a primitive-guided test-time adaptive framework for robust robotic manipulation that consistently improves robustness, long-horizon execution, and generalization ability across both diffusion and flow-based policies.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zezeng Li",
    "id": "115419471",
    "h_index": 6,
    "papers": 18
   },
   {
    "name": "Enda Xiang",
    "id": "2376540146",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "T. Tran",
    "id": "2448443607",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Di Huang",
    "id": "2372456200",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Momath Thiam",
    "id": "121743824",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Liming Chen",
    "id": "2348307858",
    "h_index": 2,
    "papers": 4
   }
  ],
  "comment": "",
  "topics": [
   "imitation-diffusion",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.07076v1",
  "pdf_url": "https://arxiv.org/pdf/2607.07076v1",
  "html_url": "https://arxiv.org/html/2607.07076v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.06988",
  "slug": "wam-ttt-steering-world-action-models-by-watching-human-play-at-test-ti",
  "title": "WAM-TTT: Steering World-Action Models by Watching Human Play at Test Time",
  "abstract": "Steering robot foundation models (RFMs) toward new task variants or user-preferred behaviors remains challenging, often requiring additional robot demonstrations, task-specific fine-tuning, or long-context conditioning. We present WAM-TTT, a test-time training framework for steering world action models from raw human videos. Rather than treating human videos as trajectories to imitate, WAM-TTT absorbs them into a lightweight adaptive memory inside a frozen WAM through self-supervised video prediction. To make this memory useful for control, we introduce a meta-training stage that aligns human demonstrations with robot behaviors using paired human-robot data and a key--value memory reconstruction objective. At test time, only unlabeled human videos are required to adapt the memory, while the pretrained WAM remains frozen. This enables efficient and reusable steering without robot actions, human-side annotations, or task-specific fine-tuning, while preserving the generalization ability of the foundation model. Extensive experiments show that WAM-TTT consistently outperforms in-context human-video conditioning baselines across diverse manipulation tasks and generalization settings.",
  "published": "2026-07-08",
  "updated": "2026-07-10",
  "year": "2026",
  "authors": [
   "Yusen Feng",
   "Bingchen Han",
   "Jiangran Lyu",
   "Kai Liu",
   "Yixin Zheng",
   "Yuxuan Wan",
   "Weiheng Liu",
   "Sun Han",
   "Ruiqin Li",
   "Yulong Zhang",
   "Fangfu Liu",
   "Xuesong Shi",
   "Libin Liu",
   "Yizhou Wang",
   "Zhizheng Zhang",
   "He Wang"
  ],
  "author_count": 16,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "WAM-TTT, a test-time training framework for steering world action models from raw human videos that absorbs them into a lightweight adaptive memory inside a frozen WAM through self-supervised video prediction, and introduces a meta-training stage that aligns human demonstrations with robot behaviors using paired human-robot data and a key--value memory reconstruction objective.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yusen Feng",
    "id": "2410458406",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Bingchen Han",
    "id": "2449503035",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Jiangran Lyu",
    "id": "2217476979",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Kai Liu",
    "id": "2377763386",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Yixin Zheng",
    "id": "2402892933",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Yuxuan Wan",
    "id": "2255592808",
    "h_index": 4,
    "papers": 14
   },
   {
    "name": "Weiheng Liu",
    "id": "2347176603",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Sun Han",
    "id": "2449599748",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Ruiqing Li",
    "id": "2446859463",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yulong Zhang",
    "id": "2445890595",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Fangfu Liu",
    "id": "2273665760",
    "h_index": 12,
    "papers": 32
   },
   {
    "name": "Xuesong Shi",
    "id": "2347163648",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Libin Liu",
    "id": "2395527849",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Yizhou Wang",
    "id": "2256800070",
    "h_index": 8,
    "papers": 24
   },
   {
    "name": "Zhizheng Zhang",
    "id": "2287041015",
    "h_index": 10,
    "papers": 29
   },
   {
    "name": "He Wang",
    "id": "2330238483",
    "h_index": 4,
    "papers": 9
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "egocentric-data",
   "foundation-pretraining",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.06988v2",
  "pdf_url": "https://arxiv.org/pdf/2607.06988v2",
  "html_url": "https://arxiv.org/html/2607.06988v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2607.06978",
  "slug": "spectra-context-conditioned-spectral-movement-primitives-for-robot-ski",
  "title": "SPECTRA: Context-Conditioned Spectral Movement Primitives for Robot Skill Generalization",
  "abstract": "Robot imitation learning for manipulation should preserve demonstrated task geometry while producing dynamically admissible robot motions. Existing pipelines often learn task-dependent trajectories and impose execution limits afterward through filtering, smoothing, clipping, or time scaling, which may distort task-critical end-effector paths. We propose the Spectral Movement Primitive (SMP), a frequency-domain imitation learning framework that couples task-space skill generation with joint-space execution regulation. Demonstrations are represented by truncated finite-horizon Fourier coefficients. An empirically selected low-frequency task band captures the dominant motion geometry, while higher harmonics contribute disproportionately to derivative growth. A frame-aware context-conditioned GMM/GMR prior predicts the task-band coefficients in a canonical task frame, and the resulting Cartesian trajectory is mapped to joint space through sequential inverse kinematics. A phase-coupled regulator then limits the requested phase progression without modifying the spectral coefficients, thereby enforcing joint velocity and acceleration limits while preserving the represented path. Experiments evaluate task-band reconstruction, robustness to composite demonstration corruption, out-of-distribution cross-board generalization, joint-space dynamic admissibility, end-effector path preservation, and deployment on a Franka Panda robot. Results show compact geometric reconstruction, consistent transfer across unseen task frames, substantial reductions in dynamic violations and jerk, and preservation of the intended end-effector path during phase regulation.",
  "published": "2026-07-08",
  "updated": "2026-07-14",
  "year": "2026",
  "authors": [
   "Boxuan Zhang",
   "Sheng Liu",
   "Chenlin Ming",
   "Ahmed Abdelrahman"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The Spectral Movement Primitive (SMP), a frequency-domain imitation learning framework that couples task-space skill generation with joint-space execution regulation, is proposed and results show compact geometric reconstruction, consistent transfer across unseen task frames, substantial reductions in dynamic violations and jerk, and preservation of the intended end-effector path during phase regulation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Boxuan Zhang",
    "id": "2315788969",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Sheng Liu",
    "id": "2448369211",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Chenglin Ming",
    "id": "2448369052",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Ahmed Abdelrahman",
    "id": "1577617526",
    "h_index": 2,
    "papers": 5
   }
  ],
  "comment": "",
  "topics": [
   "imitation-diffusion",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.06978v2",
  "pdf_url": "https://arxiv.org/pdf/2607.06978v2",
  "html_url": "https://arxiv.org/html/2607.06978v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.06882",
  "slug": "gemnav-discrete-token-visual-robot-navigation-using-a-multimodal-large",
  "title": "GemNav: Discrete-Token Visual Robot Navigation using a Multimodal Large Language Model",
  "abstract": "Visual navigation policies built on large pretrained models have so far followed a common recipe: a dedicated visual encoder, a bespoke action head, and training on thousands of hours of cross-embodiment datasets. We ask whether this recipe is necessary. In this paper, we introduce GemNav, a visual robot navigation policy that adapts a frozen Multimodal Large Language Model (MLLM) for short-to-medium horizon waypoint navigation using Low-Rank Adaptation (LoRA) on the language tower alone, with no auxiliary visual encoder and no continuous regression head. Waypoints and categorical navigation signals share a single discrete token vocabulary generated by the language-model head, and a soft-decoded auxiliary loss recovers the metric structure that pure cross-entropy training discards. On a single 8.7-hour open corpus, roughly three orders of magnitude smaller than competing training sets, the policy transfers zero-shot to four physically distinct unseen environments and stops within 0.25-0.42m of the goal across 20 real-world trials covering an open carpark, an obstacle carpark, a long outdoor chemical yard, and an indoor warehouse. Conditioning on short image histories improves offline metrics but yields no robot benefit, pointing to a ceiling on what temporal context adds once pretrained vision features are in place. These results indicate that discrete-token adaptation of frozen MLLMs can provide a data-efficient, deployable alternative for foundation model robot navigation.",
  "published": "2026-07-08",
  "updated": "2026-07-26",
  "year": "2026",
  "authors": [
   "Peter Bohm",
   "Saimunur Rahman",
   "Abdelwahed Khamis",
   "Sagun Man Singh Shrestha",
   "Chris McCool",
   "Peyman Moghadam"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "GemNav is introduced, a visual robot navigation policy that adapts a frozen Multimodal Large Language Model (MLLM) for short-to-medium horizon waypoint navigation using Low-Rank Adaptation (LoRA) on the language tower alone, and results indicate that discrete-token adaptation of frozen MLLMs can provide a data-efficient, deployable alternative for foundation model robot navigation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Peter Bohm",
    "id": "2350343551",
    "h_index": 0,
    "papers": 5
   },
   {
    "name": "Saimunur Rahman",
    "id": "2322624351",
    "h_index": 0,
    "papers": 5
   },
   {
    "name": "Abdelwahed Khamis",
    "id": "2332363556",
    "h_index": 2,
    "papers": 12
   },
   {
    "name": "Sagun Shrestha",
    "id": "2237787617",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Chris McCool",
    "id": "2273992906",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "P. Moghadam",
    "id": "2242950949",
    "h_index": 7,
    "papers": 32
   }
  ],
  "comment": "",
  "topics": [
   "navigation",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.06882v2",
  "pdf_url": "https://arxiv.org/pdf/2607.06882v2",
  "html_url": "https://arxiv.org/html/2607.06882v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.09772",
  "slug": "a-risk-field-enhanced-closed-loop-digital-twin-framework-for-autonomou",
  "title": "A Risk-Field Enhanced Closed-Loop Digital Twin Framework for Autonomous Driving Safety Validation",
  "abstract": "Autonomous driving systems require reliable safety validation before real-world deployment. However, large-scale road testing is costly, difffcult to reproduce, and inefffcient for exposing rare safety-critical scenarios. Conventional simulation improves repeatability, but an offfine simulator alone cannot continuously connect physical trafffc states, virtual reconstruction, algorithm evaluation, and scenario evolution. This paper proposes a risk-ffeld enhanced closed-loop digital twin framework for autonomous driving safety validation. The framework integrates physical data acquisition, data synchronization, virtual twin reconstruction, risk-aware scenario generation, autonomous driving algorithm evaluation, and safety analysis. A driving risk ffeld is introduced as a uniffed intermediate representation to describe obstacle, lane-departure, road-boundary, time-to-collision, and comfort-related risks around the ego vehicle. The risk ffeld ranks high-risk scenarios in the digital twin scenario library and provides dense safety guidance for reinforcement learning-based driving policies. A simulation-style evaluation protocol is designed to compare conventional reinforcement learning baselines, risk-penalty baselines, and the proposed risk-ffeld guided method. The study indicates that embedding explicit risk structure into digital twins can make autonomous driving validation more targeted, interpretable, and reusable, while its practical effectiveness remains bounded by model ffdelity, risk calibration, and sim-to-real transfer.",
  "published": "2026-07-07",
  "updated": "2026-07-07",
  "year": "2026",
  "authors": [
   "Yongzhi Liu"
  ],
  "author_count": 1,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The study indicates that embedding explicit risk structure into digital twins can make autonomous driving validation more targeted, interpretable, and reusable, while its practical effectiveness remains bounded by model ffdelity, risk calibration, and sim-to-real transfer.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yongzhi Liu",
    "id": "2449181269",
    "h_index": 0,
    "papers": 3
   }
  ],
  "comment": "",
  "topics": [
   "sim2real",
   "rl-control",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.09772v1",
  "pdf_url": "https://arxiv.org/pdf/2607.09772v1",
  "html_url": "https://arxiv.org/html/2607.09772v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.09764",
  "slug": "omniscs-omni-safety-critical-scenario-synthesis-for-autonomous-driving",
  "title": "OmniSCS: Omni Safety-Critical Scenario Synthesis for Autonomous Driving via a Fully Editable Driving World",
  "abstract": "The synthesis of safety-critical scenarios (SCS) and their evaluation through closed-loop simulations are crucial for developing robust autonomous driving systems. A key aspect of this process involves editing agent states in both appearance and trajectory levels within existing scenes. However, current methods struggle to preserve data fidelity after scene editing and fail to efficiently generate high-quality SCS through such modifications. To overcome these limitations, we propose OmniSCS, an innovative system that generates photorealistic SCS with high physical fidelity while enabling closed-loop testing in synthetic environments. OmniSCS comprises two key modules: 1) A Fully Editable Driving World Construction module that maintains high-fidelity agent appearance and background during scene editing via dual-strategy agent reconstruction and depth-refinement background reconstruction methods. 2) A SCS Synthesis module that facilitates object insertion and agent trajectory editing to synthesize diverse SCS while preserving data fidelity. Experiments on nuScenes, Waymo, and KITTI datasets show that OmniSCS outperforms state-of-the-art methods in edited scene fidelity. We further validate its ability to enhance autonomous driving algorithms and support real-time (13Hz) closed-loop testing. Overall, OmniSCS provides a safer, more effective, and cost-efficient solution for SCS optimization and testing in autonomous driving.",
  "published": "2026-07-07",
  "updated": "2026-07-07",
  "year": "2026",
  "authors": [
   "Xiaoyun Dong",
   "Qian Xu",
   "Yang Lu",
   "Yang Lou",
   "Yung-Hui Li",
   "Jianping Wang"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "OmniSCS is an innovative system that generates photorealistic SCS with high physical fidelity while enabling closed-loop testing in synthetic environments and provides a safer, more effective, and cost-efficient solution for SCS optimization and testing in autonomous driving.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xiaoyun Dong",
    "id": "2394355536",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Qian Xu",
    "id": "2149106597",
    "h_index": 9,
    "papers": 25
   },
   {
    "name": "Yang Lu",
    "id": "2396314491",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Yang Lou",
    "id": "2143489417",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Yung-Hui Li",
    "id": "2293652766",
    "h_index": 5,
    "papers": 18
   },
   {
    "name": "Jianping Wang",
    "id": "2307034068",
    "h_index": 4,
    "papers": 12
   }
  ],
  "comment": "",
  "topics": [
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.09764v1",
  "pdf_url": "https://arxiv.org/pdf/2607.09764v1",
  "html_url": "https://arxiv.org/html/2607.09764v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.06740",
  "slug": "a-continual-learning-framework-for-adaptive-control-of-modular-soft-ro",
  "title": "A Continual Learning Framework for Adaptive Control of Modular Soft Robots",
  "abstract": "Soft robots have attracted significant attention in applications such as medical intervention, rehabilitation, and robotic manipulation due to their inherent compliance, flexibility, and high degrees of freedom. Modular soft robots (MSRs), composed of multiple interconnected segments, represent an emerging class of robotic systems with highly deformable and reconfigurable structures capable of performing complex tasks. However, designing controllers for MSRs remains challenging due to their nonlinear dynamics, modeling complexity, and hyper-redundant nature. Existing approaches typically require controllers to be retrained from scratch whenever the robot morphology changes. In this work, we address these challenges through a continual learning inspired control framework capable of incrementally adapting to changes in robot morphology while preserving previously acquired knowledge. Specifically, the proposed framework enables the controller to sequentially learn new MSR configurations without forgetting previously learned ones. In addition, for MSRs with fixed configurations, the same framework can be employed in a distributed manner to learn module-specific dynamics, enabling localized control and improved precision. The proposed approach is validated through closed-loop trajectory tracking experiments in simulation using a tendon-driven soft robot, as well as on a real-world three-module pneumatic soft robotic arm. Furthermore, we demonstrate the adaptive capabilities of the framework through a reaching experiment in which the controller selectively activates only the necessary modules to reach a virtual target position, thereby reducing computational overhead.",
  "published": "2026-07-07",
  "updated": "2026-07-07",
  "year": "2026",
  "authors": [
   "Nilay Kushawaha",
   "Muhammad Sunny Nazeer",
   "Baljinder Singh Bal",
   "Cecilia Laschi",
   "Egidio Falotico"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes a continual learning inspired control framework capable of incrementally adapting to changes in robot morphology while preserving previously acquired knowledge and demonstrates the adaptive capabilities of the framework through a reaching experiment in which the controller selectively activates only the necessary modules to reach a virtual target position, thereby reducing computational overhead.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "N. Kushawaha",
    "id": "2182375121",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Muhammad Sunny Nazeer",
    "id": "2217471870",
    "h_index": 5,
    "papers": 17
   },
   {
    "name": "Baljinder Singh Bal",
    "id": "2448369242",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Cecilia Laschi",
    "id": "2264347417",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "E. Falotico",
    "id": "1714652",
    "h_index": 25,
    "papers": 154
   }
  ],
  "comment": "",
  "topics": [
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.06740v1",
  "pdf_url": "https://arxiv.org/pdf/2607.06740v1",
  "html_url": "https://arxiv.org/html/2607.06740v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.06706",
  "slug": "vision-language-action-vla-models-for-unmanned-aerial-robotics-and-bim",
  "title": "Vision Language Action (VLA) Models for Unmanned Aerial Robotics and Bimanual Manipulation: A Review",
  "abstract": "Vision Language Action (VLA) models unify visual perception, natural-language understanding, and action generation within a single foundation model, allowing a robot to follow instructions such as fold the towel or fly to the red building directly from camera images. Because VLAs inherit world knowledge from internet-scale pre-training, they have become the dominant framework for learning-based manipulation, with bimanual coordination serving as the most demanding testbed: two arms with 7 degrees of freedom each must move in concert to fold, assemble, and reorient objects. Unmanned aerial robotics faces a structurally similar challenge: a drone must coordinate thrust, attitude, and increasingly gripper commands from visual observations under strict latency and payload constraints. This review covers 183 contributions spanning 2017-2026 and organized along seven dimensions: VLA architectures, training recipes, action representations, bimanual coordination (2022-2026), unmanned aerial vehicle (UAV) navigation and control (2017-2026), language grounding, and cross-cutting concerns including memory and world models. We show that the coordination strategies, training recipes, and action representations developed for bimanual VLAs transfer to unmanned aerial systems and identify fourteen research directions across both domains.",
  "published": "2026-07-07",
  "updated": "2026-07-07",
  "year": "2026",
  "authors": [
   "Inkyu Sa",
   "Chanoh Park",
   "Hea-Min Lee",
   "Donghee Noh",
   "Ho Seok Ahn"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "Drones",
  "venue_source": "semantic-scholar",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "It is shown that the coordination strategies, training recipes, and action representations developed for bimanual VLAs transfer to unmanned aerial systems and identify fourteen research directions across both domains.",
  "doi": "10.3390/drones10060412",
  "oa_pdf": "https://doi.org/10.3390/drones10060412",
  "s2_authors": [
   {
    "name": "Inkyu Sa",
    "id": "2249529212",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Chan-Oh. Park",
    "id": "1996019865",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Hea-Min Lee",
    "id": "2178668904",
    "h_index": 4,
    "papers": 16
   },
   {
    "name": "Donghee Noh",
    "id": "2279301640",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Ho Seok Ahn",
    "id": "2249535419",
    "h_index": 4,
    "papers": 13
   }
  ],
  "comment": "56 pages, 11 figures, 16 tables",
  "topics": [
   "world-models",
   "vla",
   "navigation",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.06706v1",
  "pdf_url": "https://arxiv.org/pdf/2607.06706v1",
  "html_url": "https://arxiv.org/html/2607.06706v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2607.06699",
  "slug": "robosnap-one-shot-real-to-sim-scene-generation-for-generalizable-robot",
  "title": "RoboSnap: One-Shot Real-to-Sim Scene Generation for Generalizable Robot Learning and Evaluation",
  "abstract": "Recovering real-world scenes as interactive simulation environments can enable generalizable robot learning and reproducible policy evaluation. However, constructing scenes that are both physically stable and visually faithful remains slow and expensive. In this work, we present RoboSnap, a real-to-sim framework that turns a single RGB image into a simulation-ready scene. The key idea is a layered design that separates the physics-critical interaction area from the surrounding visual context: collision-aware foreground assets are refined for stable robot interaction, while a 3D Gaussian splatting visual layer preserves faithful background appearance under novel views. Experiments on DROID scenes and real-robot tasks show that RoboSnap achieves reliable trajectory replay in the recovered scenes, supports task-specific synthetic data generation for policy training, and yields meaningful sim-real correlation for policy evaluation. To further support real-to-sim research, we introduce DROID-Sim, a real-to-sim companion dataset constructed from 564 real-world scenes in DROID. Extensive experiments suggest that the value of real-to-sim methods lies not only in high-fidelity visual reconstruction, but in turning real environments into reusable infrastructure for robot learning and evaluation.",
  "published": "2026-07-07",
  "updated": "2026-07-07",
  "year": "2026",
  "authors": [
   "Shujie Zhang",
   "Jingkun Yi",
   "Weipeng Zhong",
   "Zirui Zhou",
   "Yangkun Zhu",
   "Hanqing Wang",
   "Xudong Xu",
   "Weinan Zhang",
   "Chunhua Shen"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 2,
  "influential_citations": 0,
  "tldr": "RoboSnap, a real-to-sim framework that turns a single RGB image into a simulation-ready scene that achieves reliable trajectory replay in the recovered scenes, supports task-specific synthetic data generation for policy training, and yields meaningful sim-real correlation for policy evaluation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Shujie Zhang",
    "id": "2382031727",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Jingkun Yi",
    "id": "2448437589",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Weipeng Zhong",
    "id": "2380988112",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Zirui Zhou",
    "id": "2351437743",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Yangkun Zhu",
    "id": "2385972163",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Hanqing Wang",
    "id": "2311307527",
    "h_index": 11,
    "papers": 27
   },
   {
    "name": "Xudong Xu",
    "id": "2383165807",
    "h_index": 3,
    "papers": 13
   },
   {
    "name": "Weinan Zhang",
    "id": "2344034124",
    "h_index": 5,
    "papers": 16
   },
   {
    "name": "Chunhua Shen",
    "id": "2257475525",
    "h_index": 7,
    "papers": 15
   }
  ],
  "comment": "24 pages, 16 figures, Project page: https://robosnap.github.io",
  "topics": [
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.06699v1",
  "pdf_url": "https://arxiv.org/pdf/2607.06699v1",
  "html_url": "https://arxiv.org/html/2607.06699v1",
  "code_url": "https://robosnap.github.io",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.48
 },
 {
  "id": "2607.06600",
  "slug": "milsd-a-micro-line-segment-detector-for-resource-constrained-devices",
  "title": "MiLSD: A Micro Line-Segment Detector for Resource-Constrained Devices",
  "abstract": "Line segment detection is a key building block in visual SLAM, 3D reconstruction, and industrial inspection. Recent deep learning methods have greatly improved accuracy, yet even the smallest models require several megabytes of memory, exceeding low-cost MCU capacity. This work investigates the maximum achievable accuracy under a sub-megabyte budget. We propose MiLSD, a detector tailored for MCU-level constraints, and systematically compare three output representations within a compact fully-convolutional backbone. Our study shows that the proposed F-Clip center-with-length-and-angle formulation learns most effectively at small model sizes. We find that 8-bit quantization preserves full-precision performance, while 4-bit quantization causes significant degradation, particularly in angle regression, with quantization-aware training recovering only part of the loss. With a one-megabyte activation budget and inference enhancements including sub-pixel decoding, test-time augmentation, and a lightweight verifier, MiLSD improves sAP10 on ShanghaiTech Wireframe from 10.6 (25k parameters, 0.25 MB) to 24.1 within 1 MB. Rather than competing with GPU-scale parsers, we map the accuracy memory trade-off across representations, bit-widths, capacities, and post-processing strategies for embedded vision systems.",
  "published": "2026-07-07",
  "updated": "2026-07-07",
  "year": "2026",
  "authors": [
   "Parsa Hassani Shariat Panahi",
   "Amir Hossein Jalilvand",
   "M. Hassan Najafi"
  ],
  "author_count": 3,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "This work proposes MiLSD, a detector tailored for MCU-level constraints, and systematically compare three output representations within a compact fully-convolutional backbone, and finds that 8-bit quantization preserves full-precision performance, while 4-bit quantization causes significant degradation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "P. Panahi",
    "id": "2260632861",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Amir Hossein Jalilvand",
    "id": "2003405849",
    "h_index": 7,
    "papers": 17
   },
   {
    "name": "M. Najafi",
    "id": "2247527356",
    "h_index": 8,
    "papers": 56
   }
  ],
  "comment": "10 pages, 12 figures, 5 tables",
  "topics": [
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.06600v1",
  "pdf_url": "https://arxiv.org/pdf/2607.06600v1",
  "html_url": "https://arxiv.org/html/2607.06600v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2607.06559",
  "slug": "rynnworld-4d-4d-embodied-world-models-for-robotic-manipulation",
  "title": "RynnWorld-4D: 4D Embodied World Models for Robotic Manipulation",
  "abstract": "Robotic manipulation in the open world requires not only recognizing what a scene looks like, but also anticipating how its 3D structure moves under interaction. We argue that synchronized RGB, depth, and optical flow, namely RGB-DF, provide a physically grounded representation that captures the underlying 4D dynamics of a scene. Compared to 2D pixel videos, this multi-modal synergy aligns visual appearance with geometric structure and temporal motion, creating a representation space significantly closer to the low-level end-effector actions demanded by robotic systems, thereby narrowing the gap between world prediction and policy learning. Building on this insight, we introduce RynnWorld-4D, a generative model that co-produces future RGB frames, depth maps, and optical flow from a single RGB-D image and a language instruction within one unified diffusion process. This 4D world model features a tri-branch architecture that integrates cross-modal attention with frame-wise 3D RoPE, ensuring that appearance, geometry, and motion evolve consistently. To supply training data at scale, we curate Rynn4DDataset 1.0, a massive dataset of over 254.4 million frames across egocentric human and robotic manipulation videos with high-quality pseudo-labels for depth and optical flow. We further propose RynnWorld-4D-Policy, an inverse dynamics head that consumes the internal 4D representations of RynnWorld-4D in a single forward pass, bypassing expensive multi-step denoising, to output robot actions in a closed-loop manner. Experiments show that RynnWorld-4D produces temporally and spatially coherent 4D predictions, and that RynnWorld-4D-Policy achieves state-of-the-art performance on real-world dexterous bimanual manipulation tasks, particularly excelling in tasks demanding spatial precision and temporal coordination.",
  "published": "2026-07-07",
  "updated": "2026-07-07",
  "year": "2026",
  "authors": [
   "Haoyu Zhao",
   "Xingyue Zhao",
   "Siteng Huang",
   "Xin Li",
   "Deli Zhao",
   "Zhongyu Li"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "RynnWorld-4D is introduced, a generative model that co-produces future RGB frames, depth maps, and optical flow from a single RGB-D image and a language instruction within one unified diffusion process, and achieves state-of-the-art performance on real-world dexterous bimanual manipulation tasks, particularly excelling in tasks demanding spatial precision and temporal coordination.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Haoyu Zhao",
    "id": "2316659455",
    "h_index": 5,
    "papers": 15
   },
   {
    "name": "Xingyue Zhao",
    "id": "2303839096",
    "h_index": 5,
    "papers": 16
   },
   {
    "name": "Siteng Huang",
    "id": "2371084328",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Xin Li",
    "id": "2376124596",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Deli Zhao",
    "id": "2303980061",
    "h_index": 13,
    "papers": 24
   },
   {
    "name": "Zhongyu Li",
    "id": "2341901903",
    "h_index": 3,
    "papers": 10
   }
  ],
  "comment": "Project Page: https://alibaba-damo-academy.github.io/RynnWorld-4D.github.io, Github: https://github.com/alibaba-damo-academy/RynnWorld-4D",
  "topics": [
   "world-models",
   "dexterous-manipulation",
   "egocentric-data"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.06559v1",
  "pdf_url": "https://arxiv.org/pdf/2607.06559v1",
  "html_url": "https://arxiv.org/html/2607.06559v1",
  "code_url": "https://alibaba-damo-academy.github.io/RynnWorld-4D.github.io",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.3
 },
 {
  "id": "2607.06558",
  "slug": "rynnworld-teleop-an-action-conditioned-world-model-for-digital-teleope",
  "title": "RynnWorld-Teleop: An Action-Conditioned World Model for Digital Teleoperation",
  "abstract": "Scaling robot learning requires massive, diverse trajectory data, yet collection is currently bottlenecked by physical teleoperation, where every demonstration binds operator time to specific hardware and workspaces. We introduce digital teleoperation, a paradigm that decouples data collection from physical constraints by replacing the real robot with a generative world model. In this framework, an operator's hand-pose stream drives a robot-centric generative world model to synthesize high-fidelity egocentric videos from a single reference image. The recorded pose stream serves as an embodiment-agnostic action label transferable to any target robot via standard retargeting, yielding complete state-action trajectories for imitation learning independent of physical hardware. We instantiate this paradigm in RynnWorld-Teleop, a system that integrates depth-aware skeletal conditioning, progressive human-to-robot training on a video Diffusion Transformer, and streaming autoregressive distillation. This pipeline compresses the generative process into a single-pass inference, enabling 40+ FPS, real-time interactive generation on a single H100 GPU. Policies trained exclusively on RynnWorld-Teleop-generated data achieve effective zero-shot Sim2Real transfer across dexterous and diverse bimanual tasks. Moreover, augmenting real-world datasets with our digitally teleoperated data consistently improves success rates, demonstrating that RynnWorld-Teleop serves as a high-fidelity, scalable data engine for the next generation of robotic agents.",
  "published": "2026-07-07",
  "updated": "2026-07-12",
  "year": "2026",
  "authors": [
   "Haoyu Zhao",
   "Xingyue Zhao",
   "Hangyu Li",
   "Biao Gong",
   "Kehan Li",
   "Siteng Huang",
   "Xin Li",
   "Deli Zhao",
   "Zhongyu Li"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Digital teleoperation, a paradigm that decouples data collection from physical constraints by replacing the real robot with a generative world model, is introduced in RynnWorld-Teleop, a system that integrates depth-aware skeletal conditioning, progressive human-to-robot training on a video Diffusion Transformer, and streaming autoregressive distillation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Haoyu Zhao",
    "id": "2316659455",
    "h_index": 5,
    "papers": 15
   },
   {
    "name": "Xingyue Zhao",
    "id": "2303839096",
    "h_index": 5,
    "papers": 16
   },
   {
    "name": "Hangyu Li",
    "id": "2448287228",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Biao Gong",
    "id": "2268398906",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Kehan Li",
    "id": "2371267604",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Siteng Huang",
    "id": "2371084328",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Xin Li",
    "id": "2376124596",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Deli Zhao",
    "id": "2303980061",
    "h_index": 13,
    "papers": 24
   },
   {
    "name": "Zhongyu Li",
    "id": "2341901903",
    "h_index": 3,
    "papers": 10
   }
  ],
  "comment": "Project Page: https://alibaba-damo-academy.github.io/RynnWorld-Teleop.github.io, Github: https://github.com/alibaba-damo-academy/RynnWorld-Teleop",
  "topics": [
   "world-models",
   "dexterous-manipulation",
   "egocentric-data",
   "sim2real",
   "imitation-diffusion",
   "data-teleop",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.06558v2",
  "pdf_url": "https://arxiv.org/pdf/2607.06558v2",
  "html_url": "https://arxiv.org/html/2607.06558v2",
  "code_url": "https://alibaba-damo-academy.github.io/RynnWorld-Teleop.github.io",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2607.06501",
  "slug": "hypothesis-driven-model-expansion-under-uncertainty-for-open-world-rob",
  "title": "Hypothesis-driven Model Expansion under Uncertainty for Open-World Robot Planning",
  "abstract": "We consider an open-world planning setting in which service robots must operate in unknown environments with incomplete knowledge of objects and actions. Traditional closed-world approaches with pre-programmed knowledge bases fail when robots encounter unexpected situations and tasks, posing a fundamental challenge for autonomous knowledge expansion in human environments. In this work, we propose an open-world planning framework that enables robots to automatically generate, verify, and update hypotheses about their abstract world models. Our key insight is to explicitly maintain uncertainty-aware knowledge expansion and integrate hypothesis verification into goal-reaching planning. The framework leverages foundation models to generate initial hypotheses over states and transitions, and applies automated planning to produce action sequences that jointly address hypothesis verification and task execution. Through iterative execution and refinement, the robot expands its knowledge by incorporating verification feedback from the foundation models when hypotheses prove incorrect. Extensive experiments in simulated and real-world environments demonstrate that our framework enables autonomous knowledge expansion and effective operation in open-world settings. These results indicate that integrating uncertainty-aware model expansion from robot foundation models with planning advances the practical deployment of household service robots.",
  "published": "2026-07-07",
  "updated": "2026-07-07",
  "year": "2026",
  "authors": [
   "Anxing Xiao",
   "Hanbo Zhang",
   "Tianrun Hu",
   "David Hsu"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "RSS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "An open-world planning framework that enables robots to automatically generate, verify, and update hypotheses about their abstract world models, and applies automated planning to produce action sequences that jointly address hypothesis verification and task execution is proposed.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Anxing Xiao",
    "id": "2322928601",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Hanbo Zhang",
    "id": "2345858566",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Tianrun Hu",
    "id": "2323755758",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "David Hsu",
    "id": "2268673376",
    "h_index": 2,
    "papers": 4
   }
  ],
  "comment": "Accepted to Robotics: Science and Systems (RSS) 2026",
  "topics": [
   "world-models",
   "foundation-pretraining",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.06501v1",
  "pdf_url": "https://arxiv.org/pdf/2607.06501v1",
  "html_url": "https://arxiv.org/html/2607.06501v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2607.06438",
  "slug": "wristmimic-full-body-humanoid-control-with-wrist-guided-manipulation",
  "title": "WristMimic: Full-Body Humanoid Control with Wrist-Guided Manipulation",
  "abstract": "Retargeting human object interaction demonstrations to physics based simulation requires reproducing not only body motion but also the object motion and contacts that make manipulation succeed. However, position only hand trajectories do not specify the contact forces needed to manipulate objects, and directly tracking them can overconstrain contact rich finger behavior. We introduce WristMimic, a wrist guided whole body control framework that explicitly separates contact free body motion from contact rich hand manipulation. The contact free body and wrist are guided by kinematic pose targets, whereas the fingers are not directly supervised by human hand pose. Instead, they learn grasping and manipulation behaviors from object tracking and contact outcomes. Our key insight is that the wrist is the natural gate between these two regimes. It is largely free from contact and can be tracked kinematically, yet it determines the global hand configuration and places the fingers within reachable grasp affordances. To ensure reliable wrist placement during interaction, we introduce wrist specific reset constraints and reward prioritization. Experiments show that WristMimic matches or surpasses methods using full finger pose supervision while enabling finger agnostic retargeting across diverse hand embodiments.",
  "published": "2026-07-07",
  "updated": "2026-07-13",
  "year": "2026",
  "authors": [
   "Wongyun Yu",
   "Youngwoon Kim",
   "Minsu Cho"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.CV",
   "cs.GR"
  ],
  "primary_category": "cs.RO",
  "venue": "ECCV 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "WristMimic, a wrist guided whole body control framework that explicitly separates contact free body motion from contact rich hand manipulation, is introduced, showing that the wrist is the natural gate between these two regimes.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Wongyun Yu",
    "id": "2375947674",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Young-Won Kim",
    "id": "2445440298",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Minsu Cho",
    "id": "2352406558",
    "h_index": 2,
    "papers": 6
   }
  ],
  "comment": "Accepted to ECCV 2026",
  "topics": [
   "dexterous-manipulation",
   "humanoids"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.06438v2",
  "pdf_url": "https://arxiv.org/pdf/2607.06438v2",
  "html_url": "https://arxiv.org/html/2607.06438v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2607.06403",
  "slug": "from-foundation-to-application-improving-vla-models-in-practice",
  "title": "From Foundation to Application: Improving VLA Models in Practice",
  "abstract": "Despite recent progress of VLA foundation models, the disparity between laboratory conditions and real-world applications continues to impede their practical implementation. To bridge this gap, we present LingBot-VLA 2.0, which advances LingBot-VLA through improvements in three functional domains. (1) Generalization across tasks and embodiments. Compared to the previous version, we revamp the data processing pipeline and curate around 60,000 hours of data for pretraining, including 50,000 hours of robot trajectories spanning 20 robot configurations and 10,000 hours of egocentric human videos. (2) Expanded action space in addition to dual-arm hardware platforms. In particular, our system accommodates degrees of freedom for the heads, waists, mobile bases, and dexterous hands, thereby empowering the robots to tackle more complex tasks in practical scenarios. (3) Predictive dynamics modeling for improved temporal reasoning. Specifically, we formulate future prediction as a proxy task, facilitated by a video representation model for semantic priors and a depth estimation model for geometric cues. Evaluations on the GM-100 benchmark, conducted in a generalist setting, validate the beneficial impact of these proposed modifications. Furthermore, benefiting from the expanded pretraining data that covers whole-body degrees of freedom, LingBot-VLA-2.0 demonstrates strong cross-embodiment long-horizon mobile manipulation capability across the two robotic platforms.",
  "published": "2026-07-07",
  "updated": "2026-07-07",
  "year": "2026",
  "authors": [
   "Wei Wu",
   "Fangjing Wang",
   "Fan Lu",
   "He Sun",
   "Shi Liu",
   "Yunnan Wang",
   "Yibin Yan",
   "Yong Wang",
   "Shuailei Ma",
   "Xinyang Wang",
   "Yibin Liu",
   "Shuai Yang",
   "Tianxiang Zhou",
   "Kejia Zhang",
   "Lei Zhou",
   "Cheng Su",
   "Nan Xue",
   "Bin Tan",
   "Han Zhang",
   "Youchao Zhang",
   "Fei Liao",
   "Xing Zhu",
   "Yujun Shen",
   "Kecheng Zheng"
  ],
  "author_count": 24,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 11,
  "influential_citations": 3,
  "tldr": "Benefiting from the expanded pretraining data that covers whole-body degrees of freedom, LingBot-VLA-2.0 demonstrates strong cross-embodiment long-horizon mobile manipulation capability across the two robotic platforms.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Wei Wu",
    "id": "2276770083",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Fangjing Wang",
    "id": "2374081930",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Fan Lu",
    "id": "2293769075",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "He Sun",
    "id": "2350521999",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Shi Liu",
    "id": "2364422725",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Yunnan Wang",
    "id": "2130933957",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Yibin Yan",
    "id": "2311827925",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Yong Wang",
    "id": "2370795438",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Shuailei Ma",
    "id": "2293403497",
    "h_index": 9,
    "papers": 17
   },
   {
    "name": "Xinyang Wang",
    "id": "2448705668",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Yibin Liu",
    "id": "2370952744",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Shuai Yang",
    "id": "2406903143",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Tianxiang Zhou",
    "id": "2268025566",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Kejia Zhang",
    "id": "2237597509",
    "h_index": 7,
    "papers": 48
   },
   {
    "name": "Lei Zhou",
    "id": "2446303770",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Che-Shih Su",
    "id": "30905943",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Nan Xue",
    "id": "2292027056",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "Bin Tan",
    "id": "2336997012",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Hanze Zhang",
    "id": "2444264539",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Youchao Zhang",
    "id": "2142599044",
    "h_index": 8,
    "papers": 21
   },
   {
    "name": "Fei Liao",
    "id": "2365256008",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Xing Zhu",
    "id": "2406915796",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Yujun Shen",
    "id": "2392945842",
    "h_index": 6,
    "papers": 21
   },
   {
    "name": "Kecheng Zheng",
    "id": "2313906719",
    "h_index": 2,
    "papers": 9
   }
  ],
  "comment": "Website: https://technology.robbyant.com/lingbot-vla-v2, Github: https://github.com/robbyant/lingbot-vla-v2, Checkpoints: https://huggingface.co/collections/robbyant/lingbot-vla-v2",
  "topics": [
   "vla",
   "dexterous-manipulation",
   "humanoids",
   "egocentric-data",
   "spatial-3d",
   "navigation",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.06403v1",
  "pdf_url": "https://arxiv.org/pdf/2607.06403v1",
  "html_url": "https://arxiv.org/html/2607.06403v1",
  "code_url": "https://github.com/robbyant/lingbot-vla-v2",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.08
 },
 {
  "id": "2607.06388",
  "slug": "learning-to-throw-objects-safely-in-multi-obstacle-environments",
  "title": "Learning to Throw Objects Safely in Multi-Obstacle Environments",
  "abstract": "Robotic throwing enables fast and efficient object placement beyond the robot's immediate workspace, but reliable throwing in cluttered environments remains underexplored. Existing approaches, such as TossingBot, learn throwing strategies from visual input but assume obstacle-free settings. In this paper, we address the problem of throwing objects into a target basket while avoiding obstacles placed randomly in the scene. We introduce a potential field state representation that compactly encodes both basket attraction and obstacle repulsion on a fixed-size grid, enabling reinforcement learning (RL) policies to generalize across arbitrary numbers and configurations of obstacles. The policy is initialized from kinesthetic demonstrations and optimized in simulation using three state-of-the-art RL algorithms (SAC, DDPG, TD3). Among these, SAC achieves the most consistent performance across scenarios. We compare the potential field representation against explicit state encodings and demonstrate that it achieves higher success rates and better scalability to unseen obstacle configurations. Real-robot experiments with unseen throwable objects confirm robust sim-to-real transfer, achieving up to $90\\%$ success in cluttered scenes. These results demonstrate that PFR provides a practical and robust representation for safe and efficient robotic throwing in unstructured environments. A video showcasing our experiments is available at: https://youtu.be/ZZnJf8ua2dE",
  "published": "2026-07-07",
  "updated": "2026-07-07",
  "year": "2026",
  "authors": [
   "Mohammadreza Kasaei",
   "Klemen Voncina",
   "Hamidreza Kasaei"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A potential field state representation is introduced that compactly encodes both basket attraction and obstacle repulsion on a fixed-size grid, enabling reinforcement learning (RL) policies to generalize across arbitrary numbers and configurations of obstacles.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "M. Kasaei",
    "id": "3160474",
    "h_index": 8,
    "papers": 27
   },
   {
    "name": "Klemen Voncina",
    "id": "147265747",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "H. Kasaei",
    "id": "69533920",
    "h_index": 17,
    "papers": 64
   }
  ],
  "comment": "This paper has been presented at the IEEE International Conference on Robotics & Automation (ICRA), 2026",
  "topics": [
   "sim2real",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.06388v1",
  "pdf_url": "https://arxiv.org/pdf/2607.06388v1",
  "html_url": "https://arxiv.org/html/2607.06388v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2607.06323",
  "slug": "lamp-latent-motion-prior-guided-real-world-learning-for-dexterous-hand",
  "title": "LAMP: Latent Motion Prior-Guided Real-World Learning for Dexterous Hand Manipulation",
  "abstract": "Real-world learning for dexterous hands remains brittle because high-dimensional hand actions amplify imitation errors and make reinforcement-learning exploration prone to contact-breaking motion. While combining imitation learning (IL) with online reinforcement learning (RL) can reduce manual supervision, unconstrained exploration in raw hand-action spaces is sample-inefficient and risky for physical hardware. We introduce a latent motion prior module (\\prior{}) that maps recent hand-action histories to a compact, history-conditioned latent prior and decodes continuous latent commands into executable high-dimensional hand targets. Built on this prior, \\method{} is a three-stage real-world dexterous learning framework: it pretrains \\prior{} from demonstrations, trains a visuomotor policy that predicts native arm commands and latent hand-action offsets, and improves the policy with online residual RL in the same latent hand-action space. This shared, decodable interface lets residual exploration make local corrections near demonstrated, contact-consistent hand motions rather than perturbing every finger joint independently. We evaluate \\method{} on four real-robot dexterous manipulation tasks against raw, linear, and discrete hand-action interfaces. Starting from small task-specific demonstration sets, \\method{} achieves a 56.25\\% average IL success rate and raises it to 98.75\\% after online RL, reaching 100\\% final success on three tasks and 95\\% on the remaining task.",
  "published": "2026-07-07",
  "updated": "2026-07-07",
  "year": "2026",
  "authors": [
   "Xinye Yang",
   "Zhiyuan Ma",
   "Hongze Yu",
   "Yuanpei Chen",
   "Yaodong Yang",
   "Xiaojie Chai",
   "Xinlei Chen",
   "Chao Yu"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A latent motion prior module (\\prior{}) is introduced that maps recent hand-action histories to a compact, history-conditioned latent prior and decodes continuous latent commands into executable high-dimensional hand targets and improves the policy with online residual RL in the same latent hand-action space.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xinye Yang",
    "id": "2448344238",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Zhiyuan Ma",
    "id": "2355382766",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Hongze Yu",
    "id": "2299277918",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Yuanpei Chen",
    "id": "2261728034",
    "h_index": 13,
    "papers": 33
   },
   {
    "name": "Yaodong Yang",
    "id": "2239164053",
    "h_index": 8,
    "papers": 21
   },
   {
    "name": "Xiaojie Chai",
    "id": "2448128546",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Xinlei Chen",
    "id": "2363524289",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Chao Yu",
    "id": "2296946468",
    "h_index": 5,
    "papers": 10
   }
  ],
  "comment": "17 pages, 11 figures",
  "topics": [
   "dexterous-manipulation",
   "imitation-diffusion",
   "rl-control",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.06323v1",
  "pdf_url": "https://arxiv.org/pdf/2607.06323v1",
  "html_url": "https://arxiv.org/html/2607.06323v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.06165",
  "slug": "eagor-embodied-reasoning-in-omni-direction",
  "title": "EAGOR: Embodied Reasoning in Omni-direction",
  "abstract": "Omni-directional (360\u00b0) cameras provide embodied agents with a holistic view of their surroundings, making them suited for directional reasoning in tasks such as navigation and object search. Existing Vision Language Models (VLMs) project 360\u00b0 observations to 2D equirectangular projection (ERP) images and process them using architectures designed for perspective images. However, they ignore the spherical nature of 360\u00b0 observations, where each pixel represents a viewing direction relative to the agent. Consequently, their direction estimates often become inconsistent under camera view transformations caused by agent motion. This limitation is particularly critical for map-free navigation, where the agent must continuously estimate the target direction in its egocentric frame. We propose EAGOR, a training-free, geometry-aware framework for embodied 360\u00b0 directional reasoning. Instead of predicting target directions as ERP image coordinates, EAGOR formulates directional reasoning as recursive Bayesian estimation directly on the sphere. It maintains a continuous belief over target directions and propagates it equivariantly under agent motion without training the backbone VLMs. To achieve this, we introduce the Spherical Harmonic Belief Field (SH-BF), whose spherical harmonic representation provides a globally defined, rotation-aware basis for directional estimation on the spherical manifold. This formulation eliminates ERP seam discontinuities, latitude distortions, and interpolation errors. We evaluate EAGOR on two benchmark datasets and real-world experiments with a legged robot across directional reasoning tasks. EAGOR consistently outperforms existing methods, achieving average relative gains of +34.4% and +45.6% on HOS and OSR-Bench, respectively, while improving navigation success by +14.6%, reducing step count by 17.7%, and lowering mean angular error by 24.5%.",
  "published": "2026-07-07",
  "updated": "2026-07-07",
  "year": "2026",
  "authors": [
   "Shriram Damodaran",
   "Soumyaratna Debnath",
   "Yan Wu",
   "Wei-Yun Yau",
   "Addison Lin Wang"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The Spherical Harmonic Belief Field (SH-BF), whose spherical harmonic representation provides a globally defined, rotation-aware basis for directional estimation on the spherical manifold, is introduced, whose formulation eliminates ERP seam discontinuities, latitude distortions, and interpolation errors.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Shriram Damodaran",
    "id": "2378875422",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Soumyaratna Debnath",
    "id": "2041267381",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Yan Wu",
    "id": "2391837584",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "W. Yau",
    "id": "145492070",
    "h_index": 39,
    "papers": 239
   },
   {
    "name": "A-dan Wang",
    "id": "15451585",
    "h_index": 2,
    "papers": 5
   }
  ],
  "comment": "12 Pages, 7 Figures, 4 Tables",
  "topics": [
   "humanoids",
   "egocentric-data",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.06165v1",
  "pdf_url": "https://arxiv.org/pdf/2607.06165v1",
  "html_url": "https://arxiv.org/html/2607.06165v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.06052",
  "slug": "thorarena-benchmarking-humanoid-physical-interaction-with-human-motion",
  "title": "ThorArena: Benchmarking Humanoid Physical Interaction with Human Motion-Force Demonstrations",
  "abstract": "Humanoid robots are increasingly expected to perform contact-rich tasks that require not only accurate whole-body motion but also robust physical interaction with surrounding objects and humans. Although recent advances in humanoid motion imitation and whole-body control have achieved remarkable tracking performance, existing datasets and benchmarks primarily focus on kinematic motion while largely overlooking synchronized interaction forces. As a result, current evaluations fail to capture how external interaction forces affect tracking accuracy, stability, and control robustness. In this paper, we present ThorArena, a benchmark for evaluating force-aware humanoid interaction based on human demonstrations with synchronized motion and force measurements. We collect a real-world interaction dataset that simultaneously captures whole-body human motion and forces exerted by both hands across six representative physical interaction tasks. Based on these demonstrations, we propose force-aware evaluation metrics that jointly assess whole-body tracking accuracy, robustness under different force levels, control effort, and episode survival through the Force-Aware Tracking Score (FATS) and complementary diagnostic metrics. We further establish a unified benchmark protocol that replays recorded interaction forces in simulation and provides a standardized evaluation interface for different humanoid control policies. Experiments on representative whole-body control policies demonstrate that force-aware evaluation reveals substantial performance differences that remain largely hidden under conventional no-force evaluation. ThorArena provides a practical and reproducible framework for studying force-aware humanoid interaction and offers a new benchmark for evaluating contact-rich humanoid behaviors.",
  "published": "2026-07-07",
  "updated": "2026-07-07",
  "year": "2026",
  "authors": [
   "Chenhao Yu",
   "Hongwu Wang",
   "Weitao Zhang",
   "Youhao Hu",
   "Jiachen Zhang",
   "Gangyang Li",
   "Alois Knoll",
   "Shaqi Luo"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "ThorArena is presented, a benchmark for evaluating force-aware humanoid interaction based on human demonstrations with synchronized motion and force measurements and a unified benchmark protocol that replays recorded interaction forces in simulation and provides a standardized evaluation interface for different humanoid control policies.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Chenhao Yu",
    "id": "2262085884",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Hongwu Wang",
    "id": "2108986618",
    "h_index": 18,
    "papers": 63
   },
   {
    "name": "Weitao Zhang",
    "id": "2220793142",
    "h_index": 2,
    "papers": 11
   },
   {
    "name": "Youhao Hu",
    "id": "152241939",
    "h_index": 6,
    "papers": 21
   },
   {
    "name": "Jiacheng Zhang",
    "id": "2265932600",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Gangyang Li",
    "id": "2267853382",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "A. Knoll",
    "id": "2199865233",
    "h_index": 12,
    "papers": 81
   },
   {
    "name": "Shaqi Luo",
    "id": "2307566436",
    "h_index": 5,
    "papers": 12
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "egocentric-data",
   "tactile",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.06052v1",
  "pdf_url": "https://arxiv.org/pdf/2607.06052v1",
  "html_url": "https://arxiv.org/html/2607.06052v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.05801",
  "slug": "trig-trajectory-rig-decoupled-metric-geometry-learning",
  "title": "TRIG: Trajectory-Rig Decoupled Metric Geometry Learning",
  "abstract": "Vision-centric autonomous driving requires accurate metric geometry and ego-motion estimation from synchronized multi-camera observations. Recent visual geometry models show strong performance in pose estimation, depth prediction, and 3D reconstruction, but are not tailored to rigid multi-camera driving systems. They often encode camera poses as entangled representations, in which time-varying ego-motion and static camera-rig geometry are jointly modeled, limiting the utilization of vehicle-side geometric priors. We propose Trajectory-Rig Decoupled Metric Geometry Learning (TRIG), a geometry perception framework for autonomous driving. TRIG factorizes camera poses into ego-trajectory and camera-rig components, enabling separate modeling of ego-motion and static multi-camera topology. We introduce decoupled pose encoding and supervision, which separately constrain trajectory evolution and rig geometry for metric-consistent learning. Moreover, sparse Temporal--Spatial attention separates cross-camera interaction from temporal aggregation, reducing global attention cost while preserving geometric reasoning. Experiments on five autonomous driving benchmarks show that TRIG achieves state-of-the-art performance in pose estimation, metric depth prediction, and 3D reconstruction.",
  "published": "2026-07-07",
  "updated": "2026-07-14",
  "year": "2026",
  "authors": [
   "Lizhou Liao",
   "Wentao Xu",
   "Handong Wang",
   "Lirong Yang",
   "Shuai Yang",
   "Weiwei Liu",
   "Chang Huang"
  ],
  "author_count": 7,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "TRIG factorizes camera poses into ego-trajectory and camera-rig components, enabling separate modeling of ego-motion and static multi-camera topology, and introduces decoupled pose encoding and supervision, which separately constrain trajectory evolution and rig geometry for metric-consistent learning.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "L. Liao",
    "id": "2150041350",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Wentao Xu",
    "id": "2448146282",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Handong Wang",
    "id": "2445725147",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Lirong Yang",
    "id": "2290972616",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Shuai Yang",
    "id": "2379968786",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Weiwei Liu",
    "id": "2378714137",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Chang Huang",
    "id": "2378958703",
    "h_index": 1,
    "papers": 3
   }
  ],
  "comment": "10 pages, 4 figures, 8 tables",
  "topics": [
   "spatial-3d",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.05801v2",
  "pdf_url": "https://arxiv.org/pdf/2607.05801v2",
  "html_url": "https://arxiv.org/html/2607.05801v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.05780",
  "slug": "forge-towards-functional-tool-use-generalization-via-keypoint-trajecto",
  "title": "FORGE: Towards Functional Tool-Use Generalization via Keypoint Trajectory Reasoning",
  "abstract": "While humans readily repurpose a book, a stone, or a shoe to drive a nail, robots trained on specific tools fail to transfer the same function to novel ones -- a gap we formalize as functional generalization. Such tools share a common functional intent that is visually recognizable, yet this perceptual similarity does not carry over to action space, where each tool demands an entirely different motor pattern. To bridge this gap, we explore intermediate representations including affordance images, human video prompts, and 2D keypoint trajectories, finding that keypoint trajectories best balance functional expressiveness and action groundability. Building on this, we propose FunctiOnal Reasoning and Grounded Execution (FORGE), a two-stage policy that decouples functional reasoning from action execution: predicting generalizable keypoint trajectories from action-free data, then grounding them into robot actions with limited demonstrations. On a seven-tool hitting-function benchmark, FORGE consistently outperforms state-of-the-art methods on unseen tools in both simulation and the real world, achieving over 2X improvement in average success rate.",
  "published": "2026-07-07",
  "updated": "2026-07-07",
  "year": "2026",
  "authors": [
   "Chuhao Zhou",
   "Liquan Wang",
   "Shuxin Cao",
   "Xiangyu Chen",
   "Yuxuan Hu",
   "Boyu Ma",
   "Animesh Garg",
   "Jianfei Yang"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes FunctiOnal Reasoning and Grounded Execution (FORGE), a two-stage policy that decouples functional reasoning from action execution: predicting generalizable keypoint trajectories from action-free data, then grounding them into robot actions with limited demonstrations.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Chuhao Zhou",
    "id": "2326548035",
    "h_index": 2,
    "papers": 17
   },
   {
    "name": "Liquang Wang",
    "id": "2108908463",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Shuxin Cao",
    "id": "65856356",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Xiangyu Chen",
    "id": "2388613119",
    "h_index": 1,
    "papers": 12
   },
   {
    "name": "Yuxuan Hu",
    "id": "2409873744",
    "h_index": 2,
    "papers": 12
   },
   {
    "name": "Boyu Ma",
    "id": "2275779097",
    "h_index": 4,
    "papers": 29
   },
   {
    "name": "Animesh Garg",
    "id": "2282540862",
    "h_index": 7,
    "papers": 18
   },
   {
    "name": "Jianfei Yang",
    "id": "2404007795",
    "h_index": 2,
    "papers": 26
   }
  ],
  "comment": "15 pages, 8 figures, 6 tables",
  "topics": [
   "egocentric-data"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.05780v1",
  "pdf_url": "https://arxiv.org/pdf/2607.05780v1",
  "html_url": "https://arxiv.org/html/2607.05780v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.05665",
  "slug": "efficient-transfer-learning-of-robot-dynamic-models-using-morphologica",
  "title": "Efficient Transfer Learning of Robot Dynamic Models Using Morphological Similarity",
  "abstract": "This study proposes a neural network-based transfer learning framework for modeling the dynamics of soft, fin-actuated underwater robots. We focus on morphologically similar robots that differ in scale and hydrodynamic properties. A model trained on data from a larger robot (source domain) is adapted to a smaller one (target domain) with limited labeled data. To enable label-efficient transfer, we develop an autoencoder-based domain adaptation approach that learns a shared latent representation aligning the dynamics of both robots. Experiments on two real underwater robots show that the proposed method enables accurate state estimation of the body-frame velocities on a target platform without labeled data, highlighting its potential for efficient cross-robot dynamics transfer among morphologically similar platforms.",
  "published": "2026-07-06",
  "updated": "2026-07-06",
  "year": "2026",
  "authors": [
   "Pavlo Kupyn",
   "Yuya Hamamatsu",
   "Roza Gkliva",
   "Asko Ristolainen",
   "Maarja Kruusmaa"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Experiments on two real underwater robots show that the proposed neural network\u2013based transfer learning framework enables accurate state estimation of the body-frame velocities on a target platform without labeled data, highlighting its potential for efficient cross-robot dynamics transfer among morphologically similar platforms.",
  "doi": "10.1109/CoDIT70676.2026.11631228",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Pavlo Kupyn",
    "id": "2343827107",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yuya Hamamatsu",
    "id": "1393463027",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Roza Gkliva",
    "id": "1557548498",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "A. Ristolainen",
    "id": "2523821",
    "h_index": 9,
    "papers": 38
   },
   {
    "name": "M. Kruusmaa",
    "id": "2329071536",
    "h_index": 3,
    "papers": 14
   }
  ],
  "comment": "Accepted for publication in the 2026 12th International Conference on Control, Decision and Information Technologies (CoDIT)",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.05665v1",
  "pdf_url": "https://arxiv.org/pdf/2607.05665v1",
  "html_url": "https://arxiv.org/html/2607.05665v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.04940",
  "slug": "closing-the-reality-gap-zero-shot-sim-to-real-deployment-for-dexterous",
  "title": "Closing the Reality Gap: Zero-Shot Sim-to-Real Deployment for Dexterous Force-Based Grasping and Manipulation",
  "abstract": "Human-like dexterous hands with multiple fingers offer human-level manipulation capabilities but remain difficult to train the control policies that can deploy on real hardware due to contact-rich physics and imperfect actuation. We present a sim-to-real reinforcement learning method that leverages dense tactile feedback combined with joint torque sensing to explicitly regulate physical interactions. To enable effective sim-to-real transfer, we introduce (i) a computationally fast tactile simulation that computes distances between dense virtual tactile units and the object via parallel forward kinematics, providing high-rate, high-resolution touch signals needed by RL; (ii) a current-to-torque calibration that eliminates the need for torque sensors on dexterous hands by mapping motor current to joint torque; and (iii) actuator dynamics modeling with randomization to account for non-ideal torque-speed effects and bridge the actuation gaps. Using an asymmetric actor-critic PPO pipeline, we train policies entirely in simulation and deploy them directly to a five-finger hand. The resulting policies demonstrate two essential human-hand skills: (1) command-based controllable grasp force tracking and (2) reorientation of objects in the hand, both of which are robustly executed without fine-tuning on the robot. By combining tactile and torque in the observation space with scalable sensing and actuation modeling, our system provides a practical solution to achieve reliable dexterous manipulation. To our knowledge, this is the first demonstration of controllable grasping on a multi-finger dexterous hand trained entirely in simulation and transferred zero-shot on real hardware.",
  "published": "2026-07-06",
  "updated": "2026-07-06",
  "year": "2026",
  "authors": [
   "Zhe Zhao",
   "Zhibin Li",
   "Yilin Ou",
   "Mengshi Qi"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 6,
  "influential_citations": 0,
  "tldr": "A practical sim-to-real reinforcement learning framework that utilizes dense tactile feedback combined with joint torque sensing to explicitly regulate physical interactions and provides a practical solution to achieve reliable dexterous manipulation.",
  "doi": "10.48550/arXiv.2601.02778",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhe Zhao",
    "id": "2312464276",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Haoyu Dong",
    "id": "2383869412",
    "h_index": 4,
    "papers": 14
   },
   {
    "name": "Zhengmao He",
    "id": "2265617102",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Yang Li",
    "id": "2390949333",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Xinyu Yi",
    "id": "2390568607",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Zhibin Li",
    "id": "2157095073",
    "h_index": 1,
    "papers": 4
   }
  ],
  "comment": "9 pages, 9 figures. arXiv admin note: substantial text overlap with arXiv:2601.02778",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "sim2real",
   "rl-control",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.04940v1",
  "pdf_url": "https://arxiv.org/pdf/2607.04940v1",
  "html_url": "https://arxiv.org/html/2607.04940v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.85
 },
 {
  "id": "2607.04927",
  "slug": "dswam-a-dual-system-world-action-foundation-model-for-fine-grained-rob",
  "title": "DSWAM: A Dual-System World Action Foundation Model for Fine-Grained Robot Manipulation",
  "abstract": "World Action Models (WAMs) provide a promising alternative to Vision-Language-Action (VLA) policies by using video-based world modeling as dense supervision for robot action learning. Existing WAMs excel at physically grounded execution, but typically lack the explicit language-level planning interface in VLM-based VLAs for decomposing coarse instructions. Such decomposition becomes important when household tasks involve complex multi-step goals, where coarse user commands need to be converted into sequences of fine-grained executable subtasks. Meanwhile, the field still lacks a fair real-robot comparison between VLA and WAM execution capabilities, since existing systems often differ in data, robot embodiments, and task protocols. To address both the decomposition gap and the need for a controlled WAM-VLA comparison, we introduce DSWAM, a Dual-System World Action Foundation Model for fine-grained robot manipulation. DSWAM keeps a System 1 WAM executor as the default control path and optionally activates a System 2 vision-language subtask planner only when task decomposition is useful. The planner predicts executable subtasks from short-term visual history and a global task prompt, while the WAM executor performs world-aware action generation for each instruction or subtask. The executor is trained with action prediction and video co-training, but inference directly predicts action chunks without explicit future video generation. To make this execution path practical on real robots, we further integrate TensorRT acceleration, asynchronous execution, and real-time chunking (RTC) so that policy queries do not block robot control. To provide a fair real-robot comparison with VLA policies, we build and evaluate DSWAM under the DeMaVLA real-world deformable manipulation setting with matched robot platform, pretraining data, post-training data, and evaluation criteria.",
  "published": "2026-07-06",
  "updated": "2026-07-06",
  "year": "2026",
  "authors": [
   "Jian Zhu",
   "Jianjun Zhang",
   "Taiyi Su",
   "Tianbin Liu",
   "Zhangyuan Wang",
   "Kai Xie",
   "Zitai Huang",
   "Chong Ma",
   "Youzhang He",
   "Tianjian Wang",
   "Hanyang Wang",
   "Weihao Ding",
   "Yi Xu"
  ],
  "author_count": 13,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "DSWAM is introduced, a Dual-System World Action Foundation Model for fine-grained robot manipulation and built and evaluated under the DeMaVLA real-world deformable manipulation setting with matched robot platform, pretraining data, post-training data, and evaluation criteria.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jian Zhu",
    "id": "2310391063",
    "h_index": 0,
    "papers": 5
   },
   {
    "name": "Jianjun Zhang",
    "id": "2265461301",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Taiyi Su",
    "id": "2158513633",
    "h_index": 4,
    "papers": 16
   },
   {
    "name": "Tianbin Liu",
    "id": "2448144562",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Zhangyuan Wang",
    "id": "2281227857",
    "h_index": 4,
    "papers": 14
   },
   {
    "name": "Kai Xie",
    "id": "2448009248",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Zitai Huang",
    "id": "2392868432",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Chong Ma",
    "id": "2286256393",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Youzhang He",
    "id": "2213262766",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Tianjian Wang",
    "id": "2329455591",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Hanyang Wang",
    "id": "2291393860",
    "h_index": 7,
    "papers": 19
   },
   {
    "name": "Weihao Ding",
    "id": "2355370275",
    "h_index": 0,
    "papers": 7
   },
   {
    "name": "Yi Xu",
    "id": "2363513002",
    "h_index": 5,
    "papers": 9
   }
  ],
  "comment": "13 pages, 1 figures",
  "topics": [
   "world-models",
   "vla",
   "foundation-pretraining",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.04927v1",
  "pdf_url": "https://arxiv.org/pdf/2607.04927v1",
  "html_url": "https://arxiv.org/html/2607.04927v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2607.04745",
  "slug": "trajectory-anchor-optimization-for-overconfident-thermal-visual-place",
  "title": "Trajectory-Anchor Optimization for Overconfident Thermal Visual Place Recognition: Zero-Leakage OOD Auditing and Kidnapped-Robot Recovery",
  "abstract": "Modern thermal visual place recognition (TIR-VPR) frontends based on foundation models achieve remarkable closed-set retrieval but suffer from an overconfident forced-matching failure mode. Under out-of-distribution (OOD) or unmapped conditions, they generate highly plausible yet false loop candidates without a drop in similarity scores. While classical multi-hypothesis tracking (MHT) backends can mitigate these ambiguities by maintaining divergent trajectory beliefs, their exponential computational overhead violates real-time robotic constraints. To bridge this gap, we present Trajectory-Anchor Optimization (TAO). To counter the combinatorial challenge of evaluating parallel hypotheses (e.g., K=100), TAO compresses multi-view temporal verification into a batched SE(2) Procrustes alignment problem. By leveraging tensor-level vectorization and single-invocation batched SVD, this formulation bypasses the dynamic tree expansion of MHT, guaranteeing a strictly bounded per-frame execution loop of O(KN). Under a strict zero-leakage evaluation protocol, we show that while a passive geometric backend cannot mathematically separate metric localization errors from coherent hallucinations at a micro-scale (<5m) due to local visual ambiguities, TAO serves as an efficient fail-safe filter at a macro-scale. Within a 5m radius, hallucinations often possess a locally consistent geometry that deceives rigid alignment. However, beyond this threshold, the K=100 disparate hypotheses disperse spatially across the global map. This dispersion breaks the rigid temporal co-visibility constraint within the sliding window (N=20), causing the joint optimization residual to escalate sharply. Consequently, TAO establishes a distinct macroscopic convergence basin (10m) where multi-view geometric consistency reliably isolates catastrophic topological breaks and suppresses critical false acceptances.",
  "published": "2026-07-06",
  "updated": "2026-07-06",
  "year": "2026",
  "authors": [
   "Zhiyuan Lu",
   "Kanji Tanaka"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Trajectory-Anchor Optimization (TAO) is presented, showing that while a passive geometric backend cannot mathematically separate metric localization errors from coherent hallucinations at a micro-scale (<5m) due to local visual ambiguities, TAO serves as an efficient fail-safe filter at a macro-scale.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhiyuan Lu",
    "id": "2448110937",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Kanji Tanaka",
    "id": "2276489952",
    "h_index": 2,
    "papers": 18
   }
  ],
  "comment": "11 pages, 5 figures, technical report",
  "topics": [
   "sim2real",
   "foundation-pretraining",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.04745v1",
  "pdf_url": "https://arxiv.org/pdf/2607.04745v1",
  "html_url": "https://arxiv.org/html/2607.04745v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.04689",
  "slug": "a-reliable-context-aware-and-temporal-planning-framework-for-autonomou",
  "title": "A Reliable Context-Aware and Temporal Planning Framework for Autonomous Driving",
  "abstract": "Safe operation of autonomous vehicles in dense urban traffic depends on perception and planning that remain reliable when onboard sensing is degraded. In real driving conditions, camera observations are frequently corrupted by occlusion, motion blur, illumination change, and sensor noise, and when such degraded observations are aggregated indiscriminately over time, trajectory planning becomes unstable and collision risk rises for both the ego vehicle and surrounding road users. Recent Bird's-Eye-View (BEV) approaches unify perception and planning through a shared spatial representation, but most fuse temporal information across frames without assessing the reliability of the underlying observations. We present a Reliable Context-Aware and Temporal Planning framework for Autonomous Driving (RCT-AD) that explicitly models feature quality and temporal consistency to support safer, more consistent planning. A Reliable Context Awareness module scores per-frame reliability and selectively retains trustworthy features through a quality-gated First-In-Last-Out (FILO) memory mechanism, reconstructing degraded observations from reliable historical context so that corrupted inputs do not destabilize the scene representation. A Temporal Trajectory Planner captures long-term dependencies and multi-agent interactions to produce smoother, safety-aware trajectories, while a joint detection-and-segmentation head injects semantic and motion cues into the shared BEV space to strengthen scene understanding. Experiments on the nuScenes autonomous driving benchmark show that RCT-AD improves perception accuracy, motion prediction, and planning robustness over recent end-to-end baselines, achieving 61.5 nuScenes Detection Score, 52.9 mean Average Precision, and 52.3 mean Intersection over Union, while maintaining competitive computational efficiency suitable for real-time deployment.",
  "published": "2026-07-06",
  "updated": "2026-07-06",
  "year": "2026",
  "authors": [
   "Argho Dey",
   "Yunfei Yin",
   "Swachha Ray",
   "Md Minhazul Islam",
   "Zheng Yuan",
   "Sijing Xiong",
   "Hongyu Liu",
   "Zhiqiu Huang"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Experiments show that RCT-AD improves perception accuracy, motion prediction, and planning robustness over recent end-to-end baselines, while maintaining competitive computational efficiency suitable for real-time deployment.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Argho Dey",
    "id": "2360323756",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Yunfei Yin",
    "id": "2237299002",
    "h_index": 6,
    "papers": 21
   },
   {
    "name": "Swachha Ray",
    "id": "2399564107",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Md Minhazul Islam",
    "id": "2448362921",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Zheng Yuan",
    "id": "2302511959",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Sijing Xiong",
    "id": "5046850",
    "h_index": 19,
    "papers": 27
   },
   {
    "name": "Hongyu Liu",
    "id": "2446287149",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Zhiqiu Huang",
    "id": "2448021157",
    "h_index": 0,
    "papers": 1
   }
  ],
  "comment": "Submitted to IEEE Transactions on Intelligent Transportation Systems. 12 pages, 6 figures",
  "topics": [
   "spatial-3d",
   "navigation",
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.04689v1",
  "pdf_url": "https://arxiv.org/pdf/2607.04689v1",
  "html_url": "https://arxiv.org/html/2607.04689v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.04554",
  "slug": "hugs-guiding-unified-dexterous-grasp-synthesis-across-modes-and-scales",
  "title": "HUGS: Guiding Unified Dexterous Grasp Synthesis Across Modes and Scales via Learned Human Priors",
  "abstract": "Dexterous grasping across diverse object scales requires contact modes ranging from two-finger pinches to bimanual grasps. Existing dexterous grasp synthesis methods reduce the high-dimensional optimization space with manually designed expected contacts and initialization heuristics, which struggle to balance synthesis success rate and diversity. We present HUGS (Human-prior-guided Unified Dexterous Grasp Synthesis), a human-prior-guided framework for unified dexterous grasp synthesis across modes and scales. Instead of directly retargeting human demonstrations, HUGS learns an object-conditioned human prior that captures human grasp preferences and guides downstream force-closure-aware optimization. The prior is trained on a compact self-collected human grasp dataset with 1.8K grasps over 304 objects, providing broad coverage of object scales and contact modes. During synthesis, HUGS adaptively proposes contact modes and wrist initializations, substantially improving the balance between contact-mode coverage and synthesis success rate over heuristic-based methods. With HUGS, we synthesize 3.2M robotic grasps over 157K scenes, spanning object half-diagonal lengths from 2 cm to 30 cm and modes from two-finger to bimanual grasps. Models trained on the synthesized dataset autonomously select appropriate contact modes in the real world, enabling grasping from screws to large boxes.",
  "published": "2026-07-06",
  "updated": "2026-07-06",
  "year": "2026",
  "authors": [
   "Mingrui Yu",
   "Yongpeng Jiang",
   "Yongyi Jia",
   "Kangchen Lv",
   "Li Huang",
   "Yi Ren",
   "Xiang Li"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "HUGS (Human-prior-guided Unified Dexterous Grasp Synthesis), a human-prior-guided framework for unified dexterous grasp synthesis across modes and scales, and adaptively proposes contact modes and wrist initializations, substantially improving the balance between contact-mode coverage and synthesis success rate over heuristic-based methods.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Mingrui Yu",
    "id": "2047407832",
    "h_index": 9,
    "papers": 32
   },
   {
    "name": "Yongpeng Jiang",
    "id": "2258802294",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Yongyi Jia",
    "id": "2198480005",
    "h_index": 5,
    "papers": 17
   },
   {
    "name": "Kangchen Lv",
    "id": "1648745587",
    "h_index": 8,
    "papers": 15
   },
   {
    "name": "Li Huang",
    "id": "2448343273",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yi Ren",
    "id": "2276490366",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Xiang Li",
    "id": "2258789571",
    "h_index": 6,
    "papers": 14
   }
  ],
  "comment": "The first two authors contributed equally. Project website: https://hugs-dex.github.io/",
  "topics": [
   "dexterous-manipulation",
   "egocentric-data"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.04554v1",
  "pdf_url": "https://arxiv.org/pdf/2607.04554v1",
  "html_url": "https://arxiv.org/html/2607.04554v1",
  "code_url": "https://hugs-dex.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2607.04546",
  "slug": "mask2real-wm-segmentation-masks-as-a-sim-to-real-bridge-for-controllab",
  "title": "Mask2Real-WM: Segmentation Masks as a Sim-to-Real Bridge for Controllable Dexterous World Models",
  "abstract": "Action-conditioned world models allow robots to predict the future consequences of candidate actions without additional physical interaction, supporting policy evaluation, planning, and data augmentation. We present Mask2Real-WM, a two-stage action-conditioned world model for dexterous manipulation that decouples pixel prediction into a dynamics model and a rendering model. The dynamics model predicts future segmentation masks from past masks and 23-DoF action sequences. The rendering model maps the predicted masks to photorealistic RGB using a ControlNet-augmented Stable Video Diffusion backbone. The smaller sim-to-real gap in segmentation space enables the dynamics model to benefit from large-scale pretraining on over 50 h of synthetic simulation data, followed by fine-tuning on fewer than 2.5 h of real demonstrations. Experiments on a dexterous pick-and-place benchmark show that mask conditioning and simulation pretraining are both required for per-DoF action controllability across all 23 degrees of freedom. In contrast, monolithic baselines capture broad hand and end-effector trajectories but do not reliably reflect fine-grained, per-joint action effects.",
  "published": "2026-07-05",
  "updated": "2026-08-18",
  "year": "2026",
  "authors": [
   "Riccardo O. Feingold",
   "Davide Liconti",
   "Chenyu Yang",
   "Robert K. Katzschmann"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Mask2Real-WM is presented, a two-stage action-conditioned world model for dexterous manipulation that decouples pixel prediction into a dynamics model and a rendering model that shows that mask conditioning and simulation pretraining are both required for per-DoF action controllability across all 23 degrees of freedom.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Riccardo Feingold",
    "id": "2340019802",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Davide Liconti",
    "id": "2298269450",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Chenyu Yang",
    "id": "2329514965",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Robert K. Katzschmann",
    "id": "2139596377",
    "h_index": 16,
    "papers": 58
   }
  ],
  "comment": "23 pages, 24 figures, 4 tables. Preprint. Project page: https://srl-ethz.github.io/Mask2Real-WM/",
  "topics": [
   "world-models",
   "dexterous-manipulation",
   "sim2real",
   "foundation-pretraining",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.04546v2",
  "pdf_url": "https://arxiv.org/pdf/2607.04546v2",
  "html_url": "https://arxiv.org/html/2607.04546v2",
  "code_url": "https://srl-ethz.github.io/Mask2Real-WM/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2607.04426",
  "slug": "ace-brain-0-5-a-unified-embodied-foundational-model-for-physical-agent",
  "title": "ACE-Brain-0.5: A Unified Embodied Foundational Model for Physical Agentic AI",
  "abstract": "Embodied AI is moving from isolated perception or action modules toward physical agents that understand, plan under goals, act through robot bodies, monitor progress, and improve from experience. Existing systems address this loop only in parts: end-to-end policies generate actions but often lack spatial reasoning, planning, and execution assessment, while robot-agent systems orchestrate tools or specialists but do not learn a shared representation. This fragmentation limits general Physical Agentic AI. We present ACE-Brain-0.5, a unified embodied foundation model that organizes robot intelligence into five coupled functions: spatial perception, decision making, embodied interaction, self-monitoring, and self-improvement. Built on ACE-Brain-0, which established spatial intelligence as a shared scaffold across robot platforms, ACE-Brain-0.5 extends an understanding-centric model into a closed-loop foundation model. A single 8B backbone instantiates the first four functions: grounding objects and affordances, reasoning over 3D and egocentric spatial relations, decomposing instructions into subgoals, generating navigation and manipulation actions, and estimating progress for verification and recovery. To unify these capabilities without cross-task interference, we introduce SSR+, which extends Scaffold-Specialize-Reconcile with a Reactivate stage after task-vector merging. The fifth function, self-improvement, is realized by a companion framework that updates external execution state, including task schemas, spatial memory, and failure-recovery cases, from rollouts. Across fifteen benchmarks, ACE-Brain-0.5 improves over ACE-Brain-0 on 14 of 18 spatial perception and grounding benchmarks, achieves competitive navigation and manipulation performance, and provides strong progress estimation in ID and OOD settings. Together, these results mark an early step toward general Physical Agentic AI.",
  "published": "2026-07-05",
  "updated": "2026-07-05",
  "year": "2026",
  "authors": [
   "ACE-Brain Team",
   " :",
   "Ziyang Gong",
   "Haoming Gu",
   "Zehang Luo",
   "Tianyi Zhang",
   "Tao Tao",
   "Yixiao Chi",
   "Zhe Liu",
   "Lingsi Zhu",
   "Jingyuan Liu",
   "Anke Tang",
   "Songze Li",
   "Yilun Kong",
   "Ningjing Liu",
   "Tianyu Zhu",
   "Yunpeng Qing",
   "Shuang Luo",
   "Xiang Liu",
   "Shi Fu",
   "Dawei Nie",
   "Sixiang Liu",
   "Zhexi Wen",
   "Feng Pan",
   "Xiaofeng Wang",
   "Zhi Hou",
   "Chunxiao Liu",
   "Xue Yang",
   "Junchi Yan",
   "Hengshuang Zhao",
   "Dacheng Tao",
   "Xiaogang Wang"
  ],
  "author_count": 32,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 2,
  "influential_citations": 0,
  "tldr": "ACE-Brain-0.5 is presented, a unified embodied foundation model that organizes robot intelligence into five coupled functions: spatial perception, decision making, embodied interaction, self-monitoring, and self-improvement, and SSR+, which extends Scaffold-Specialize-Reconcile with a Reactivate stage after task-vector merging.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "ACE-Brain Team Ziyang Gong",
    "id": "2448010188",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Haoming Gu",
    "id": "2448014287",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Zehang Luo",
    "id": "2403015372",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Tianyi Zhang",
    "id": "2327147044",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Tao Tao",
    "id": "2324113672",
    "h_index": 1,
    "papers": 19
   },
   {
    "name": "Yixiao Chi",
    "id": "2366016380",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Zhe Liu",
    "id": "2404004658",
    "h_index": 2,
    "papers": 13
   },
   {
    "name": "Lingsi Zhu",
    "id": "2448017019",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Jingyuan Liu",
    "id": "2448653304",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "An-Liu Tang",
    "id": "1384642730",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Song Li",
    "id": "2447750974",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yilun Kong",
    "id": "2267333752",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Ningjing Liu",
    "id": "2448018852",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Tianyu Zhu",
    "id": "2448019047",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yunpeng Qing",
    "id": "2190750836",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Shuang Luo",
    "id": "2361707616",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Xiang Liu",
    "id": "2448164327",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Shi Fu",
    "id": "2284759814",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "D. Nie",
    "id": "2064196109",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Sixiang Liu",
    "id": "2448368599",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Zhexi Wen",
    "id": "2448015606",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Fengyi Pan",
    "id": "2446502700",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Xiaofeng Wang",
    "id": "2242976725",
    "h_index": 17,
    "papers": 57
   },
   {
    "name": "Zhi Hou",
    "id": "2068251872",
    "h_index": 9,
    "papers": 16
   },
   {
    "name": "Chunxiao Liu",
    "id": "2107926923",
    "h_index": 9,
    "papers": 17
   },
   {
    "name": "Xue Yang",
    "id": "2353618153",
    "h_index": 4,
    "papers": 18
   },
   {
    "name": "Junchi Yan",
    "id": "2344324060",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Hengshuang Zhao",
    "id": "2310758544",
    "h_index": 9,
    "papers": 26
   },
   {
    "name": "Dacheng Tao",
    "id": "2281943279",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Xiaogang Wang",
    "id": "2448923880",
    "h_index": 0,
    "papers": 2
   }
  ],
  "comment": "",
  "topics": [
   "egocentric-data",
   "spatial-3d",
   "navigation",
   "foundation-pretraining",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.04426v1",
  "pdf_url": "https://arxiv.org/pdf/2607.04426v1",
  "html_url": "https://arxiv.org/html/2607.04426v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.48
 },
 {
  "id": "2607.04144",
  "slug": "semantic-guided-progressive-object-removal-with-gaussian-splatting",
  "title": "Semantic-Guided Progressive Object Removal with Gaussian Splatting",
  "abstract": "Removing unwanted objects from reconstructed 3D scenes is an important task in computer vision, supporting applications in AR/VR, robotics, and digital content creation. Existing methods typically complete the entire masked region in a single step and without effectively utilizing semantic information from other views, leading to difficulties in handling complex geometric details and textures. In this work, we propose a novel framework that integrates Semantic-guided Block Matching (SBM) and Region-Wise Progressive Refinement (RPR) for high-quality 3D object removal. First, we leverage DINOv2 to encode semantic guidance from multi-view observations, and the best match tokens are decoded to complete missing regions in the target view while maintaining cross-view consistency. Second, we introduce a RPR strategy that segments the target mask into multiple subregions and selectively refines those with poor visual quality. Our method is built upon Gaussian Splatting, ensuring high-fidelity scene reconstruction with efficient computation. Experimental results demonstrate that our approach outperforms existing Gaussian-based methods in terms of perceptual quality and coherence in 3D object removal.",
  "published": "2026-07-05",
  "updated": "2026-07-05",
  "year": "2026",
  "authors": [
   "Xianliang Huang",
   "Chen Xiao",
   "Yuanxiang Ni",
   "Guanming Liu",
   "Mingkai Liu",
   "Dikai Fan",
   "Xiao Liu",
   "Hao Zhang"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes a novel framework that integrates Semantic-guided Block Matching (SBM) and Region-Wise Progressive Refinement (RPR) for high-quality 3D object removal, built upon Gaussian Splatting, ensuring high-fidelity scene reconstruction with efficient computation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xianliang Huang",
    "id": "2386423849",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Chen Xiao",
    "id": "2448015708",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yuanxiang Ni",
    "id": "2448007579",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Guanming Liu",
    "id": "2448022844",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Mingkai Liu",
    "id": "2340328236",
    "h_index": 2,
    "papers": 15
   },
   {
    "name": "Dikai Fan",
    "id": "2386017557",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Xiao Liu",
    "id": "2386429545",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Hao Zhang",
    "id": "2448019407",
    "h_index": 0,
    "papers": 2
   }
  ],
  "comment": "8 pages, 4 figures",
  "topics": [
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.04144v1",
  "pdf_url": "https://arxiv.org/pdf/2607.04144v1",
  "html_url": "https://arxiv.org/html/2607.04144v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.04127",
  "slug": "real-time-lidar-gaussian-splatting-slam",
  "title": "Real-Time LiDAR Gaussian Splatting SLAM",
  "abstract": "We present a real-time LiDAR-based framework for Gaussian Splatting SLAM that tightly couples fast G-ICP registration with spherical rasterization-based dense mapping for large-scale sequences. Leveraging LiDAR geometry rather than appearance, we reuse tracking-estimated local covariances to initialize Gaussians with range-aware scales and to derive surface normals for geometry-aware map optimization. We further introduce a covariance-derived geometry score that measures local complexity and drives pruning in planar regions and selective densification in structurally rich areas, while optimized Gaussians and LiDAR-specific confidence cues are fed back to improve tracking robustness. On the Newer College dataset, our method achieves an F-score of 86.78\\% using purely online trajectories at real-time speed ($>$20 FPS), and additional experiments on other datasets confirm its stability and scalability.",
  "published": "2026-07-05",
  "updated": "2026-07-05",
  "year": "2026",
  "authors": [
   "Seungjun Tak",
   "Yewon Jeon",
   "Jaeik Hwang",
   "SukMin Hwang",
   "Seongbo Ha",
   "Hyeonwoo Yu"
  ],
  "author_count": 6,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A real-time LiDAR-based framework for Gaussian Splatting SLAM that tightly couples fast G-ICP registration with spherical rasterization-based dense mapping for large-scale sequences and introduces a covariance-derived geometry score that measures local complexity and drives pruning in planar regions and selective densification in structurally rich areas.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Seungjun Tak",
    "id": "2393209665",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Y. Jeon",
    "id": "2448001927",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Jaeik Hwang",
    "id": "2448163359",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "SukMin Hwang",
    "id": "2448111927",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Seongbo Ha",
    "id": "2292348138",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Hyeonwoo Yu",
    "id": "2296087424",
    "h_index": 3,
    "papers": 13
   }
  ],
  "comment": "18 pages, 5 figures",
  "topics": [
   "spatial-3d",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.04127v1",
  "pdf_url": "https://arxiv.org/pdf/2607.04127v1",
  "html_url": "https://arxiv.org/html/2607.04127v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.03964",
  "slug": "worldscape-moe-a-unified-mixture-of-experts-world-model-for-scalable-h",
  "title": "Worldscape-MoE: A Unified Mixture-of-Experts World Model for Scalable Heterogeneous Action Control",
  "abstract": "World models are rapidly becoming a core infrastructure for embodied intelligence and interactive agents: they provide controllable simulators in which agents can perceive, act, forecast, and acquire scalable experience. Yet current video generation world models are still organized around isolated control interfaces, such as camera trajectories, robot actions, or hand-joint signals. This fragmentation is increasingly a scaling bottleneck. The central challenge is not the absence of controllable generators, but the lack of a unified and extensible learning framework that can absorb heterogeneous action supervision while preserving a shared model of world dynamics. In this work, we introduce Worldscape-MoE, a Mixture-of-Experts world model built on Diffusion Transformers for scalable heterogeneous action control. Our key observation is that different controls specify different interfaces to the same underlying world: although their representations differ, they constrain shared physical regularities, scene dynamics, and interaction semantics. Worldscape-MoE operationalizes this observation through modality-aware control injection, shared and control-specific experts, and a progressive MoE tuning strategy that supports continual extension to new action modalities. Experiments across locomotion, robotic manipulation, and egocentric hand control show that heterogeneous supervision improves rather than interferes with individual control capabilities. Worldscape-MoE achieves strong results on WorldArena, improves locomotion and hand-control metrics, exhibits robust out-of-distribution generalization, and demonstrates scaling behavior as additional control data and experts are integrated.",
  "published": "2026-07-04",
  "updated": "2026-07-04",
  "year": "2026",
  "authors": [
   "Jianjie Fang",
   "Yongyan Xu",
   "Ziyou Wang",
   "Chen Gao",
   "Yuchao Huang",
   "Zhaolu Wang",
   "Rongze Tang",
   "Mingyuan Jia",
   "Baining Zhao",
   "Weichen Zhang",
   "Xin Zhang",
   "Haisheng Su",
   "Yu Shang",
   "Wei Wu",
   "Xinlei Chen",
   "Yong Li"
  ],
  "author_count": 16,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work introduces Worldscape-MoE, a Mixture-of-Experts world model built on Diffusion Transformers for scalable heterogeneous action control that achieves strong results on WorldArena, improves locomotion and hand-control metrics, exhibits robust out-of-distribution generalization, and demonstrates scaling behavior as additional control data and experts are integrated.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jianjie Fang",
    "id": "2326466107",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Yongyang Xu",
    "id": "2327617757",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Ziyou Wang",
    "id": "2349399845",
    "h_index": 5,
    "papers": 16
   },
   {
    "name": "Chen Gao",
    "id": "2351212267",
    "h_index": 7,
    "papers": 26
   },
   {
    "name": "Yuchao Huang",
    "id": "2291442503",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Zhaolu Wang",
    "id": "2283761750",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Rongze Tang",
    "id": "2326410337",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Ming-Ming Jia",
    "id": "2322839115",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Baining Zhao",
    "id": "2215234748",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Weichen Zhang",
    "id": "2325910749",
    "h_index": 10,
    "papers": 41
   },
   {
    "name": "Xin Zhang",
    "id": "2333419153",
    "h_index": 4,
    "papers": 22
   },
   {
    "name": "Haisheng Su",
    "id": "2363482540",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Yu Shang",
    "id": "2325021108",
    "h_index": 8,
    "papers": 19
   },
   {
    "name": "Wei Wu",
    "id": "2302827202",
    "h_index": 5,
    "papers": 15
   },
   {
    "name": "Xinlei Chen",
    "id": "2300421419",
    "h_index": 7,
    "papers": 25
   },
   {
    "name": "Yong Li",
    "id": "2300459663",
    "h_index": 6,
    "papers": 22
   }
  ],
  "comment": "36 pages",
  "topics": [
   "world-models",
   "humanoids",
   "egocentric-data",
   "sim2real",
   "video-generation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.03964v1",
  "pdf_url": "https://arxiv.org/pdf/2607.03964v1",
  "html_url": "https://arxiv.org/html/2607.03964v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.03828",
  "slug": "objretarget-an-object-aware-motion-retargeting-framework-with-anthropo",
  "title": "ObjRetarget: An Object-Aware Motion Retargeting Framework with Anthropomorphic Arm Constraints and Polyhedral Hand Modeling",
  "abstract": "Learning robot dexterous manipulation from human manipulation videos requires reliably retargeting human intent to executable robot actions while maintaining stable hand-object contact, which remains a key challenge in embodied intelligence. Existing retargeting methods often ignore explicit contact modeling or rely on reinforcement learning, resulting in limited accuracy and generalization. To address this, we propose ObjRetarget, a human-to-robot motion retargeting framework for learning robot dexterous manipulation from human videos, which integrates anthropomorphic arm trajectory constraints with structured hand-object geometric modeling. For arm motion, reference trajectories extracted from human videos are used for initialization, followed by anthropomorphic constraints and redundancy-aware optimization to generate natural and accurate movements. For hand manipulation, ObjRetarget represents multi-finger contacts using polytope clusters and preserves contact structure through geometric invariants to improve stability. Experiments on real robots show that ObjRetarget improves manipulation success rates and contact stability across multiple dexterous tasks, and generalizes well to different demonstrations, object poses, and task settings.",
  "published": "2026-07-04",
  "updated": "2026-07-04",
  "year": "2026",
  "authors": [
   "Yuanchuan Lai",
   "Qing Gao",
   "Ziyan Liang",
   "Junjie Hu",
   "Zhaojie Ju"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yuanchuan Lai",
    "id": "2335167546",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Qing Gao",
    "id": "2260320039",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Ziyan Liang",
    "id": "2184599199",
    "h_index": 7,
    "papers": 27
   },
   {
    "name": "Junjie Hu",
    "id": "1409846329",
    "h_index": 17,
    "papers": 51
   },
   {
    "name": "Zhaojie Ju",
    "id": "2260351212",
    "h_index": 4,
    "papers": 26
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "egocentric-data",
   "rl-control",
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.03828v1",
  "pdf_url": "https://arxiv.org/pdf/2607.03828v1",
  "html_url": "https://arxiv.org/html/2607.03828v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.03723",
  "slug": "omnitactune-policy-agnostic-real-world-rl-for-tactile-residual-adaptat",
  "title": "OmniTacTune: Policy-Agnostic Real-World RL for Tactile Residual Adaptation of Visual Policies",
  "abstract": "Visual policies learned from human videos, teleoperation, and robot demonstrations offer scalable motion priors, but often fail in contact-rich manipulation, where success significantly depends on local force and contact geometry. Tactile sensing provides these complementary signals, yet tactile data remain costly to collect and hard to generalize across sensors, robots, and tasks. We introduce OmniTacTune, a policy-agnostic real-world RL pipeline that adapts tactile feedback to pretrained visual policies through residual correction. OmniTacTune uses a two-stage design: it first bootstraps tactile-aware learning from autonomous base-policy rollouts, then learns a lightweight tactile residual policy through online interaction. Extensive experiments show that OmniTacTune generalizes across diverse contact-rich tasks, visual base policies, and tactile representations. Across four real-world contact-rich tasks, it improves visual base policies from 5-40% success to 85-100% within 40-80 minutes, demonstrating an efficient path for adapting tactile feedback to scalable visual robot policies. Project page: https://colinyu1.github.io/omnitactune-site/",
  "published": "2026-07-04",
  "updated": "2026-07-04",
  "year": "2026",
  "authors": [
   "Kelin Yu",
   "Haode Zhang",
   "Harish Ravichandar",
   "Yunhai Han",
   "Ruohan Gao"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "OmniTacTune is introduced, a policy-agnostic real-world RL pipeline that adapts tactile feedback to pretrained visual policies through residual correction and generalizes across diverse contact-rich tasks, visual base policies, and tactile representations.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Kelin Yu",
    "id": "2261909068",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Haode Zhang",
    "id": "2135719214",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "H. Ravichandar",
    "id": "2138933",
    "h_index": 17,
    "papers": 74
   },
   {
    "name": "Yunhai Han",
    "id": "1995513527",
    "h_index": 8,
    "papers": 14
   },
   {
    "name": "Ruohan Gao",
    "id": "2376472371",
    "h_index": 2,
    "papers": 4
   }
  ],
  "comment": "Project page: https://colinyu1.github.io/omnitactune-site/",
  "topics": [
   "egocentric-data",
   "tactile",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.03723v1",
  "pdf_url": "https://arxiv.org/pdf/2607.03723v1",
  "html_url": "https://arxiv.org/html/2607.03723v1",
  "code_url": "https://colinyu1.github.io/omnitactune-site/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2607.03570",
  "slug": "cross-embodiment-robot-manipulation-via-a-unified-hand-action-space",
  "title": "Cross-Embodiment Robot Manipulation via a Unified Hand Action Space",
  "abstract": "Robot manipulation policies are typically tied to specific robotic hand embodiments, limiting the transfer of learned behaviors across platforms with different kinematic structures. In this work, we propose the Unified Hand Action Space (UHAS), a sphere-based unified action representation for cross-embodiment dexterous manipulation. UHAS represents robotic hand actions as geometric deformations of a canonical sphere and uses a Cascade Inverse Kinematics (CIK) algorithm to map the shared representation to embodiment-specific joint configurations. Using reinforcement learning, we train dexterous manipulation policies directly in the proposed action space for in-hand cube reorientation tasks. We evaluate our method in both simulation and real-world experiments across multiple robotic hands, including the Allegro Hand, LEAP Hand, Shadow Hand, and MANO Human Hand. Experimental results demonstrate effective dexterous manipulation, zero-shot transfer to unseen hands, rapid finetuning across embodiments, and successful real-world deployment. Our experiments show that the proposed UHAS representation enables stable dexterous control and cross-embodiment policy transfer across robotic hands.",
  "published": "2026-07-03",
  "updated": "2026-07-03",
  "year": "2026",
  "authors": [
   "Luis Felipe Casas",
   "Robert Teal",
   "Keval Shah",
   "Abhijit Tadepalli",
   "Wanxin Jin",
   "Yu Xiang"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Experimental results demonstrate effective dexterous manipulation, zero-shot transfer to unseen hands, rapid finetuning across embodiments, and successful real-world deployment of the proposed UHAS representation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Luis Felipe Casas",
    "id": "2395570707",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Robert Teal",
    "id": "2448007004",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Keval Shah",
    "id": "2448007531",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Abhijit Tadepalli",
    "id": "2424473424",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Wanxin Jin",
    "id": "2258938465",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Yu Xiang",
    "id": "2292006241",
    "h_index": 6,
    "papers": 17
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "rl-control",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.03570v1",
  "pdf_url": "https://arxiv.org/pdf/2607.03570v1",
  "html_url": "https://arxiv.org/html/2607.03570v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.03529",
  "slug": "current-as-touch-proprioceptive-contact-feedback-for-compliant-dextero",
  "title": "Current as Touch: Proprioceptive Contact Feedback for Compliant Dexterous Manipulation",
  "abstract": "Compliance is essential for dexterous manipulation, yet existing solutions often rely on external tactile or force sensors that are costly, fragile, and difficult to deploy on low-cost robot hands. We propose a proprioception-driven framework that learns contact-aware compliance cues from motor current and joint states. Since motor current is closely related to actuator torque, it provides an intrinsic signal for perceiving contact force, object resistance, and grasp stability without additional sensing hardware. Rather than estimating external wrenches or commanding torque, our method predicts a compliance reference position: an ideal joint-position target for a standard PD controller whose induced position error generates appropriate grasping force. This position-based formulation is compatible with mainstream teleoperation and policy-learning pipelines, while enabling the robot to adapt interaction forces from real-time proprioceptive feedback. Thus, motor current serves not only as a force proxy but also as a learnable proprioceptive contact signal for compliance reference prediction. Experiments on multiple dexterous hands and contact-rich tasks, including fragile object handling, sustained surface contact, thin-object retrieval, and dynamic load adaptation, show stable compliant grasping, safer and more efficient teleoperation, and improved downstream policy learning without external tactile or force sensors.",
  "published": "2026-07-03",
  "updated": "2026-07-03",
  "year": "2026",
  "authors": [
   "Chenyang Ma",
   "Yunchao Yao",
   "Zhenyu Wei",
   "Ruogu Li",
   "Daniel Szafir",
   "Mingyu Ding"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 2,
  "influential_citations": 0,
  "tldr": "A proprioception-driven framework that learns contact-aware compliance cues from motor current and joint states, which serves not only as a force proxy but also as a learnable proprioceptive contact signal for compliance reference prediction.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Chenyang Ma",
    "id": "2292350179",
    "h_index": 5,
    "papers": 19
   },
   {
    "name": "Yunchao Yao",
    "id": "2352910959",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Zhenyu Wei",
    "id": "2394124879",
    "h_index": 4,
    "papers": 16
   },
   {
    "name": "Ruogu Li",
    "id": "2448107268",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "D. Szafir",
    "id": "2259503818",
    "h_index": 6,
    "papers": 21
   },
   {
    "name": "Mingyu Ding",
    "id": "2346837065",
    "h_index": 4,
    "papers": 15
   }
  ],
  "comment": "Project website: https://cat.chenyangma.com",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "data-teleop",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.03529v1",
  "pdf_url": "https://arxiv.org/pdf/2607.03529v1",
  "html_url": "https://arxiv.org/html/2607.03529v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.48
 },
 {
  "id": "2607.03338",
  "slug": "closed-loop-vs-open-loop-kalman-filter-architectures-in-airborne-aided",
  "title": "Closed-loop vs. Open-loop Kalman Filter Architectures in Airborne Aided Inertial Navigation",
  "abstract": "Closed-loop (or feedback) error-state Kalman filters with their relatives and offspring are the state-of-the-art in modern aided inertial navigation research. Estimated inertial navigation system (INS) errors are continually fed back to the INS to correct the nominal system state before subsequent predictions. Conversely, in safety-critical aeronautical applications, open-loop (or feedforward) systems are an undisputed standard, where the inertial mechanization is strictly decoupled to allow for operational independence and fault isolation of computing units. We assess the performance impacts of this architectural choice beyond qualitative system-safety justifications using a standard inertial mechanization in geodetic coordinates and direct position aiding. Simulations using a variety of inertial sensor error characteristics, ranging from consumer to navigation grade systems, showcase the trade-off between smooth information fusion for high-end IMUs using an open-loop filter and the inherent long-term stability of the closed-loop architecture.",
  "published": "2026-07-03",
  "updated": "2026-07-03",
  "year": "2026",
  "authors": [
   "Antonia Hager",
   "Torleiv H. Bryne"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Simulations using a variety of inertial sensor error characteristics, ranging from consumer to navigation grade systems, showcase the trade-off between smooth information fusion for high-end IMUs using an open-loop filter and the inherent long-term stability of the closed-loop architecture.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Antonia Hager",
    "id": "2385805745",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "T. Bryne",
    "id": "9187557",
    "h_index": 13,
    "papers": 65
   }
  ],
  "comment": "6 pages, 6 figures, submission to IEEE Navicon",
  "topics": [
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.03338v1",
  "pdf_url": "https://arxiv.org/pdf/2607.03338v1",
  "html_url": "https://arxiv.org/html/2607.03338v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.03289",
  "slug": "invisible-strings-deriving-puppetry-principles-and-their-hidden-connec",
  "title": "Invisible Strings: Deriving Puppetry Principles and their Hidden Connections to Robot Behavior Design",
  "abstract": "When designing robots' nonverbal behaviors, many researchers have turned to arts-based insights, such as Disney's Animation Principles. Yet, while these principles bear key insights into the design of like-life characters, their application to robot design is inherently limited, in part because animation is not constrained by real-world physics, and in part because animation principles focus on low level animation mechanics and not high-level design considerations for physically embodied, interactive characters. In contrast, little attention has been paid to art forms like puppetry, despite their long history of exploring morphological, behavior, and interaction design of physically embodied, interactive characters. As such, in this work we leverage puppetry texts and practicing puppeteers' expert knowledge knowledge to derive a set of puppetry principles with key insights for robot design. As we show, these insights go beyond -- and uniquely complement -- the prior insights provided by theater, dance, and animation.",
  "published": "2026-07-03",
  "updated": "2026-07-03",
  "year": "2026",
  "authors": [
   "Claire Lewis",
   "Sawyer Collins",
   "Alyssa Hanson",
   "Johanna Smith",
   "Tom Williams"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.HC"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Puppetry texts and practicing puppeteers' expert knowledge knowledge are leveraged to derive a set of puppetry principles with key insights for robot design that go beyond -- and uniquely complement -- the prior insights provided by theater, dance, and animation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Claire Lewis",
    "id": "2303279652",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Sawyer Collins",
    "id": "39212150",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Alyssa Hanson",
    "id": "2211103300",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Johanna Smith",
    "id": "2448138008",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Tom Williams",
    "id": "2387634730",
    "h_index": 0,
    "papers": 3
   }
  ],
  "comment": "22 pages, 10 figures",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.03289v1",
  "pdf_url": "https://arxiv.org/pdf/2607.03289v1",
  "html_url": "https://arxiv.org/html/2607.03289v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.03209",
  "slug": "fast-3d-foundation-model-initialized-gaussian-splatting",
  "title": "Fast 3D Foundation Model Initialized Gaussian Splatting",
  "abstract": "This paper introduces a fast method for high-quality 3D Gaussian Splatting (3DGS) reconstruction without traditional Structure-from-Motion (SfM). The proposed approach leverages 3D Foundation Models (3DFMs) for camera pose and point-cloud initialization, then jointly optimizes both camera poses and Gaussian primitives using a depth-guided loss function. This enables fast convergence even from rough initialization with as few as 50-60 input views. To further improve reconstruction quality in sparse-view scenarios, an MLP-based pose refinement module is introduced alongside depth-guided supervision from the foundation model. Extensive experiments on Mip-NeRF 360, Tanks and Temples, and RobustNeRF demonstrate that the proposed method achieves competitive reconstruction quality (23.61 dB PSNR, 0.19 LPIPS) while reducing training time to approximately three minutes per scene. The proposed method produces ready-to-use 3DGS models at a fraction of the time required by existing pipelines, making it suitable for near real-time applications in robotics, VR, and autonomous navigation.",
  "published": "2026-07-03",
  "updated": "2026-07-03",
  "year": "2026",
  "authors": [
   "Anurag Dalal",
   "Daniel Hagen",
   "Kjell G. Robbersmyr",
   "Kristian Muri Knausg\u00e5rd"
  ],
  "author_count": 4,
  "categories": [
   "cs.CV",
   "cs.GR",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The proposed approach leverages 3D Foundation Models for camera pose and pointcloud initialization, then jointly optimizes both camera poses and Gaussian primitives using a depth-guided loss function to enable fast convergence even from rough initialization with as few as 50-60 input views.",
  "doi": "10.1109/ICECET65726.2026.11633076",
  "oa_pdf": "https://doi.org/10.48550/arxiv.2607.03209",
  "s2_authors": [
   {
    "name": "Anurag Dalal",
    "id": "2300098561",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Daniel Hagen",
    "id": "2183413930",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "K. Robbersmyr",
    "id": "1709298",
    "h_index": 25,
    "papers": 174
   },
   {
    "name": "Kristian Muri Knausg\u00e5rd",
    "id": "1414896180",
    "h_index": 6,
    "papers": 17
   }
  ],
  "comment": "8 pages, 5 figures, 5 tables. Accepted at ICECET 2026",
  "topics": [
   "spatial-3d",
   "navigation",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.03209v1",
  "pdf_url": "https://arxiv.org/pdf/2607.03209v1",
  "html_url": "https://arxiv.org/html/2607.03209v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.02764",
  "slug": "morphquad-morphable-quadrotor-for-superhuman-maneuverability-manipulat",
  "title": "MorphQuad: Morphable Quadrotor for Superhuman Maneuverability, Manipulation, and Resiliency",
  "abstract": "Infrastructure maintenance, contact-based inspection, and emergency response can benefit from aerial vehicles that act as a flying human hand with extreme maneuverability, manipulation, and resiliency (MMR): maneuverability to fly in arbitrary orientations to reach remote and tight locations; manipulation to point sensors, turn valves, and press tools at arbitrary orientations; resiliency to maintain accurate motion and force control despite disturbances from arbitrary directions, such as wind, ground effects, and friction. Realizing MMR on aerial vehicles requires not only omnidirectional flight; it also requires (I) vectoring of maximum thrust in any direction, to maximize capacity for contact-force application and disturbance rejection, (II) global stability, to enable control over any orientation/position, and (III) compact, standard designs that build upon platforms such as quadrotors to inherit technological know-how. No current aerial vehicle simultaneously enables I--III, due to structural and control limitations that constrain actuation. We present MorphQuad: a morphable quadrotor that enjoys MMR. Key to our approach is a hardware and control co-design: on hardware, we independently articulate each of the four rotor systems via two-axis gimbals; on control, we introduce globally-stable control, and energy-optimal thrust allocation that permits inter-rotor thrust cancellations only to avoid downwash interference and gimbal lock. With fully-onboard autonomy, MorphQuad demonstrates multi-revolution rotation while translating or hovering, for pipe inspection and target tracking (maneuverability); valve turning, perching, and object pressing and pushing with human-level strengths (manipulation); and wind rejection from any direction, even directed to a single rotor, and push-pull recovery (resiliency).",
  "published": "2026-07-02",
  "updated": "2026-07-29",
  "year": "2026",
  "authors": [
   "Jose Diaz Peon Gonzalez Pacheco",
   "Jiawei Xu",
   "Andrew Zhao",
   "Hongyu Zhou",
   "Atharva Navsalkar",
   "Andrew Scheffer",
   "Amrith Malli Reddi",
   "Sashreek Shankar",
   "Yuqing Bao",
   "Vasileios Tzoumas"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "10.31390/lsucontrol.26.35",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jose Diaz Peon Gonzalez Pacheco",
    "id": "2451866823",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Jiawei Xu",
    "id": "2110476942",
    "h_index": 6,
    "papers": 20
   },
   {
    "name": "Andrew Zhao",
    "id": "2054417511",
    "h_index": 6,
    "papers": 18
   },
   {
    "name": "Hongyu Zhou",
    "id": "2249544952",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Atharva Navsalkar",
    "id": "2154617829",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Andrew Scheffer",
    "id": "2412070612",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Amrith Malli Reddi",
    "id": "2451871119",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Sashreek Shankar",
    "id": "2363493527",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Yuqing Bao",
    "id": "2451888052",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Vasileios Tzoumas",
    "id": "1886606",
    "h_index": 20,
    "papers": 53
   }
  ],
  "comment": "",
  "topics": [
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.02764v4",
  "pdf_url": "https://arxiv.org/pdf/2607.02764v4",
  "html_url": "https://arxiv.org/html/2607.02764v4",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.02417",
  "slug": "lime-learning-intent-aware-camera-motion-from-egocentric-video",
  "title": "LIME: Learning Intent-aware Camera Motion from Egocentric Video",
  "abstract": "Autonomous robots often need to move their camera before they can act: to inspect an object, reveal an occluded region, or obtain a view that responds to a user's intent. While vision-language navigation translates instructions to base motion and vision-language-action policies map instructions to manipulation actions, language-conditioned camera motion remains comparatively underexplored as a first-class action. We formulate language-conditioned camera motion generation: given a current RGB observation and a free-form natural-language intent, predict a relative target camera pose for the next observation. This task is inherently non-trivial: viewpoint changes are driven by latent perceptual intentions, and a valid motion may operate at different semantic granularity, from entering a room to looking around a corner, inspecting a visible object, or revealing an occluded detail. To model this structure, we mine multi-intention camera-motion supervision from egocentric video, pairing plausible intents and observation-gain descriptions with relative SE(3) target poses. We propose LIME, a vision-language camera-motion generator that combines an auto-regressive observation-gain output with a continuous flow-matching pose head. This design lets the model jointly predict what the next view should reveal while representing multi-hypothesis target views. Across experiments and downstream robotic tasks, we show that LIME can learn to actively choose camera poses from passive human video, turning ordinary egocentric recordings into supervision for intent-aware active perception.",
  "published": "2026-07-02",
  "updated": "2026-07-02",
  "year": "2026",
  "authors": [
   "Boyang Sun",
   "Jiajie Li",
   "Yung-Hsu Yang",
   "Chenyangguang Zhang",
   "Tim Engelbracht",
   "Sunghwan Hong",
   "Cesar Cadena",
   "Marc Pollefeys",
   "Hermann Blum"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "LIME is proposed, a vision-language camera-motion generator that combines an auto-regressive observation-gain output with a continuous flow-matching pose head that can learn to actively choose camera poses from passive human video, turning ordinary egocentric recordings into supervision for intent-aware active perception.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Boyang Sun",
    "id": "2256489737",
    "h_index": 7,
    "papers": 23
   },
   {
    "name": "Jiajie Li",
    "id": "2309076585",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Yung-Hsu Yang",
    "id": "2321645939",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Chenyangguang Zhang",
    "id": "2400048384",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Tim Engelbracht",
    "id": "2363878466",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Sung\u2010Jin Hong",
    "id": "2153119782",
    "h_index": 13,
    "papers": 32
   },
   {
    "name": "C\u00e9sar Cadena",
    "id": "2586813",
    "h_index": 39,
    "papers": 119
   },
   {
    "name": "Marc Pollefeys",
    "id": "2263467446",
    "h_index": 19,
    "papers": 98
   },
   {
    "name": "Hermann Blum",
    "id": "2253614133",
    "h_index": 9,
    "papers": 25
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "egocentric-data",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.02417v1",
  "pdf_url": "https://arxiv.org/pdf/2607.02417v1",
  "html_url": "https://arxiv.org/html/2607.02417v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.01860",
  "slug": "dl-slam-enabling-high-fidelity-gaussian-splatting-slam-in-dynamic-envi",
  "title": "DL-SLAM: Enabling High-Fidelity Gaussian Splatting SLAM in Dynamic Environments based on Dual-Level Probability",
  "abstract": "Recent advances in 3D Gaussian Splatting (3DGS) have enabled significant progress in dense dynamic Simultaneous Localization And Mapping (SLAM). Prevailing methods typically discard predefined dynamic objects, ignoring that transiently static objects offer valuable geometric constraints for pose estimation. A recent work attempts to leverage this potential by employing per-pixel uncertainty maps to quantify the magnitude of motion. While this approach enables transiently static objects to enhance pose estimation, it erroneously integrates these objects into the static map, resulting in persistent artifacts. Moreover, its reliance on purely geometric information leads to ambiguous object boundaries in the uncertainty maps. To overcome these limitations, we present DL-SLAM, a monocular Gaussian Splatting SLAM system built upon a novel dual-level probabilistic framework. Our method computes dynamic probability maps by combining semantic and geometric information. These pixel-level probabilities are lifted to 3D and aggregated to derive an object-level dynamic probability for each instance. Object-level probability enables the categorical pruning of dynamic Gaussians, resulting in an artifact-free static map. The static map, in turn, provides a geometrically consistent guidance to refine the pixel-wise probabilities, enhancing their reliability. Experimental results demonstrate that DL-SLAM outperforms existing approaches, improving tracking accuracy by up to 13\\% while generating high-fidelity semantic maps.",
  "published": "2026-07-02",
  "updated": "2026-07-03",
  "year": "2026",
  "authors": [
   "Ziheng Xu",
   "Qingfeng Li",
   "Xuefeng Liu",
   "Chen Chen",
   "Jianwei Niu"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "DL-SLAM is presented, a monocular Gaussian Splatting SLAM system built upon a novel dual-level probabilistic framework that outperforms existing approaches, improving tracking accuracy by up to 13\\% while generating high-fidelity semantic maps.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ziheng Xu",
    "id": "2277568424",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Qingfeng Li",
    "id": "2244628158",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Xuefeng Liu",
    "id": "48033068",
    "h_index": 16,
    "papers": 98
   },
   {
    "name": "Chen Chen",
    "id": "2277451300",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Jianwei Niu",
    "id": "2315312838",
    "h_index": 2,
    "papers": 8
   }
  ],
  "comment": "",
  "topics": [
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.01860v2",
  "pdf_url": "https://arxiv.org/pdf/2607.01860v2",
  "html_url": "https://arxiv.org/html/2607.01860v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.01803",
  "slug": "pixgs-pixel-space-diffusion-for-direct-3d-gaussian-splat-generation",
  "title": "PixGS: Pixel-Space Diffusion for Direct 3D Gaussian Splat Generation",
  "abstract": "Recent advances in 3D content generation from text or images have achieved impressive results, yet view inconsistency from 2D generators and the scarcity of high-quality 3D data remain significant bottlenecks. Existing solutions typically adapt large-scale pre-trained text-to-image latent diffusion models to generate 3D Gaussian Splats (3DGS). However, these approaches often rely on training complex cascade pipelines that are computationally expensive and scalability-limited. Most critically, the quality of generated 3D assets is inherently constrained by each component capacity and compressed latent space, leading to decoding artifacts and accumulated errors. To address these limitations, we propose PixGS, a single-stage pipeline for direct high-quality 3DGS generation, which leverages recent advances in pixel-space diffusion to bypass lossy latent compression while still benefiting from the vast 2D generative priors. By directly denoising 3D Gaussian attributes at each timestep, our method enables precise, splat-level regularization of both appearance and geometry. Furthermore, we introduce a comprehensive supervision strategy that incorporates surface normals, depth, and high-frequency structural information, which is often overlooked in prior works. Experiments demonstrate that PixGS outperforms current state-of-the-art methods while maintaining a fast inference speed (1s on a single A100 GPU), offering a robust and efficient alternative to multi-stage generation pipelines.",
  "published": "2026-07-02",
  "updated": "2026-07-04",
  "year": "2026",
  "authors": [
   "Cao Duy",
   "Phong Nguyen-Ha"
  ],
  "author_count": 2,
  "categories": [
   "cs.CV",
   "cs.GR",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "ECCV 2026",
  "venue_source": "arxiv-comment",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "PixGS is proposed, a single-stage pipeline for direct high-quality 3DGS generation, which leverages recent advances in pixel-space diffusion to bypass lossy latent compression while still benefiting from the vast 2D generative priors, offering a robust and efficient alternative to multi-stage generation pipelines.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "C. Duy",
    "id": "2352324326",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Phong Nguyen-Ha",
    "id": "1405405282",
    "h_index": 3,
    "papers": 8
   }
  ],
  "comment": "Accepted at ECCV 2026",
  "topics": [
   "spatial-3d",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.01803v2",
  "pdf_url": "https://arxiv.org/pdf/2607.01803v2",
  "html_url": "https://arxiv.org/html/2607.01803v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.8
 },
 {
  "id": "2607.01367",
  "slug": "simulation-based-reward-function-validation-for-multi-agent-on-orbit-i",
  "title": "Simulation Based Reward Function Validation for Multi-Agent On Orbit Inspection",
  "abstract": "A proposed method for the control of groups of inspection spacecraft is Multi-Agent Reinforcement Learning (MARL). While MARL has already been employed for this purpose in previous work, the reward functions used focus on reaching a finite set of predetermined inspection points around the target. In this work, we study and develop a generalized reward function for the MARL inspection task informed by the analysis of 3D reconstructions of inspected objects in orbit. Because the reward function is generalized such that any number of images at arbitrary locations may evaluated, we also allow trained agents to have complete control over when images are collected. With this approach, we gather insights into best practices for not only the specific MARL inspection task, but also gain key takeaways informative to the broader inspection task outside of a MARL context.",
  "published": "2026-07-01",
  "updated": "2026-07-01",
  "year": "2026",
  "authors": [
   "Patrick Quinn",
   "Bala Prenith Reddy Gopu",
   "George M. Nehma",
   "Madhur Tiwari"
  ],
  "author_count": 4,
  "categories": [
   "cs.MA",
   "cs.RO"
  ],
  "primary_category": "cs.MA",
  "venue": "AIAA SCITECH 2026 Forum",
  "venue_source": "semantic-scholar",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "This work studies and develops a generalized reward function for the MARL inspection task informed by the analysis of 3D reconstructions of inspected objects in orbit and allows trained agents to have complete control over when images are collected.",
  "doi": "10.2514/6.2026-2042 10.2514/6.2026-2042.c1",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Patrick Quinn",
    "id": "2343638282",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Bala Vikram Reddy Gopu",
    "id": "150934637",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "George M. Nehma",
    "id": "2238621316",
    "h_index": 3,
    "papers": 14
   },
   {
    "name": "Madhur Tiwari",
    "id": "2238569121",
    "h_index": 3,
    "papers": 18
   }
  ],
  "comment": "13 pages, 6 figures. This submission integrates a published correction made to the original manuscript. The DOIs for both the original manuscript as well as the correction are provided",
  "topics": [
   "rl-control",
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.01367v1",
  "pdf_url": "https://arxiv.org/pdf/2607.01367v1",
  "html_url": "https://arxiv.org/html/2607.01367v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.8
 },
 {
  "id": "2607.01200",
  "slug": "fastbridge-closing-the-model-based-realization-gap-in-safety-filters-o",
  "title": "FastBridge: Closing the Model-Based Realization Gap in Safety Filters on 3D Gaussian Splatting for Fast Quadrotor Flight",
  "abstract": "Fast quadrotor flight requires safe obstacle avoidance under tight onboard compute limits. While 3D Gaussian Splatting (3DGS) provides a continuous, geometry-aware scene representation for perception-driven navigation, existing 3DGS safety filters use reduced-order models such as single- and double-integrators that ignore actuator limits and assume commanded accelerations are realized instantaneously. Building on an analytic collision cone barrier for 3DGS, we introduce a nonlinear, actuator-aware safety filter enforced through the full quadrotor dynamics. We derive a high-relative-degree collision cone exponential CBF and a backup CBF that preserves QP feasibility under input constraints using a forward-simulated backup policy. Compared with a state-of-the-art 3DGS safety filter, our approach reduces trajectory jerk by 47% and runs 2.25 times faster. We validate the method in simulation and on hardware for real-time navigation in cluttered, perception-derived environments.",
  "published": "2026-07-01",
  "updated": "2026-07-01",
  "year": "2026",
  "authors": [
   "Tscholl Dario",
   "Nakka Yashwanth Kumar",
   "Gunter Brian"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work derives a high-relative-degree collision cone exponential CBF and a backup CBF that preserves QP feasibility under input constraints using a forward-simulated backup policy and validate the method in simulation and on hardware for real-time navigation in cluttered, perception-derived environments.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Tscholl Dario",
    "id": "2446040753",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Nakka Yashwanth Kumar",
    "id": "2446316709",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Gunter Brian",
    "id": "2446036870",
    "h_index": 0,
    "papers": 1
   }
  ],
  "comment": "preprint, 9 pages, 4 figures",
  "topics": [
   "spatial-3d",
   "navigation",
   "hardware-codesign",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.01200v1",
  "pdf_url": "https://arxiv.org/pdf/2607.01200v1",
  "html_url": "https://arxiv.org/html/2607.01200v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2607.01067",
  "slug": "human-centric-transferable-tactile-pre-training-for-dexterous-robotic",
  "title": "Human-Centric Transferable Tactile Pre-Training for Dexterous Robotic Manipulation",
  "abstract": "As an essential modality for dexterous and contact-rich tasks, tactile sensing provides precise force feedback that cannot be reliably inferred from vision. However, limited by hardware and data collection systems, existing datasets with tactility remain small in scale and narrow in contact coverage. Meanwhile, Vision-Language-Action (VLA) models with tactile modality are constrained on dynamics-agnostic post-training, which limits the performance ceiling on downstream tasks. In this paper, we present H-Tac, a large-scale tactile-action dataset with 160-hour egocentric human videos containing more than 300 tasks and 135k episodes. Building upon this, we propose Transferable Tactile Pre-Training (TTP), a system of tactile-based pre-training on human data for fine-grained robotic tasks. To bridge the gap between humans and robots, we use unified tactile and action spaces throughout the pre-training and post-training phases, preserving prior knowledge during human-to-robot transfer. By leveraging a tactile expert for future tactile prediction, our framework explicitly models the contact dynamics and precise physical interactions. Extensive experiments in simulation and on real robots demonstrate that our model achieves superior performance, exhibiting robust generalization and fine-grained manipulation capabilities. TTP paves the way for scalable tactile pre-training via human-to-robot transfer.",
  "published": "2026-07-01",
  "updated": "2026-07-01",
  "year": "2026",
  "authors": [
   "Chi Zhang",
   "Penglin Cai",
   "Ziheng Xi",
   "Haoqi Yuan",
   "Hao Luo",
   "Wanpeng Zhang",
   "Sipeng Zheng",
   "Chaoyi Xu",
   "Zongqing Lu"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 3,
  "influential_citations": 0,
  "tldr": "Transferable Tactile Pre-Training (TTP) is proposed, a system of tactile-based pre-training on human data for fine-grained robotic tasks that paves the way for scalable tactile pre-training via human-to-robot transfer.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Chi Zhang",
    "id": "2269711204",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Penglin Cai",
    "id": "2139688739",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Ziheng Xi",
    "id": "2405809088",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Haoqi Yuan",
    "id": "1429192914",
    "h_index": 12,
    "papers": 37
   },
   {
    "name": "Hao Luo",
    "id": "2199828096",
    "h_index": 9,
    "papers": 20
   },
   {
    "name": "Wanpeng Zhang",
    "id": "2324490975",
    "h_index": 8,
    "papers": 23
   },
   {
    "name": "Sipeng Zheng",
    "id": "2258682220",
    "h_index": 11,
    "papers": 24
   },
   {
    "name": "Chaoyi Xu",
    "id": "2373397887",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Zongqing Lu",
    "id": "2258676670",
    "h_index": 29,
    "papers": 144
   }
  ],
  "comment": "The first two authors contribute equally. Orders are decided by flipping a coin",
  "topics": [
   "vla",
   "dexterous-manipulation",
   "egocentric-data",
   "tactile",
   "foundation-pretraining",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.01067v1",
  "pdf_url": "https://arxiv.org/pdf/2607.01067v1",
  "html_url": "https://arxiv.org/html/2607.01067v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.6
 },
 {
  "id": "2607.00776",
  "slug": "from-prediction-uncertainty-to-conformalized-distance-fields-for-safe",
  "title": "From Prediction Uncertainty to Conformalized Distance Fields for Safe Motion Planning",
  "abstract": "Safe motion planning in dynamic environments requires reasoning about the uncertainty in predicted obstacle motion without sacrificing real-time performance. Existing conformal approaches conformalize a scalar score that aggregates per-obstacle prediction errors, losing spatial coherence and scaling poorly with scene density. We instead conformalize the entire predicted distance field at once. This functional conformal prediction (FCP) framework yields a distribution-free, field-level lower bound, from which safety follows uniformly: any trajectory satisfying the resulting constraint is certified safe, independent of how the control space is sampled. The key enabler is that the residual distance field is empirically low-rank and approximately time-invariant, which makes the bound decomposable in coefficient space. An envelope is fitted offline via functional PCA and a Gaussian-mixture inductive conformal procedure, then refined online by a lightweight adaptive functional conformal (AFCP) update on a low-dimensional vector. This keeps the per-step cost largely insensitive to obstacle count and retains long-run field coverage under distribution shift. We embed the envelope as a tightened safety constraint in a sampling-based model predictive controller, FCP-MPC. On the ETH--UCY pedestrian benchmarks and a dense 3D quadrotor task with up to 280 dynamic obstacles, FCP-MPC attains a favorable balance of safety, feasibility, and efficiency, reaching goals where pointwise and egocentric conformal baselines become too conservative or too expensive, while keeping per-step computation far below online uncertainty-reasoning baselines.",
  "published": "2026-07-01",
  "updated": "2026-07-01",
  "year": "2026",
  "authors": [
   "Jaeuk Shin",
   "Yoonseok Ra",
   "Insoon Yang"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "FCP-MPC attains a favorable balance of safety, feasibility, and efficiency, reaching goals where pointwise and egocentric conformal baselines become too conservative or too expensive, while keeping per-step computation far below online uncertainty-reasoning baselines.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jaeuk Shin",
    "id": "77965190",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Yoonseok Ra",
    "id": "2446035325",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Insoon Yang",
    "id": "2353070260",
    "h_index": 1,
    "papers": 5
   }
  ],
  "comment": "",
  "topics": [
   "egocentric-data",
   "rl-control",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.00776v1",
  "pdf_url": "https://arxiv.org/pdf/2607.00776v1",
  "html_url": "https://arxiv.org/html/2607.00776v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2607.00673",
  "slug": "path-planning-in-physically-viable-world-models",
  "title": "Path Planning in Physically Viable World Models",
  "abstract": "Robots deployed in unstructured outdoor environments often plan from scene reconstructions collected before deployment because operators cannot remap large or remote sites before every mission. As a result, robots must make long-horizon planning decisions using stale maps that assume the terrain remains unchanged, even though physical changes to the environment may render previously feasible routes unsafe or unreachable at execution time. We present a physically viable world model for evaluating what-if queries for robot navigation under future terrain change. The system augments reconstructed 3D Gaussian splat scenes with physics-based simulation to generate physically modified versions of the same environment without recollecting sensor data or rebuilding the map. We then implement a terrain-aware planner that accounts for physical events, obstacles, and deformations that are simulated by the world model. This allows robots and human operators to evaluate whether planned routes remain feasible before committing to a planned route, particularly in constrained environments where retreat or recovery may become impossible once conditions change. We evaluate the system on a real outdoor field site in Central Texas using simulated flooding across multiple severity levels. We measure route and mission feasibility as terrain conditions deteriorate under physically simulated interventions. Our results show that physically viable world models expose long-horizon route failures and rerouting behavior that are not apparent when planning only on the original reconstructed environment, allowing robots to evaluate how future terrain changes may affect route feasibility before deployment.",
  "published": "2026-07-01",
  "updated": "2026-07-01",
  "year": "2026",
  "authors": [
   "Su Ann Low",
   "Cheng-Hsi Hsiao",
   "Xingjian Li",
   "Adam J. Thorpe",
   "Ufuk Topcu",
   "Krishna Kumar"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A physically viable world model is presented for evaluating what-if queries for robot navigation under future terrain change and shows that physically viable world models expose long-horizon route failures and rerouting behavior that are not apparent when planning only on the original reconstructed environment.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "S. Low",
    "id": "1721044620",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Cheng-Hsi Hsiao",
    "id": "2298266356",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Xingjian Li",
    "id": "2355767966",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Adam J. Thorpe",
    "id": "2243272973",
    "h_index": 4,
    "papers": 14
   },
   {
    "name": "U. Topcu",
    "id": "3199888",
    "h_index": 54,
    "papers": 610
   },
   {
    "name": "Krishna Kumar",
    "id": "2323907405",
    "h_index": 3,
    "papers": 17
   }
  ],
  "comment": "18 pages, 7 figures, submitted to CORL",
  "topics": [
   "world-models",
   "spatial-3d",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.00673v1",
  "pdf_url": "https://arxiv.org/pdf/2607.00673v1",
  "html_url": "https://arxiv.org/html/2607.00673v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2607.00424",
  "slug": "robust-operational-space-control-with-conformal-disturbance-bounds-for",
  "title": "Robust Operational Space Control with Conformal Disturbance Bounds for Safe Redundant Manipulation",
  "abstract": "Redundant robotic manipulators operating in constrained and human-interactive environments require accurate task-space tracking together with rigorous safety guarantees under dynamic uncertainties. Classical operational space computed torque controller (OSCTC) relies on accurate dynamic models and degrades in the presence of disturbances. In contrast, the data-driven paradigm of residual learning approximates disturbances as functions learned from full-state measurements, which are often noisy in practice, lack rigorous theoretical guarantees, and introduce additional design complexity. This paper proposes a robust OSCTC framework that integrates an extended state observer (ESO) with conformal prediction to combine model-based robustness and data-driven adaptability. The ESO estimates lumped disturbances directly in operational space without requiring full-state measurements as in residual learning, and a robust control barrier function (CBF) is constructed to enforce safety under uncertainty. However, robust CBFs require a known disturbance-variation bound to guarantee absolute safety, which often leads to conservatism in practice. To address this limitation, we further employ a sliding-window conformal prediction mechanism to estimate the bound online in a distribution-free manner, thereby achieving practical probabilistic safety guarantees. Experiments on a 7-DoF Franka Research 3 manipulator demonstrate millimeter-level tracking accuracy and real-time safe control at 1~kHz under various disturbances.",
  "published": "2026-07-01",
  "updated": "2026-07-01",
  "year": "2026",
  "authors": [
   "Wenhua Liu",
   "Fan Zhang",
   "Qin Lin"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "A robust OSCTC framework that integrates an extended state observer (ESO) with conformal prediction with conformal prediction to combine model-based robustness and data-driven adaptability is proposed, thereby achieving practical probabilistic safety guarantees.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Wenhua Liu",
    "id": "2446307293",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Fan Zhang",
    "id": "2352524005",
    "h_index": 0,
    "papers": 5
   },
   {
    "name": "Qin Lin",
    "id": "2303522877",
    "h_index": 2,
    "papers": 8
   }
  ],
  "comment": "Paper accepted to IROS 2026",
  "topics": [
   "rl-control",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.00424v1",
  "pdf_url": "https://arxiv.org/pdf/2607.00424v1",
  "html_url": "https://arxiv.org/html/2607.00424v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.8
 },
 {
  "id": "2607.00160",
  "slug": "distributed-multi-robot-lunar-cargo-transportation-via-phase-decompose",
  "title": "Distributed Multi Robot Lunar Cargo Transportation via Phase Decomposed Reinforcement Learning",
  "abstract": "Modular reconfigurable robotic systems provide a scalable solution for cooperative surface operations in future lunar missions. However, cooperative cargo transportation remains challenging due to morphology-dependent topology changes, strong payload-induced coupling, long-horizon decision making, and safety constraints. This paper proposes a phase-decomposed reinforcement learning framework for cooperative cargo transport with distributed robotic units. The task is decomposed into lifting, transportation, and placement, each optimized with a dedicated joint-state policy capturing inter-agent coupling. Centralized training promotes stable convergence, while deployment uses onboard proprioception for control and OptiTrack motion capture for ground-truth evaluation and post-processed metrics. A deterministic phase controller expressed in Markov state representation regulates transitions between stages, and a failure-sensitive synchronization mechanism ensures coordinated progression and safety-aware halting during real-world execution. The framework is evaluated in simulation and through controlled field experiments at a JAXA space exploration test facility. Results demonstrate reliable cooperative transport across all stages in both simulation and hardware experiments.",
  "published": "2026-06-30",
  "updated": "2026-06-30",
  "year": "2026",
  "authors": [
   "Ashutosh Mishra",
   "Elian Neppel",
   "Shreya Santra",
   "Antoine Jonqui\u00e8res",
   "Muhammad Athallah Naufal",
   "Kentaro Uno",
   "Kazuya Yoshida"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A phase-decomposed reinforcement learning framework for cooperative cargo transport with distributed robotic units, decomposed into lifting, transportation, and placement, each optimized with a dedicated joint-state policy capturing inter-agent coupling is proposed.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ashutosh Mishra",
    "id": "2344597510",
    "h_index": 2,
    "papers": 11
   },
   {
    "name": "Elian Neppel",
    "id": "2372258648",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Shreya Santra",
    "id": "2343401068",
    "h_index": 2,
    "papers": 17
   },
   {
    "name": "Antoine Jonquieres",
    "id": "2446040249",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "M. Naufal",
    "id": "2446009340",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Kentaro Uno",
    "id": "2131338407",
    "h_index": 8,
    "papers": 63
   },
   {
    "name": "Kazuya Yoshida",
    "id": "2237807517",
    "h_index": 6,
    "papers": 43
   }
  ],
  "comment": "8 pages, 9 Figures, Accepted at IROS2026",
  "topics": [
   "egocentric-data",
   "rl-control",
   "navigation",
   "hardware-codesign",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.00160v1",
  "pdf_url": "https://arxiv.org/pdf/2607.00160v1",
  "html_url": "https://arxiv.org/html/2607.00160v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.32009",
  "slug": "human-as-humanoid-enabling-zero-shot-humanoid-learning-from-ego-exo-hu",
  "title": "Human-as-Humanoid: Enabling Zero-Shot Humanoid Learning from Ego-Exo Human Videos with Human-Aligned Embodiments",
  "abstract": "Vision-language-action (VLA) models across robot embodiments require high-quality observation--action supervision to learn deployable action distributions, yet scaling such robot data remains difficult, especially for high-DoF humanoids. Teleoperation provides controller-aligned supervision, while human egocentric videos capture diverse bimanual manipulation but do not directly provide executable robot actions. We introduce Human-as-Humanoid, a human-to-humanoid supervision framework that enables near-real-time human-centric action generation, making human demonstrations usable for high-DoF humanoid VLA training by jointly aligning the robot embodiment, the sensing setup, and the action-label interface. Built on PrimeU, a human-aligned 60-DoF upper-body humanoid, Human-as-Humanoid uses synchronized ego-exo videos to pair deployment-aligned egocentric observations with exocentric motion recovery, retargets the recovered human motion through staged Inverse Kinematics (IK) into controller-aligned 60-DoF action chunks, and trains the VLA model with Forward Kinematics (FK)-aware supervision to preserve wrist and fingertip task-space geometry. This converts large-scale human demonstrations from visual observations into executable observation--action supervision for the target humanoid. Experiments validate the conversion chain at the motion-recovery, robot-action-space, and real-robot deployment levels. Human-as-Humanoid yields a 4.8--7.2x raw demonstration-throughput gain over humanoid teleoperation in our data-collection analysis, and on several downstream tasks, policies post-trained only with the converted human labels generalize to real-robot deployment without target-task robot demonstrations. The official project website is available at https://zgc-embodyai.github.io/Human-as-Humanoid.",
  "published": "2026-06-30",
  "updated": "2026-06-30",
  "year": "2026",
  "authors": [
   "Xiaopeng Lin",
   "Ruoqi Yang",
   "Shijie Lian",
   "Zhaolong Shen",
   "Bin Yu",
   "Changti Wu",
   "Haibao Liu",
   "Yuxiang Zhang",
   "Hong Li",
   "Qiyuan Su",
   "Haochen Liu",
   "Xuguo He",
   "Yukun Shi",
   "Cong Huang",
   "Zhirui Zhang",
   "Bojun Cheng",
   "Kai Chen"
  ],
  "author_count": 17,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Human-as-Humanoid is introduced, a human-to-humanoid supervision framework that enables near-real-time human-centric action generation, making human demonstrations usable for high-DoF humanoid VLA training by jointly aligning the robot embodiment, the sensing setup, and the action-label interface.",
  "doi": "10.48550/arXiv.2606.32009",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xiaopeng Lin",
    "id": "2406490921",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Ruoqi Yang",
    "id": "2377921274",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Shijie Lian",
    "id": "2225550405",
    "h_index": 7,
    "papers": 24
   },
   {
    "name": "Zhaolong Shen",
    "id": "2406950797",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Bin Yu",
    "id": "2332297389",
    "h_index": 6,
    "papers": 17
   },
   {
    "name": "Changti Wu",
    "id": "2293554408",
    "h_index": 6,
    "papers": 17
   },
   {
    "name": "Haibao Liu",
    "id": "2445892353",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yuxiang Zhang",
    "id": "2445513479",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Hong Li",
    "id": "2445889260",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Qi Su",
    "id": "2349847584",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Haochen Liu",
    "id": "2256317177",
    "h_index": 9,
    "papers": 21
   },
   {
    "name": "Xu He",
    "id": "2277992784",
    "h_index": 3,
    "papers": 19
   },
   {
    "name": "Yukun Shi",
    "id": "2404257413",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Cong Huang",
    "id": "2399268226",
    "h_index": 4,
    "papers": 22
   },
   {
    "name": "Zhirui Zhang",
    "id": "2445881492",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Bo Cheng",
    "id": "2213461789",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Kai Chen",
    "id": "2359596863",
    "h_index": 6,
    "papers": 22
   }
  ],
  "comment": "20 pages, 9 figures",
  "topics": [
   "vla",
   "humanoids",
   "egocentric-data",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.32009v1",
  "pdf_url": "https://arxiv.org/pdf/2606.32009v1",
  "html_url": "https://arxiv.org/html/2606.32009v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.31909",
  "slug": "codex-learning-compositional-dexterous-functional-manipulation-without",
  "title": "CoDex: Learning Compositional Dexterous Functional Manipulation without Demonstrations",
  "abstract": "In this work, we study Compositional Dexterous Functional Object Manipulation (CD-FOM): tasks such as aiming and actuating a spray bottle on a plant or a glue gun on wood, which require both actuating an object's internal mechanism and controlling its pose to apply the object's function to the environment. These tasks pose significant challenges for robots due to the demanding integration of semantic understanding of the object's function, actuation mode, and application area with intricate physical dexterity to manage grasp stability, movement trajectory, and actuation. We introduce CoDex, a zero-demonstration framework that autonomously discovers CD-FOM manipulation strategies. CoDex uses vision-language models (VLMs) to infer semantic constraints from the task and scene. These constraints guide analytic constrained optimization to generate a short list of functional grasp candidates that can be efficiently refined with reinforcement learning to generate full grasp-move-actuate policies transferable from simulation to the real world. We evaluate CoDex on a 7-DoF robot arm with a 16-DoF multi-fingered hand across six CD-FOM tasks involving previously unseen objects with internal mechanisms, including spray bottles, hot glue guns, air dusters, flashlights, and pepper grinders, and their application to unseen target objects, showcasing its ability to autonomously discover and execute complex, physically viable dexterous behaviors without human demonstrations. More information at https://robin-lab.cs.utexas.edu/CoDex/.",
  "published": "2026-06-30",
  "updated": "2026-06-30",
  "year": "2026",
  "authors": [
   "Bowen Jiang",
   "William Painter Reger",
   "Roberto Martin-Martin"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work introduces CoDex, a zero-demonstration framework that autonomously discovers CD-FOM manipulation strategies using vision-language models to infer semantic constraints from the task and scene and generates a short list of functional grasp candidates that can be efficiently refined with reinforcement learning to generate full grasp-move-actuate policies transferable from simulation to the real world.",
  "doi": "10.48550/arXiv.2606.31909",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Bowen Jiang",
    "id": "2153963881",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "William Painter Reger",
    "id": "2445666383",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Roberto Mart\u00edn-Mart\u00edn",
    "id": "2316638007",
    "h_index": 6,
    "papers": 12
   }
  ],
  "comment": "IEEE International Conference on Robotics and Automation (ICRA) 2026. Project page: https://robin-lab.cs.utexas.edu/CoDex/",
  "topics": [
   "dexterous-manipulation",
   "egocentric-data",
   "rl-control",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.31909v1",
  "pdf_url": "https://arxiv.org/pdf/2606.31909v1",
  "html_url": "https://arxiv.org/html/2606.31909v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2606.31836",
  "slug": "robotacdex-a-dexterous-visual-tactile-action-dataset-for-humanoid-mani",
  "title": "RoboTacDex: A Dexterous Visual-Tactile-Action Dataset for Humanoid Manipulation",
  "abstract": "In the field of robot learning, large-scale and diverse demonstration trajectories provide the fundamental basis for enhancing robotic manipulation ability. We introduce RoboTacDex, a large, multi-modal, and diverse dataset of dexterous manipulation behaviors performed with a humanoid robot. Built on the publicly accessible humanoid robot Unitree G1, RoboTacDex consists of 6k trajectories covering 19 tasks, 23 skills, and interactions with 22 objects. RoboTacDex provides comprehensive records including multi-view RGB and depth information, tactile feedback, and detailed semantic annotations. Furthermore, the dataset features a variety of relatively challenging tasks that can only be completed by dual arms and dexterous hands, aiming to mimic human-like operational logic and simulate real-world manipulation complexity. To ensure data collection quality, we develop an improved multi-camera synchronization system to enable millisecond data synchronization and recording of modalities. In our experiments, we evaluate three representative imitation learning models on our dataset, analyzing their performance as well as their respective strengths and limitations across different task categories. Successful trial results and a moderate level of generalization capabilities across a suite of tasks indicate the effectiveness and diversity of the collected dataset. Our dataset will be open-sourced soon.",
  "published": "2026-06-30",
  "updated": "2026-06-30",
  "year": "2026",
  "authors": [
   "Xinyi Wang",
   "Donghan Li",
   "Zi'Ang Chen",
   "Chong Yu",
   "Chen Xin",
   "Peng Ye",
   "Yingkai Sun",
   "Tao Chen"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work introduces RoboTacDex, a large, multi-modal, and diverse dataset of dexterous manipulation behaviors performed with a humanoid robot, and develops an improved multi-camera synchronization system to enable millisecond data synchronization and recording of modalities.",
  "doi": "10.48550/arXiv.2606.31836",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xinyi Wang",
    "id": "2445918416",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Donghan Li",
    "id": "2445889521",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Ziang Chen",
    "id": "2445893501",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Chong Yu",
    "id": "2261393532",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Chen Xin",
    "id": "2445801737",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Peng Ye",
    "id": "2290162992",
    "h_index": 10,
    "papers": 30
   },
   {
    "name": "Ying Sun",
    "id": "2339340425",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Tao Chen",
    "id": "2353266444",
    "h_index": 3,
    "papers": 12
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "humanoids",
   "tactile",
   "imitation-diffusion",
   "data-teleop"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2606.31836v1",
  "pdf_url": "https://arxiv.org/pdf/2606.31836v1",
  "html_url": "https://arxiv.org/html/2606.31836v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2606.31723",
  "slug": "unitacvla-unified-tactile-understanding-and-prediction-in-vision-langu",
  "title": "UniTacVLA: Unified Tactile Understanding and Prediction in Vision Language Action Models",
  "abstract": "Vision-language-action (VLA) models have achieved strong performance in many robotic manipulation tasks, yet remain limited in contact-rich dexterous manipulation. To overcome this limitation, recent vision-tactile-language-action (VTLA) methods incorporate tactile sensing into VLA models to provide direct contact information. However, they typically treat tactile signals as passive auxiliary inputs, making it difficult to model tactile semantics and future physical interactions. To this end, we propose a unified tactile learning framework for contact-rich manipulation that models tactile signals as dynamic interaction cues for both contact understanding and prediction. Specifically, we construct a unified tactile latent space and jointly model current tactile states and future contact changes through tactile chain-of-thought reasoning and coarse-to-fine future tactile prediction, thereby forming a state-aware and dynamics-aware tactile prior. Based on this prior, we introduce a tactile-action mixed controller that combines real-time and predicted tactile feedback to refine low-frequency action chunks with high-frequency corrections. Real-world experiments on four categories of contact-rich tasks, including adjustment, insertion, wiping, and assembly, under both clean and externally perturbed settings, show that our method improves success rate, manipulation accuracy, and contact robustness over existing methods, demonstrating its effectiveness in dexterous physical interaction.",
  "published": "2026-06-30",
  "updated": "2026-06-30",
  "year": "2026",
  "authors": [
   "Xidong Zhang",
   "Yichi Zhang",
   "Jiaxin Shi",
   "Fucai Zhu",
   "Siyu Zhu",
   "Michael Yu Wang",
   "Xiaojun Wu",
   "Weihao Yuan"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 2,
  "influential_citations": 0,
  "tldr": "This work proposes a unified tactile learning framework for contact-rich manipulation that models tactile signals as dynamic interaction cues for both contact understanding and prediction and introduces a tactile-action mixed controller that combines real-time and predicted tactile feedback to refine low-frequency action chunks with high-frequency corrections.",
  "doi": "10.48550/arXiv.2606.31723",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xidong Zhang",
    "id": "2373529061",
    "h_index": 2,
    "papers": 12
   },
   {
    "name": "Yichi Zhang",
    "id": "2365993236",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Jiaxin Shi",
    "id": "2292341907",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Fucai Zhu",
    "id": "30772053",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Siyu Zhu",
    "id": "2374489685",
    "h_index": 1,
    "papers": 11
   },
   {
    "name": "Michael Yu Wang",
    "id": "2248999920",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Xiaojun Wu",
    "id": "2445885992",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Weihao Yuan",
    "id": "11349534",
    "h_index": 15,
    "papers": 24
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "dexterous-manipulation",
   "tactile",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.31723v1",
  "pdf_url": "https://arxiv.org/pdf/2606.31723v1",
  "html_url": "https://arxiv.org/html/2606.31723v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.48
 },
 {
  "id": "2606.31037",
  "slug": "labimus-a-simulation-and-benchmark-for-humanoid-dexterous-manipulation",
  "title": "Labimus: A Simulation and Benchmark for Humanoid Dexterous Manipulation in Chemical Laboratory",
  "abstract": "Laboratory automation has made remarkable progress through robotic platforms and AI-driven scientific reasoning. However, many laboratory operations (e.g., solid--solid transfer) remain inherently dynamic and require real-time adaptation to different materials and experimental conditions. Such precision-critical manipulations are difficult to standardize, motivating the use of humanoid robots with dexterous hands. Despite this opportunity, no existing benchmark evaluates humanoid manipulation in precision-critical laboratory environments. We present Labimus, to our knowledge, the first benchmark for humanoid dexterous manipulation in organic chemistry laboratories. Labimus reconstructs over 30 functionally faithful assets from real organic chemistry workstations through real-to-sim modeling, collectively covering the core operations of routine organic chemistry experiments. The benchmark integrates articulated laboratory instruments, particle-based powder physics, and closed-loop instrument readouts, enabling a complete manipulation-to-measurement pipeline. It further defines six atomic operations and a seven-step solid-weighing workflow derived from real laboratory standard operating procedures. We introduce a precision-aware evaluation protocol designed to jointly measure task completion, experimental precision, and long-horizon execution. We benchmark three representative policies under procedural layouts and environmental perturbations. Results reveal a precision gap: policies that successfully complete laboratory tasks can still fail to satisfy the quantitative tolerances required by experimental protocols. Our benchmark exposes a fundamental disconnect between task completion and experimental validity, providing a new testbed for developing reliable humanoid robots for scientific laboratories.",
  "published": "2026-06-30",
  "updated": "2026-07-01",
  "year": "2026",
  "authors": [
   "Yuhan Wu",
   "Zhao Jin",
   "Tao Li",
   "Yuheng Zhang",
   "Zhichao Wang",
   "Shuo Wang",
   "Jun Jiang",
   "Xiaobo Li",
   "Yanyong Zhang",
   "Jian Tang",
   "Zhengping Che",
   "Yan Xia"
  ],
  "author_count": 12,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "",
  "doi": "10.48550/arXiv.2606.31037",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yuhan Wu",
    "id": "2281899017",
    "h_index": 5,
    "papers": 38
   },
   {
    "name": "Zhao Jin",
    "id": "2279864362",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Tao Li",
    "id": "2324418062",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Yuheng Zhang",
    "id": "2366307710",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Zhengping Che",
    "id": "1939695",
    "h_index": 26,
    "papers": 88
   },
   {
    "name": "Jian Tang",
    "id": "2152779004",
    "h_index": 12,
    "papers": 33
   },
   {
    "name": "Zhichao Wang",
    "id": "2308348037",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Shuoyi Wang",
    "id": "2446317953",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Junyun Jiang",
    "id": "2379832105",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Xiaobo Li",
    "id": "2445893658",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Yanyong Zhang",
    "id": "2341792088",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Yan Xia",
    "id": "2239199844",
    "h_index": 7,
    "papers": 25
   }
  ],
  "comment": "Project page: https://labimus.github.io/",
  "topics": [
   "dexterous-manipulation",
   "humanoids",
   "sim2real"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.31037v2",
  "pdf_url": "https://arxiv.org/pdf/2606.31037v2",
  "html_url": "https://arxiv.org/html/2606.31037v2",
  "code_url": "https://labimus.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.3
 },
 {
  "id": "2606.30900",
  "slug": "the-quadruped-soft-tail-compliant-grasping-and-swabbing-for-contaminat",
  "title": "The Quadruped Soft Tail: Compliant Grasping and Swabbing for Contamination Surveys in Harsh Environments",
  "abstract": "Beryllium contamination surveys in radioactive areas are challenging for robots in environments cluttered with cables and electronics. To address this problem, we have developed a novel quadruped system augmentation: A lightweight, soft, and compliant tendon-actuated robotic tail mounted on a quadruped robot. The tail features a hollow, flexible backbone and a tendon-actuated soft gripper that enables the robot to pick up sampling tissues, swab contaminated surfaces, and release the tissues at designated collection locations for subsequent beryllium analysis. To enable intuitive teleoperation, a closed-form kinematic model and a singularity-robust task-space controller are developed. Experimental results demonstrate that gripper actuation has a negligible effect on robot shape, while common-mode tendon actuation provides an effective mechanism for stiffness modulation and preload control. Furthermore, experimental validation indicates that the proposed kinematic model provides a suitable basis for real-time task-space control. The proposed system combines the agility of legged locomotion with the compliance of soft robotic manipulation, enabling the complete contamination-survey procedure to be performed without human exposure. While motivated by beryllium contamination surveys at CERN, the proposed quadruped soft-tail concept is broadly applicable to legged robots operating in cluttered, confined, or hazardous environments where conventional rigid-link manipulators are undesirable.",
  "published": "2026-06-29",
  "updated": "2026-07-01",
  "year": "2026",
  "authors": [
   "Harald Minde Hansen",
   "Nandita Gallacher",
   "Kristin Y. Pettersen",
   "Jan Tommy Gravdahl",
   "Mario di Castro"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "10.48550/arXiv.2606.30900",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Harald Minde Hansen",
    "id": "2424680399",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Nandita Gallacher",
    "id": "2424682226",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "K. Y. Pettersen",
    "id": "2267964617",
    "h_index": 6,
    "papers": 46
   },
   {
    "name": "J. Gravdahl",
    "id": "1739946",
    "h_index": 45,
    "papers": 385
   },
   {
    "name": "Mario Di Castro",
    "id": "2268128426",
    "h_index": 2,
    "papers": 6
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "humanoids",
   "data-teleop",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.30900v2",
  "pdf_url": "https://arxiv.org/pdf/2606.30900v2",
  "html_url": "https://arxiv.org/html/2606.30900v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.30809",
  "slug": "gausslite-online-task-conditioned-3d-gaussian-splatting-for-real-time",
  "title": "GaussLite: Online Task-Conditioned 3D Gaussian Splatting for Real-Time Robotic Mapping",
  "abstract": "Existing 3D Gaussian Splatting (3DGS) systems distribute representation capacity uniformly across a scene, ignoring the fact that many downstream robotic tasks engage only a fraction of the reconstructed geometry. This causes valuable onboard compute to be allocated towards optimizing irrelevant parts of the scene, either limiting online capacity or under-optimizing the most relevant parts of the scene. We introduce GaussLite, a task-driven 3DGS mapping system that conditions its representation density on a natural-language task specification. Given a posed RGB-D stream and a task such as \"prepare to pick up the object on the desk,\" GaussLite uses a one-shot LLM parser to extract target and anchor objects, which are grounded per-frame by an open-vocabulary detector and segmented to produce per-pixel relevance masks in real time. The mapper allocates seeding density, gradient flow and scaling by task relevance. At matched Gaussian budget and real-time mapping at 4 Hz on resource-constrained hardware, GaussLite outperforms baselines on ROI PSNR on the Replica Dataset by an average +2.72 dB and on a real-hardware demonstration in indoor and outdoor settings by +2.23 dB. We further show that two task-specialized agents' maps can be fused into a single shared map via per-voxel voting on active-optimization counts in real time, outperforming concatenation by +3.42 dB while only sharing an average 7.08% of the map.",
  "published": "2026-06-29",
  "updated": "2026-06-29",
  "year": "2026",
  "authors": [
   "Annika Thomas",
   "Mason Peterson",
   "Jonathan P. How"
  ],
  "author_count": 3,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work introduces GaussLite, a task-driven 3DGS mapping system that conditions its representation density on a natural-language task specification, and shows that two task-specialized agents'maps can be fused into a single shared map via per-voxel voting on active-optimization counts in real time.",
  "doi": "10.48550/arXiv.2606.30809",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Annika Thomas",
    "id": "2265614445",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Mason B. Peterson",
    "id": "2171740944",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Jonathan P. How",
    "id": "2264754352",
    "h_index": 7,
    "papers": 22
   }
  ],
  "comment": "",
  "topics": [
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.30809v1",
  "pdf_url": "https://arxiv.org/pdf/2606.30809v1",
  "html_url": "https://arxiv.org/html/2606.30809v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.30754",
  "slug": "streaming-gaussian-encoding-for-4d-panoptic-occupancy-tracking",
  "title": "Streaming Gaussian Encoding for 4D Panoptic Occupancy Tracking",
  "abstract": "Camera-based 4D panoptic occupancy tracking (4D-POT) is a promising paradigm for holistic scene understanding from multi-view imagery, enabling joint reasoning about geometry, semantics, and object identities across time. Recent mask-based pipelines achieve strong performance by propagating instance queries across frames. However, their underlying volumetric representations are typically recomputed at each timestep, limiting geometric temporal consistency, particularly under occlusion and for static scene elements. To address this limitation, we propose a streaming Gaussian encoder that maintains a persistent volumetric scene representation for 4D-POT. Our method models the scene as a fixed-size set of latent Gaussian queries that are propagated via ego-motion compensation and refreshed under a confidence-guided budget constraint. Crucially, we shape Gaussian opacities through depth-based supervision to serve as proxy for visibility, enabling confidence to accumulate as a temporally aggregated measure of persistent scene support. Together with a warmup-based multi-frame training strategy, this yields representation-level temporal coherence beyond decoder-only tracking. Extensive experiments on Occ3D-extended nuScenes and Waymo establish a new state-of-the-art for camera-based 4D-POT, improving tracking consistency with negligible computational overhead while remaining fully compatible with existing mask-based pipelines. We provide code and models at https://sge.cs.uni-freiburg.de.",
  "published": "2026-06-29",
  "updated": "2026-06-29",
  "year": "2026",
  "authors": [
   "Maximilian Luz",
   "Thomas N\u00fcrnberg",
   "Yakov Miron",
   "Abhinav Valada"
  ],
  "author_count": 4,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A streaming Gaussian encoder that maintains a persistent volumetric scene representation for camera-based 4D-POT, improving tracking consistency with negligible computational overhead while remaining fully compatible with existing mask-based pipelines.",
  "doi": "10.48550/arXiv.2606.30754",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Maximilian Luz",
    "id": "2189375847",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "T. Nurnberg",
    "id": "101529627",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Yakov Miron",
    "id": "2385465861",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Abhinav Valada",
    "id": "2609831",
    "h_index": 36,
    "papers": 119
   }
  ],
  "comment": "Accepted to the 2026 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS 2026)",
  "topics": [
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.30754v1",
  "pdf_url": "https://arxiv.org/pdf/2606.30754v1",
  "html_url": "https://arxiv.org/html/2606.30754v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2606.30749",
  "slug": "from-grasps-to-dexterity-large-scale-grasp-pretraining-for-dexterous-m",
  "title": "From Grasps to Dexterity: Large-Scale Grasp Pretraining for Dexterous Manipulation",
  "abstract": "Large-scale dexterous grasp datasets encode rich priors over hand-object interaction, but their use has largely been confined to grasp generation and pick-and-place manipulation. We study whether such data can instead support functional dexterity in articulated tool use, where a robot must acquire a tool, maintain contact, and operate its functional moving parts. We adapt a hierarchical imitation learning framework that combines high-level hand sub-goal prediction with a low-level goal-conditioned controller. We construct a 355k-trajectory grasp-pretraining dataset from large-scale dexterous grasp annotations and use it to pretrain the low-level controller. The controller is then fine-tuned on downstream task demonstrations. To evaluate this setting, we introduce DexCraft, a simulation benchmark with six articulated tool-use tasks requiring coordinated finger motion. Across simulation and real-world experiments, our approach outperforms end-to-end diffusion policy baselines and hierarchical policies trained from scratch. In the real world, it improves full-task success by 33.3 percentage points over DP3. These results show that grasp datasets can serve not only as resources for grasp synthesis, but also as scalable pretraining data for contact-rich dexterous manipulation. Videos are shown on https://yingyuan0414.github.io/grasp2dexterity/ .",
  "published": "2026-06-29",
  "updated": "2026-06-29",
  "year": "2026",
  "authors": [
   "Ying Yuan",
   "Xinyu Liu",
   "Sriram Krishna",
   "David Held"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work adapts a hierarchical imitation learning framework that combines high-level hand sub-goal prediction with a low-level goal-conditioned controller to serve not only as resources for grasp synthesis, but also as scalable pretraining data for contact-rich dexterous manipulation.",
  "doi": "10.48550/arXiv.2606.30749",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ying Yuan",
    "id": "2325901709",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Xinyu Liu",
    "id": "2445899668",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Sriram Krishna",
    "id": "2062575864",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "David Held",
    "id": "2277741823",
    "h_index": 3,
    "papers": 10
   }
  ],
  "comment": "Project page: https://yingyuan0414.github.io/grasp2dexterity/",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "sim2real",
   "imitation-diffusion",
   "foundation-pretraining",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.30749v1",
  "pdf_url": "https://arxiv.org/pdf/2606.30749v1",
  "html_url": "https://arxiv.org/html/2606.30749v1",
  "code_url": "https://yingyuan0414.github.io/grasp2dexterity/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2606.30645",
  "slug": "vlk-learning-humanoid-loco-manipulation-from-synthetic-interactions-in",
  "title": "VLK: Learning Humanoid Loco-Manipulation from Synthetic Interactions in Reconstructed Scenes",
  "abstract": "Perception-based humanoid loco-manipulation requires connecting egocentric observations and task instructions to whole-body motion. Learning this mapping requires synchronized egocentric images, language commands, and robot-compatible kinematic trajectories, yet no existing data source provides this complete tuple at scale. We address this bottleneck by generating vision-language-kinematics (VLK) supervision synthetically in reconstructed scenes. Our pipeline leverages 3D Gaussian Splatting to reconstruct metric-scale indoor environments, synthesizes navigation and object-interaction trajectories using privileged scene information, and renders paired egocentric observations after the fact. We produce 48,000 paired trajectories with no human intervention and train a VLK policy that predicts short-horizon whole-body kinematic trajectories. A whole-body tracker converts these predictions into actions on the physical humanoid. We evaluate on the physical Unitree G1 performing navigation and single-object transport, demonstrating that synthesized interactions in reconstructed scenes provide effective supervision for sim-to-real perception-based humanoid loco-manipulation. Project Website: https://vision-language-kinematics.github.io/",
  "published": "2026-06-29",
  "updated": "2026-06-29",
  "year": "2026",
  "authors": [
   "Yen-Jen Wang",
   "Jiaman Li",
   "Sirui Chen",
   "Takara E. Truong",
   "Pei Xu",
   "Pieter Abbeel",
   "Rocky Duan",
   "Koushil Sreenath",
   "Angjoo Kanazawa",
   "Carmelo Sferrazza",
   "Guanya Shi",
   "Karen Liu"
  ],
  "author_count": 12,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.GR",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work generates vision-language-kinematics (VLK) supervision synthetically in reconstructed scenes and evaluates on the physical Unitree G1 performing navigation and single-object transport, demonstrating that synthesized interactions in reconstructed scenes provide effective supervision for sim-to-real perception-based humanoid loco-manipulation.",
  "doi": "10.48550/arXiv.2606.30645",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yen-Jen Wang",
    "id": "2115740911",
    "h_index": 11,
    "papers": 17
   },
   {
    "name": "Jiaman Li",
    "id": "22133106",
    "h_index": 16,
    "papers": 22
   },
   {
    "name": "Sirui Chen",
    "id": "2209905328",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Takara Truong",
    "id": "2369918131",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Pei Xu",
    "id": "2335601372",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Pieter Abbeel",
    "id": "2381724317",
    "h_index": 9,
    "papers": 17
   },
   {
    "name": "Rocky Duan",
    "id": "2381724728",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "K. Sreenath",
    "id": "144116765",
    "h_index": 55,
    "papers": 230
   },
   {
    "name": "Angjoo Kanazawa",
    "id": "20615377",
    "h_index": 60,
    "papers": 126
   },
   {
    "name": "Carmelo Sferrazza",
    "id": "47218071",
    "h_index": 21,
    "papers": 44
   },
   {
    "name": "Guanya Shi",
    "id": "2384824402",
    "h_index": 9,
    "papers": 24
   },
   {
    "name": "Karen Liu",
    "id": "2408789535",
    "h_index": 2,
    "papers": 4
   }
  ],
  "comment": "19 pages, 7 figures, 4 tables",
  "topics": [
   "humanoids",
   "egocentric-data",
   "sim2real",
   "spatial-3d",
   "navigation"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2606.30645v1",
  "pdf_url": "https://arxiv.org/pdf/2606.30645v1",
  "html_url": "https://arxiv.org/html/2606.30645v1",
  "code_url": null,
  "presented": [
   {
    "club": "saturday-robotics",
    "club_name": "Saturday Robotics",
    "session_number": 21,
    "session_title": "Robotics & World Models Reading Club 21: Vision-Language-Kinematics Supervision for Perception-Based Humanoid Loco-Manipulation. SF 8/1",
    "date_text": "Saturday, August 1, 2026 | 2:00 PM \u2013 5:00 PM",
    "city": "San Francisco, CA",
    "url": "https://lu.ma/4xkibxbh",
    "listed_as": "Project: https://vision-language-kinematics.github.io"
   }
  ],
  "club_note": "Video: https://youtu.be/ZB6k\\iMJP7M",
  "featured": true,
  "signal": 3.5
 },
 {
  "id": "2606.30457",
  "slug": "behavior-prompting-policy-demonstrations-as-prompts-for-manipulation",
  "title": "Behavior Prompting Policy: Demonstrations as Prompts for Manipulation",
  "abstract": "We study behavior prompting, a paradigm that enables robots to perform new tasks at inference time given a single human demonstration, which we call a behavior prompt. To enable this capability, we present contributions in algorithm, data, and evaluation. For algorithm, we introduce Behavior Prompting Policy (BPP), an in-context visuomotor architecture that translates the behavior prompt and the current observation into robot actions. For data, we identify that task diversity is the primary driver of the prompting capability and introduce iPhUMI, a handheld manipulation interface for collecting diverse training data. For evaluation, we introduce DrawAnything and LIBERO-Gen to evaluate test-time adaptation to unseen drawing and tabletop manipulation tasks. We also demonstrate that iPhUMI serves as a practical interface for specifying behavior prompts at test time, enabling a human to command a robot via a single demonstration to complete known tasks or to define new robot capabilities. Altogether, behavior prompting provides a flexible and scalable way to teach robots new skills without the need for expensive fine-tuning. Our project website is located at https://behavior-prompting.github.io/ .",
  "published": "2026-06-29",
  "updated": "2026-06-29",
  "year": "2026",
  "authors": [
   "Austin Patel",
   "Ben Pekarek",
   "Joel Enrique Castro Hernandez",
   "Shuran Song"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 2,
  "influential_citations": 0,
  "tldr": "",
  "doi": "10.48550/arXiv.2606.30457",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Austin Patel",
    "id": "2034257941",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Ben Pekarek",
    "id": "2378986255",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "J. Hernandez",
    "id": "2445709300",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Shuran Song",
    "id": "2364257433",
    "h_index": 6,
    "papers": 17
   }
  ],
  "comment": "",
  "topics": [
   "egocentric-data",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.30457v1",
  "pdf_url": "https://arxiv.org/pdf/2606.30457v1",
  "html_url": "https://arxiv.org/html/2606.30457v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.48
 },
 {
  "id": "2606.30367",
  "slug": "futurenav-unified-world-action-modeling-for-vision-and-language-naviga",
  "title": "FutureNav: Unified World-Action Modeling for Vision-and-Language Navigation",
  "abstract": "Vision-and-language navigation (VLN) in continuous environments requires an agent to ground instructions in egocentric observations while maintaining spatial understanding across long action sequences. Recent navigation foundation models have shown strong progress by scaling vision-language models, but they often learn navigation primarily as direct action generation, without explicitly modeling world states or predicting their future evolution. We introduce FutureNav, a VLM-based unified world-action modeling framework for vision-and-language navigation. Specifically, FutureNav jointly encodes text, visual, and spatial features and feeds them into the LLM, and optimizes four objectives for simultaneous world and action modeling: an action policy objective for navigation action prediction, inverse and forward dynamics objectives for modeling state transitions, and a future generation objective for predicting future spatial states. This unified architecture strengthens action prediction while explicitly modeling the world, without sacrificing inference speed. Extensive experiments show that, with only a 4B-scale backbone, FutureNav achieves state-of-the-art performance on multiple VLN benchmarks and substantially outperforms prior VLN methods, paving the way toward future world-action models for VLN. We will release the code and models to support future research.",
  "published": "2026-06-29",
  "updated": "2026-06-29",
  "year": "2026",
  "authors": [
   "Lingfeng Zhang",
   "Zeying Gong",
   "Xiaoshuai Hao",
   "Haoxiang Fu",
   "Qiang Zhang",
   "Mingliang Zhou",
   "Hangjun Ye",
   "Xiaojun Liang",
   "Junwei Liang",
   "Wenbo Ding"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "FutureNav is introduced, a VLM-based unified world-action modeling framework for vision-and-language navigation that achieves state-of-the-art performance on multiple VLN benchmarks and substantially outperforms prior VLN methods, paving the way toward future world-action models for VLN.",
  "doi": "10.48550/arXiv.2606.30367",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Lingfeng Zhang",
    "id": "2366951551",
    "h_index": 8,
    "papers": 22
   },
   {
    "name": "Zeying Gong",
    "id": "2249762165",
    "h_index": 6,
    "papers": 21
   },
   {
    "name": "Xiaoshuai Hao",
    "id": "2386788371",
    "h_index": 5,
    "papers": 21
   },
   {
    "name": "Haoxiang Fu",
    "id": "2376055009",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Qiang Zhang",
    "id": "2155842084",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Mingliang Zhou",
    "id": "2381726519",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Hangjun Ye",
    "id": "2384401186",
    "h_index": 7,
    "papers": 34
   },
   {
    "name": "Xiaojun Liang",
    "id": "2394119119",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Junwei Liang",
    "id": "2333968399",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "Wenbo Ding",
    "id": "2375173478",
    "h_index": 5,
    "papers": 21
   }
  ],
  "comment": "",
  "topics": [
   "egocentric-data",
   "navigation",
   "foundation-pretraining",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.30367v1",
  "pdf_url": "https://arxiv.org/pdf/2606.30367v1",
  "html_url": "https://arxiv.org/html/2606.30367v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2606.30290",
  "slug": "x-morph-human-motion-priors-for-scalable-robot-learning-across-morphol",
  "title": "X-Morph: Human Motion Priors for Scalable Robot Learning Across Morphologies",
  "abstract": "Recent progress in humanoid behavior models has been driven in large part by abundant human motion data, but comparable motion data is scarce for non-humanoid legged robots such as quadrupeds, hexapods, and quadruped manipulators. A promising alternative is to repurpose human motion across embodiments; however, direct retargeting often produces motions that are visually plausible yet physically inconsistent or difficult to track under robot dynamics. We present X-Morph, a human-motion-to-robot-behavior pipeline that converts human motion into deployable locomotion and loco-manipulation policies for diverse non-humanoid legged morphologies. A cross-morphology retargeting stage converts human motions into kinematically plausible, intent-preserving robot references, which are then tracked by a privileged RL policy and distilled into a causal student policy. We evaluate X-Morph on three morphologically distinct platforms: a quadruped, a hexapod, and a quadruped equipped with a manipulator. The resulting policies track diverse retargeted motions, generalize to unseen human motions, and support downstream use cases including video-based teleoperation, behavior-prior control, and text-conditioned motion generation. These results suggest that large-scale human motion can serve as a substrate for learning broad, reusable behavior priors beyond humanoid robots. Project page: https://maker-rat.github.io/morph/",
  "published": "2026-06-29",
  "updated": "2026-06-29",
  "year": "2026",
  "authors": [
   "Ritwik Sharma",
   "Shivam Sood",
   "Arhaan Jain",
   "Shyam Charan Kesavamoorthi",
   "Chengyang He",
   "Guillaume Sartoretti"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "X-Morph is presented, a human-motion-to-robot-behavior pipeline that converts human motion into deployable locomotion and loco-manipulation policies for diverse non-humanoid legged morphologies and suggests that large-scale human motion can serve as a substrate for learning broad, reusable behavior priors beyond humanoid robots.",
  "doi": "10.48550/arXiv.2606.30290",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ritwik Sharma",
    "id": "2445706965",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Shivam Sood",
    "id": "2221118825",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Arhaan Jain",
    "id": "2438776085",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "S. Kesavamoorthi",
    "id": "2444206464",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Chengyang He",
    "id": "2257636330",
    "h_index": 5,
    "papers": 17
   },
   {
    "name": "G. Sartoretti",
    "id": "2292917033",
    "h_index": 11,
    "papers": 67
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "data-teleop",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.30290v1",
  "pdf_url": "https://arxiv.org/pdf/2606.30290v1",
  "html_url": "https://arxiv.org/html/2606.30290v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.29940",
  "slug": "warp-whole-body-retargeting-for-learning-from-offline-human-demonstrat",
  "title": "WARP: Whole-Body Retargeting for Learning from Offline Human Demonstrations",
  "abstract": "Direct transfer from human demonstration to learnable robot action is a crucial step towards scalable whole-body mobile manipulation. While human data scales better than mobile teleoperation, it requires overcoming significant embodiment gaps. Existing retargeting methods yield imprecise or inconsistent solutions, causing action multi-modality that prevents supervised policies from reliably converging. We present Whole-body-Aware Retargeting from human Pose (WARP), an offline pipeline that explicitly models embodiment differences to extract precise, unique whole-body actions. WARP leverages a closed-form Shoulder-Elbow-Wrist (SEW) geometric solver for exact end-effector tracking while preserving whole-body structural intent. Paired with lazy mobile-base control, it extracts accurate, consistent robot trajectories. Evaluations show WARP provides highly reliable data for open-loop real-world replay. To our knowledge, WARP is the first framework to achieve zero-shot whole-body mobile manipulation directly from offline human demonstrations, eliminating the need for human-in-the-loop teleoperation action data. More details on https://warp-retargeting.github.io/",
  "published": "2026-06-29",
  "updated": "2026-08-19",
  "year": "2026",
  "authors": [
   "Zhenyang Chen",
   "Chuizheng Kong",
   "Chuye Zhang",
   "Yuanshao Yang",
   "Lawrence Y. Zhu",
   "Shreyas Kousik",
   "Danfei Xu"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "Warp-retargeting from human Pose (WARP) is the first framework to achieve zero-shot whole-body mobile manipulation directly from offline human demonstrations, eliminating the need for human-in-the-loop teleoperation action data.",
  "doi": "10.48550/arXiv.2606.29940",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhenyang Chen",
    "id": "2367302562",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Chuizheng Kong",
    "id": "2286875386",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Chuye Zhang",
    "id": "2215500204",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Yuanshao Yang",
    "id": "2344198233",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Lawrence Y. Zhu",
    "id": "2379184973",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Shreyas Kousik",
    "id": "7880672",
    "h_index": 13,
    "papers": 50
   },
   {
    "name": "Danfei Xu",
    "id": "2322756258",
    "h_index": 5,
    "papers": 12
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "egocentric-data",
   "navigation",
   "data-teleop",
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.29940v2",
  "pdf_url": "https://arxiv.org/pdf/2606.29940v2",
  "html_url": "https://arxiv.org/html/2606.29940v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2606.29937",
  "slug": "repair-bench-a-benchmark-for-robot-error-perception-and-interaction-re",
  "title": "REPAIR-Bench: A Benchmark for Robot Error Perception And Interaction Recovery",
  "abstract": "Understanding how users perceive and respond to robot failures is essential for building robust and trustworthy robot systems. Prior work, however, (i) often treats failures as independent events, (ii) emphasizes binary failure detection, (iii) with rule-based recovery modeling. We present REPAIR-Bench, built on 214 interaction trials from 41 participants, the benchmark spans four induced failure types and provides synchronized facial action units, head pose, speech transcripts, and post-interaction affect and recovery reports. The benchmark spans three novel evaluation tasks that jointly capture the lifecycle of failure in human-robot interaction (HRI): (i) failure detection over inter-dependent interaction sessions, modeling longitudinal user adaptation across repeated failures; (ii) visual failure-type classification beyond binary success/failure formulations; and (iii) user-centered recovery prediction, inferring users' preferred recovery strategies from interaction context rather than relying on manually designed or rule-based strategies. In baseline experiments, hierarchical recurrent modeling improved failure detection over a single-session model (strict F1: 0.80 vs. 0.68), achieved a failure localization mean signed error of -0.51 s, median absolute error of 2.97 s and, for recovery prediction, a QLoRA-tuned Mistral-7B reached Hit@5=0.76 and F1@5=0.32. REPAIR-Bench provides both the HRI and Medical HRI communities with a standardized framework for (1) evaluating robot failures and (2) building transparent, adaptive, and trustworthy recovery systems.",
  "published": "2026-06-29",
  "updated": "2026-06-29",
  "year": "2026",
  "authors": [
   "Giuliano Pioldi",
   "Yashika Batra",
   "Arman Ibrayeva",
   "Yuanchen Bai",
   "Purnjay Maruur",
   "Promise Ekpo",
   "Angelique Taylor"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "REPAIR-Bench provides both the HRI and Medical HRI communities with a standardized framework for evaluating robot failures and building transparent, adaptive, and trustworthy recovery systems.",
  "doi": "10.48550/arXiv.2606.29937",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Giuliano Pioldi",
    "id": "2422647004",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Yashika Batra",
    "id": "2422646806",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Arman Ibrayeva",
    "id": "4628130",
    "h_index": 4,
    "papers": 14
   },
   {
    "name": "Yuanchen Bai",
    "id": "2342823238",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Purnjay Maruur",
    "id": "2422646737",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Promise Ekpo",
    "id": "2366163054",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "A. Taylor",
    "id": "2347008072",
    "h_index": 3,
    "papers": 9
   }
  ],
  "comment": "",
  "topics": [
   "sim2real",
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.29937v1",
  "pdf_url": "https://arxiv.org/pdf/2606.29937v1",
  "html_url": "https://arxiv.org/html/2606.29937v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.29936",
  "slug": "openspm-an-environment-transferable-robotic-key-spatial-pose-memory-an",
  "title": "OpenSPM: An Environment-Transferable Robotic Key Spatial Pose Memory and Closed-Loop High-Frequency Flow-Matching Action Generation Model",
  "abstract": "Open-environment tabletop robotic manipulation requires systems to possess semantic understanding, precise geometric pose estimation, and high-frequency action generation. While end-to-end vision-language-action (VLA) models excel at semantic generalization, they often lack explicit geometric constraints for fine-grained tasks and require costly training. To bridge the gap between high-level semantics and low-level physical execution, we propose OpenSPM, an open environment spatial persistent memory framework consisting of spatial pose memory and flow-matching action generation model. OpenSPM first leverages semantically conditioned 3D perception and Kalman filtering to track continuous 6D poses. It then extracts key spatial poses from human demonstrations, keeping them as transferable, object-centric spatial persistent memory entries. During inference, OpenSPM retrieves relevant memory entries in terms of natural language instructions, transfers the spatial poses to new scenes using SE(3) transformations, and generates high-frequency action chunks via a lightweight conditional flow-matching model. Combined with real-time proprioceptive state feedback and terminal residual correction, the system effectively suppresses trajectory error accumulation. Evaluated on ten LIBERO-GOAL tasks, OpenSPM achieves an 85.6% success rate and an equivalent control frequency of 1033.3 Hz, while requiring minimal inference AI computing power. Extensive ablations illustrate that structured spatial persistent memory and closed-loop residual correction play a crucial role in reliable, high-frequency robotic manipulation.",
  "published": "2026-06-29",
  "updated": "2026-07-05",
  "year": "2026",
  "authors": [
   "Iok Tong Lei",
   "Qingchen Xie",
   "Yifan Wang",
   "Yap Ying Jie",
   "Zhidong Deng"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "OpenSPM, an open environment spatial persistent memory framework consisting of spatial pose memory and flow-matching action generation model, is proposed, which achieves an 85.6% success rate and an equivalent control frequency, while requiring minimal inference AI computing power.",
  "doi": "10.48550/arXiv.2606.29936",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Iok Tong Lei",
    "id": "2445688588",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Qingchen Xie",
    "id": "2445704623",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yifan Wang",
    "id": "2446321830",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yap Ying Jie",
    "id": "2445686073",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Zhidong Deng",
    "id": "2302705800",
    "h_index": 2,
    "papers": 9
   }
  ],
  "comment": "Preprint",
  "topics": [
   "vla",
   "egocentric-data"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.29936v2",
  "pdf_url": "https://arxiv.org/pdf/2606.29936v2",
  "html_url": "https://arxiv.org/html/2606.29936v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.29783",
  "slug": "falcontrack-photorealistic-auto-labeled-perception-and-physics-aware-v",
  "title": "FalconTrack: Photorealistic Auto-Labeled Perception and Physics-Aware Vision-Based Aerial Tracking",
  "abstract": "Vision-based aerial tracking is critical in GPS-denied environments. Reliable perception for tracking depends on large-scale labeled data, yet most photorealistic datasets rely on heavy manual annotation and are time-consuming to produce. We present FalconTrack, a unified perception-and-tracking framework that (i) leverages a photorealistic editable simulator for automated label generation and (ii) combines multi-head perception with physics-aware tracking for zero-shot sim-to-real transfer. FalconTrack provides an automated labeling pipeline in a Gaussian Splatting simulator that isolates target Gaussians from short object videos and composites them with randomized backgrounds to generate RGB, mask, class, and 6-DoF pose labels, producing about 10k labeled images in under 20 minutes. Using this dataset, we train a multi-head perception module with staged learning and reprojection consistency, and fuse its outputs with class-conditioned dynamics priors in an EKF for tracking. Our perception model outperforms two baselines and reaches 96-100% class accuracy in zero-shot sim-to-real transfer on three geometrically diverse objects and two environments, while maintaining consistent performance in unseen simulated and real scenes. In real hardware closed-loop visual tracking, the onboard system runs at about 25 Hz and achieves 100% success in sim-to-real F1-tenth and gate tracking in five trajectories across two environments, while a mask-centered vision baseline drops to 60% success on F1-tenth during fast out-of-view scenarios.",
  "published": "2026-06-29",
  "updated": "2026-06-29",
  "year": "2026",
  "authors": [
   "Yan Miao",
   "Karteek Gandiboyina",
   "Noah Giles",
   "Hideki Okamoto",
   "Bardh Hoxha",
   "Georgios Fainekos",
   "Sayan Mitra"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": " FalconTrack is presented, a unified perception-and-tracking framework that (i) leverages a photorealistic editable simulator for automated label generation and (ii) combines multi-head perception with physics-aware tracking for zero-shot sim-to-real transfer.",
  "doi": "10.48550/arXiv.2606.29783",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yan Miao",
    "id": "2332294391",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Karteek Gandiboyina",
    "id": "2445701568",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Noah Giles",
    "id": "2438591474",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Hideki Okamoto",
    "id": "2256348492",
    "h_index": 6,
    "papers": 26
   },
   {
    "name": "Bardh Hoxha",
    "id": "1729953",
    "h_index": 19,
    "papers": 99
   },
   {
    "name": "Georgios Fainekos",
    "id": "1682745",
    "h_index": 41,
    "papers": 203
   },
   {
    "name": "Sayan Mitra",
    "id": "2315472917",
    "h_index": 3,
    "papers": 16
   }
  ],
  "comment": "",
  "topics": [
   "sim2real",
   "spatial-3d",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.29783v1",
  "pdf_url": "https://arxiv.org/pdf/2606.29783v1",
  "html_url": "https://arxiv.org/html/2606.29783v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2606.29738",
  "slug": "mygo-splat-multi-objective-closed-loop-geometric-feedback-for-rgb-only",
  "title": "MyGO-Splat: Multi-Objective Closed-Loop Geometric Feedback for RGB-Only Gaussian SLAM",
  "abstract": "Real-time monocular Simultaneous Localization and Mapping (SLAM) fundamentally suffers from scale ambiguity and a lack of geometric self-correction. While 3D Gaussian Splatting (3DGS) enables high-fidelity rendering, existing RGB-only systems remain open-loop because depth priors are injected into mapping but refined geometry cannot effectively regulate tracking drift. We present MyGO-Splat, a closed-loop Gaussian SLAM framework that analytically rasterizes Gaussian primitives into pixel-wise depth and surface normals, allowing the map to actively supervise camera pose optimization. To bridge monocular priors and scale consistency, our framework introduces scale-aware adaptive alignment that projects foundation-model depth estimates into the globally optimized Gaussian space, forming a self-correcting cycle for scale feedback. Extensive evaluations show that this closed-loop design improves scale stability and appearance-geometry consistency, achieving performance comparable to RGB-D methods while using only monocular input.",
  "published": "2026-06-29",
  "updated": "2026-06-29",
  "year": "2026",
  "authors": [
   "Fan Zhu",
   "Ziyu Chen",
   "Zhenjun Zhao",
   "Zhisong Xu",
   "Hui Zhu",
   "Mingrui Li",
   "Chunmao Jiang",
   "Javier Civera"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "MyGO-Splat is a closed-loop Gaussian SLAM framework that analytically rasterizes Gaussian primitives into pixel-wise depth and surface normals, allowing the map to actively supervise camera pose optimization, achieving performance comparable to RGB-D methods while using only monocular input.",
  "doi": "10.48550/arXiv.2606.29738",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Fan Zhu",
    "id": "2302523558",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Ziyu Chen",
    "id": "2278840704",
    "h_index": 4,
    "papers": 21
   },
   {
    "name": "Zhenjun Zhao",
    "id": "2152618896",
    "h_index": 9,
    "papers": 20
   },
   {
    "name": "Zhisong Xu",
    "id": "2299327365",
    "h_index": 3,
    "papers": 15
   },
   {
    "name": "Hui Zhu",
    "id": "2264712840",
    "h_index": 5,
    "papers": 18
   },
   {
    "name": "Mingrui Li",
    "id": "2237955481",
    "h_index": 11,
    "papers": 26
   },
   {
    "name": "Chunmao Jiang",
    "id": "2261096846",
    "h_index": 4,
    "papers": 25
   },
   {
    "name": "Javier Civera",
    "id": "2315512243",
    "h_index": 2,
    "papers": 7
   }
  ],
  "comment": "IROS 2026",
  "topics": [
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.29738v1",
  "pdf_url": "https://arxiv.org/pdf/2606.29738v1",
  "html_url": "https://arxiv.org/html/2606.29738v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2606.29731",
  "slug": "real-time-compliance-and-position-control-of-a-hyper-redundant-soft-ro",
  "title": "Real-Time Compliance and Position Control of a Hyper-redundant Soft Robotic Arm",
  "abstract": "Robots working in unstructured or partially unobservable environments must combine accurate motion with physical compliance that can passively correct contact misalignment. Soft robots provide this compliance but have struggled to precisely control their tip compliance and position. This paper presents a robot architecture designed around that control problem: a 7-link arm whose six articulated joints provide twelve independently driven revolute axes, each actuated by an antagonistic pair of pneumatic muscles, so that every axis can simultaneously change its angle and linearly adjust its stiffness. The rigid articulated backbone makes the tip compliance and position of the arm predictable enough to be commanded quantitatively in real time. The robot employs a unified iterative inverse-kinematics and inverse-compliance controller to achieve simultaneous, quantitative control of both compliance and position. The task-space compliance and kinematics models and the control law are derived and verified on both the physical arm and a matched simulation. Simulation is then used to study how the same framework extends to other arm morphologies. Finally, the arm demonstrates tasks that have been difficult for both rigid and soft arms: rejecting disturbances while writing on a moving whiteboard, and passively correcting hidden misalignment during a key-insertion and drawer-opening task. That these tasks succeed under so straightforward a controller is evidence for the advantage of this algorithm-informed structural design.",
  "published": "2026-06-29",
  "updated": "2026-06-29",
  "year": "2026",
  "authors": [
   "Runze Zuo",
   "Tianhua Zou",
   "Naike Wu",
   "Mingyuan Li",
   "Daniel Bruder"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "10.48550/arXiv.2606.29731",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Runze Zuo",
    "id": "2128374703",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Tian Zou",
    "id": "2303578573",
    "h_index": 5,
    "papers": 19
   },
   {
    "name": "Naike Wu",
    "id": "2316761615",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Ming-Yang Li",
    "id": "2448598133",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Daniel Bruder",
    "id": "2337107894",
    "h_index": 1,
    "papers": 6
   }
  ],
  "comment": "",
  "topics": [
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.29731v1",
  "pdf_url": "https://arxiv.org/pdf/2606.29731v1",
  "html_url": "https://arxiv.org/html/2606.29731v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.29677",
  "slug": "lateral-string-stability-for-vehicle-platoons",
  "title": "Lateral String Stability for Vehicle Platoons",
  "abstract": "Connected and automated vehicle (CAV) platooning promises gains in energy efficiency and traffic throughput and, most critically, in safety. These safety benefits hinge on string stability, which determines how disturbances propagate along a platoon. While longitudinal string stability is well studied, lateral string stability, which governs the propagation of path-tracking errors that can lead to unsafe deviations from the intended path, remains underexplored. Its importance is increasing as autonomous vehicles rely more heavily on onboard sensing and map-free navigation, where sensor occlusion and dense formations amplify safety risks. This paper presents a new framework for lateral string stability that directly addresses safety-critical path-relative tracking errors and enables consistent comparison across vehicles following the same road geometry. Central to this framework is an arc-length (Eulerian) viewpoint, a departure from traditional analyses, that clarifies how tracking errors at a given point on the path propagate from one vehicle to the next. A formal definition of lateral string stability is introduced along with two control strategies: an onboard-sensing-only controller and a novel learn-from-predecessor approach utilizing vehicle-to-vehicle (V2V) communication. We show that onboard sensing alone cannot guarantee attenuation of path-tracking errors, imposing a fundamental safety limitation, whereas V2V communication enables true error attenuation.",
  "published": "2026-06-29",
  "updated": "2026-06-29",
  "year": "2026",
  "authors": [
   "Sixu Li",
   "Swaroop Darbha",
   "Yang Zhou"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "",
  "doi": "10.48550/arXiv.2605.01731",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Sixu Li",
    "id": "2268545632",
    "h_index": 7,
    "papers": 18
   },
   {
    "name": "S. Darbha",
    "id": "2204712",
    "h_index": 33,
    "papers": 212
   },
   {
    "name": "Yang Zhou",
    "id": "2308713489",
    "h_index": 2,
    "papers": 8
   }
  ],
  "comment": "",
  "topics": [
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.29677v1",
  "pdf_url": "https://arxiv.org/pdf/2606.29677v1",
  "html_url": "https://arxiv.org/html/2606.29677v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2606.29548",
  "slug": "vista-dz-visual-semantic-trajectory-adaptation-for-personalized-dilemm",
  "title": "VISTA-DZ: Visual Semantic Trajectory Adaptation for Personalized Dilemma Zone Prediction",
  "abstract": "Driver decision making in the dilemma zone at signalized intersections is safety critical, as vehicles approaching a yellow signal must decide whether to stop or proceed within limited time and distance margins. Accurate prediction of both stop-go decisions and decision timing is important for adaptive signal control, advanced driver assistance systems, and human-centered intelligent transportation applications. However, dilemma zone behavior is strongly driver dependent. Similar approach trajectories may lead to different decisions across drivers because of differences in risk preference, braking habit, and decision threshold. Existing personalized models often rely on handcrafted scalar descriptors, which provide useful but limited summaries of individual behavior. This paper proposes VISTA-DZ, a semantic-profile-conditioned framework for personalized stop-go and decision-time prediction. Historical trajectories are converted into visual representations, interpreted by a vision-language model to generate behavioral profiles, and encoded as semantic embeddings to condition a dual-output prediction network. The final model combines a bidirectional GRU encoder, driver-conditioned multi-head cross-attention, and Feature-wise Linear Modulation for temporal evidence selection and feature adaptation. Experiments on the SDZ dataset and a newly collected FDZ dataset show that VISTA-DZ outperforms trajectory-only and handcrafted personalization baselines, achieving 93.26% in-domain simulation accuracy and 90.22% mean accuracy across 20 held-out simulation drivers. Cross-domain results further show feasible zero-shot simulation-to-real transfer and better real-world generalization when simulation data are combined with limited field data.",
  "published": "2026-06-28",
  "updated": "2026-06-28",
  "year": "2026",
  "authors": [
   "Chuheng Wei",
   "Ziye Qin",
   "Ziran Wang",
   "Guoyuan Wu"
  ],
  "author_count": 4,
  "categories": [
   "cs.LG",
   "cs.AI",
   "cs.DB",
   "cs.HC",
   "cs.RO"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "VISTA-DZ is proposed, a semantic-profile-conditioned framework for personalized stop-go and decision-time prediction that outperforms trajectory-only and handcrafted personalization baselines, and cross-domain results further show feasible zero-shot simulation-to-real transfer and better real-world generalization when simulation data are combined with limited field data.",
  "doi": "10.48550/arXiv.2606.29548",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Chuheng Wei",
    "id": "2284284573",
    "h_index": 7,
    "papers": 25
   },
   {
    "name": "Ziye Qin",
    "id": "2295562577",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Ziran Wang",
    "id": "4141749",
    "h_index": 34,
    "papers": 159
   },
   {
    "name": "Guoyuan Wu",
    "id": "2267489645",
    "h_index": 7,
    "papers": 27
   }
  ],
  "comment": "This manuscript is currently under review",
  "topics": [
   "video-generation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.29548v1",
  "pdf_url": "https://arxiv.org/pdf/2606.29548v1",
  "html_url": "https://arxiv.org/html/2606.29548v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.29517",
  "slug": "core-common-outcome-regularities-from-action-free-visual-demonstration",
  "title": "CORE: Common Outcome Regularities from Action-Free Visual Demonstrations for Robot Manipulation",
  "abstract": "Robot imitation learning often relies on costly robot demonstrations, while abundant action-free visual demonstrations, such as human videos, are difficult to use because they lack robot-executable actions and suffer from embodiment gaps. We propose CORE, a policy learning framework that extracts Common Outcome Regularities (CORE) from visual demonstrations. Rather than transferring explicit actions across embodiments, CORE exploits a key observation: although successful trajectories for the same task can be diverse, their terminal states often share stable object configurations, spatial relations, and contact constraints. CORE first trains a terminal outcome encoder with contrastive and auxiliary temporal objectives, then aggregates successful terminal embeddings into visual goal prototypes, and finally injects these prototypes as global goal conditions into robot policies. Compared with language instructions, visual goal prototypes provide more concrete geometric and physical constraints for task completion. Across Meta-World, RoboTwin 2.0, and real-world manipulation, CORE improves the average success rate of the corresponding policy backbones by up to +3.9, +11.1, and +17.0 percentage points, respectively, and outperforms text-conditioned variants under the evaluated settings. The project and code are available at https://logssim.github.io/CORE.github.io/.",
  "published": "2026-06-28",
  "updated": "2026-08-03",
  "year": "2026",
  "authors": [
   "Juyi Sheng",
   "Mingxin Tan",
   "Jincheng Li",
   "Mengyuan Liu"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "CORE is a policy learning framework that extracts Common Outcome Regularities (CORE) from visual demonstrations and improves the average success rate of the corresponding policy backbones by up to +3.9, +11.1, and +17.0 percentage points, respectively.",
  "doi": "10.48550/arXiv.2606.29517",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Juyi Sheng",
    "id": "2357933816",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Jincheng Li",
    "id": "2445726576",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Mingxing Tan",
    "id": "2249538482",
    "h_index": 7,
    "papers": 17
   },
   {
    "name": "Mengyuan Liu",
    "id": "2358260559",
    "h_index": 3,
    "papers": 15
   }
  ],
  "comment": "",
  "topics": [
   "egocentric-data",
   "imitation-diffusion"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.29517v2",
  "pdf_url": "https://arxiv.org/pdf/2606.29517v2",
  "html_url": "https://arxiv.org/html/2606.29517v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.29237",
  "slug": "mope-motion-permanence-for-robust-monocular-gaussian-mapping-in-dynami",
  "title": "MoPe: Motion Permanence for Robust Monocular Gaussian Mapping in Dynamic Environments",
  "abstract": "Robust robot autonomy depends on scene representations that remain stable enough to support localization, navigation, and downstream decision making in dynamic environments. Monocular Gaussian Splatting SLAM provides high-fidelity mapping, but current uncertainty-aware methods still treat dynamic regions largely as per-frame observations. This makes the representation effectively memoryless: when a pedestrian slows, pauses, or reappears after occlusion, the current frame may look static, allowing dynamic content to be absorbed into the map and leaving persistent ghosting artifacts. We argue that this failure reflects a representation-level mismatch. Dynamic-ness is not an instantaneous appearance property, but a temporal property defined by motion history. Building on this view, we introduce Motion Permanence: the principle that an object's dynamic identity should persist over time rather than be re-decided from each frame independently. We realize this principle in MoPe, a memory-aware uncertainty filter for monocular Gaussian mapping. MoPe propagates the historical dynamic posterior through geometry-consistent SE(3) warping and fuses it with current-frame evidence using bounded Bayesian log-odds updates. The resulting persistent posterior guides tracking, mapping, dynamic-aware Gaussian insertion, and Gaussian-level post-cleanup. On Wild-SLAM, Bonn, and TUM sequences, MoPe improves tracking robustness and reduces residual ghosting, with the strongest gains on dynamic-human scenes that most directly violate the memoryless assumption. These results show that maintaining temporal dynamic state inside the scene representation is a practical step toward more reliable representation-centric autonomy in changing real-world environments.",
  "published": "2026-06-28",
  "updated": "2026-06-28",
  "year": "2026",
  "authors": [
   "Qixin Xiao"
  ],
  "author_count": 1,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "RSS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "MoPe, a memory-aware uncertainty filter for monocular Gaussian mapping that propagates the historical dynamic posterior through geometry-consistent SE(3) warping and fuses it with current-frame evidence using bounded Bayesian log-odds updates, shows that maintaining temporal dynamic state inside the scene representation is a practical step toward more reliable representation-centric autonomy in changing real-world environments.",
  "doi": "10.48550/arXiv.2606.29237",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Qixin Xiao",
    "id": "2397572710",
    "h_index": 2,
    "papers": 7
   }
  ],
  "comment": "RSS 2026 Workshop",
  "topics": [
   "spatial-3d",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.29237v1",
  "pdf_url": "https://arxiv.org/pdf/2606.29237v1",
  "html_url": "https://arxiv.org/html/2606.29237v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2606.28995",
  "slug": "hj-safedmp-hamilton-jacobi-reachability-guided-dynamic-movement-primit",
  "title": "HJ-SafeDMP: Hamilton-Jacobi Reachability-Guided Dynamic Movement Primitives for Provably Safe Robot Motion",
  "abstract": "Robots deployed in safety-critical environments must execute motions that are simultaneously robust to disturbances and provably safe from collisions. Dynamic Movement Primitives (DMPs) offer inherent stability, temporal flexibility, and efficient trajectory generalization from single demonstrations, but they lack formal safety certificates. Conversely, Hamilton-Jacobi (HJ) Reachability analysis provides a principled framework for computing worst-case safety margins and forward-invariant safe sets, but classical grid-based methods suffer from the curse of dimensionality and are impractical for real-time control. This paper introduces HJ-SafeDMP, a framework that integrates DMPs with learned HJ Reachability-based safety value functions to achieve provably safe, robust, and computationally efficient robot motion. We learn a Control Barrier Value Function (CBVF) from offline demonstration data using a model-free, finite-difference HJ recursion and deploy it as a real-time safety filter via a closed-form control law that modulates the DMP output. Unlike optimization-based CBF-QP approaches, our method achieves safety filtering without online quadratic program solves, preserving the computational efficiency of DMPs. We further incorporate an expectile-based offline learning objective that avoids querying out-of-distribution actions, and a conformal prediction calibration step that provides finite-sample probabilistic safety coverage. Experimental evaluation on a 7-DOF robot manipulator demonstrates that HJ-SafeDMP achieves formal safety guarantees with orders-of-magnitude faster execution than optimization-based baselines, while maintaining the robustness and adaptability of DMPs for human-robot interaction.",
  "published": "2026-06-27",
  "updated": "2026-06-27",
  "year": "2026",
  "authors": [
   "Siddhanth Ramesh",
   "Ravi Prakash"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Experimental evaluation on a 7-DOF robot manipulator demonstrates that HJ-SafeDMP achieves formal safety guarantees with orders-of-magnitude faster execution than optimization-based baselines, while maintaining the robustness and adaptability of DMPs for human-robot interaction.",
  "doi": "10.48550/arXiv.2606.28995",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Siddhanth Ramesh",
    "id": "2445690112",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Ravi Prakash",
    "id": "2350512711",
    "h_index": 1,
    "papers": 7
   }
  ],
  "comment": "8 pages, 1 figure",
  "topics": [
   "rl-control",
   "hri",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.28995v1",
  "pdf_url": "https://arxiv.org/pdf/2606.28995v1",
  "html_url": "https://arxiv.org/html/2606.28995v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.28813",
  "slug": "human2any-human-to-robot-transfer-via-constraint-aware-compositional-p",
  "title": "Human2Any: Human-to-Robot Transfer via Constraint-Aware Compositional Planning",
  "abstract": "Human videos are a scalable source of supervision for robot manipulation, as they are abundant and naturally capture rich object interactions. However, transferring human demonstrations to robots remains challenging due to embodiment mismatch, scene variation, and robot-specific feasibility constraints. We present Human2Any, a framework for learning reusable object-centric interaction priors from human videos without requiring real-world robot demonstrations in the target task contexts. Human2Any represents manipulation through object-object interaction motion, capturing task-relevant scene changes while abstracting away embodiment-specific details. It composes learned interaction priors with robot-side feasibility reasoning and motion planning, allowing the same human-derived knowledge to adapt to different embodiments, scene geometries, and task contexts. We validate Human2Any across diverse manipulation settings, including real-world experiments on a Franka tabletop setup and an RBY-1 humanoid mobile robot, demonstrating robust interaction-centric manipulation without real-world robot training data. Project website: https://human2any.github.io/.",
  "published": "2026-06-27",
  "updated": "2026-06-27",
  "year": "2026",
  "authors": [
   "Shuo Cheng",
   "Chuye Zhang",
   "Alfred Cueva",
   "Caelan Garrett",
   "Ajay Mandlekar",
   "Danfei Xu"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Human2Any is presented, a framework for learning reusable object-centric interaction priors from human videos without requiring real-world robot demonstrations in the target task contexts, and validated across diverse manipulation settings.",
  "doi": "10.48550/arXiv.2606.28813",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Shuo Cheng",
    "id": "2232588215",
    "h_index": 6,
    "papers": 16
   },
   {
    "name": "Chuye Zhang",
    "id": "2215500204",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "A. Cueva",
    "id": "104623039",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Caelan Reed Garrett",
    "id": "1834058",
    "h_index": 24,
    "papers": 59
   },
   {
    "name": "A. Mandlekar",
    "id": "49686756",
    "h_index": 36,
    "papers": 67
   },
   {
    "name": "Danfei Xu",
    "id": "2260291195",
    "h_index": 7,
    "papers": 14
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "egocentric-data"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.28813v1",
  "pdf_url": "https://arxiv.org/pdf/2606.28813v1",
  "html_url": "https://arxiv.org/html/2606.28813v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.28804",
  "slug": "vipsim-collaborating-visual-and-parameter-spaces-for-consistent-long-h",
  "title": "ViPSim: Collaborating Visual and Parameter Spaces for Consistent Long-Horizon Embodied World Models",
  "abstract": "Embodied World Models (EWMs) have emerged as a scalable and risk-free paradigm for advancing embodied intelligence, enabling the safety-critical evaluation of Vision-Language-Action systems. However, their reliability as evaluation benchmarks and foundational simulators is often hindered by the representation gap between low-dimensional actions and high-dimensional video synthesis. This gap results in a lack of geometric correspondence, manifesting as accumulated trajectory drift and inconsistent robot-object interactions during long-horizon rollouts. To bridge this gap, we propose ViPSim, a framework that achieves consistent long-horizon generation through the synergistic collaboration of Visual and Parameter Spaces. We define the Visual Space as a domain of explicit spatial priors, integrating pixel-aligned projections of end-effector pose, camera perspectives, depth-informed scene geometry, and robotic morphological masks to provide dense structural grounding. Concurrently, the Parameter Space serves as a domain of numerical drivers, injecting raw action sequences and camera matrices to provide precise motion guidance. By unifying these two spaces, ViPSim ensures that the generated states are simultaneously anchored by geometric boundaries and steered by numerical commands. Extensive experiments demonstrate that ViPSim is backbone-agnostic and significantly enhances trajectory consistency. Notably, our approach exhibits emergent capabilities in generating complex interactions with deformable objects (e.g., cloth folding) and maintains robust performance in out-of-distribution and cross-embodiment scenarios, providing a high-fidelity foundation for the automated evaluation and predictive control of embodied agents.",
  "published": "2026-06-27",
  "updated": "2026-06-27",
  "year": "2026",
  "authors": [
   "Longyu Chen",
   "Heng Li",
   "Wei Yang",
   "Manqi Zhao",
   "Dongsheng Jiang"
  ],
  "author_count": 5,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "RSS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes ViPSim, a framework that achieves consistent long-horizon generation through the synergistic collaboration of Visual and Parameter Spaces, and exhibits emergent capabilities in generating complex interactions with deformable objects and maintains robust performance in out-of-distribution and cross-embodiment scenarios.",
  "doi": "10.48550/arXiv.2606.28804",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Long Chen",
    "id": "2450163513",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Heng Li",
    "id": "2445715950",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Wei Yang",
    "id": "2266191697",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Manqi Zhao",
    "id": "2354127475",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Dongsheng Jiang",
    "id": "2316516611",
    "h_index": 4,
    "papers": 12
   }
  ],
  "comment": "Accepted to Robotics: Science and Systems (RSS) 2026",
  "topics": [
   "world-models",
   "vla",
   "sim2real",
   "foundation-pretraining",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.28804v1",
  "pdf_url": "https://arxiv.org/pdf/2606.28804v1",
  "html_url": "https://arxiv.org/html/2606.28804v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2606.28760",
  "slug": "vision-language-models-for-deployable-social-robot-navigation-bridging",
  "title": "Vision-Language Models for Deployable Social Robot Navigation: Bridging Semantic Reasoning and Low-Level Control",
  "abstract": "Social robot navigation (SRN) requires more than geometric path planning; it demands understanding human intentions, social norms, and contextual cues to generate socially compliant behaviors. Although classical navigation methods provide reliable metric planning and collision avoidance, they often lack the semantic reasoning capabilities necessary for operation in complex human-centered environments. Recent advances in Vision-Language Models (VLMs) have opened new opportunities for SRN by enabling high-level VLM understanding, commonsense reasoning, and natural language interaction. However, a fundamental challenge remains: how to integrate VLMs into real-time, safety-critical navigation systems and reliably translate their high-level reasoning into grounded navigation actions. In this survey, we present a unified perspective of VLM-based SRN and organize existing approaches into three interconnected components: high-level VLM reasoning, low-level planning and control, and intermediate mechanisms that bridge reasoning and action. Based on this perspective, we propose a structured roadmap for coupling VLMs with navigation systems, covering semantic reasoning, evaluators, spatial grounding, intermediate representations, and control modules. The roadmap highlights both the strengths of VLMs and the necessity of hybrid architectures for practical deployment. We further review representative datasets and evaluation platforms developed for SRN. Finally, we discuss key open challenges. This survey aims to provide a foundation for building reliable, socially compliant, and deployable VLM-enabled navigation systems.",
  "published": "2026-06-27",
  "updated": "2026-06-27",
  "year": "2026",
  "authors": [
   "Runji Cai",
   "Toshihiko Yamasaki",
   "Ling Xiao"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A unified perspective of VLM-based SRN is presented and a structured roadmap for coupling VLMs with navigation systems is proposed, covering semantic reasoning, evaluators, spatial grounding, intermediate representations, and control modules.",
  "doi": "10.48550/arXiv.2606.28760",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Runji Cai",
    "id": "2392872228",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Toshihiko Yamasaki",
    "id": "2269145188",
    "h_index": 4,
    "papers": 17
   },
   {
    "name": "Ling Xiao",
    "id": "2269704272",
    "h_index": 5,
    "papers": 30
   }
  ],
  "comment": "",
  "topics": [
   "navigation",
   "hri",
   "video-generation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.28760v1",
  "pdf_url": "https://arxiv.org/pdf/2606.28760v1",
  "html_url": "https://arxiv.org/html/2606.28760v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.28757",
  "slug": "a-physics-grounded-benchmark-for-multi-agent-dynamics-in-world-models",
  "title": "A Physics-Grounded Benchmark for Multi-Agent Dynamics in World Models",
  "abstract": "Generative world models hold immense promise as scalable simulators for autonomous systems, particularly for synthesizing rare but safety-critical multi-agent interactions, such as vehicle collisions. However, current evaluation paradigms index heavily on visual fidelity and semantic alignment, leaving a critical blind spot: they cannot reliably quantify whether generated dynamics actually obey the fundamental physical laws required for reliable simulation. Assessing this physical plausibility is inherently difficult due to a lack of physical metrics and the challenge of extracting metric-scale kinematics from uncalibrated video rollouts. To bridge this gap, we introduce CrashTwin, a physics-grounded evaluation framework designed to stress-test the physical trustworthiness of world models. CrashTwin couples a diverse dataset of multi-agent collision scenarios, comprising 25K controllable synthetic and 12K in-the-wild real-world collision sequences with a novel calibration-free reconstruction pipeline, enabling the recovery of 3D physical attributes directly from world model rollouts. We propose a diagnostic suite that systematically evaluates three dimensions: spatio-temporal consistency, momentum and kinetic energy conservation, and world-dynamics integrity. Extensive benchmarking of state-of-the-art models reveals a crucial insight: high perceptual quality frequently masks severe physical violations during complex interactions. By quantitatively exposing these failure modes, CrashTwin provides a vital diagnostic tool for developing physically grounded world models capable of reliable real-world simulation.",
  "published": "2026-06-27",
  "updated": "2026-07-07",
  "year": "2026",
  "authors": [
   "Nuo Chen",
   "Lulin Liu",
   "Zihao Li",
   "Ziyao Zeng",
   "Zihao Zhu",
   "Wenyan Cong",
   "Junyuan Hong",
   "Yunhao Yang",
   "Zhengzhong Tu",
   "Yan Wang",
   "Boris Ivanovic",
   "Marco Pavone",
   "Zhangyang Wang",
   "Yang Zhou",
   "Zhiwen Fan"
  ],
  "author_count": 15,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "CrashTwin is introduced, a physics-grounded evaluation framework designed to stress-test the physical trustworthiness of world models, and a diagnostic suite that systematically evaluates three dimensions: spatio-temporal consistency, momentum and kinetic energy conservation, and world-dynamics integrity.",
  "doi": "10.48550/arXiv.2606.28757",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Nuo Chen",
    "id": "2257286549",
    "h_index": 5,
    "papers": 15
   },
   {
    "name": "Lu Liu",
    "id": "2330453339",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Zihao Li",
    "id": "2445705752",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Ziyao Zeng",
    "id": "2282099512",
    "h_index": 6,
    "papers": 21
   },
   {
    "name": "Zihao Zhu",
    "id": "2385506413",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Wenyan Cong",
    "id": "2292160328",
    "h_index": 7,
    "papers": 16
   },
   {
    "name": "Junyuan Hong",
    "id": "2344645677",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Yunhao Yang",
    "id": "2292090208",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Zhengzhong Tu",
    "id": "2292032805",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Yan Wang",
    "id": "2353108236",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "B. Ivanovic",
    "id": "145156173",
    "h_index": 36,
    "papers": 95
   },
   {
    "name": "Marco Pavone",
    "id": "2237790577",
    "h_index": 26,
    "papers": 65
   },
   {
    "name": "Zhangyang Wang",
    "id": "2336081383",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Yang Zhou",
    "id": "2292121408",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Zhiwen Fan",
    "id": "2346644913",
    "h_index": 5,
    "papers": 20
   }
  ],
  "comment": "34 pages, 9 figures, 12 tables",
  "topics": [
   "world-models",
   "sim2real",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.28757v2",
  "pdf_url": "https://arxiv.org/pdf/2606.28757v2",
  "html_url": "https://arxiv.org/html/2606.28757v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2606.28720",
  "slug": "cubifygs-object-centric-3d-gaussian-splatting-for-lifelong-dynamic-sce",
  "title": "CubifyGS: Object-Centric 3D Gaussian Splatting for Lifelong Dynamic Scene Maintenance",
  "abstract": "Lifelong scene mapping under rigid object rearrangement remains a fundamental challenge in robotics. While 3D Gaussian Splatting (3DGS) enables high-fidelity modeling, primitive-level updates often cause persistent ghosting and slow recovery. We propose CubifyGS, an object-level mapping framework that shifts dynamic maintenance from passive re-optimization to active asset management. CubifyGS models movable instances as reusable Gaussian assets, detects object appearance and disappearance, and updates maps through asset retrieval, rigid transformation, and explicit pruning rather than reconstruction from scratch. To address geometric voids and local photometric mismatch after such edits, we further propose an event-triggered adaptive optimization strategy that focuses computation on affected regions. We validate our approach on a newly constructed high-fidelity dynamic benchmark, demonstrating that CubifyGS improves artifact suppression and maintenance efficiency over representative reproducible baselines in the evaluated object-rearrangement setting.",
  "published": "2026-06-27",
  "updated": "2026-07-04",
  "year": "2026",
  "authors": [
   "Bohan Ren",
   "Dianyi Yang",
   "Shiyang Liu",
   "Yu Gao",
   "Jiadong Tang",
   "Zhilin Lai",
   "Yi Yang",
   "Mengyin Fu"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes CubifyGS, an object-level mapping framework that shifts dynamic maintenance from passive re-optimization to active asset management, and validate the approach on a newly constructed high-fidelity dynamic benchmark, demonstrating that CubifyGS improves artifact suppression and maintenance efficiency over representative reproducible baselines in the evaluated object-rearrangement setting.",
  "doi": "10.48550/arXiv.2606.28720",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Bohan Ren",
    "id": "2373599171",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Dianyi Yang",
    "id": "2311192060",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Shiyang Liu",
    "id": "2373889634",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Yu Gao",
    "id": "2311320606",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Jiadong Tang",
    "id": "2311446130",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Zhilin Lai",
    "id": "2445686644",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yi Yang",
    "id": "2374467530",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Mengyin Fu",
    "id": "2240528626",
    "h_index": 6,
    "papers": 33
   }
  ],
  "comment": "Accepted to IROS 2026. 8 pages, 5 figures, 4 tables",
  "topics": [
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.28720v2",
  "pdf_url": "https://arxiv.org/pdf/2606.28720v2",
  "html_url": "https://arxiv.org/html/2606.28720v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2606.28323",
  "slug": "dexcompose-reusing-dexterous-policies-for-multi-task-manipulation-with",
  "title": "DexCompose: Reusing Dexterous Policies for Multi-Task Manipulation with a Single Hand",
  "abstract": "Dexterous manipulation policies can solve individual skills, but composing them to perform multiple tasks with a single hand remains challenging. Adding a new task on top of an existing manipulation skill often imposes conflicting demands on overlapping fingers and contact modes, causing destructive interference between preserving an existing manipulation outcome and executing a new one. We propose DexCompose, a role-aware residual composition framework that reuses pretrained dexterous policies for multi-task manipulation through explicit finger-level action ownership. Given two pretrained full-hand policies, DexCompose first collects successful post-task states from the first skill and performs release tests over candidate finger masks to identify which fingers are necessary for maintaining the established skill state. It then trains two asymmetric residual modules: a bounded residual stabilizer for task preservation, and a context-aware residual that adapts the frozen downstream policy only within the action subspace assigned to the new task. We evaluate the framework on 16 composite dexterous manipulation tasks spanning four object-retention skills and four downstream interactions. DexCompose achieves a 77.4% average composite success rate, demonstrating that structural action ownership with dual residuals offers a promising direction for composing dexterous skills beyond conventional policy chaining.",
  "published": "2026-06-26",
  "updated": "2026-06-26",
  "year": "2026",
  "authors": [
   "Dihong Huang",
   "Zhenyu Wei",
   "Zhuxiu Xu",
   "Yunchao Yao",
   "Sikai Li",
   "Mingyu Ding"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 3,
  "influential_citations": 0,
  "tldr": "",
  "doi": "10.48550/arXiv.2606.28323",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Di Huang",
    "id": "2292899023",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Zhenyu Wei",
    "id": "2394124879",
    "h_index": 4,
    "papers": 16
   },
   {
    "name": "Zhuxiu Xu",
    "id": "2308983686",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Yunchao Yao",
    "id": "2352910959",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Sikai Li",
    "id": "2283135687",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Mingyu Ding",
    "id": "2346837065",
    "h_index": 4,
    "papers": 15
   }
  ],
  "comment": "Project page: https://devon018.github.io/DexCompose-Webpage/",
  "topics": [
   "dexterous-manipulation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.28323v1",
  "pdf_url": "https://arxiv.org/pdf/2606.28323v1",
  "html_url": "https://arxiv.org/html/2606.28323v1",
  "code_url": "https://devon018.github.io/DexCompose-Webpage/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.6
 },
 {
  "id": "2606.28237",
  "slug": "unleashing-infinite-motion-scaling-expressive-quadrupedal-motion-via-g",
  "title": "Unleashing Infinite Motion: Scaling Expressive Quadrupedal Motion via Generative Video Priors",
  "abstract": "Quadruped robots have achieved remarkable locomotion, yet their behavioral repertoire remains confined to a few gaits--far from the expressive, companion-like presence long envisioned for them. Attempts to import the humanoid recipe of large-scale motion data have inherited one tacit assumption: that robot motion must first pass through an animal body, making data collection dependent on cooperative animals, reconstruction fragile across species, and retargeting ill-posed across incompatible morphologies. We propose Uni-Mo, a fully automated pipeline that removes the animal from the loop by reframing data scarcity as a generation problem: an LLM proposes motion prompts, a video diffusion model synthesizes the corresponding robot behaviors, and the generated videos are lifted into 3D reference trajectories used to train tracking policies deployed on a real Unitree Go2. To make naively-drifting generations reliably extractable, we introduce an Identity Consistency Loss that enforces appearance coherence across frames. We release Quad-Imaginarium at https://github.com/GaoLii/Quad-Imaginarium.git, the resulting open-source dataset of 7,488 language-annotated quadruped motions (18.5 hours) spanning acrobatic and performative behaviors. We validate 392 randomly sampled motions on a real Unitree Go2 with a 96.7% deployment success rate, complemented by a 97.6% success rate across the full dataset in simulation.",
  "published": "2026-06-26",
  "updated": "2026-06-26",
  "year": "2026",
  "authors": [
   "Youzhi Liu",
   "Li Gao",
   "Yifei Qian",
   "Liu Liu",
   "Yang Cai",
   "Ziqiao Li"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "Uni-Mo, a fully automated pipeline that removes the animal from the loop by reframing data scarcity as a generation problem, is proposed and an Identity Consistency Loss that enforces appearance coherence across frames is introduced.",
  "doi": "10.48550/arXiv.2606.28237",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Youzhi Liu",
    "id": "2317117846",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Li Gao",
    "id": "2382320989",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Yifei Qian",
    "id": "2445495605",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Liu Liu",
    "id": "2332597209",
    "h_index": 8,
    "papers": 31
   },
   {
    "name": "Yang Cai",
    "id": "2383107401",
    "h_index": 5,
    "papers": 16
   },
   {
    "name": "Ziqiao Li",
    "id": "40862330",
    "h_index": 8,
    "papers": 37
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "data-teleop",
   "video-generation"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2606.28237v1",
  "pdf_url": "https://arxiv.org/pdf/2606.28237v1",
  "html_url": "https://arxiv.org/html/2606.28237v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.8
 },
 {
  "id": "2606.28196",
  "slug": "learning-stable-in-grasp-manipulation-in-a-non-dropping-action-space",
  "title": "Learning Stable In-Grasp Manipulation in a Non-Dropping Action Space",
  "abstract": "Traditionally, dexterous manipulation controllers are designed using analytic models constrained by strong assumptions about the hand and the objects being manipulated. Reinforcement learning (RL) has become another common approach in which skills are explored openly in an end-to-end manner but is inefficient because of unnoticeable instability and conflicts in learning objectives. This paper attempts to efficiently explore stable and accurate manipulation skills by decomposing dexterous skills into multiple simpler/analyzable components. Each skill component is subsequently learned with constraints and guidance from classical physics and control theory. Our work shows that for stable grasp, in-grasp reposition/reorientation with different objects, sensor/motor noise, latency, and frictional conditions, skill learning becomes efficient and stable with prior knowledge from theory.",
  "published": "2026-06-26",
  "updated": "2026-07-18",
  "year": "2026",
  "authors": [
   "Ha Thang Long Doan",
   "Hikaru Arita",
   "Kazuto Nakashima",
   "Kenji Tahara"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work shows that for stable grasp, in-grasp reposition/reorientation with different objects, sensor/motor noise, latency, and frictional conditions, skill learning becomes efficient and stable with prior knowledge from theory.",
  "doi": "10.48550/arXiv.2606.28196",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hackley Doan",
    "id": "144822290",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Hikaru Arita",
    "id": "2281154695",
    "h_index": 2,
    "papers": 26
   },
   {
    "name": "Kazuto Nakashima",
    "id": "2452223",
    "h_index": 10,
    "papers": 52
   },
   {
    "name": "Kenji Tahara",
    "id": "2274004194",
    "h_index": 2,
    "papers": 26
   }
  ],
  "comment": "This work has been submitted to the Taylor & Francis for possible publication",
  "topics": [
   "dexterous-manipulation",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.28196v2",
  "pdf_url": "https://arxiv.org/pdf/2606.28196v2",
  "html_url": "https://arxiv.org/html/2606.28196v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.27861",
  "slug": "ppo-eal-exact-augmented-lagrangian-proximal-policy-optimization-for-sa",
  "title": "PPO-EAL: Exact Augmented Lagrangian Proximal Policy Optimization for Safe Robotic Control",
  "abstract": "Reinforcement learning (RL) has emerged as a promising solution to accomplish complex robotic control tasks; however, most of the current work ignores the safety requirements. Safe RL seeks to maximize task performance while satisfying explicit physical constraints, but current algorithms struggle to learn the policy efficiently with precise constraint satisfaction. This work proposes PPO-EAL, a novel first-order constrained policy optimization framework that integrates exact augmented Lagrangian optimization into proximal policy optimization for safe robotic control. By combining clipped policy updates with exact quadratic penalty terms, PPO-EAL achieves theoretically grounded constraint enforcement without requiring impractically large penalty factors. A momentum-regulated multiplier update further improves dual-variable stability, reducing constraint oscillation and unsafe behavior while preserving task performance. We provide exactness and convergence analysis under standard stochastic approximation assumptions. Extensive validation across diverse GPU-accelerated robotic benchmarks-including cart-pole balancing, cart-double-pendulum stabilization, 7-DoF Franka end-effector reaching, and quadrupedal locomotion-demonstrates superior safety precision and reward performance compared with state-of-the-art first-order safe RL baselines. Finally, we demonstrate zero-shot sim-to-real deployment in a contact-rich gear assembly task, where PPO-EAL substantially improves task success, reduces peak contact force, and enhances operational robustness. These results establish PPO-EAL as a general and practically deployable safe RL framework for diverse safety-critical robotic systems.",
  "published": "2026-06-26",
  "updated": "2026-06-26",
  "year": "2026",
  "authors": [
   "Jiatao Ding",
   "Songqun Gao",
   "Andrea Del Prete",
   "Matteo Saveriano"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes PPO-EAL, a novel first-order constrained policy optimization framework that integrates exact augmented Lagrangian optimization into proximal policy optimization for safe robotic control and establishes PPO-EAL as a general and practically deployable safe RL framework for diverse safety-critical robotic systems.",
  "doi": "10.48550/arXiv.2606.27861",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jiatao Ding",
    "id": "2244168269",
    "h_index": 7,
    "papers": 16
   },
   {
    "name": "Songqun Gao",
    "id": "2221143826",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Andrea Del Prete",
    "id": "2239201923",
    "h_index": 5,
    "papers": 22
   },
   {
    "name": "Matteo Saveriano",
    "id": "2296186061",
    "h_index": 6,
    "papers": 23
   }
  ],
  "comment": "11 pages, 8 figures and 8 tables",
  "topics": [
   "humanoids",
   "tactile",
   "sim2real",
   "rl-control",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.27861v1",
  "pdf_url": "https://arxiv.org/pdf/2606.27861v1",
  "html_url": "https://arxiv.org/html/2606.27861v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.27813",
  "slug": "booster-lab-a-data-centric-pipeline-for-learning-deployable-humanoid-l",
  "title": "Booster Lab: A Data-Centric Pipeline for Learning Deployable Humanoid Locomotion Policies",
  "abstract": "Humanoid robot motion learning requires not only task-oriented control policies but also physically feasible and natural behaviors that can be transferred to real robots. However, robot-feasible motion data are often scarce: raw human demonstrations may be incompatible with the robot morphology, open-source clips vary in quality, and simulation-collected robot trajectories still require feasibility checking. To address these challenges, we propose a data-centric training and deployment pipeline that integrates motion data curation, real-to-sim model adaptation, AMP-based reinforcement learning, and sim-to-real deployment. We validate the framework on the Booster T1 robot and further provide preliminary cross-platform validation on Booster K1.",
  "published": "2026-06-26",
  "updated": "2026-06-26",
  "year": "2026",
  "authors": [
   "Penghui Chen",
   "Tinglong Zheng",
   "Yufeng Zhang",
   "Mingguo Zhao"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A data-centric training and deployment pipeline that integrates motion data curation, real-to-sim model adaptation, AMP-based reinforcement learning, and sim-to-real deployment is proposed.",
  "doi": "10.48550/arXiv.2606.27813",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Penghui Chen",
    "id": "2347860386",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Tinglong Zheng",
    "id": "2387013378",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Yufeng Zhang",
    "id": "2446332527",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Mingguo Zhao",
    "id": "2239286448",
    "h_index": 4,
    "papers": 14
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "egocentric-data",
   "sim2real",
   "rl-control",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.27813v1",
  "pdf_url": "https://arxiv.org/pdf/2606.27813v1",
  "html_url": "https://arxiv.org/html/2606.27813v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.27766",
  "slug": "rs-diffuser-risk-sensitive-diffusion-planning-with-distributional-valu",
  "title": "RS-Diffuser: Risk-Sensitive Diffusion Planning with Distributional Value Guidance",
  "abstract": "Offline reinforcement learning enables policy learning from fixed datasets without additional environment interaction, making it appealing for safety-critical applications where online exploration is costly or unsafe. Diffusion-based decision-making methods have recently achieved strong performance in offline RL by modeling rich, multimodal trajectory distributions. However, existing diffusion planners are typically risk-neutral and therefore may overlook rare but catastrophic outcomes that are crucial in real-world deployment. In this work, we propose RS-Diffuser, a risk-sensitive offline diffusion planning framework that combines diffusion-based trajectory generation with distributional value critics. RS-Diffuser learns a diffusion planner over future state trajectories, a separate inverse dynamics model for action decoding, and a Monte Carlo distributional critic that estimates the full return distribution of candidate plans through quantile regression. At sampling time, we incorporate a risk-sensitive guidance signal into the denoising process, using gradients computed from tail-aware objectives such as Conditional Value at Risk to steer generation toward desired risk profiles. As a result, a single trained model can flexibly produce risk-averse, risk-neutral, or risk-seeking behaviors by changing only the inference-time risk parameter. Extensive experiments on risk-sensitive D4RL and risky robot navigation benchmarks demonstrate that RS-Diffuser achieves state-of-the-art performance, improving both overall return and worst-case robustness while reducing safety violations.",
  "published": "2026-06-26",
  "updated": "2026-06-26",
  "year": "2026",
  "authors": [
   "Shiqiang Gong"
  ],
  "author_count": 1,
  "categories": [
   "cs.LG",
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "RS-Diffuser is proposed, a risk-sensitive offline diffusion planning framework that combines diffusion-based trajectory generation with distributional value critics, and achieves state-of-the-art performance, improving both overall return and worst-case robustness while reducing safety violations.",
  "doi": "10.1007/978-981-92-3381-6_30",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Shiqi Gong",
    "id": "2386814321",
    "h_index": 0,
    "papers": 6
   }
  ],
  "comment": "ICIC 2026 Oral",
  "topics": [
   "rl-control",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.27766v1",
  "pdf_url": "https://arxiv.org/pdf/2606.27766v1",
  "html_url": "https://arxiv.org/html/2606.27766v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.27581",
  "slug": "scenebot-contact-prompted-general-humanoid-whole-body-tracking-with-sc",
  "title": "SceneBot: Contact-Prompted General Humanoid Whole Body Tracking with Scene-Interaction",
  "abstract": "Current humanoid reinforcement-learning policies excel at free-space motions but struggle with contact-rich tasks, as pure kinematic tracking cannot resolve the physical ambiguities of interacting with objects and uneven terrain. To address this, we introduce SceneBot, a unified motion-tracking framework capable of handling freespace locomotion, terrain traversal, and whole-body manipulation. SceneBot conditions a single policy on both reference motions and per-link contact labels, explicitly defining expected environmental interactions. To overcome the lack of annotated interaction data, we propose a hindsight scene reconstruction approach that infers scene-interaction graphs from retargeted human motion. Trained on 7.5 hours of this reconstructed, contact-rich data, SceneBot successfully generalizes to unseen motions and environments. Our results demonstrate that SceneBot is the first general framework to seamlessly unify free-space and contact-rich behaviors executing complex, long-horizon tasks like carrying a box upstairs and establishing contact conditioning as a powerful interface for humanoid control. All code and data will be open-sourced. More demos and information are available at: https://ericcsr.github.io/scenebot/",
  "published": "2026-06-25",
  "updated": "2026-06-25",
  "year": "2026",
  "authors": [
   "Sirui Chen",
   "Shibo Zhao",
   "Zhen Wu",
   "Jiaman Li",
   "Guanya Shi",
   "C. Karen Liu"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 2,
  "influential_citations": 1,
  "tldr": "The results demonstrate that SceneBot is the first general framework to seamlessly unify free-space and contact-rich behaviors executing complex, long-horizon tasks like carrying a box upstairs and establishing contact conditioning as a powerful interface for humanoid control.",
  "doi": "10.48550/arXiv.2606.27581",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Sirui Chen",
    "id": "2209905328",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Shibo Zhao",
    "id": "2322805160",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Zhen Wu",
    "id": "2308574851",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Jiaman Li",
    "id": "22133106",
    "h_index": 16,
    "papers": 22
   },
   {
    "name": "Guanya Shi",
    "id": "2384824402",
    "h_index": 9,
    "papers": 24
   },
   {
    "name": "C. K. Liu",
    "id": "2325109818",
    "h_index": 3,
    "papers": 9
   }
  ],
  "comment": "15 pages 10 figures",
  "topics": [
   "humanoids",
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.27581v1",
  "pdf_url": "https://arxiv.org/pdf/2606.27581v1",
  "html_url": "https://arxiv.org/html/2606.27581v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.48
 },
 {
  "id": "2606.27374",
  "slug": "world-action-models-enable-continual-imitation-learning-with-recurrent",
  "title": "World Action Models Enable Continual Imitation Learning with Recurrent Generative Replays",
  "abstract": "Going beyond predicting robot actions, World Action Models (WAMs) can also generate future visual observations. We build on this generative capability to propose Recurrent Generative Replay (REGEN), a continual imitation learning framework that synthesizes pseudo-replay trajectories, enabling a robot policy to rehearse previously learned tasks without storing their original human demonstrations. During continual adaptation, REGEN recursively queries the WAM to synthesize pseudo-replay trajectories conditioned only on prior task instructions and current-task observations. Experiments in both simulation and real-world manipulation settings show that REGEN reduces catastrophic forgetting by up to $50\\%$ relative to sequential fine-tuning, while approaching the performance of privileged experience replay methods that require access to real replay data. Finally, we analyze the factors limiting generated replay, identifying long-horizon visual degradation and action-observation inconsistency as the primary bottlenecks. Our results establish WAMs as a promising foundation for continual robot learning without stored demonstrations.",
  "published": "2026-06-25",
  "updated": "2026-06-25",
  "year": "2026",
  "authors": [
   "Manish Kumar Govind",
   "Dominick Reilly",
   "Smit Patel",
   "Hieu Le",
   "Srijan Das"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "Recurrent Generative Replay (REGEN) is proposed, a continual imitation learning framework that synthesizes pseudo-replay trajectories, enabling a robot policy to rehearse previously learned tasks without storing their original human demonstrations.",
  "doi": "10.48550/arXiv.2606.27374",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Manish Kumar Govind",
    "id": "2306265813",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Dominick Reilly",
    "id": "2160860706",
    "h_index": 5,
    "papers": 21
   },
   {
    "name": "Smit Patel",
    "id": "2109497867",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Hieu Le",
    "id": "2393015202",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Srijan Das",
    "id": "2264489626",
    "h_index": 4,
    "papers": 16
   }
  ],
  "comment": "",
  "topics": [
   "egocentric-data",
   "imitation-diffusion"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.27374v1",
  "pdf_url": "https://arxiv.org/pdf/2606.27374v1",
  "html_url": "https://arxiv.org/html/2606.27374v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2606.27344",
  "slug": "vibeact-vibration-to-actions-for-contact-rich-reactive-robot-dexterity",
  "title": "VibeAct: Vibration to Actions for Contact-Rich Reactive Robot Dexterity",
  "abstract": "Dexterous manipulation depends on contact events that are fast, local, and often visually occluded. Piezoelectric microphones offer a compact and high-bandwidth way to sense these interactions, but the resulting vibro-acoustic signals are difficult to simulate faithfully enough for end-to-end sim-to-real policy learning on dexterous robot hands. We propose VibeAct, a framework that bridges real vibrotactile sensing and simulation-based reinforcement learning through a shared physical representation of contact and slip. In the real world, we embed piezoelectric microphones into a dexterous robot hand and collect vibro-acoustic data through teleoperation, then replay the recordings in a calibrated digital clone to automatically label per-finger contact and slip. A tactile estimator learns to predict contact and slip from real microphone waveforms, while manipulation policies are trained in simulation on the same representation computed directly from simulated contacts. This decoupling lets policies exploit rapid tactile feedback without simulating raw audio. Across five contact-rich tasks spanning regrasping, in-hand reorientation, and insertion, VibeAct consistently outperforms a proprioception-and-point-cloud baseline in simulation, with the largest gains on tasks requiring sustained reactive control, where the continuous slip-magnitude channel proves the most informative observation. The learned policies transfer to a physical dexterous hand-arm platform, improving success rates on deployed tasks. Project videos and additional details are at https://vibeact.github.io/.",
  "published": "2026-06-25",
  "updated": "2026-06-25",
  "year": "2026",
  "authors": [
   "Yuemin Mao",
   "Uksang Yoo",
   "Jean Oh",
   "Jonathan Francis",
   "Jeffrey Ichnowski"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Across five contact-rich tasks spanning regrasping, in-hand reorientation, and insertion, VibeAct consistently outperforms a proprioception-and-point-cloud baseline in simulation, with the largest gains on tasks requiring sustained reactive control.",
  "doi": "10.48550/arXiv.2606.27344",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yuemin Mao",
    "id": "2301143224",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Uksang Yoo",
    "id": "2049034998",
    "h_index": 8,
    "papers": 26
   },
   {
    "name": "Jean Oh",
    "id": "2244824559",
    "h_index": 6,
    "papers": 14
   },
   {
    "name": "Jonathan Francis",
    "id": "2314826620",
    "h_index": 5,
    "papers": 15
   },
   {
    "name": "Jeffrey Ichnowski",
    "id": "2269146110",
    "h_index": 9,
    "papers": 29
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "sim2real",
   "rl-control",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.27344v1",
  "pdf_url": "https://arxiv.org/pdf/2606.27344v1",
  "html_url": "https://arxiv.org/html/2606.27344v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.27163",
  "slug": "learning-to-fold-prizewinning-solution-at-lehome-challenge-2026-1st-pl",
  "title": "Learning to Fold: prizewinning solution at LeHome Challenge 2026 (1st place online, 2nd offline)",
  "abstract": "I describe my solution to the LeHome Challenge 2026, an ICRA 2026 competition on bimanual garment folding. The system placed 1st of 62 teams in the online (simulation) round and 2nd in the real-world final. It improves a vision-language-action (VLA) policy with a reinforcement-learning loop. The policy is its own value function: the same network that predicts actions also predicts success, progress, and a few task-relevant future quantities, and those predictions drive advantage estimation, live failure detection, and candidate selection. The work mostly recombines existing RL ideas with engineering and optimization contributions that can be used together as one recipe or individually: AWR + RECAP combined for flow-matching VLA; an asynchronous distributed training / rollout pipeline through HuggingFace Hub; inference-time hyperparameters optimization via Thompson sampling; a sim-to-real recipe with camera-alignment tooling, heavy augmentation and DAgger-like HIL data collection.",
  "published": "2026-06-25",
  "updated": "2026-07-18",
  "year": "2026",
  "authors": [
   "Ilia Larchenko"
  ],
  "author_count": 1,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The work improves a vision-language-action (VLA) policy with a reinforcement-learning loop that predicts success, progress, and a few task-relevant future quantities and drives advantage estimation, live failure detection, and candidate selection in the LeHome Challenge 2026.",
  "doi": "10.48550/arXiv.2606.27163",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "I. Larchenko",
    "id": "12445350",
    "h_index": 1,
    "papers": 6
   }
  ],
  "comment": "Solution of the LeHome Challenge at ICRA 2026",
  "topics": [
   "vla",
   "sim2real",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.27163v2",
  "pdf_url": "https://arxiv.org/pdf/2606.27163v2",
  "html_url": "https://arxiv.org/html/2606.27163v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2606.27128",
  "slug": "flamevqa-a-physically-grounded-uav-wildfire-vqa-benchmark-with-radiome",
  "title": "FlameVQA: A Physically-Grounded UAV Wildfire VQA Benchmark with Radiometric Thermal Supervision",
  "abstract": "Wildfire monitoring from UAVs requires reliable reasoning over complex aerial scenes, where smoke, scale variation, and occlusions often limit RGB-only interpretation. We introduce FlameVQA, a multiple-choice visual question answering benchmark for UAV-based wildfire intelligence built on FLAME 3, leveraging paired RGB imagery and radiometric thermal TIFFs for temperature-grounded, safety-critical reasoning. FlameVQA includes 34 multiple-choice questions per image spanning six operational capability groups, covering tasks such as detection, localization, distribution/coverage estimation, cross-modal reasoning, and flight planning. To ensure label reliability, we combine MLLM-assisted annotation with deterministic thermal rules and cross-question consistency checks, followed by human auditing. We also evaluate representative MLLMs on FlameVQA to provide baselines for future work. Results show strong performance when explicit cross-modal cues are available, but notable failures on presence detection under heavy smoke and on coverage estimation. These findings suggest that current MLLMs require domain-specific adaptation to better support disaster and wildfire monitoring. The dataset and benchmark code are open-source at github.com/mobiiin/WildFire_VQA",
  "published": "2026-06-25",
  "updated": "2026-06-25",
  "year": "2026",
  "authors": [
   "Mobin Habibpour",
   "John Spodnik",
   "Niloufar Alipour Talemi",
   "Fatemeh Afghah"
  ],
  "author_count": 4,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "FlameVQA is introduced, a multiple-choice visual question answering benchmark for UAV-based wildfire intelligence built on FLAME 3, leveraging paired RGB imagery and radiometric thermal TIFFs for temperature-grounded, safety-critical reasoning.",
  "doi": "10.48550/arXiv.2606.27128",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Mobin Habibpour",
    "id": "2327347004",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "J. Spodnik",
    "id": "16756379",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Niloufar Alipour Talemi",
    "id": "2204866947",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Fatemeh Afghah",
    "id": "2331754448",
    "h_index": 3,
    "papers": 9
   }
  ],
  "comment": "",
  "topics": [
   "navigation",
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.27128v1",
  "pdf_url": "https://arxiv.org/pdf/2606.27128v1",
  "html_url": "https://arxiv.org/html/2606.27128v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.27123",
  "slug": "proposal-conditioned-latent-diffusion-for-closed-loop-traffic-scenario",
  "title": "Proposal-Conditioned Latent Diffusion for Closed-Loop Traffic Scenario Generation",
  "abstract": "Closed-loop traffic simulation remains challenging because it must generate interactive multi-agent behaviors that are scene-consistent and controllable throughout rollout. Prior diffusion-based approaches achieve strong realism, but their computational cost can hinder deployment in time-constrained replanning loops for autonomous vehicle planning and simulation. We present a diffusion-based scenario generation framework conditioned on instance-centric scene context and multimodal proposal priors, with optional test-time guidance for shaping safety-critical behaviors. A compact action-latent representation and proposal-based initialization improve sampling efficiency and reduce per-step runtime without retraining. Experiments on the Waymo Open Motion Dataset demonstrate a favorable balance among realism, safety, and controllability across diverse interactive scenarios, while showing that test-time guidance enables systematic trade-offs among competing objectives.",
  "published": "2026-06-25",
  "updated": "2026-06-25",
  "year": "2026",
  "authors": [
   "Shubham Vaijanath Phoolari",
   "Aleyna Kara",
   "Christoph Lauer",
   "Steven Peters"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A diffusion-based scenario generation framework conditioned on instance-centric scene context and multimodal proposal priors, with optional test-time guidance for shaping safety-critical behaviors is presented, with a favorable balance among realism, safety, and controllability across diverse interactive scenarios.",
  "doi": "10.48550/arXiv.2606.27123",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Shubham Vaijanath Phoolari",
    "id": "2042888246",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Aleyna Kara",
    "id": "2142735826",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "C. Lauer",
    "id": "46705625",
    "h_index": 8,
    "papers": 21
   },
   {
    "name": "Steven Peters",
    "id": "2367046731",
    "h_index": 1,
    "papers": 3
   }
  ],
  "comment": "Accepted for publication at the IEEE International Conference on Intelligent Transportation Systems (ITSC), 2026",
  "topics": [
   "video-generation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.27123v1",
  "pdf_url": "https://arxiv.org/pdf/2606.27123v1",
  "html_url": "https://arxiv.org/html/2606.27123v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.26855",
  "slug": "humanoid-dart-humanoid-loco-manipulation-using-diffusion-guided-augmen",
  "title": "Humanoid-DART: Humanoid Loco-Manipulation using Diffusion-guided Augmentation through Relabeling and Tracking",
  "abstract": "Imitating human demonstrations has emerged as a dominant paradigm for learning humanoid loco-manipulation policies. However, scaling these approaches remains challenging due to the high cost of collecting diverse demonstrations and the need for continual human intervention to correct policy failures. In this paper, we present a self-supervised framework that bootstraps from sparse demonstrations and progressively expands its behavioral repertoire, enabling the learning of a goal-conditioned policy that automatically explores the goal space with minimal expert supervision. Our approach combines diffusion-based trajectory generation with reinforcement learning, where the latter is used to track goal-conditioned trajectories produced by the diffusion model for a range of loco-manipulation skills. Through extensive ablation studies and comparisons with state-of-the-art methods, we demonstrate the effectiveness of our framework on multiple humanoid loco-manipulation skills.",
  "published": "2026-06-25",
  "updated": "2026-06-25",
  "year": "2026",
  "authors": [
   "Pranav Debbad",
   "Kanish Thiagarajan",
   "Victor Dh\u00e9din",
   "Shafeef Omar",
   "Majid Khadiv"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "This paper presents a self-supervised framework that bootstraps from sparse demonstrations and progressively expands its behavioral repertoire, enabling the learning of a goal-conditioned policy that automatically explores the goal space with minimal expert supervision.",
  "doi": "10.48550/arXiv.2606.26855",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Pranav Debbad",
    "id": "2359291435",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "K. Thiagarajan",
    "id": "2279048772",
    "h_index": 1,
    "papers": 9
   },
   {
    "name": "V. Dh\u00e9din",
    "id": "2187056367",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Shafeef Omar",
    "id": "2224615527",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "M. Khadiv",
    "id": "8134198",
    "h_index": 19,
    "papers": 89
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "egocentric-data",
   "rl-control",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.26855v1",
  "pdf_url": "https://arxiv.org/pdf/2606.26855v1",
  "html_url": "https://arxiv.org/html/2606.26855v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2606.26661",
  "slug": "lamp-lane-aligned-motion-primitives-for-feasible-trajectory-prediction",
  "title": "LAMP: Lane-Aligned Motion Primitives for Feasible Trajectory Prediction",
  "abstract": "Motion forecasting is essential for autonomous driving systems to enable safe decision-making and planning in complex driving scenarios. While existing predictors excel at minimizing standard displacement errors, they often overlook the adherence to lane topology of multimodal predictions, particularly for lower-probability modes. Consequently, predicted trajectories may violate physical and logical constraints, making the prediction set unreliable for safety-critical planning. In this paper, we propose LAMP (Lane-Aligned Motion Primitives), a topology-aware forecasting framework that anchors multimodal prediction to structured motion primitives aligned with lane topology. Specifically, we use a VQ-VAE to learn shape-aware motion primitives as discrete intention queries, capturing spatiotemporal patterns beyond endpoint-based intentions. We further introduce a feasibility-aware intention selector trained with a lane-topology prior for filtering unreachable intention queries, guiding the decoder to prioritize topology-consistent intentions while preserving behavioral diversity. Extensive experiments on the Argoverse 2 dataset demonstrate that LAMP achieves prediction accuracy comparable to state-of-the-art baselines while outperforming them in feasibility and diversity metrics.",
  "published": "2026-06-25",
  "updated": "2026-06-25",
  "year": "2026",
  "authors": [
   "Sangjin Han",
   "Hoseong Jung",
   "Jeongtae Her",
   "Changhyun Choi",
   "H. Jin Kim"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A VQ-VAE is used to learn shape-aware motion primitives as discrete intention queries, capturing spatiotemporal patterns beyond endpoint-based intentions, and a feasibility-aware intention selector trained with a lane-topology prior for filtering unreachable intention queries is introduced.",
  "doi": "10.48550/arXiv.2606.26661",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Sang-Min Han",
    "id": "2220215056",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Hoseong Jung",
    "id": "2211894041",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Jeongtae Her",
    "id": "2444867307",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Changhyun Choi",
    "id": "2257134105",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "H. J. Kim",
    "id": "2367176226",
    "h_index": 1,
    "papers": 6
   }
  ],
  "comment": "IEEE ITSC 2026, 6 pages",
  "topics": [
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.26661v1",
  "pdf_url": "https://arxiv.org/pdf/2606.26661v1",
  "html_url": "https://arxiv.org/html/2606.26661v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.26215",
  "slug": "tasknpoint-how-to-teach-your-humanoid-to-hit-a-backhand-in-minutes",
  "title": "TaskNPoint: How to Teach Your Humanoid to Hit a Backhand in Minutes",
  "abstract": "How do we learn to hit a tennis backhand? Not from a thousand hours of tennis tournaments on TV - we work with a coach and practice. We argue this is also the right recipe for teaching dynamic skills to humanoid robots. This follows from a structural property of dynamic skills: the outcome is decided by a short, crucial portion of the trajectory - for a backhand, the ~20cm of racket travel around ball contact. Getting this interaction window right requires coordinating the whole motion, so that control, physics, and morphology act in concert. Learning thus reduces to mastering a handful of distinct actions and, for each, practicing until the window comes out right. To this end, we introduce TaskNPoint, a training protocol which makes the coach-learner division of labor explicit. The human coach contributes four inputs: a discrete set of skills (e.g. different shots), one demonstration per skill, identification of the interaction window, and the goal. Learning in a physically realistic simulation environment fills in each action trajectory and provides robustness to unmodeled events. Crucially, randomized target sampling during training lets a single demonstration generalize zero-shot to unseen goal locations. We test this approach on a Unitree G1 humanoid that hits forehands and backhands against balls thrown by a human, kicks incoming soccer balls, and picks and places boxes from novel locations. We find that learning is successful from short human video demonstrations and under an hour of training on a single GPU, with no per-task reward tuning.",
  "published": "2026-06-24",
  "updated": "2026-06-24",
  "year": "2026",
  "authors": [
   "Blake Werner",
   "Ilona Demler",
   "Pietro Perona",
   "Aaron D. Ames"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "TaskNPoint, a training protocol which makes the coach-learner division of labor explicit, is introduced and it is found that learning is successful from short human video demonstrations and under an hour of training on a single GPU, with no per-task reward tuning.",
  "doi": "10.48550/arXiv.2606.26215",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Blake Werner",
    "id": "2362086530",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Ilona A. Demler",
    "id": "2386988871",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Pietro Perona",
    "id": "2286494182",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Aaron D. Ames",
    "id": "2338277217",
    "h_index": 4,
    "papers": 23
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "egocentric-data",
   "hardware-codesign",
   "safety-eval"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2606.26215v1",
  "pdf_url": "https://arxiv.org/pdf/2606.26215v1",
  "html_url": "https://arxiv.org/html/2606.26215v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2606.26188",
  "slug": "morphology-specific-closed-loop-control-of-logarithmic-spiral-continuu",
  "title": "Morphology-Specific Closed-Loop Control of Logarithmic-Spiral Continuum Arms via Online Jacobian Error Compensation",
  "abstract": "Logarithmic spirals are ubiquitous in biological appendages and provide an attractive morphology for continuum manipulators capable of reaching, wrapping, and grasping. Recently reported logarithmic-spiral robots demonstrated scalable fabrication and versatile grasping but lacked inverse kinematics and closed-loop control. This work presents the first morphology-specific closed-loop task-space control framework for logarithmic-spiral continuum arms. A segmented tendon-driven model with a centerline backbone and equilateral tendon routing is developed in MuJoCo to capture tapered compliance and contact dynamics. An analytical task-space Jacobian is derived directly from the logarithmic-spiral kinematics and combined with online Jacobian error compensation using a Broyden secant update and Kalman-filter estimation. The resulting controller continuously corrects modeling errors arising from nonlinear deformation, contact, and geometric mismatch. The framework is validated through planar and spatial simulations, including trajectory tracking, attitude regulation, disturbance rejection, three-dimensional position tracking, and simultaneous position-orientation control. Compared with a piecewise-constant-curvature (PCC) baseline, the proposed method consistently reduces tracking errors, suppresses attitude drift, and maintains a bounded Jacobian estimation error. The controller is further applied to morphology-enabled manipulation tasks, including obstacle-assisted reach-wrap-release motions, adaptive whole-arm grasping, and cooperative multi-arm object handling. Results demonstrate that combining logarithmic-spiral morphology with online Jacobian compensation enables accurate, robust, and scalable control of highly underactuated continuum manipulators. The proposed framework establishes a physics-grounded baseline for future hardware implementation and learning-augmented soft robotic control.",
  "published": "2026-06-24",
  "updated": "2026-06-24",
  "year": "2026",
  "authors": [
   "Partha Datta",
   "Yi Jin",
   "Wei Lin",
   "C. Chase Cao"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "physics.app-ph"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "10.48550/arXiv.2606.26188",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "P. Datta",
    "id": "40417132",
    "h_index": 13,
    "papers": 23
   },
   {
    "name": "Yi Jin",
    "id": "2370510613",
    "h_index": 2,
    "papers": 11
   },
   {
    "name": "Wei Lin",
    "id": "2108811098",
    "h_index": 11,
    "papers": 26
   },
   {
    "name": "C. Cao",
    "id": "2064672461",
    "h_index": 11,
    "papers": 20
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.26188v1",
  "pdf_url": "https://arxiv.org/pdf/2606.26188v1",
  "html_url": "https://arxiv.org/html/2606.26188v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.26175",
  "slug": "rmtl-reinforced-micro-task-learning-for-long-horizon-manipulation-with",
  "title": "RMTL: Reinforced Micro-task Learning for Long-Horizon Manipulation with VLM Rewards",
  "abstract": "Reinforcement learning (RL) for robotic manipulation often requires manually designing a dense reward function, which is difficult to tune and often fragile, or learning a reward from human demonstrations or preferences, which can be expensive. A recent line of work uses pretrained vision-language models (VLMs) as zero-shot reward models, replacing these costs with a single text prompt. However, we argue that a single global prompt is too coarse for long-horizon manipulation tasks with randomized initial conditions. The single-prompt VLM reward is near-flat for much of the trajectory, making early progress hard for the agent to detect. We propose Reinforced Micro-Task Learning (RMTL), an approach that decomposes a manipulation task into a small set of language-described micro-tasks and trains the agent to switch between them. At each step, the agent receives a multi-view VLM reward computed using the prompt of the currently active micro-task and averaged across multiple camera views to reduce the effect of view-specific occlusions. A reverse curriculum gradually exposes the agent to harder initial conditions, while a PPO worker is first trained with a fixed distance-based rule that selects the active micro-task. We then replace this rule with a learned hierarchical manager, turning rule-based phase selection into a fully learned hierarchical policy. We instantiate RMTL on the Fetch manipulation environment using three short stage-specific prompts and without additional prompt tuning. Experiments show that RMTL provides more informative reward signals than single-prompt VLM rewards, enabling faster learning. These results suggest that decomposing VLM rewards into micro-task-specific language prompts can substantially improve the scalability of language-guided reinforcement learning for robotic manipulation.",
  "published": "2026-06-24",
  "updated": "2026-06-24",
  "year": "2026",
  "authors": [
   "An\u0131l Can Ate\u015f",
   "Orhan Kahraman",
   "Cihan Topal"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Reinforced Micro-Task Learning (RMTL), an approach that decomposes a manipulation task into a small set of language-described micro-tasks and trains the agent to switch between them, and shows that RMTL provides more informative reward signals than single-prompt VLM rewards, enabling faster learning.",
  "doi": "10.48550/arXiv.2606.26175",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Anil Can Ates",
    "id": "2456469465",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "O. Kahraman",
    "id": "2135520680",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Cihan Topal",
    "id": "2391984018",
    "h_index": 0,
    "papers": 3
   }
  ],
  "comment": "16 pages, 11 figures",
  "topics": [
   "egocentric-data",
   "rl-control",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.26175v1",
  "pdf_url": "https://arxiv.org/pdf/2606.26175v1",
  "html_url": "https://arxiv.org/html/2606.26175v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.26093",
  "slug": "forceband-learning-forceful-manipulation-with-semg",
  "title": "ForceBand: Learning Forceful Manipulation with sEMG",
  "abstract": "Human demonstrations are a scalable data source for learning robot manipulation policies. However, common sources of human demonstration data, such as motion-capture trajectories and internet videos, capture mostly motion and appearance while missing the contact forces that are critical for force-sensitive manipulation. In this paper, we introduce ForceBand, a low-cost wrist-worn sEMG system that turns human muscle activity into force-enriched demonstrations. We first collect a 10-hour multimodal dataset containing egocentric video, sEMG, IMU, and fingertip force measurements across diverse actions and objects. Using this dataset, we pre-train an EMG2Force model that predicts per-finger forces from sEMG and IMU signals. After a short user-specific calibration, users can collect target-task demonstrations using only ForceBand and video; EMG2Force then labels these demonstrations with per-finger force traces, producing force-augmented demonstrations for robot policy learning. Experiments show that ForceBand recovers fine-grained fingertip interactions with over 50% lower force prediction error than vision-based baselines and achieves an 87% success rate on pick, squeeze, and place tasks that require object-specific force control across objects with diverse shapes, sizes, and weights. Project website: https://forceband-emg.github.io",
  "published": "2026-06-24",
  "updated": "2026-06-24",
  "year": "2026",
  "authors": [
   "Botao He",
   "Zhi Wang",
   "Linna Kuang",
   "Ishaan Ghosh",
   "Jitendra Malik",
   "Cornelia Fermuller",
   "Tingfan Wu",
   "Jiayuan Mao",
   "Ruoshi Liu",
   "Haozhi Qi",
   "Yiannis Aloimonos"
  ],
  "author_count": 11,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "ForceBand is introduced, a low-cost wrist-worn sEMG system that turns human muscle activity into force-enriched demonstrations that achieves an 87% success rate on pick, squeeze, and place tasks that require object-specific force control across objects with diverse shapes, sizes, and weights.",
  "doi": "10.48550/arXiv.2606.26093",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Botao He",
    "id": "2296438839",
    "h_index": 5,
    "papers": 16
   },
   {
    "name": "Zhi Wang",
    "id": "2282565866",
    "h_index": 3,
    "papers": 15
   },
   {
    "name": "Linna Kuang",
    "id": "2444885776",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Ishaan Ghosh",
    "id": "2444884820",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Jitendra Malik",
    "id": "2242761335",
    "h_index": 15,
    "papers": 32
   },
   {
    "name": "Cornelia Fermuller",
    "id": "2305680792",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Tingfan Wu",
    "id": "2254158966",
    "h_index": 12,
    "papers": 20
   },
   {
    "name": "Jiayuan Mao",
    "id": "2323437497",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Ruoshi Liu",
    "id": "2143183492",
    "h_index": 14,
    "papers": 24
   },
   {
    "name": "Haozhi Qi",
    "id": "2247951244",
    "h_index": 11,
    "papers": 18
   },
   {
    "name": "Y. Aloimonos",
    "id": "1697493",
    "h_index": 55,
    "papers": 418
   }
  ],
  "comment": "",
  "topics": [
   "egocentric-data"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.26093v1",
  "pdf_url": "https://arxiv.org/pdf/2606.26093v1",
  "html_url": "https://arxiv.org/html/2606.26093v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.26048",
  "slug": "deep-reinforcement-learning-enhanced-event-triggered-data-driven-predi",
  "title": "Deep Reinforcement Learning-Enhanced Event-Triggered Data-Driven Predictive Control for a 3D Cable-Driven Soft Robotic Arm",
  "abstract": "Soft robots are challenging to control due to their nonlinear and time-varying dynamics. Data-enabled predictive control (DeePC) offers a model-free alternative by directly leveraging measured input-output trajectories to construct a predictive controller. However, its receding-horizon formulation requires solving a constrained optimization problem at every sampling instant, which can be computationally demanding for real-time deployment on resource-limited robotic platforms. To address this limitation, we propose an adaptive reinforcement-learning-based event-triggered DeePC (RL-ET-DeePC) framework for soft robotic control. A model-free RL policy is trained to determine when to invoke the DeePC optimizer based on the current system state representation, thereby reducing unnecessary optimization calls while preserving closed-loop performance. Simulation results show that RL-ET-DeePC reduces optimization frequency by up to 66% compared to periodic DeePC, while maintaining comparable tracking accuracy. Hardware experiments on a three-dimensional cable-driven soft robotic arm demonstrate zero-shot transfer, achieving a 34% reduction in optimization frequency with tracking accuracy comparable to periodic DeePC and more consistent performance than a static threshold-based event-triggered baseline.",
  "published": "2026-06-24",
  "updated": "2026-06-26",
  "year": "2026",
  "authors": [
   "Cheng Ouyang",
   "Moeen Ul Islam",
   "Kaixiang Zhang",
   "Zhaojian Li",
   "Xiaobo Tan",
   "Dong Chen"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "An adaptive reinforcement-learning-based event-triggered DeePC (RL-ET-DeePC) framework for soft robotic control is proposed, where a model-free RL policy is trained to determine when to invoke the DeePC optimizer based on the current system state representation, thereby reducing unnecessary optimization calls while preserving closed-loop performance.",
  "doi": "10.48550/arXiv.2606.26048",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ouyang Cheng",
    "id": "2385422309",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Moeen Ul Islam",
    "id": "2288158401",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Kaixiang Zhang",
    "id": "2153281148",
    "h_index": 15,
    "papers": 46
   },
   {
    "name": "Zhaojian Li",
    "id": "2265540422",
    "h_index": 8,
    "papers": 32
   },
   {
    "name": "Xiaobo Tan",
    "id": "2244580501",
    "h_index": 4,
    "papers": 16
   },
   {
    "name": "Dong Chen",
    "id": "2279473220",
    "h_index": 6,
    "papers": 12
   }
  ],
  "comment": "",
  "topics": [
   "rl-control",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.26048v2",
  "pdf_url": "https://arxiv.org/pdf/2606.26048v2",
  "html_url": "https://arxiv.org/html/2606.26048v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.26047",
  "slug": "learning-robot-visual-navigation-in-crowds-via-intention-aware-scene-r",
  "title": "Learning Robot Visual Navigation in Crowds via Intention-Aware Scene Representations",
  "abstract": "Robot crowd navigation requires the ability to infer human intentions while accounting for the structural constraints of the environment. Currently, deep reinforcement learning (DRL) provides a promising method for learning navigation policies that understand human intentions. However, most of them rely on limited scene representations, treating pedestrians as simple 2D points and ignoring rich visual cues from both humans and the environment. To address this issue, we introduce iCrowdNav, a novel visual crowd navigation method with intention-aware scene representations, to encode behavioral and structural context from egocentric visual observations. Our method employs two key components: a spatio-temporal encoder for extracting occupancy features of the scene, and Intent-Interact Former (I$^2$ Former), an attention-based module that encodes human poses to infer pedestrians' motion intentions. These features are integrated into a compact state embedding that supports effective DRL policy training. Extensive experiments show that our method achieves superior performance over baselines, and real-world deployment demonstrates vision-based crowd navigation.",
  "published": "2026-06-24",
  "updated": "2026-06-24",
  "year": "2026",
  "authors": [
   "Han Bao",
   "Bingyi Xia",
   "Hanjing Ye",
   "Yu Zhan",
   "Hao Cheng",
   "Baozhi Jia",
   "Wenjun Xu",
   "Jiankun Wang"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "RA-L",
  "venue_source": "semantic-scholar",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "iCrowdNav, a novel visual crowd navigation method with intention-aware scene representations, is introduced to encode behavioral and structural context from egocentric visual observations and achieves superior performance over baselines and real-world deployment demonstrates vision-based crowd navigation.",
  "doi": "10.1109/LRA.2026.3677748",
  "oa_pdf": "https://arxiv.org/pdf/2606.26047",
  "s2_authors": [
   {
    "name": "Han Bao",
    "id": "47469612",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Bingyi Xia",
    "id": "2348400978",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Hanjing Ye",
    "id": "2160747744",
    "h_index": 7,
    "papers": 27
   },
   {
    "name": "Yu Zhan",
    "id": "2243336940",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Hao Cheng",
    "id": "2384997472",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Baozhi Jia",
    "id": "38113606",
    "h_index": 5,
    "papers": 18
   },
   {
    "name": "Wenjun Xu",
    "id": "2379807474",
    "h_index": 1,
    "papers": 11
   },
   {
    "name": "Jiankun Wang",
    "id": "51068901",
    "h_index": 25,
    "papers": 126
   }
  ],
  "comment": "",
  "topics": [
   "egocentric-data",
   "rl-control",
   "spatial-3d",
   "navigation",
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.26047v1",
  "pdf_url": "https://arxiv.org/pdf/2606.26047v1",
  "html_url": "https://arxiv.org/html/2606.26047v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2606.26046",
  "slug": "roboatlas-contextual-active-slam",
  "title": "RoboAtlas: Contextual Active SLAM",
  "abstract": "We present RoboAtlas, a contextual Active SLAM framework that adaptively balances geometric exploration and semantic reasoning using a scalable 3D semantic mapping system, OpenRoboVox. RoboAtlas integrates frontier exploration, global semantic-map reasoning, and egocentric VLM-based reasoning through a contextual multi-armed bandit that transitions from exploration to semantically guided navigation as scene understanding improves. We evaluate the system in simulation and on a Unitree Go2 robot in large-scale real-world environments exceeding 1800 m2 with approx. 30k mapped semantic instances, achieving a 100% task success rate. On the GOAT-Bench \"Val Unseen\" benchmark, RoboAtlas achieves state-of-the-art performance with highest reported success rate (SR) of 90.6%, using GPT-4o, improving over the strongest prior baseline by 17.8 percentage points in SR. Using the much smaller Qwen2.5-VL-7B model, it still achieves 88.8% SR, outperforming all baselines using GPT-4o in SR, and revealing the importance of the information gained by our semantic mapping framework over simply replacing the underlying foundation model. The results demonstrate that grounding foundation models with large-scale 3D semantic maps enables robust and efficient contextual Active SLAM.",
  "published": "2026-06-24",
  "updated": "2026-08-07",
  "year": "2026",
  "authors": [
   "Alexander Schperberg",
   "Shivam K. Panda",
   "Abraham P. Vinod",
   "Rokaha Bhagawan",
   "M. K. Jawed",
   "Stefano Di Cairano"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The results demonstrate that grounding foundation models with large-scale 3D semantic maps enables robust and efficient contextual Active SLAM and reveals the importance of the information gained by the semantic mapping framework over simply replacing the underlying foundation model.",
  "doi": "10.48550/arXiv.2606.26046",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Alexander Schperberg",
    "id": "1838808377",
    "h_index": 7,
    "papers": 16
   },
   {
    "name": "S. K. Panda",
    "id": "1575855790",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Abraham P. Vinod",
    "id": "2300478379",
    "h_index": 4,
    "papers": 21
   },
   {
    "name": "Rokaha Bhagawan",
    "id": "2456629679",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "M. Jawed",
    "id": "2409832148",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "S. Cairano",
    "id": "2864341",
    "h_index": 40,
    "papers": 304
   }
  ],
  "comment": "Alexander Schperberg and Shivam K. Panda made equal contribution",
  "topics": [
   "egocentric-data",
   "spatial-3d",
   "navigation",
   "foundation-pretraining"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2606.26046v2",
  "pdf_url": "https://arxiv.org/pdf/2606.26046v2",
  "html_url": "https://arxiv.org/html/2606.26046v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2606.26025",
  "slug": "in-context-world-modeling-for-robotic-control",
  "title": "In-Context World Modeling for Robotic Control",
  "abstract": "Modern Vision-Language-Action (VLA) models often fail to generalize to novel setups, such as altered camera viewpoints or robot morphologies, because they are typically conditioned only on current observations and language instructions. By ignoring the underlying system configuration as a variable, these models implicitly assume a fixed execution context encountered during training, necessitating data-intensive fine-tuning for any new environment. In this work, we introduce In-Context World Modeling (ICWM), a framework that treats system identification as an in-context adaptation problem. ICWM enables robot policies to autonomously infer essential system variables from a short history of self-generated, task-agnostic interactions. Unlike traditional In-Context Learning that uses demonstrations to specify what task to perform, ICWM leverages the context window to understand how the system operates. By processing these interactions before task execution, the model implicitly captures the world dynamics of the current system, enabling adaptation to novel configurations without parameter updates. Extensive experiments in simulation and on real-world robot platforms demonstrate that ICWM significantly outperforms standard VLA baselines on novel camera viewpoints.",
  "published": "2026-06-24",
  "updated": "2026-07-03",
  "year": "2026",
  "authors": [
   "Siyin Wang",
   "Junhao Shi",
   "Senyu Fei",
   "Zhaoyang Fu",
   "Li Ji",
   "Jingjing Gong",
   "Xipeng Qiu"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work introduces In-Context World Modeling (ICWM), a framework that treats system identification as an in-context adaptation problem and enables robot policies to autonomously infer essential system variables from a short history of self-generated, task-agnostic interactions.",
  "doi": "10.48550/arXiv.2606.26025",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Siyin Wang",
    "id": "2182224120",
    "h_index": 9,
    "papers": 24
   },
   {
    "name": "Junhao Shi",
    "id": "2348888995",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Senyu Fei",
    "id": "2385785349",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Zhao-Yang Fu",
    "id": "46905137",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Li Ji",
    "id": "2371306698",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Jingjing Gong",
    "id": "2371292918",
    "h_index": 6,
    "papers": 20
   },
   {
    "name": "Xipeng Qiu",
    "id": "2406448211",
    "h_index": 2,
    "papers": 11
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "vla"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.26025v3",
  "pdf_url": "https://arxiv.org/pdf/2606.26025v3",
  "html_url": "https://arxiv.org/html/2606.26025v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.25629",
  "slug": "event-adaptive-motion-planning-with-distilled-vision-language-model-in",
  "title": "Event-Adaptive Motion Planning with Distilled Vision-Language Model in Safety-Critical Situations",
  "abstract": "Robot navigation in safety-critical scenarios faces significant challenges from unforeseen semantic events, where collisions arise primarily from the unpredictable behaviors of dynamic agents rather than unseen objects. While large vision-language models (VLMs) offer remarkable capabilities in commonsense reasoning, frequently invoking them within the continuous control loop introduces severe computational latency, fundamentally destabilizing physical execution. To address these challenges, we propose event-adaptive motion planning (EAMP), an efficient framework for VLM-based robot navigation. Specifically, a prompt-configurable semantic event trigger (PC-SET) selectively activates semantic intervention by continuously monitoring short temporal clips for behavioral anomalies. Upon triggering, an event-triggered distilled SemNav-VLM, fine-tuned via physically verified semantic distillation, maps detected anomalies into discrete strategy-level decisions. Subsequently, a semantic model predictive control (SMPC) module translates these strategies into dynamic reconfigurations of optimization objectives and geometric references. Extensive experiments in safety-critical logistics scenarios demonstrate that EAMP effectively aligns high-level reasoning with low-level control, significantly improving dynamic safety margins over existing baselines while preserving real-time efficiency.",
  "published": "2026-06-24",
  "updated": "2026-06-24",
  "year": "2026",
  "authors": [
   "Zhenwei Huang",
   "Changsheng You",
   "Shuai Wang",
   "Chao Zhou",
   "Wei Xu",
   "Yi Gong"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "eess.SP"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Extensive experiments in safety-critical logistics scenarios demonstrate that EAMP effectively aligns high-level reasoning with low-level control, significantly improving dynamic safety margins over existing baselines while preserving real-time efficiency.",
  "doi": "10.48550/arXiv.2606.25629",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhe Huang",
    "id": "2321334083",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Changsheng You",
    "id": "2318981067",
    "h_index": 4,
    "papers": 24
   },
   {
    "name": "Shuai Wang",
    "id": "2290101718",
    "h_index": 4,
    "papers": 14
   },
   {
    "name": "Chao Zhou",
    "id": "2349764004",
    "h_index": 5,
    "papers": 19
   },
   {
    "name": "Wei Xu",
    "id": "2290366257",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Ying Gong",
    "id": "2423996564",
    "h_index": 0,
    "papers": 2
   }
  ],
  "comment": "8 pages, 8 figures, 4 tables. Accepted by IROS 2026",
  "topics": [
   "rl-control",
   "navigation",
   "video-generation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.25629v1",
  "pdf_url": "https://arxiv.org/pdf/2606.25629v1",
  "html_url": "https://arxiv.org/html/2606.25629v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2606.25497",
  "slug": "sage-nav-leveraging-llm-planning-and-alignment-fusion-for-hierarchical",
  "title": "SAGE-Nav: Leveraging LLM Planning and Alignment Fusion for Hierarchical Scene Graph-Guided Navigation",
  "abstract": "Object-Goal Navigation (ObjNav) requires embodied agents to autonomously locate specified targets using only egocentric visual observations. Existing monolithic methods struggle with long-horizon reasoning and generalize poorly to novel environments. To address these limitations, we propose SAGE-Nav, a novel hierarchical framework that integrates the reasoning capabilities of Large Language Models (LLMs) with dynamic scene graphs. Crucially, it decouples asynchronous global semantic planning from the high-frequency reactive control loop. The LLM serves as a global planner, decomposing abstract instructions into a sequence of semantically grounded waypoints. To translate these plans into dense multi-modal guidance, we design a Hierarchical Scene Graph Encoder (HSGE) that leverages relational graph convolutions to produce structure-aware embeddings preserving both semantic and spatial topology. Furthermore, we develop the Goal-aware Alignment-Fusion Network (GAFN) to dynamically fuse real-time perception with these structural priors. Using an adaptive gating mechanism with an explicit inductive bias, GAFN ensures robust visual-topological alignment for the low-level policy. Extensive evaluations in the i-THOR and RoboTHOR environments demonstrate that SAGE-Nav achieves state-of-the-art performance, delivering substantial gains in navigation efficiency and zero-shot generalization while maintaining the low control latency required for physical robotic deployment.",
  "published": "2026-06-24",
  "updated": "2026-06-24",
  "year": "2026",
  "authors": [
   "Hao Su",
   "Yuehao Huang",
   "Yukai Ma",
   "Yong Liu",
   "Jiajun Lv"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "SAGE-Nav is proposed, a novel hierarchical framework that integrates the reasoning capabilities of Large Language Models with dynamic scene graphs with dynamic scene graphs, and develops the Goal-aware Alignment-Fusion Network (GAFN) to dynamically fuse real-time perception with these structural priors.",
  "doi": "10.48550/arXiv.2606.25497",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hao Su",
    "id": "2336954031",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Yuehao Huang",
    "id": "2282599261",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Yukai Ma",
    "id": "2125737851",
    "h_index": 13,
    "papers": 39
   },
   {
    "name": "Yong Liu",
    "id": "2317960641",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Jiajun Lv",
    "id": "2054668538",
    "h_index": 12,
    "papers": 32
   }
  ],
  "comment": "Accepted by IROS 2026",
  "topics": [
   "egocentric-data",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.25497v1",
  "pdf_url": "https://arxiv.org/pdf/2606.25497v1",
  "html_url": "https://arxiv.org/html/2606.25497v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.8
 },
 {
  "id": "2606.25348",
  "slug": "self-capacitive-tactile-sensor-system-designed-for-companion-robots",
  "title": "Self Capacitive Tactile Sensor System designed for Companion Robots",
  "abstract": "Tactile sensing is essential for humanoid robots to achieve safe physical interaction, dexterous manipulation, and truly human-like responsiveness. However, the design of such systems remains challenging. Conventional approaches often suffer from complex multilayer structures, intricate wiring, high cost, and poor scalability, making it difficult to realize full-body tactile sensing with real-time, low-latency detection while maintaining minimal computational load on the robot's main processor. In this work, we present a simple, scalable and hardware friendly tactile sensing system for a companion humanoid robot based on the self-capacitance principle. The proposed sensor system employs a single conductive fabric layer with a conductive fabric wire architecture and does not require intricate electrode patterning. Scalability was demonstrated by fabricating a 100-point sensor array on a flexible printed circuit (FPC). Evaluation across sampling frequencies showed that 10 Hz is insufficient and misses transient events, whereas 100 Hz and 1000 Hz reliably capture and clearly distinguish all interaction types: gentle touch, slow tapping, fast tapping, and hitting. A decision-tree classifier was implemented directly on the FPGA, offloading real-time inference from the Raspberry Pi 4 with minimal latency and negligible power overhead. This design fully meets the tactile sensing requirements of the HIRO-chan robot and is well-suited for full-body tactile sensing in HIRO-chan and other companion robots.",
  "published": "2026-06-24",
  "updated": "2026-06-24",
  "year": "2026",
  "authors": [
   "Mohsin Ali",
   "Hidenobu Sumioka",
   "Shuhei Ikemoto"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work presents a simple, scalable and hardware friendly tactile sensing system for a companion humanoid robot based on the self-capacitance principle that fully meets the tactile sensing requirements of the HIRO-chan robot and is well-suited for full-body tactile sensing in HIRO-chan and other companion robots.",
  "doi": "10.48550/arXiv.2606.25348",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Mohsin Ali",
    "id": "2299237423",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "H. Sumioka",
    "id": "2177583",
    "h_index": 21,
    "papers": 157
   },
   {
    "name": "Shuhei Ikemoto",
    "id": "2304955242",
    "h_index": 1,
    "papers": 6
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "humanoids",
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.25348v1",
  "pdf_url": "https://arxiv.org/pdf/2606.25348v1",
  "html_url": "https://arxiv.org/html/2606.25348v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.25179",
  "slug": "learning-perceptive-platform-adaptive-locomotion-controllers-for-quadr",
  "title": "Learning Perceptive Platform Adaptive Locomotion Controllers for Quadrupedal Robots",
  "abstract": "Universal quadrupedal locomotion remains limited by the difficulty of integrating perception across diverse robot morphologies. State-of-the-art controllers rely on single-robot training or blind policies that omit real-time perception, leading to poor cross-embodiment generalization. Designing locomotion policies that remain robust across related quadruped morphologies while incorporating perception is challenging. Moreover, fully perceptive policies are often sensitive to noise, whereas blind controllers lack terrain awareness. In this work, we study how perception should be integrated into morphology-aware reinforcement learning architectures for deployable quadrupedal control. Building on MorAL, we train morphology-specialized universal controllers on multiple reference quadrupeds using adaptive terrain curricula. We compare a blind baseline, a critic-perceptive variant (MorAL+), and a fully perceptive actor-critic (PPAL). Policies are evaluated in simulation on flat and rough terrains, and deployed on ANYmal hardware. Results show that critic-only perception improves robustness and tracking consistency over blind baselines while remaining more stable than fully perceptive policies under perception noise. These findings highlight that perception placement and curriculum design are key factors for scalable, morphology-aware locomotion.",
  "published": "2026-06-23",
  "updated": "2026-06-23",
  "year": "2026",
  "authors": [
   "David Rytz",
   "Kim Tien Ly",
   "Ioannis Havoutis"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Results show that critic-only perception improves robustness and tracking consistency over blind baselines while remaining more stable than fully perceptive policies under perception noise, highlighting that perception placement and curriculum design are key factors for scalable, morphology-aware locomotion.",
  "doi": "10.48550/arXiv.2606.25179",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "David Rytz",
    "id": "46432408",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "K. Ly",
    "id": "2248272872",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Ioannis Havoutis",
    "id": "2281743264",
    "h_index": 4,
    "papers": 9
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "rl-control",
   "foundation-pretraining",
   "hardware-codesign",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.25179v1",
  "pdf_url": "https://arxiv.org/pdf/2606.25179v1",
  "html_url": "https://arxiv.org/html/2606.25179v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.25160",
  "slug": "toward-low-latency-vision-language-models-with-doubly-correct-predicti",
  "title": "Toward Low-Latency Vision-Language Models with Doubly-Correct Predictions in Egocentric Visual Understanding",
  "abstract": "The rapid rise of Vision-Language Models (VLMs) in egocentric visual understanding has made low-latency inference in human-robot collaborative (HRC) tasks increasingly critical. Weight pruning techniques developed for VLMs to shrink model size and computation can be readily applied to satisfy the efficiency demands of on-board processing and real-time interactive robotics. Moreover, safe human-robot interaction demands pruning strategies that preserve doubly-correct predictions; outputs must be both accurate and evidentially grounded to mitigate risks and ensure user trust. In this paper, we present a new study of VLM pruning through the lens of doubly-correct prediction. Our experiments surprisingly show that existing pruning methods often preserve the right evidence localization but undermine correct prediction. To address this, we propose a rationale-informed pruning strategy that better aligns evidence with decisions. Benchmark results on egocentric video datasets demonstrate that our method not only achieves the highest prediction accuracy but also outperforms existing approaches in attaining doubly-correct predictions. We aim to stimulate research on efficient and reliable VLMs, ensuring accuracy-driven advances align with the transparency, auditability, and safety required for responsible human-robot interaction and embodied intelligence.",
  "published": "2026-06-23",
  "updated": "2026-06-23",
  "year": "2026",
  "authors": [
   "Qitong Wang",
   "Fan Du",
   "Pranav Maneriker",
   "Jihui Jin",
   "Christopher Rasmussen"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper proposes a rationale-informed pruning strategy that better aligns evidence with decisions in VLM pruning, and achieves the highest prediction accuracy and outperforms existing approaches in attaining doubly-correct predictions.",
  "doi": "10.48550/arXiv.2606.25160",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Qitong Wang",
    "id": "2335989946",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Fan Du",
    "id": "2401266319",
    "h_index": 1,
    "papers": 11
   },
   {
    "name": "Pranav Maneriker",
    "id": "8394636",
    "h_index": 9,
    "papers": 36
   },
   {
    "name": "Jihui Jin",
    "id": "7703157",
    "h_index": 5,
    "papers": 20
   },
   {
    "name": "C. Rasmussen",
    "id": "2288261859",
    "h_index": 2,
    "papers": 6
   }
  ],
  "comment": "International Conference on Intelligent Robots and Systems (IROS) 2026",
  "topics": [
   "egocentric-data",
   "hri",
   "video-generation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.25160v1",
  "pdf_url": "https://arxiv.org/pdf/2606.25160v1",
  "html_url": "https://arxiv.org/html/2606.25160v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2606.25134",
  "slug": "causality-based-parametric-control-barrier-function-for-safe-multi-veh",
  "title": "Causality-Based Parametric Control Barrier Function for Safe Multi-Vehicle Interaction",
  "abstract": "Safe control has been widely studied in various safety-critical applications, for instance, autonomous driving. In order to ensure the autonomous vehicle does not collide with other vehicles, it is essential to obtain an accurate expectation of surrounding vehicles' behavior and react adaptively. Instead of assuming fully cooperative and homogeneous vehicles using the same safety-critical controllers, recent works have been exploring different data-driven approaches to model the neighboring vehicles' underlying controllers with observed data. However, existing works either suffer from 1) the inter-vehicle influence during the multi-vehicle interaction, which makes it hard to determine the causality of surrounding vehicles' behavior in controller modeling, or 2) being dominated by the worst-case analysis, which may lead to overly conservative behavior. In this paper, we extend the prior work on Parametric-Control Barrier Function (Parametric-CBF) to multi-robot interactions with embedded causality inference to explicitly reason over the inter-vehicle influence. Given the learned Causality-based Parametric-CBF, we present an adaptive safety-critical controller that allows the ego vehicle to safely react to surrounding vehicles with the learned expectation. We demonstrate that by leveraging the motion flexibility among multi-vehicle systems, task efficiency can be greatly improved in various interaction-intensive scenarios.",
  "published": "2026-06-23",
  "updated": "2026-06-23",
  "year": "2026",
  "authors": [
   "Yiwei Lyu",
   "Caleb Chang",
   "John M. Dolan"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper extends the prior work on Parametric-Control Barrier Function to multi-robot interactions with embedded causality inference to explicitly reason over the inter-vehicle influence and presents an adaptive safety-critical controller that allows the ego vehicle to safely react to surrounding vehicles with the learned expectation.",
  "doi": "10.48550/arXiv.2606.25134",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yiwei Lyu",
    "id": "2307081934",
    "h_index": 4,
    "papers": 15
   },
   {
    "name": "Caleb Chang",
    "id": "2367346732",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "John M. Dolan",
    "id": "2264995948",
    "h_index": 2,
    "papers": 5
   }
  ],
  "comment": "accepted ICRA 2026",
  "topics": [
   "rl-control",
   "navigation",
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.25134v1",
  "pdf_url": "https://arxiv.org/pdf/2606.25134v1",
  "html_url": "https://arxiv.org/html/2606.25134v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2606.24884",
  "slug": "insight-self-guided-skill-acquisition-via-steerable-vlas",
  "title": "InSight: Self-Guided Skill Acquisition via Steerable VLAs",
  "abstract": "Vision-language-action (VLA) models can learn manipulation skills from demonstrations, but their capabilities are bounded by the skills in the training data. We present InSight, a framework that unlocks autonomous skill acquisition by rendering VLAs steerable at the primitive-action level (e.g., \"move gripper to the bowl\", \"lift upward\", \"pour the bottle\"). InSight consists of two primary stages: (1) an automated segmentation pipeline that partitions demonstrations into labeled primitives via VLM plan decomposition and end-effector poses to enable VLA primitive steerability, and (2) a VLM-guided data flywheel that identifies missing primitives required to accomplish a novel task, autonomously attempts demonstrations of the missing primitives with VLM-proposed low-level control, and automatically labels, stores, and integrates successful demonstrations into the VLA training set. We evaluate InSight across simulation and real-world manipulation tasks, including block flipping, drawer closing, sweeping, twisting, and pouring, without any human demonstrations of these target skills. Once learned, these primitives can be composed to execute novel, long-horizon tasks without additional human demonstrations. Our findings demonstrate that primitive steerability provides a practical foundation for continual skill acquisition in VLA policies. Project website: https://insight-vla.github.io.",
  "published": "2026-06-23",
  "updated": "2026-06-23",
  "year": "2026",
  "authors": [
   "Maggie Wang",
   "Lars Osterberg",
   "Stephen Tian",
   "Ola Shorinwa",
   "Jiajun Wu",
   "Mac Schwager"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "InSight is presented, a framework that unlocks autonomous skill acquisition by rendering VLAs steerable at the primitive-action level and demonstrates that primitive steerability provides a practical foundation for continual skill acquisition in VLA policies.",
  "doi": "10.48550/arXiv.2606.24884",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "M. Wang",
    "id": "2027032530",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Lars W. Osterberg",
    "id": "2323996376",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Stephen Tian",
    "id": "2264968694",
    "h_index": 9,
    "papers": 14
   },
   {
    "name": "O. Shorinwa",
    "id": "116069035",
    "h_index": 14,
    "papers": 37
   },
   {
    "name": "Jiajun Wu",
    "id": "2326444500",
    "h_index": 35,
    "papers": 214
   },
   {
    "name": "Mac Schwager",
    "id": "2285268697",
    "h_index": 7,
    "papers": 21
   }
  ],
  "comment": "Project website: https://insight-vla.github.io",
  "topics": [
   "vla",
   "egocentric-data",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.24884v1",
  "pdf_url": "https://arxiv.org/pdf/2606.24884v1",
  "html_url": "https://arxiv.org/html/2606.24884v1",
  "code_url": "https://insight-vla.github.io",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2606.24628",
  "slug": "artitwinsplat-interactable-digital-twin-reconstruction-via-gaussian-sp",
  "title": "ArtiTwinSplat: Interactable Digital Twin Reconstruction via Gaussian Splatting from RGB-D videos",
  "abstract": "Deploying robots in unstructured real-world environments needs accurate, interactive models of the objects. Constructing these models at scale remains a critical bottleneck for robotic system integration. We present ArtiTwinSplat, a framework that automatically constructs articulated, photo-realistic digital twins of objects directly from RGB-D videos, requiring no CAD models, simulation assets, or manual annotations. Our method is built on 3D Gaussian Splatting that preserve geometric fidelity and photometric realism, coupled with an unsupervised articulation discovery pipeline that recovers part structure and joint kinematics from observed motion alone. With tracking and optimization stages our method provides stable, queryable digital twins that support real-time rendering, viewpoint control, and interactive manipulation. Unlike prior methods confined to simulation, ArtiTwinSplat operates directly on real-world observations and produces twins that are immediately usable by downstream robot planning and learning systems. This method offers a practical, scalable pathway toward digital twin construction, lowering the integration barrier for articulated object manipulation in embodied AI and human-robot collaboration contexts.",
  "published": "2026-06-23",
  "updated": "2026-06-23",
  "year": "2026",
  "authors": [
   "Pranjal Mishra",
   "Ren\u00e9 Zurbr\u00fcgg",
   "Max Wilder-Smith",
   "Marco Hutter",
   "Marc Pollefeys",
   "Zuria Bauer",
   "Hermann Blum"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "ArtiTwinSplat is a framework that automatically constructs articulated, photo-realistic digital twins of objects directly from RGB-D videos, requiring no CAD models, simulation assets, or manual annotations, and operates directly on real-world observations.",
  "doi": "10.48550/arXiv.2606.24628",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Pranjal Mishra",
    "id": "2118703465",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Ren'e Zurbrugg",
    "id": "2297668196",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Maxium Wilder-Smith",
    "id": "2282076724",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Marco Hutter",
    "id": "2381057021",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Marc Pollefeys",
    "id": "2263467446",
    "h_index": 19,
    "papers": 98
   },
   {
    "name": "Z. Bauer",
    "id": "51521118",
    "h_index": 7,
    "papers": 36
   },
   {
    "name": "Hermann Blum",
    "id": "2349540902",
    "h_index": 2,
    "papers": 5
   }
  ],
  "comment": "Presented at the ICRA 2026 Workshop on Advances and Challenges in AI-Driven Automation and Robotic System Integration with Digital Twins, Vienna, June 2026",
  "topics": [
   "spatial-3d",
   "data-teleop",
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.24628v1",
  "pdf_url": "https://arxiv.org/pdf/2606.24628v1",
  "html_url": "https://arxiv.org/html/2606.24628v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2606.24450",
  "slug": "nocontactnoworries-estimating-contact-through-vision-and-proprioceptio",
  "title": "NoContactNoWorries: Estimating Contact through Vision and Proprioception for In-Hand Dexterous Manipulation",
  "abstract": "Perceiving physical contact is fundamental to dexterous manipulation. While robots often rely on dedicated hardware tactile sensors, humans exhibit a remarkable ability to infer contact by integrating visual information with an innate sense of their body's pose and movement. Inspired by this embodied perceptual skill, we investigate whether a robot can learn to infer contact from vision, an approach that also offers a scalable alternative to tactile hardware specifically for binary contact estimation, which faces practical challenges in cost, fragility, and integration. We present NoContactNoWorries, a transformer-based multimodal framework that fuses RGB-D vision with the robot's proprioception to infer binary contact states as a pseudo-tactile signal for hand-object interactions. We validate by training a single contact prediction model on multiple objects and show that the inferred contact signal supports downstream reinforcement learning agents for in-hand object reorientation, generalizing to novel objects. Experiments in both simulation and on a real-world robot validate our approach, highlighting the feasibility of inferring contact from vision and proprioception. Project Page: https://soham2560.github.io/no-contact-no-worries/",
  "published": "2026-06-23",
  "updated": "2026-06-23",
  "year": "2026",
  "authors": [
   "Soham Patil",
   "Avirup Das",
   "Sourabh Bhosale",
   "Spandan Roy"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "NoContactNoWorries is presented, a transformer-based multimodal framework that fuses RGB-D vision with the robot's proprioception to infer binary contact states as a pseudo-tactile signal for hand-object interactions and shows that the inferred contact signal supports downstream reinforcement learning agents for in-hand object reorientation, generalizing to novel objects.",
  "doi": "10.48550/arXiv.2606.24450",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Soham Patil",
    "id": "2328056243",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Avirup Das",
    "id": "2307550817",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Sourabh P. Bhosale",
    "id": "2162911288",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Spandan Roy",
    "id": "2499037",
    "h_index": 28,
    "papers": 105
   }
  ],
  "comment": "Accepted to IEEE/RSJ International Conference on Intelligent Robots and Systems(IROS) 2026",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.24450v1",
  "pdf_url": "https://arxiv.org/pdf/2606.24450v1",
  "html_url": "https://arxiv.org/html/2606.24450v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2606.24377",
  "slug": "pds-joint-a-parametric-double-spiral-joint-tailored-for-dexterous-hand",
  "title": "PDS Joint: A Parametric Double-Spiral Joint Tailored for Dexterous Hands",
  "abstract": "Compliant joints can embed safety and adaptability into dexterous hands, but achieving large-stroke anthropomorphic motion while maintaining joint-specific, directiondependent stiffness and reliable proprioception remains challenging. This paper presents the PDS joint, a parametric doublespiral (PDS) compliant joint that enables systematic shaping of directional stiffness across multiple deformation modes, including flexion/extension, abduction/adduction, and pronation/supination. We instantiate the joint using Archimedean and logarithmic spiral templates for different hand joints and introduce an asymmetry ratio to tailor stiffness distributions for both grasp stability and hyperextension resistance. To make the joint practically usable under large deformation, we co-design embedded inductive proprioception and propose a learningbased calibration pipeline that maps raw inductive signals to joint states using ArUco-marker tracking. Experiments characterize the stiffness landscapes across geometric parameters and demonstrate a non-monotonic dependence of lateral support on asymmetry, indicating the importance of principled parameter tuning. For joint-state estimation in the most challenging abduction/adduction motion, a learned multilayer-perceptron (MLP) mapping reduces the error compared with conventional curve fitting by 41.6%. Finally, we integrate the proposed joints into an open-source dexterous hand as a demonstration platform, on which the hand grasps a set of nine everyday objects and performs safe, contact-rich human-involved interactions.",
  "published": "2026-06-23",
  "updated": "2026-06-23",
  "year": "2026",
  "authors": [
   "Haoyang Li",
   "Yibo Wen",
   "Yixiang Fan",
   "Yiheng Xu",
   "Yufeng Yue"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.AR"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "10.48550/arXiv.2606.24377",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Haoyang Li",
    "id": "2445705368",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yibo Wen",
    "id": "2398864454",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Yixiang Fan",
    "id": "2109755771",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Yiheng Xu",
    "id": "2445293943",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yufeng Yue",
    "id": "2242944661",
    "h_index": 12,
    "papers": 79
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "hardware-codesign",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.24377v1",
  "pdf_url": "https://arxiv.org/pdf/2606.24377v1",
  "html_url": "https://arxiv.org/html/2606.24377v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.24068",
  "slug": "obsgraph-hierarchical-observation-representation-for-embodied-reasonin",
  "title": "ObsGraph: Hierarchical Observation Representation for Embodied Reasoning and Exploration",
  "abstract": "Embodied reasoning and exploration are increasingly considered crucial abilities for robots operating in complex and unfamiliar environments. To accomplish tasks in such settings, an agent must identify and acquire the information necessary for the task through exploration. We propose ObsGraph, an observation-centric hierarchical scene graph that unifies scene representation, retrieval, and exploration. It retains visual evidence and organizes it into room-view-object layers: rooms provide coarse semantic anchors, views preserve contextual object covisibility, and objects store fine-grained details. On top of this representation, we perform coarse-to-fine hierarchical retrieval under a bounded budget, and crucially use retrieval outcomes to structure the exploration candidate space--activating room-level exploration, view refinement, or frontier exploration--thereby tightly coupling representation, retrieval, and adaptive multi-scale exploration. Experiments across embodied reasoning and exploration benchmarks demonstrate improved success and efficiency, highlighting the benefits of structured scene representation and more targeted information gathering driven by identified evidence gaps.",
  "published": "2026-06-23",
  "updated": "2026-06-23",
  "year": "2026",
  "authors": [
   "Taekbeom Lee",
   "Youngseok Jang",
   "Jeonghwa Heo",
   "Jeongjun Choi",
   "H. Jin Kim"
  ],
  "author_count": 5,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "ObsGraph is proposed, an observation-centric hierarchical scene graph that unifies scene representation, retrieval, and exploration and demonstrates improved success and efficiency, highlighting the benefits of structured scene representation and more targeted information gathering driven by identified evidence gaps.",
  "doi": "10.48550/arXiv.2606.24068",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Taekbeom Lee",
    "id": "2297391769",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Youngseok Jang",
    "id": "30684744",
    "h_index": 5,
    "papers": 21
   },
   {
    "name": "Jeong-Dae Heo",
    "id": "2281078680",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Jeongjun Choi",
    "id": "2118016621",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "H. J. Kim",
    "id": "2261895627",
    "h_index": 2,
    "papers": 10
   }
  ],
  "comment": "",
  "topics": [
   "spatial-3d",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.24068v1",
  "pdf_url": "https://arxiv.org/pdf/2606.24068v1",
  "html_url": "https://arxiv.org/html/2606.24068v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.24039",
  "slug": "turbompc-fast-scalable-and-differentiable-model-predictive-control-on",
  "title": "TurboMPC: Fast, Scalable, and Differentiable Model Predictive Control on the GPU",
  "abstract": "Robotics increasingly relies on GPUs for parallel simulation, large-scale learning, and neural-network inference. For model predictive control (MPC) to scale with this paradigm, solvers must run efficiently on this hardware while remaining fast, differentiable, and compatible with expressive MPC formulations used in robotics. We present TurboMPC, a differentiable MPC solver that runs entirely on the GPU and supports state and control inequality constraints, implicit integrators, cross-time-coupled costs, and slack variables. TurboMPC combines sequential quadratic programming (SQP), an alternating direction method of multipliers (ADMM) inner solver, implicit differentiation, and a co-designed JAX-CUDA implementation for efficiency and ease of use. In simulation, we validate TurboMPC on constrained planning, humanoid imitation learning, and reinforcement learning with neural-network cost function tasks, achieving up to $15\\times$ and $58\\times$ speedups over state-of-the-art CPU and GPU differentiable solvers, respectively. We deploy TurboMPC on a full-scale car for minimum-time racing and find that batched, GPU-accelerated tuning of MPC parameters via Bayesian optimization yields significantly faster driving than a hand-tuned baseline. TurboMPC also scales to planning horizons of over $8000$ knot points while maintaining control of the vehicle. We open-source TurboMPC at: https://github.com/ToyotaResearchInstitute/turbompc",
  "published": "2026-06-23",
  "updated": "2026-06-23",
  "year": "2026",
  "authors": [
   "Gabriel Bravo-Palacios",
   "Jianghan Zhang",
   "Zachary Pestrikov",
   "Brian Plancher",
   "Thomas Lew"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.LG",
   "eess.SY",
   "math.OC"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 1,
  "tldr": "TurboMPC is presented, a differentiable MPC solver that runs entirely on the GPU and supports state and control inequality constraints, implicit integrators, cross-time-coupled costs, and slack variables.",
  "doi": "10.48550/arXiv.2606.24039",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Gabriel Bravo-Palacios",
    "id": "1500418939",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Jianghan Zhang",
    "id": "30773631",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Zachary Pestrikov",
    "id": "2444873449",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Brian Plancher",
    "id": "10803865",
    "h_index": 14,
    "papers": 27
   },
   {
    "name": "T. Lew",
    "id": "2327053448",
    "h_index": 5,
    "papers": 23
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "imitation-diffusion",
   "rl-control",
   "hardware-codesign"
  ],
  "orgs": [
   "Toyota Research Institute"
  ],
  "abs_url": "https://arxiv.org/abs/2606.24039v1",
  "pdf_url": "https://arxiv.org/pdf/2606.24039v1",
  "html_url": "https://arxiv.org/html/2606.24039v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.8
 },
 {
  "id": "2607.00033",
  "slug": "learning-dexterous-manipulation-using-contact-wrench-guidance-from-hum",
  "title": "Learning Dexterous Manipulation Using Contact Wrench Guidance From Human Demonstration",
  "abstract": "Dexterous robot manipulation can benefit from the abundance of human demonstrations, but transferring such demonstrations to robot policies remains challenging. We present Contact Wrench Guidance from Human Demonstration in Robotic Dexterous Manipulation (CHORD), a framework for long-horizon manipulation of rigid and articulated objects with reinforcement learning. The key idea is object-centric contact wrench space guidance: we represent human and robot motions by the forces and torques they can induce on the object, enabling similarity to be measured by the induced instantaneous motions. This guidance makes reinforcement learning more scalable for contact-rich dexterous manipulation. We further introduce a large-scale simulation benchmark with 4,739 bimanual dexterous manipulation tasks, constructed from motion-capture datasets and reconstructed in-house videos. Evaluated on 1,831 benchmark tasks, CHORD achieves an average success rate of 82.12%, demonstrating strong scalability. CHORD also generalizes to whole-body manipulation from hand-only and third-person demonstrations, achieving a 90.77% success rate, and the learned policies transfer to the real world in both open-loop and closed-loop settings.",
  "published": "2026-06-22",
  "updated": "2026-08-14",
  "year": "2026",
  "authors": [
   "Xinghao Zhu",
   "Zixi Liu",
   "Shalin Jain",
   "Chenran Li",
   "Milad Noori",
   "Michael Andres Lin",
   "Huihua Zhao",
   "John Welsh",
   "Mrinal Verghese",
   "Wei Liu",
   "Tingwu Wang",
   "Xingye Da",
   "Zhengyi Luo",
   "Vishal Kulkarni",
   "Naema Bhatti",
   "Yuke Zhu",
   "Linxi Fan",
   "Bowen Wen",
   "Danfei Xu",
   "Soha Pouya",
   "Yan Chang"
  ],
  "author_count": 21,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 2,
  "influential_citations": 0,
  "tldr": "This work presents Contact Wrench Guidance from Human Demonstration in Robotic Dexterous Manipulation (CHORD), a framework for long-horizon manipulation of rigid and articulated objects with reinforcement learning, and introduces a large-scale simulation benchmark.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xinghao Zhu",
    "id": "8362363",
    "h_index": 14,
    "papers": 29
   },
   {
    "name": "Zixi Liu",
    "id": "2117942490",
    "h_index": 11,
    "papers": 17
   },
   {
    "name": "Shalin Jain",
    "id": "2339727283",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Chenran Li",
    "id": "2242186839",
    "h_index": 14,
    "papers": 69
   },
   {
    "name": "Milad Noori",
    "id": "2446050004",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Hui Zhao",
    "id": "2311464898",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "John Welsh",
    "id": "2322098872",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Michael A. Lin",
    "id": "2391690840",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Wei Liu",
    "id": "2327422244",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Tingwu Wang",
    "id": "2392415176",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Xingye Da",
    "id": "2350863929",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Zhengyi Luo",
    "id": "2329051397",
    "h_index": 9,
    "papers": 15
   },
   {
    "name": "Vishal Kulkarni",
    "id": "2446022253",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Naema Bhatti",
    "id": "2446035499",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yuke Zhu",
    "id": "2338857250",
    "h_index": 7,
    "papers": 14
   },
   {
    "name": "Linxi (Jim) Fan",
    "id": "3275727",
    "h_index": 18,
    "papers": 22
   },
   {
    "name": "Bowen Wen",
    "id": "2261740421",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Danfei Xu",
    "id": "2264393671",
    "h_index": 4,
    "papers": 15
   },
   {
    "name": "Soha Pouya",
    "id": "2653368",
    "h_index": 14,
    "papers": 38
   },
   {
    "name": "Yan Chang",
    "id": "2322368474",
    "h_index": 5,
    "papers": 9
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "humanoids",
   "egocentric-data",
   "tactile",
   "sim2real",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2607.00033v2",
  "pdf_url": "https://arxiv.org/pdf/2607.00033v2",
  "html_url": "https://arxiv.org/html/2607.00033v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.48
 },
 {
  "id": "2606.23686",
  "slug": "libero-safety-a-comprehensive-benchmark-for-physical-and-semantic-safe",
  "title": "LIBERO-Safety: A Comprehensive Benchmark for Physical and Semantic Safety in Vision-Language-Action Models",
  "abstract": "Despite the impressive manipulation capabilities of Vision-Language-Action (VLA) models, their operational safety under strict constraints remains largely unverified. To address this, we introduce a parametric safety benchmark to procedurally generate safety-critical scenarios with comprehensive stochasticity. To overcome the scalability bottlenecks of human teleoperation, we develop a novel keypose-driven data generation pipeline. Leveraging this infrastructure, we curate a large-scale dataset of 19,664 strictly collision-free demonstrations with extensive domain randomization. We then conduct a systematic cross-paradigm evaluation of eight VLA and two embodied foundation models. Our analysis reveals a critical generalization-safety tension: although high-diversity training fosters safer trajectories, task success remains fundamentally bottlenecked by sub-optimal trajectory synthesis and semantic misalignment. By providing a scalable pipeline, a robust dataset, and profound failure-mode insights, LIBERO-Safety establishes a crucial foundation for developing safe and reliable VLA models.",
  "published": "2026-06-22",
  "updated": "2026-06-26",
  "year": "2026",
  "authors": [
   "Rongxu Cui",
   "Zongzheng Zhang",
   "Jingrui Pang",
   "Haohan Chi",
   "Jinbang Guo",
   "Saining Zhang",
   "Shaoxuan Xie",
   "Xin Jin",
   "Yao Mu",
   "Jiaolong Yang",
   "Guocai Yao",
   "Xianyuan Zhan",
   "Ya-Qin Zhang",
   "Hao Zhao"
  ],
  "author_count": 14,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "ECCV 2026",
  "venue_source": "arxiv-comment",
  "citations": 2,
  "influential_citations": 0,
  "tldr": "A parametric safety benchmark is introduced to procedurally generate safety-critical scenarios with comprehensive stochasticity to overcome the scalability bottlenecks of human teleoperation, and establishes a crucial foundation for developing safe and reliable VLA models.",
  "doi": "10.48550/arXiv.2606.23686",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Rongxu Cui",
    "id": "2370135064",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Zongzheng Zhang",
    "id": "2294931371",
    "h_index": 7,
    "papers": 22
   },
   {
    "name": "Jin-Li Pang",
    "id": "2328913612",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Haohan Chi",
    "id": "2348541743",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Jinbang Guo",
    "id": "2438710168",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Saining Zhang",
    "id": "2233844460",
    "h_index": 7,
    "papers": 28
   },
   {
    "name": "Shaoxuan Xie",
    "id": "2387649455",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Xin Jin",
    "id": "2365147330",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Yao Mu",
    "id": "2348161293",
    "h_index": 2,
    "papers": 18
   },
   {
    "name": "Jiaolong Yang",
    "id": "2237946707",
    "h_index": 10,
    "papers": 22
   },
   {
    "name": "Guocai Yao",
    "id": "2376597393",
    "h_index": 6,
    "papers": 23
   },
   {
    "name": "Xianyuan Zhan",
    "id": "2242851906",
    "h_index": 21,
    "papers": 55
   },
   {
    "name": "Ya-Qin Zhang",
    "id": "2383105350",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Haokun Zhao",
    "id": "2296746816",
    "h_index": 3,
    "papers": 16
   }
  ],
  "comment": "Accepted by ECCV 2026, Project Page: https://libero-safety.github.io/",
  "topics": [
   "vla",
   "sim2real",
   "foundation-pretraining",
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.23686v2",
  "pdf_url": "https://arxiv.org/pdf/2606.23686v2",
  "html_url": "https://arxiv.org/html/2606.23686v2",
  "code_url": "https://libero-safety.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.98
 },
 {
  "id": "2606.23685",
  "slug": "last-hd-learning-latent-physical-reasoning-from-scalable-human-data-fo",
  "title": "LaST-HD: Learning Latent Physical Reasoning from Scalable Human Data for Robot Manipulation",
  "abstract": "Human-hand demonstrations provide a direct and scalable source of physical interaction data for robot learning. While manual retargeting is indispensable for establishing kinematic action correspondence across different morphologies, robust transfer requires going beyond geometry to address the underlying alignment of physical dynamics between human and robot manipulation. To address this, we introduce LaST-HD, a novel human-to-robot action learning paradigm that extends reasoning-before-acting VLA by aligning human-hand and robot demonstrations in a shared latent reasoning space. Rather than mimicking human kinematics, LaST-HD trains an auxiliary action-conditioned world model on unpaired human-hand and robot trajectories to synthesize unified latent targets. After aligning cross-embodiment representations in this shared forward-dynamics space, these targets supervise LaST-HD's latent reasoning process, enabling it to internalize shared physical dynamics and drive efficient human-hand action learning. Moreover, we develop Out-of-Lab (OOL) Glove, a low-cost motion-capture glove tailored to LaST-HD for human-hand data collection. The captured human data provide precise keypoints and serve as universal action supervision across grippers and dexterous hands. Armed with the aligned latent space and high-fidelity human-hand data, we develop a progressive mixed-to-human training recipe comprising mixed human-robot co-training and human-hand online correction post-training. Through mixed co-training, LaST-HD improves generalization to novel objects, scenes, and positions using only human-hand demonstrations. With online correction, LaST-HD further adapts to novel environments and achieves over 90\\% accuracy using only 20 minutes of OOL glove data.",
  "published": "2026-06-22",
  "updated": "2026-06-22",
  "year": "2026",
  "authors": [
   "Jiaming Liu",
   "Yinxi Wang",
   "Chenyang Gu",
   "Siyuan Qian",
   "Xiangju Mi",
   "Hao Chen",
   "Jiawei Chen",
   "Qingpo Wuwu",
   "Xiaoqi Li",
   "Nuowei Han",
   "Yiming Zhang",
   "Xuheng Zhang",
   "Yang Yue",
   "Yeqing Yang",
   "Lei Wang",
   "Peng Jia",
   "Hao Tang",
   "Shanghang Zhang"
  ],
  "author_count": 18,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "A novel human-to-robot action learning paradigm that extends reasoning-before-acting VLA by aligning human-hand and robot demonstrations in a shared latent reasoning space, and a progressive mixed-to-human training recipe comprising mixed human-robot co-training and human-hand online correction post-training.",
  "doi": "10.48550/arXiv.2606.23685",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jiaming Liu",
    "id": "2258602418",
    "h_index": 14,
    "papers": 37
   },
   {
    "name": "Yinxi Wang",
    "id": "2432672689",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Chenyang Gu",
    "id": "2332540856",
    "h_index": 9,
    "papers": 24
   },
   {
    "name": "Siyuan Qian",
    "id": "1610531198",
    "h_index": 5,
    "papers": 15
   },
   {
    "name": "Xiangju Mi",
    "id": "2401829719",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Hao Chen",
    "id": "2383319906",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Jiawei Chen",
    "id": "2445398928",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Qingpo Wuwu",
    "id": "2303654921",
    "h_index": 6,
    "papers": 17
   },
   {
    "name": "Xiaoqi Li",
    "id": "2243138342",
    "h_index": 8,
    "papers": 21
   },
   {
    "name": "Nuowei Han",
    "id": "2359146831",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Yiming Zhang",
    "id": "2445493688",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Xuheng Zhang",
    "id": "2359397486",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Yang Yue",
    "id": "2392691042",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Yeqing Yang",
    "id": "2382037532",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Lei Wang",
    "id": "2221467193",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Peng Jia",
    "id": "2333234654",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Hao-Ran Tang",
    "id": "2444277817",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Shanghang Zhang",
    "id": "2346116279",
    "h_index": 16,
    "papers": 51
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "vla",
   "dexterous-manipulation",
   "foundation-pretraining",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.23685v1",
  "pdf_url": "https://arxiv.org/pdf/2606.23685v1",
  "html_url": "https://arxiv.org/html/2606.23685v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2606.23680",
  "slug": "coordex-coordinating-body-and-hand-priors-for-continuous-dexterous-hum",
  "title": "CoorDex: Coordinating Body and Hand Priors for Continuous Dexterous Humanoid Loco-Manipulation",
  "abstract": "Humanoid loco-manipulation is often simplified into a stop-and-go process: walking to an object, stopping to manipulate it, and then resuming locomotion. It also commonly relies on low degree-of-freedom (DoF) end effectors that behave like an open-close grasp primitive. We introduce CoorDex, a learning pipeline that converts high-dimensional body and dexterous hand control into coordinated latent residual control, enabling high-DoF dexterous loco-manipulation on the move. Starting from simulated whole-body and hand demonstrations, CoorDex trains privileged motion tracking teachers for the humanoid body and dexterous hand, distills them into proprioception-conditioned latent priors, and uses the frozen priors as the action space for downstream residual reinforcement learning. A coordinated latent residual policy composes these priors through shared task context and separate body-hand residual heads, preserving natural whole-body motion while improving finger-level contact reliability. CoorDex enables a Unitree G1 humanoid with a 20-DoF WUJI hand to execute dexterous manipulation while in motion, including non-stop bottle grasping and carrying, fridge door opening on the move, and cube pick-and-turn. Ablations on the walk-grasp-carry task show that joint-space PPO, joint-space hand control, and monolithic latent prediction all fail under the same reward budget, while the latent-prior interface and coordinated residual structure make high-dimensional contact-rich loco-manipulation trainable. Project Page: https://skevinci.github.io/coordex/",
  "published": "2026-06-22",
  "updated": "2026-06-22",
  "year": "2026",
  "authors": [
   "Sikai Li",
   "Shuning Li",
   "Zhenyu Wei",
   "Yunchao Yao",
   "Chenran Li",
   "Mingyu Ding"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 5,
  "influential_citations": 0,
  "tldr": "CoorDex enables a Unitree G1 humanoid with a 20-DoF WUJI hand to execute dexterous manipulation while in motion, including non-stop bottle grasping and carrying, fridge door opening on the move, and cube pick-and-turn.",
  "doi": "10.48550/arXiv.2606.23680",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Sikai Li",
    "id": "2283135687",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Shuning Li",
    "id": "2445383029",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Zhenyu Wei",
    "id": "2394124879",
    "h_index": 4,
    "papers": 16
   },
   {
    "name": "Yunchao Yao",
    "id": "2352910959",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Chenran Li",
    "id": "2242186839",
    "h_index": 14,
    "papers": 69
   },
   {
    "name": "Mingyu Ding",
    "id": "2346837065",
    "h_index": 4,
    "papers": 15
   }
  ],
  "comment": "Project page: https://skevinci.github.io/coordex/",
  "topics": [
   "dexterous-manipulation",
   "humanoids",
   "tactile",
   "rl-control",
   "data-teleop"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2606.23680v1",
  "pdf_url": "https://arxiv.org/pdf/2606.23680v1",
  "html_url": "https://arxiv.org/html/2606.23680v1",
  "code_url": "https://skevinci.github.io/coordex/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.28
 },
 {
  "id": "2606.23531",
  "slug": "bilivla-scene-aware-vision-language-action-model-with-reinforcement-le",
  "title": "BiliVLA: Scene-Aware Vision-Language-Action Model with Reinforcement Learning for Autonomous Biliary Endoscopic Navigation",
  "abstract": "Endoscopic retrograde cholangiopancreatography (ERCP) demands precise endoscopic navigation and stable biliary cannulation within a narrow monocular field characterized by specular reflections, partial occlusions, and frequent tissue contact. Although recent robotic systems and vision-based assistance techniques improve operator ergonomics and provide perceptual cues, their performance degrades under pronounced anatomical variability and safety-critical visual artifacts, which hinders reliable autonomy in cannulation-grade procedures. Here, we present BiliVLA, a scene-aware Vision-Language-Action (VLA) framework that formulates biliary endoscopic navigation as an instruction-conditioned visuomotor learning problem. Given an endoscopic observation and a stage-specific language instruction, BiliVLA jointly predicts the target category, a grounded bounding box, and a discrete three-degree-of-freedom (3-DoF) motor command for a continuum endoscope. The proposed framework incorporates scene-aware supervision to improve semantic target consistency and safety-aware recovery supervision to induce conservative retreat behaviors under luminal wall contact. A key component of BiliVLA is a two-stage training paradigm that combines grounding-enhanced supervised fine-tuning (SFT) with Group Relative Policy Optimization (GRPO), thereby improving action reliability and decision consistency during closed-loop navigation. Across three ERCP subtasks, BiliVLA achieves the best overall performance in physical phantom experiments, with a total mIoU of 0.9625, an overall action precision of 91.96\\%, and an overall success rate (SR) of 84.85\\%. These results indicate that integrating semantic grounding, scene-aware learning, and reward-guided optimization strengthens perception--action alignment and enables more robust autonomous biliary endoscopic navigation.",
  "published": "2026-06-22",
  "updated": "2026-07-15",
  "year": "2026",
  "authors": [
   "Jinsong Lin",
   "Chi Kit Ng",
   "Zhiyong Xiong",
   "Zikang Pan",
   "Yihan Hu",
   "Tabassum Tamima",
   "Ziyi Hao",
   "Eddie Cheung",
   "Jiewen Lai",
   "Huxin Gao",
   "Hongliang Ren"
  ],
  "author_count": 11,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "BiliVLA is a scene-aware Vision-Language-Action (VLA) framework that formulates biliary endoscopic navigation as an instruction-conditioned visuomotor learning problem that improves action reliability and decision consistency during closed-loop navigation.",
  "doi": "10.48550/arXiv.2606.23531",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jinsong Lin",
    "id": "2445481374",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Chikit Ng",
    "id": "2307447792",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Zhiyong Xiong",
    "id": "47845657",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Zikang Pan",
    "id": "1720879996",
    "h_index": 9,
    "papers": 17
   },
   {
    "name": "Yihan Hu",
    "id": "2325677836",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Tabassum Tamima",
    "id": "148246163",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Ziyi Hao",
    "id": "2088548476",
    "h_index": 5,
    "papers": 18
   },
   {
    "name": "E. Cheung",
    "id": "143893830",
    "h_index": 6,
    "papers": 17
   },
   {
    "name": "Jiewen Lai",
    "id": "2269206316",
    "h_index": 3,
    "papers": 23
   },
   {
    "name": "Huxin Gao",
    "id": "1653073415",
    "h_index": 13,
    "papers": 49
   },
   {
    "name": "Hongliang Ren",
    "id": "2260612957",
    "h_index": 13,
    "papers": 53
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "rl-control",
   "navigation",
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.23531v4",
  "pdf_url": "https://arxiv.org/pdf/2606.23531v4",
  "html_url": "https://arxiv.org/html/2606.23531v4",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2606.23431",
  "slug": "dexteleop-0-force-aware-bimanual-dexterous-teleoperation-with-ego-cent",
  "title": "DexTeleop-0: Force-Aware Bimanual Dexterous Teleoperation with Ego-Centric Perception towards Shared Autonomy",
  "abstract": "Fine-grained, bimanual dexterous manipulation remains a foundational challenge in robotics. Traditional teleoperation systems often fail in contact-rich tasks because embodiment gaps hinder accurate kinematic mapping, while tactile and force feedback remain absent. Consequently, data collection efficiency for high-precision tasks remains prohibitively low. To address these limitations, we propose a tactile-driven adaptation strategy designed to enable fine-grained manipulation on top of teleoperation pipelines. Instantiated within our bimanual dexterous framework, DexTeleop-0, this strategy introduces a real-time optimization loop that bridges the embodiment gap by translating coarse human tracking intents into precise, force-compliant robotic commands with tactile sensing. By estimating accurate contact points and leveraging a tactile-enabled fingertip force-sensing profile, the system dynamically computes localized corrections using the operational space Jacobian with respect to joint angle updates. We rigorously evaluate this tactile-driven adaptation strategy across both simulated environments and real-world hardware. Compared with representative baselines, the proposed method consistently achieves higher task success rates and improved execution efficiency in robust grasping, disturbance-resilient manipulation, and complex dexterous tasks.",
  "published": "2026-06-22",
  "updated": "2026-06-22",
  "year": "2026",
  "authors": [
   "Haichao Liu",
   "Yuyao Jiang",
   "Hyunsun Park",
   "Yuanjiang Xue",
   "Ziwei Wang"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "This work proposes a tactile-driven adaptation strategy designed to enable fine-grained manipulation on top of teleoperation pipelines, and consistently achieves higher task success rates and improved execution efficiency in robust grasping, disturbance-resilient manipulation, and complex dexterous tasks.",
  "doi": "10.48550/arXiv.2606.23431",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Haichao Liu",
    "id": "2396080282",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Yuyao Jiang",
    "id": "2445395263",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Hyunsun Park",
    "id": "2301605662",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Yuanjiang Xue",
    "id": "2409064900",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Ziwei Wang",
    "id": "2379837969",
    "h_index": 2,
    "papers": 11
   }
  ],
  "comment": "15 pages, 6 figures, 5 tables",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "data-teleop",
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.23431v1",
  "pdf_url": "https://arxiv.org/pdf/2606.23431v1",
  "html_url": "https://arxiv.org/html/2606.23431v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2606.23085",
  "slug": "foresight-failure-detection-for-long-horizon-robotic-manipulation-with",
  "title": "Foresight: Failure Detection for Long-Horizon Robotic Manipulation with Action-Conditioned World Model Latents",
  "abstract": "Long-horizon tasks are common in real-world robotic deployments, yet failure detection for such tasks remains underexplored. Detecting failures in long-horizon robotic tasks is particularly challenging because failure onset is often ambiguous and dense temporal annotations are typically unavailable. We present Foresight, a failure detection framework that monitors manipulation trajectories using latent representations from an action-conditioned world model. Foresight is trained using only final task-level success or failure labels. By leveraging predictive world-model embeddings, our method provides a unified framework for failure detection across different policies. We further use functional conformal prediction (FCP) to calibrate detection thresholds adaptively. We evaluate Foresight with state-of-the-art vision-language-action policies in simulation on LIBERO-Long, ManiSkill-Long, and BEHAVIOR-1K, compare it against state-of-the-artfailure detection methods, and validate it on real robots with three long-horizon tasks on a ReactorX-200 arm and one task on a Franka arm. Our results suggest that action-conditioned world-model embeddings provide a scalable representation for reliable failure monitoring in long-horizon manipulation.",
  "published": "2026-06-22",
  "updated": "2026-06-22",
  "year": "2026",
  "authors": [
   "Haoran Zhang",
   "Yifu Lu",
   "Boyang Wang",
   "Xuhui Kang",
   "Yen-Ling Kuo",
   "Zezhou Cheng",
   "Mengdi Wang",
   "Odest Chadwicke Jenkins"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 1,
  "tldr": "This work presents Foresight, a failure detection framework that monitors manipulation trajectories using latent representations from an action-conditioned world model, and suggests that action-conditioned world-model embeddings provide a scalable representation for reliable failure monitoring in long-horizon manipulation.",
  "doi": "10.48550/arXiv.2606.23085",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Haoran Zhang",
    "id": "2386932321",
    "h_index": 2,
    "papers": 11
   },
   {
    "name": "Yifu Lu",
    "id": "2327002550",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Boyang Wang",
    "id": "2371132207",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Xuhui Kang",
    "id": "2325953899",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Yen-Ling Kuo",
    "id": "2325949200",
    "h_index": 4,
    "papers": 16
   },
   {
    "name": "Zezhou Cheng",
    "id": "2332680558",
    "h_index": 5,
    "papers": 15
   },
   {
    "name": "Mengdi Wang",
    "id": "2381580282",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "O. C. Jenkins",
    "id": "1792217",
    "h_index": 37,
    "papers": 199
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "vla",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.23085v1",
  "pdf_url": "https://arxiv.org/pdf/2606.23085v1",
  "html_url": "https://arxiv.org/html/2606.23085v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2606.22998",
  "slug": "texedo-test-time-scaling-for-controller-aware-language-conditioned-hum",
  "title": "TEXEDO : Test Time Scaling for Controller-aware Language-conditioned Humanoid Motion Generation",
  "abstract": "Text-conditioned motion generation is a promising interface for programming humanoid robots, yet current generators are often trained on human motion datasets retargeted to robot morphologies. Although such data provides rich semantic and kinematic priors, it fails to capture the nuances of whole-body tracking controllers, including balance, contact dynamics, actuation limits, and controller-specific failure modes. As a result, generated motions can be semantically plausible but difficult or impossible for the robot to execute. We introduce TEXEDO, a test-time scaling framework for humanoid motion generation that improves motion quality without requiring a stronger underlying generator. Given a text prompt, TEXEDO samples multiple candidate motions from a pretrained text-conditioned generator and selects the best motion that is both executable and task-aligned. The reward model combines a dynamic feasibility verifier, distilled from whole-body tracking rollouts to predict physical executability, with a semantic alignment verifier that measures text-motion alignment in a learned co-embedding space. Our pipeline treats dynamic feasibility as a hard constraint and semantic alignment as the selection objective within the feasible set. Through large-scale simulation studies and real-world deployment on a Unitree G1 humanoid robot, we show that TEXEDO consistently improves both tracking fidelity and text alignment. These results demonstrate that grounded verification is an effective path toward deployable language-guided humanoid motion generation. Project website: https://jianuocao.github.io/TEXEDO/",
  "published": "2026-06-22",
  "updated": "2026-06-23",
  "year": "2026",
  "authors": [
   "Jianuo Cao",
   "Yuxin Chen",
   "Yuzhen Song",
   "Masayoshi Tomizuka",
   "Chenran Li",
   "Thomas Tian"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Through large-scale simulation studies and real-world deployment on a Unitree G1 humanoid robot, it is shown that TEXEDO consistently improves both tracking fidelity and text alignment, demonstrating that grounded verification is an effective path toward deployable language-guided humanoid motion generation.",
  "doi": "10.48550/arXiv.2606.22998",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jian Cao",
    "id": "2388683227",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Yuxin Chen",
    "id": "2257096146",
    "h_index": 4,
    "papers": 19
   },
   {
    "name": "Yuzhen Song",
    "id": "2445570752",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Masayoshi Tomizuka",
    "id": "2322446616",
    "h_index": 3,
    "papers": 15
   },
   {
    "name": "Chenran Li",
    "id": "2242186839",
    "h_index": 14,
    "papers": 69
   },
   {
    "name": "T. Tian",
    "id": "2253675978",
    "h_index": 4,
    "papers": 7
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "rl-control",
   "safety-eval"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2606.22998v2",
  "pdf_url": "https://arxiv.org/pdf/2606.22998v2",
  "html_url": "https://arxiv.org/html/2606.22998v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2606.22471",
  "slug": "scalable-multi-task-data-generation-via-reinforcement-learning-for-lan",
  "title": "Scalable Multi-Task Data Generation via Reinforcement Learning for Language-Conditioned Bimanual Dexterous Manipulation",
  "abstract": "A key bottleneck in training generalist policies for bimanual dexterous manipulation is the lack of large-scale, high-quality datasets. Synthetic data generation in simulation provides a scalable alternative to human video demonstrations by overcoming challenges such as morphology mismatch, missing physical interactions, and the generation of robot actions. However, existing approaches based on human teleoperation offer limited task diversity, as object-centric trajectory matching often neglects the feasibility of robot execution. Reinforcement learning (RL) enables broader scalability but is often constrained by handcrafted, task-specific rewards. In this work, we propose a systematic RL-based data generation pipeline that integrates generalizable reward design, effective domain randomization, and language-conditioned task annotations. This pipeline synthesizes diverse, high-quality datasets for dexterous bimanual manipulation and enables training of language-conditioned multi-task policies. Our experiments show that the generated data significantly improves generalization across three representative manipulation tasks.",
  "published": "2026-06-21",
  "updated": "2026-06-29",
  "year": "2026",
  "authors": [
   "Zechu Li",
   "Yufeng Jin",
   "Puze Liu",
   "Jan Peters",
   "Georgia Chalvatzaki"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes a systematic RL-based data generation pipeline that integrates generalizable reward design, effective domain randomization, and language-conditioned task annotations and enables training of language-conditioned multi-task policies.",
  "doi": "10.48550/arXiv.2606.22471",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zechu Li",
    "id": "2290135844",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Yufeng Jin",
    "id": "2333435216",
    "h_index": 3,
    "papers": 14
   },
   {
    "name": "Puze Liu",
    "id": "2108920051",
    "h_index": 9,
    "papers": 27
   },
   {
    "name": "Jan Peters",
    "id": "2337797547",
    "h_index": 2,
    "papers": 15
   },
   {
    "name": "G. Chalvatzaki",
    "id": "1989757",
    "h_index": 20,
    "papers": 53
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "egocentric-data",
   "sim2real",
   "rl-control",
   "data-teleop",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.22471v2",
  "pdf_url": "https://arxiv.org/pdf/2606.22471v2",
  "html_url": "https://arxiv.org/html/2606.22471v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.22397",
  "slug": "do-rigid-body-simulators-dream-of-soft-robots-learning-contact-rich-ma",
  "title": "Do Rigid-Body Simulators Dream of Soft Robots? Learning Contact-Rich Manipulation for Tendon-Driven Continuum Robots",
  "abstract": "Learning contact-rich, whole-body manipulation for soft continuum robots is held back by the lack of simulation infrastructure that has accelerated rigid-robot manipulation. Existing soft robot simulators are physically grounded but lack the contact handling, actuation support, or learning integration needed for contact-rich manipulation; rigid-body approximations offer these capabilities but sacrifice physical grounding. We bridge this gap for tendon-driven continuum robots (TDCRs) by deriving a continuum-mechanics-informed discretization that places the soft robot natively inside MuJoCo, unifying tendon forces, body contact, and dynamics in a single physics pipeline. We validate the simulator against a Cosserat rod reference (static and dynamic) and real TDCR hardware. We then train state-based imitation learning policies via teleoperation in simulation and deploy them zero-shot to a physical 3-segment TDCR on a 7-DoF Franka arm across two contact-rich manipulation tasks. To our knowledge, this is the first demonstration of sim-to-real transfer for contact-rich manipulation with continuum robots.",
  "published": "2026-06-21",
  "updated": "2026-06-21",
  "year": "2026",
  "authors": [
   "Chengnan Shentu",
   "Nicholas Baldassini",
   "Tongjia Zheng",
   "Priyanka Rao",
   "Jessica Burgner-Kahrs"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work derives a continuum-mechanics-informed discretization that places the soft robot natively inside MuJoCo, unifying tendon forces, body contact, and dynamics in a single physics pipeline, and demonstrates the first demonstration of sim-to-real transfer for contact-rich manipulation with continuum robots.",
  "doi": "10.48550/arXiv.2606.22397",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Chengnan Shentu",
    "id": "2185646120",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "N. Baldassini",
    "id": "2264128024",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Tongjia Zheng",
    "id": "10031971",
    "h_index": 9,
    "papers": 33
   },
   {
    "name": "Priyanka Rao",
    "id": "2041661592",
    "h_index": 8,
    "papers": 18
   },
   {
    "name": "J. Burgner-Kahrs",
    "id": "1403632880",
    "h_index": 29,
    "papers": 119
   }
  ],
  "comment": "Project Page: https://continuumroboticslab.github.io/opencr-mujoco/",
  "topics": [
   "humanoids",
   "tactile",
   "sim2real",
   "imitation-diffusion",
   "data-teleop",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.22397v1",
  "pdf_url": "https://arxiv.org/pdf/2606.22397v1",
  "html_url": "https://arxiv.org/html/2606.22397v1",
  "code_url": "https://continuumroboticslab.github.io/opencr-mujoco/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2606.22332",
  "slug": "tactile-genesis-exploring-tactile-sensors-at-scale-for-learning-dexter",
  "title": "Tactile Genesis: Exploring Tactile Sensors at Scale for Learning Dexterous Tasks",
  "abstract": "Tactile sensing is critical for contact-rich dexterous manipulation, yet it remains unclear which tactile abstractions a policy needs and when richer tactile fields justify their hardware cost. This is hard to study empirically: each sensor effectively defines a new robot, and no lab can replicate the same learning experiment across all of them. We present Tactile Genesis, a GPU-parallel tactile sensor simulation platform that exposes binary contact, contact depth, per-taxel kinematic force/torque, elastomer marker displacement, geometry-aware proximity, contact audio, and a voxelized temperature field (the first of its kind in robot learning physics simulation platforms) under a common interface, with configurable placement, resolution, and a realistic noise model (drift, hysteresis, dead taxels, crosstalk). It scales past 20,000 parallel environments and 1,000 taxels on a single GPU, improving throughput by 3 to 20 times over previous tactile simulators. We train teacher-student policies on three dexterous tasks, ablating sensor type, placement, resolution, and noise, and verify transfer to the real XHand1. Proprioception alone is insufficient on every task. Sensor placement dominates sensor type: fingertip-only coverage trails whole-hand coverage by a wide margin, while adding the palm and proximal phalanges closes most of the gap to the privileged teacher. Resolution matters far less than coverage: placing 200 taxels across the whole hand suffices across tasks. We find that force/torque per taxel is consistently the most useful sensor type. These results give concrete guidance for both future tactile hardware design for improving robot hands and policy-side observation choice in dexterous manipulation. https://neuroagents-lab.github.io/tactile-genesis/",
  "published": "2026-06-21",
  "updated": "2026-07-09",
  "year": "2026",
  "authors": [
   "Trinity Chung",
   "Kashu Yamazaki",
   "Dhruv Patel",
   "Alexis Duburcq",
   "Yiling Qiao",
   "Katerina Fragkiadaki",
   "Aran Nayebi"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A GPU-parallel tactile sensor simulation platform that exposes binary contact, contact depth, per-taxel kinematic force/torque, elastomer marker displacement, geometry-aware proximity, contact audio, and a voxelized temperature field under a common interface, with configurable placement, resolution, and a realistic noise model.",
  "doi": "10.48550/arXiv.2606.22332",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "T. Chung",
    "id": "2265754339",
    "h_index": 4,
    "papers": 14
   },
   {
    "name": "Kashu Yamazaki",
    "id": "1556433845",
    "h_index": 14,
    "papers": 32
   },
   {
    "name": "Dhruv Patel",
    "id": "2328566408",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Alexis Duburcq",
    "id": "49720341",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Yilin Qiao",
    "id": "2350170271",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Katerina Fragkiadaki",
    "id": "2371999789",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Aran Nayebi",
    "id": "2238276749",
    "h_index": 3,
    "papers": 11
   }
  ],
  "comment": "24 pages, 8 figures, 12 tables",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "sim2real",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.22332v2",
  "pdf_url": "https://arxiv.org/pdf/2606.22332v2",
  "html_url": "https://arxiv.org/html/2606.22332v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.22278",
  "slug": "any-body-guard-universal-safeguarding-for-manipulation-policies-via-ac",
  "title": "Any-Body Guard: Universal Safeguarding for Manipulation Policies via Action Masking",
  "abstract": "Ensuring safety of learning-enabled robotic manipulation across diverse embodiments and tasks still requires significant manual engineering. Existing approaches typically rely on heuristically designed fallback controllers or complex forward invariance assessments. These methods are often too conservative for task success, too computationally expensive for real-time execution, too heuristic to provide useful safety guarantees, or too engineering-heavy to transfer between setups. In this paper, we propose a universal safeguarding approach, X-Safe, which reasons directly in the robot's configuration space to provide formal probabilistic guarantees for collision avoidance. By operating in the configuration space, our method transfers across embodiments while relying solely on an object-based, quasi-static scene representation and a forward kinematics model of the robotic manipulator. Thus, X-Safe provides useful formal safety guarantees without requiring additional data, or engineering effort for different embodiments or scenes. We demonstrate X-Safe for diverse embodiments and policies, both in simulation and on hardware. We observe less degradation in task performance compared to state-of-the-art safeguarding, no collisions on hardware experiments, and empirically corroborate our formal guarantees.",
  "published": "2026-06-21",
  "updated": "2026-06-21",
  "year": "2026",
  "authors": [
   "Alex Beaudin",
   "Hanna Krasowski",
   "Kartik Nagpal",
   "Sanjit A. Seshia",
   "Murat Arcak",
   "Negar Mehr"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.LG",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A universal safeguarding approach, X-Safe, is proposed, which reasons directly in the robot's configuration space to provide formal probabilistic guarantees for collision avoidance, and is demonstrated for diverse embodiments and policies, both in simulation and on hardware.",
  "doi": "10.48550/arXiv.2606.22278",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "A. Beaudin",
    "id": "2407570139",
    "h_index": 1,
    "papers": 10
   },
   {
    "name": "Hanna Krasowski",
    "id": "2042798834",
    "h_index": 9,
    "papers": 34
   },
   {
    "name": "Kartik Nagpal",
    "id": "2089898621",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "S. Seshia",
    "id": "1775517",
    "h_index": 72,
    "papers": 433
   },
   {
    "name": "Murat Arcak",
    "id": "2335659565",
    "h_index": 1,
    "papers": 13
   },
   {
    "name": "Negar Mehr",
    "id": "2270679762",
    "h_index": 6,
    "papers": 11
   }
  ],
  "comment": "",
  "topics": [
   "spatial-3d",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.22278v1",
  "pdf_url": "https://arxiv.org/pdf/2606.22278v1",
  "html_url": "https://arxiv.org/html/2606.22278v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.22136",
  "slug": "wh0-generative-world-models-as-scalable-sources-of-egocentric-human-ha",
  "title": "Wh0: Generative World Models as Scalable Sources of Egocentric Human Hand Manipulation Data",
  "abstract": "Scaling dexterous manipulation requires generalization across objects, scenes, and tasks, yet existing data sources face a trade-off between scale and scene/embodiment alignment: teleoperation data is well aligned with robot deployment but expensive to collect; simulation is scalable but limited by the sim-to-real gap; and real egocentric videos scale effectively but remain misaligned with robot deployment. We propose Wh0, a framework that uses generative video world models as scalable and controllable sources of egocentric human-hand manipulation data to unlock the manipulation capabilities of pretrained dexterous VLA models. Conditioned on language, objects, and scenes, Wh0 uses a generative world model to produce WM-H, a 50k-episode dataset of egocentric human-object interaction videos. Wh0 then converts the generated videos into robot-trainable supervision through hand motion reconstruction and visual editing. Co-trained with a limited amount of real robot data, WM-H adapts pretrained VLA models to dexterous manipulation deployment. Across 18 real-world dexterous manipulation tasks, compared with a model post-trained only on robot data, Wh0 improves zero-shot success on unseen tasks from 8.3% to 38.9%. Ablation studies further show that scalable generation and scene/embodiment alignment are key drivers of performance gains. Videos and open-source code can be found on our project website: https://chenyt31.github.io/wh0.github.io/.",
  "published": "2026-06-20",
  "updated": "2026-06-23",
  "year": "2026",
  "authors": [
   "Yangtao Chen",
   "Zixuan Chen",
   "Peiyang Wang",
   "Yong-Lu Li",
   "Jing Huo",
   "Jieqi Shi",
   "Yang Gao"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 2,
  "influential_citations": 0,
  "tldr": "Wh0, a framework that uses generative video world models as scalable and controllable sources of egocentric human-hand manipulation data to unlock the manipulation capabilities of pretrained dexterous VLA models, is proposed.",
  "doi": "10.48550/arXiv.2606.22136",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yang Chen",
    "id": "2261894562",
    "h_index": 7,
    "papers": 12
   },
   {
    "name": "Zixuan Chen",
    "id": "2247578676",
    "h_index": 5,
    "papers": 19
   },
   {
    "name": "PeiYang Wang",
    "id": "2265425936",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Yong-Lu Li",
    "id": "2395808377",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Jing Huo",
    "id": "2055851838",
    "h_index": 17,
    "papers": 84
   },
   {
    "name": "Jieqi Shi",
    "id": "2323526727",
    "h_index": 4,
    "papers": 19
   },
   {
    "name": "Yang Gao",
    "id": "2382886824",
    "h_index": 2,
    "papers": 8
   }
  ],
  "comment": "Under review. The first three authors contributed equally to this work",
  "topics": [
   "world-models",
   "vla",
   "dexterous-manipulation",
   "egocentric-data",
   "sim2real",
   "data-teleop",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.22136v2",
  "pdf_url": "https://arxiv.org/pdf/2606.22136v2",
  "html_url": "https://arxiv.org/html/2606.22136v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.48
 },
 {
  "id": "2606.22091",
  "slug": "acesplat-accelerated-3d-gaussian-scene-regression-via-rgb-and-poses-on",
  "title": "ACEsplat: Accelerated 3D Gaussian Scene Regression via RGB and Poses Only",
  "abstract": "Per-scene 3D Gaussian Splatting (3DGS) enables high-fidelity rendering, but practical robotic and AR scene capture pipelines often depend on external geometric initialization (e.g., SfM point clouds or depth estimates), which can be slow and brittle in on-site deployment. We present ACEsplat, a fast per-scene optimization framework that reconstructs 3D Gaussian representations from RGB images and camera poses only, without requiring external 3D priors (e.g., precomputed SfM models or supervised depth maps). ACEsplat uses a two-stage pipeline: (1) a self-supervised scene coordinate regression (SCR) module builds an internal geometry prior within 4--5 minutes; (2) SCR features and coordinate priors are fused by a lightweight Gaussian initialization head, followed by per-scene 3DGS optimization. On static-view rendering, ACEsplat achieves 29.11 dB PSNR on Wayspots with real-time SLAM poses and 33.20 dB on Cambridge Landmarks with SfM-refined poses. On RealEstate10K sparse-view novel view synthesis, it achieves competitive image fidelity under a challenging 2-view setting. ACEsplat completes scene-specific SCR mapping and 3DGS reconstruction within 15--25 minutes on a single GPU, making it a practical RGB+pose-only solution for rapid scene setup in robotics and mixed-reality applications.",
  "published": "2026-06-20",
  "updated": "2026-06-20",
  "year": "2026",
  "authors": [
   "Mingkai Liu",
   "Haohua Que",
   "Dikai Fan",
   "Haojia Gao",
   "Tianle Zhu",
   "Handong Yao",
   "Qian Zhang",
   "Ruopeng Zhang",
   "Xianliang Huang",
   "Fei Qiao"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "ACEsplat is presented, a fast per-scene optimization framework that reconstructs 3D Gaussian representations from RGB images and camera poses only, without requiring external 3D priors, making it a practical RGB+pose-only solution for rapid scene setup in robotics and mixed-reality applications.",
  "doi": "10.48550/arXiv.2606.22091",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Mingkai Liu",
    "id": "2340328236",
    "h_index": 2,
    "papers": 15
   },
   {
    "name": "Haohua Que",
    "id": "2200627928",
    "h_index": 3,
    "papers": 22
   },
   {
    "name": "Dikai Fan",
    "id": "2386017557",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Haojia Gao",
    "id": "2331263068",
    "h_index": 2,
    "papers": 15
   },
   {
    "name": "Tianle Zhu",
    "id": "1453038311",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Handong Yao",
    "id": "2262521303",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Qian Zhang",
    "id": "2261816376",
    "h_index": 11,
    "papers": 24
   },
   {
    "name": "Ruopeng Zhang",
    "id": "2303327824",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Xianliang Huang",
    "id": "2386423849",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Fei Qiao",
    "id": "2112050329",
    "h_index": 5,
    "papers": 10
   }
  ],
  "comment": "",
  "topics": [
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.22091v1",
  "pdf_url": "https://arxiv.org/pdf/2606.22091v1",
  "html_url": "https://arxiv.org/html/2606.22091v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.21866",
  "slug": "surge-surrogate-gradient-guided-evolution-for-co-design-of-legged-robo",
  "title": "SurGE: Surrogate Gradient-guided Evolution for Co-design of Legged Robots with Parallel Elasticity",
  "abstract": "Co-design of legged robots with elastic elements is challenging due to the non-differentiability of contact dynamics and mechanism engagement. This paper presents SurGE, a framework that computes surrogate gradients of the design objective through a differentiable pipeline consisting of a kinodynamic single-rigid-body (Kino-SRB) model and a design-aware control policy, and injects them into CMA-ES via mean shift with cosine-annealed step decay. On a 4-DOF design space of a hopping robot with unidirectional parallel spring, SurGE achieves 6 times lower cross-seed standard deviation and 18% tighter population concentration compared to vanilla CMA-ES, while matching or improving the best objective. Hardware experiments on a 2D design subspace show that, starting from a hand-tuned initial design, SurGE reduces the design objective by 37.65% on hardware, with the improvement trend identified in simulation transferring consistently to the physical system. SurGE provides the potential to accelerate non-differentiable co-design problems in legged robots via surrogate model gradients.",
  "published": "2026-06-20",
  "updated": "2026-06-20",
  "year": "2026",
  "authors": [
   "Yulun Zhuang",
   "Yue Qin",
   "Justin Lu",
   "Zelin Shen",
   "Yichen Wang",
   "Sicheng He",
   "Yanran Ding"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "SurGE is presented, a framework that computes surrogate gradients of the design objective through a differentiable pipeline consisting of a kinodynamic single-rigid-body model and a design-aware control policy, and injects them into CMA-ES via mean shift with cosine-annealed step decay.",
  "doi": "10.48550/arXiv.2606.21866",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yulun Zhuang",
    "id": "2152482461",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Yue Qin",
    "id": "2445446280",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Justin Lu",
    "id": "2911044",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Zelin Shen",
    "id": "2346905034",
    "h_index": 1,
    "papers": 9
   },
   {
    "name": "Yichen Wang",
    "id": "2383806762",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Sicheng He",
    "id": "2445229338",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yanran Ding",
    "id": "2277577437",
    "h_index": 2,
    "papers": 6
   }
  ],
  "comment": "8 pages, 7 figures. Accepted for publication at IROS 2026. Website at https://arcad-lab-um.github.io/surge-codesign/",
  "topics": [
   "humanoids",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.21866v1",
  "pdf_url": "https://arxiv.org/pdf/2606.21866v1",
  "html_url": "https://arxiv.org/html/2606.21866v1",
  "code_url": "https://arcad-lab-um.github.io/surge-codesign/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.5
 },
 {
  "id": "2606.21737",
  "slug": "programmable-magnetic-soft-robots-with-controlled-locomotion-and-direc",
  "title": "Programmable magnetic soft robots with controlled locomotion and directional liquid cargo release",
  "abstract": "Magnetically programmable soft elastomers enable complex shape morphing and locomotion dynamics in small scale soft robots under external magnetic fields. Benefiting from their programmed deformation and wireless actuation capabilities, magnetic soft robots have emerged as promising platforms for targeted drug delivery, especially in human gastrointestinal tract. However, achieving controlled directional liquid cargo release toward desired tissue interface while preserving the encoded shape morphing and locomotion capabilities remain a significant challenge. Here, we report a new design strategy that employs an optimized magnetization profile to enable controlled directional release of aqueous cargo without compromising shape morphing and locomotion capabilities. Magnetic soft robots with a specific spatially distributed magnetization profile allow directional alignment of the release interface with the orientation of the external magnetic field. This orientation control ensures active alignment of the release interface toward the intestinal wall prior to drug release. An interconnected microporous elastomer is embedded within the robot for aqueous cargo storage, while a thin microcrystalline wax layer seals the release opening hole to isolate the stored liquid cargo from external environment during transport. Triggered release is achieved by mechanically rupturing the wax sealing layer under a higher magnitude external magnetic field. Controlled directional flipping, locomotion, and triggered release are decoupled through external magnetic field's direction and strength. The controlled directional release strategy reported here integrates directional targeted liquid cargo release, shape morphing, and locomotion, which establishes the groundwork for target drug delivery in gastrointestinal tract applications.",
  "published": "2026-06-19",
  "updated": "2026-06-19",
  "year": "2026",
  "authors": [
   "Youyi Zhou",
   "Zoe Evelyn Gureno",
   "Meghna Majumder",
   "Yunus Alapan"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The controlled directional release strategy reported here integrates directional targeted liquid cargo release, shape morphing, and locomotion, which establishes the groundwork for target drug delivery in gastrointestinal tract applications.",
  "doi": "10.48550/arXiv.2606.21737",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Youyi Zhou",
    "id": "2373572775",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Zoe Evelyn Gureno",
    "id": "2444882812",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "M. Majumder",
    "id": "2217488927",
    "h_index": 0,
    "papers": 5
   },
   {
    "name": "Yunus Alapan",
    "id": "2373094133",
    "h_index": 0,
    "papers": 7
   }
  ],
  "comment": "This manuscript has been accepted to the IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS 2026)",
  "topics": [
   "humanoids",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.21737v1",
  "pdf_url": "https://arxiv.org/pdf/2606.21737v1",
  "html_url": "https://arxiv.org/html/2606.21737v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2606.21527",
  "slug": "logos-lidar-only-gaussian-elevation-splatting-for-unified-tiny-obstacl",
  "title": "LOGOS: LiDAR-Only Gaussian Elevation Splatting for Unified Tiny Obstacle Segmentation",
  "abstract": "Robust obstacle segmentation is essential for the safety of intelligent robots, where LiDAR-based perception systems play a fundamental role in the robot-environment interaction. While extensive LiDAR-based approaches have demonstrated high performance on common obstacles in urban scenarios, their results on tiny obstacles such as curbs, gravel, and potholes remain unsatisfactory due to the significant similarity between tiny obstacles and inherent road undulations. Moreover, their segmentation accuracy even deteriorates sharply when the LiDAR scans suffer from degradation in challenging off-road scenes. To overcome these bottlenecks, we propose LOGOS, a LiDAR-only unified tiny obstacle segmentation system, which models the road surface as a continuous mixture of 2D Gaussian primitives and distinguishes tiny obstacles via high-presicion elevation estimation. Unlike existing Gaussian splatting methods that rely on iterative RGB training, LOGOS is a backpropagation-free LiDAR-only approach. It directly estimates Gaussian parameters via a freespace-aware initialization by incrementally pruning non-road primitives using smoothness constraints. Subsequently, pointwise signed distances are computed via a novel normal-aware elevation splatting function, ensuring robustness to both flat and sloped terrains. We evaluate LOGOS on a highly heterogeneous benchmark of point cloud frames collected from urban mobility scenarios and mining haulage off-road environments. These data are practically acquired using different LiDAR sensors and exhibit large variations in point density, terrain roughness, and obstacle types. Experiments on the road and off-road scenes demonstrate that LOGOS significantly outperforms other state-of-the-art methods, particularly in degraded point cloud regions and challenging off-road scenarios, while maintaining real-time efficiency.",
  "published": "2026-06-19",
  "updated": "2026-06-19",
  "year": "2026",
  "authors": [
   "Nan Ming",
   "Yeqiang Qian",
   "Chunxiang Wang",
   "Ming Yang"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "LOGOS, a LiDAR-only unified tiny obstacle segmentation system, which models the road surface as a continuous mixture of 2D Gaussian primitives and distinguishes tiny obstacles via high-presicion elevation estimation and ensures robustness to both flat and sloped terrains is proposed.",
  "doi": "10.48550/arXiv.2606.21527",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Nan Ming",
    "id": "2366391215",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Yeqiang Qian",
    "id": "22187872",
    "h_index": 15,
    "papers": 55
   },
   {
    "name": "Chunxiang Wang",
    "id": "2265423969",
    "h_index": 5,
    "papers": 44
   },
   {
    "name": "Ming Yang",
    "id": "50367252",
    "h_index": 28,
    "papers": 146
   }
  ],
  "comment": "",
  "topics": [
   "spatial-3d",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.21527v1",
  "pdf_url": "https://arxiv.org/pdf/2606.21527v1",
  "html_url": "https://arxiv.org/html/2606.21527v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.21501",
  "slug": "univiewvla-a-unified-multiview-vision-language-action-model-with-world",
  "title": "UniviewVLA: A Unified Multiview Vision-Language-Action Model with World Modeling",
  "abstract": "Occluded tasks remain a bottleneck in robot manipulation. Existing solutions either deploy additional physical cameras requiring training-inference camera parity, or rely on explicit 3D reconstruction with high computational cost. Moreover, both approaches rely on standard agent-view and wrist-view observations, while failing to capture occlusion information and future scene evolution. To this end, we propose UniviewVLA, a unified multiview Vision-Language-Action model with world modeling, which infers multiview scene evolution for action prediction from only standard two-camera observations. We demonstrate that by leveraging generated multiview future views from the world model, UniviewVLA reveals occluded cues and models future scene evolution, improving action prediction and removing the need for extra hardware or explicit reconstruction. Besides, to accelerate inference while preserving prediction accuracy, UniviewVLA develops Motion-Informative Token Compression, which compresses each generated view from 625 to 16 tokens and reduces per-view latency from 6-7s to 0.2-0.3s. UniviewVLA also proposes training-free Action-Entropy View Selection, which dynamically identifies the most action-informative view at different inference stages. Extensive experiments show that UniviewVLA achieves 95.8% on LIBERO and 4.60 on CALVIN ABCD to D, both standard occlusion-free benchmarks. On customized occlusion-focused tasks, it improves success rate from 40.0% to 73.3%, and average real-robot success rate by 33.4 points, demonstrating stronger occlusion-focused performance without sacrificing standard occlusion-free benchmarks.",
  "published": "2026-06-19",
  "updated": "2026-06-19",
  "year": "2026",
  "authors": [
   "Tao Xu",
   "Runhao Zhang",
   "Zhijian Huang",
   "Jiayi Guan",
   "Jiaxin Wang",
   "Yifan Ding",
   "Yong-Lu Li",
   "Long Chen",
   "Guang Chen",
   "Jinghui Lu"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "UniviewVLA is proposed, a unified multiview Vision-Language-Action model with world modeling, which infers multiview scene evolution for action prediction from only standard two-camera observations, and dynamically identifies the most action-informative view at different inference stages.",
  "doi": "10.48550/arXiv.2606.21501",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Tao Xu",
    "id": "2335524476",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Runhao Zhang",
    "id": "2294776527",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Zhijian Huang",
    "id": "2393917972",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Jiayi Guan",
    "id": "2303657832",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Jiaxin Wang",
    "id": "2360237805",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Yifan Ding",
    "id": "2356785654",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Yong-Lu Li",
    "id": "2395808377",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Long Chen",
    "id": "2366273398",
    "h_index": 6,
    "papers": 27
   },
   {
    "name": "Guangcheng Chen",
    "id": "2161031516",
    "h_index": 4,
    "papers": 14
   },
   {
    "name": "Jinghui Lu",
    "id": "2390142796",
    "h_index": 2,
    "papers": 7
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "vla",
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.21501v1",
  "pdf_url": "https://arxiv.org/pdf/2606.21501v1",
  "html_url": "https://arxiv.org/html/2606.21501v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.21135",
  "slug": "odoriko-a-shape-aware-multimodal-diffusion-framework-for-human-motion",
  "title": "Odoriko: A Shape-Aware Multimodal Diffusion Framework for Human Motion",
  "abstract": "Human motion generation has been widely studied across diverse input modalities, text, music, and video, and recent efforts have unified these into single multimodal frameworks. However, while morphological factors such as gender and body shape are known to produce distinct kinematic signatures, no existing unified framework incorporates this into generation, treating all subjects as morphologically equivalent. We present Odoriko, the first unified multimodal motion generation framework that reflects subject bio-morphological information directly in synthesized motion output. Rather than averaging over subject variation, Odoriko generates motion that is consistent with who is moving, not just what they are asked to do, across text, music, and video conditions within a single model. When explicit morphological information is unavailable, Odoriko additionally recovers subject morphology alongside motion, unifying estimation and generation in one framework. Extensive experiments across text-to-motion, music-to-dance, and video-to-motion benchmarks demonstrate that Odoriko matches or exceeds prior specialized models on standard metrics, while enabling morphology-consistent generation that no existing unified framework supports.",
  "published": "2026-06-19",
  "updated": "2026-06-19",
  "year": "2026",
  "authors": [
   "Dongseok Shim",
   "Julian Tanke",
   "Kengo Uchida",
   "Christian Simon",
   "Koichi Saito",
   "Takashi Shibuya",
   "Shusuke Takahashi",
   "Yuki Mitsufuji"
  ],
  "author_count": 8,
  "categories": [
   "cs.CV",
   "cs.GR",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "ECCV 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Odoriko is presented, the first unified multimodal motion generation framework that reflects subject bio-morphological information directly in synthesized motion output, enabling morphology-consistent generation that no existing unified framework supports.",
  "doi": "10.48550/arXiv.2606.21135",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "D. Shim",
    "id": "6981689",
    "h_index": 6,
    "papers": 27
   },
   {
    "name": "Julian Tanke",
    "id": "2361504089",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Kengo Uchida",
    "id": "2186082380",
    "h_index": 6,
    "papers": 21
   },
   {
    "name": "Christian Simon",
    "id": "2323740466",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Koichi Saito",
    "id": "2308100287",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Takashi Shibuya",
    "id": "47720660",
    "h_index": 16,
    "papers": 56
   },
   {
    "name": "Shusuke Takahashi",
    "id": "2110724776",
    "h_index": 16,
    "papers": 56
   },
   {
    "name": "Yuki Mitsufuji",
    "id": "2373457235",
    "h_index": 5,
    "papers": 15
   }
  ],
  "comment": "ECCV 2026",
  "topics": [
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.21135v1",
  "pdf_url": "https://arxiv.org/pdf/2606.21135v1",
  "html_url": "https://arxiv.org/html/2606.21135v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2606.20958",
  "slug": "learning-based-modeling-of-soft-robots-via-cosserat-rod-theory",
  "title": "Learning-Based Modeling of Soft Robots via Cosserat Rod Theory",
  "abstract": "Modeling soft robot dynamics is challenging due to their continuum structure and typically nonlinear dynamics. Creating models based on first-order principles is typically time-demanding, and their expressiveness is limited, whereas data-driven models lack interpretability and physical consistency. This work aims to overcome these challenges by introducing a port-Hamiltonian Gaussian Process Regression framework for learning and simulating the dynamics of planar, rod-like soft robots. In detail, the proposed model integrates Cosserat rod theory and Hamiltonian physics with data-driven inference to preserve the system's energy structure while accurately learning the rod dynamics. Numerical simulations show that we can achieve accurate and energy-consistent representations of a rod-like soft robot, showing the potential for a robust and interpretable pathway for modeling complex continuum mechanics.",
  "published": "2026-06-18",
  "updated": "2026-06-18",
  "year": "2026",
  "authors": [
   "Mohammad Ali",
   "Nithin Senthur Kumar",
   "Eric J. Barth",
   "Thomas Beckers"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Numerical simulations show that the proposed port-Hamiltonian Gaussian Process Regression framework can achieve accurate and energy-consistent representations of a rod-like soft robot, showing the potential for a robust and interpretable pathway for modeling complex continuum mechanics.",
  "doi": "10.48550/arXiv.2606.20958",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "M. Ali",
    "id": "145701358",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Nithin S. Kumar",
    "id": "2267728559",
    "h_index": 1,
    "papers": 10
   },
   {
    "name": "Eric J. Barth",
    "id": "2267533901",
    "h_index": 1,
    "papers": 9
   },
   {
    "name": "Thomas Beckers",
    "id": "2238951414",
    "h_index": 3,
    "papers": 11
   }
  ],
  "comment": "8 pages, 6 figures",
  "topics": [
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.20958v1",
  "pdf_url": "https://arxiv.org/pdf/2606.20958v1",
  "html_url": "https://arxiv.org/html/2606.20958v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.20424",
  "slug": "lit-gs-lidar-inertial-thermal-gaussian-splatting-for-illumination-robu",
  "title": "LIT-GS: LiDAR-Inertial-Thermal Gaussian Splatting for Illumination-Robust Mapping",
  "abstract": "Gaussian Splatting has enabled real-time neural rendering, yet existing LiDAR-inertial-visual (LIV) Gaussian mapping pipelines remain fragile under illumination changes and texture-deficient scenes due to their reliance on RGB photometric cues. We present LIT-GS, a LiDAR-inertial-thermal Gaussian Splatting framework that injects LiDAR-derived plane geometry as an explicit constraint in both pose/structure refinement and Gaussian optimization. Specifically, we exploit LIV visual map points as confidence-aware cross-modal anchors to establish reliable thermal-LiDAR associations, and incorporate weighted LiDAR point-to-plane residuals into bundle adjustment to jointly refine camera poses and 3D points under weak thermal supervision. Building on the refined structure, we further introduce a LiDAR-plane-regularized differentiable splatting objective that constrains rendered 3D points to align with locally observed planes, mitigating surface thickening and structural drift in low-contrast thermal imagery. Experiments on proprietary sequences and public datasets demonstrate that LIT-GS consistently improves geometric accuracy and rendering quality over state-of-the-art LIV-based Gaussian Splatting baselines, particularly in challenging lighting conditions.",
  "published": "2026-06-18",
  "updated": "2026-06-18",
  "year": "2026",
  "authors": [
   "Shikuan Shi",
   "Chunran Zheng",
   "Jiaming Xu",
   "Tianyong Ye",
   "Tao Yu",
   "Yukang Cui"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "LIT-GS is presented, a LiDAR-inertial-thermal Gaussian Splatting framework that injects LiDAR-derived plane geometry as an explicit constraint in both pose/structure refinement and Gaussian optimization and consistently improves geometric accuracy and rendering quality over state-of-the-art LIV-based Gaussian Splatting baselines.",
  "doi": "10.48550/arXiv.2606.20424",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Shikuan Shi",
    "id": "2431855700",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Chunran Zheng",
    "id": "2276211455",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Jiaming Xu",
    "id": "2350860237",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Tianyong Ye",
    "id": "2319413727",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Tao Yu",
    "id": "2305631110",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Yukang Cui",
    "id": "2319400412",
    "h_index": 4,
    "papers": 18
   }
  ],
  "comment": "Accepted to IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS 2026)",
  "topics": [
   "spatial-3d",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.20424v1",
  "pdf_url": "https://arxiv.org/pdf/2606.20424v1",
  "html_url": "https://arxiv.org/html/2606.20424v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2606.20322",
  "slug": "towards-3d-karst-underwater-scene-reconstruction-from-rotating-sonar-d",
  "title": "Towards 3D karst underwater scene reconstruction from rotating sonar data",
  "abstract": "Karst aquifers provide critical freshwater resources but pose significant hazards due to their complex and poorly understood subsurface geometry. Mapping these environments is challenging because sonar data from underwater exploration is sparse and noisy, while navigation estimates suffer from drift limiting standard 3D reconstruction methods. We present a pipeline for reconstructing underwater karst conduits from a sonar profiler. We combine a continuous-time SLAM approach to correct trajectory drift with a novel two-stage deep learning method for surface reconstruction, producing an immersive and navigable 3D mesh for hydrogeological analysis.",
  "published": "2026-06-18",
  "updated": "2026-06-18",
  "year": "2026",
  "authors": [
   "Georgios Evangelos Margaritis",
   "Lionel Lapierre",
   "Simon Rohou",
   "Zhi Yan",
   "Andreas N\u00fcchter",
   "Fran\u00e7ois Goulette"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "10.48550/arXiv.2606.20322",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "G. Margaritis",
    "id": "46606848",
    "h_index": 4,
    "papers": 23
   },
   {
    "name": "Lionel Lapierre",
    "id": "2289314777",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Simon Rohou",
    "id": "8451835",
    "h_index": 6,
    "papers": 22
   },
   {
    "name": "Zhichong Yan",
    "id": "2379455919",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Andreas N\u00fcchter",
    "id": "2248820505",
    "h_index": 5,
    "papers": 31
   },
   {
    "name": "Fran\u00e7ois Goulette",
    "id": "1698805",
    "h_index": 22,
    "papers": 76
   }
  ],
  "comment": "1st Workshop on Long-term Deployments in the Wild (LoWi)",
  "topics": [
   "spatial-3d",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.20322v1",
  "pdf_url": "https://arxiv.org/pdf/2606.20322v1",
  "html_url": "https://arxiv.org/html/2606.20322v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.20193",
  "slug": "belt-finger-an-affordable-soft-belt-driven-gripper-for-dexterous-in-ha",
  "title": "Belt-Finger: An Affordable Soft Belt-Driven Gripper for Dexterous In-Hand Manipulation",
  "abstract": "Parallel-jaw grippers are the default manipulator choice in robotics because they are simple, robust, and inexpensive. Their limited in-hand mobility, however, often forces large arm motions and restricts dexterous manipulation in confined workspaces. We present a parallel-gripper upgrade: a double-soft-belt-based finger module that preserves standard opening/closing while adding three in-hand degrees of freedom (DoF): translation, pitch, and roll. The mechanism is deliberately kept simple and engineered for inexpensive manufacturing and straightforward integration, preserving the reliability and precise control of traditional parallel grippers while greatly broadening the range of manipulation capabilities. To demonstrate the utility of the added DoFs, we integrate the gripper in two control pipelines. First, we adapt a model predictive controller for in-hand manipulation of known objects. Second, we introduce a lightweight teleoperation interface that enables simultaneous control of the robot arm and gripper (10 DoFs total) with minimal hardware. Across a suite of challenging manipulation tasks executed via teleoperation, MPC, and trained policies, the proposed gripper consistently improves dexterity and task feasibility compared to a conventional parallel gripper",
  "published": "2026-06-18",
  "updated": "2026-06-18",
  "year": "2026",
  "authors": [
   "Boya Zhang",
   "Andreas Zell",
   "Georg Martius"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work presents a parallel-gripper upgrade: a double-soft-belt-based finger module that preserves standard opening/closing while adding three in-hand degrees of freedom (DoF): translation, pitch, and roll.",
  "doi": "10.48550/arXiv.2606.20193",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Boya Zhang",
    "id": "2348064232",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Andreas Zell",
    "id": "2354557044",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "G. Martius",
    "id": "2244622593",
    "h_index": 9,
    "papers": 28
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "rl-control",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.20193v1",
  "pdf_url": "https://arxiv.org/pdf/2606.20193v1",
  "html_url": "https://arxiv.org/html/2606.20193v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.19874",
  "slug": "mmd-slam-structure-enhanced-multi-meta-gaussian-distribution-guided-vi",
  "title": "MMD-SLAM: Structure-Enhanced Multi-Meta Gaussian Distribution-Guided Visual SLAM",
  "abstract": "3D Gaussian Splatting (3DGS) has significantly boosted novel view synthesis and high-fidelity scene reconstruction, expanding the potential of 3DGS-based Visual Simultaneous Localization and Mapping (SLAM) methods. However, most existing systems fail to fully exploit the underlying structural information, which limits rendering quality and often leads to inconsistent maps. To address these limitations, we propose MMD-SLAM, a structure-enhanced Visual SLAM framework that leverages the Atlanta World (AW) assumption to guide a Multi-Meta Gaussian representation for photorealistic mapping. First, we introduce a point-line fusion strategy for pose optimization, where 3D line segments are incorporated to improve tracking robustness and provide additional constraints for mapping. Second, we design a Multi-Meta Gaussian representation with dominant directions, explicitly encoding structural priors from the AW hypothesis. Finally, we propose a Gaussian evolution strategy that adapts to scene geometry and incorporates structural cues into global optimization. Extensive experiments demonstrate that these innovations enable MMD-SLAM to achieve state-of-the-art performance in both tracking accuracy and mapping quality. e.g., our method achieves a 48.56% reduction in ATE RMSE on ScanNet and a 5.71% improvement in PSNR on Replica, compared with MonoGS.",
  "published": "2026-06-18",
  "updated": "2026-06-18",
  "year": "2026",
  "authors": [
   "Fan Zhu",
   "Ziyu Chen",
   "Peichen Liu",
   "Yifan Zhao",
   "Zhisong Xu",
   "Hui Zhu",
   "Hongxing Zhou",
   "Sixun Liu",
   "Chunmao Jiang"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA 2026",
  "venue_source": "arxiv-comment",
  "citations": 2,
  "influential_citations": 0,
  "tldr": "This work proposes MMD-SLAM, a structure-enhanced Visual SLAM framework that leverages the Atlanta World (AW) assumption to guide a Multi-Meta Gaussian representation for photorealistic mapping, and designs a Multi-Meta Gaussian representation with dominant directions.",
  "doi": "10.48550/arXiv.2606.19874",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Fan Zhu",
    "id": "2302523558",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Ziyu Chen",
    "id": "2278840704",
    "h_index": 4,
    "papers": 21
   },
   {
    "name": "Peichen Liu",
    "id": "2403437699",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Yifan Zhao",
    "id": "2345330117",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Zhisong Xu",
    "id": "2299327365",
    "h_index": 3,
    "papers": 15
   },
   {
    "name": "Hui Zhu",
    "id": "2264712840",
    "h_index": 5,
    "papers": 18
   },
   {
    "name": "Hongxing Zhou",
    "id": "2218610425",
    "h_index": 2,
    "papers": 13
   },
   {
    "name": "Sixun Liu",
    "id": "2155857452",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Chunmao Jiang",
    "id": "2261096846",
    "h_index": 4,
    "papers": 25
   }
  ],
  "comment": "ICRA 2026",
  "topics": [
   "spatial-3d",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.19874v1",
  "pdf_url": "https://arxiv.org/pdf/2606.19874v1",
  "html_url": "https://arxiv.org/html/2606.19874v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.98
 },
 {
  "id": "2606.19586",
  "slug": "one-demo-is-worth-a-thousand-trajectories-action-view-augmentation-for",
  "title": "One Demo is Worth a Thousand Trajectories: Action-View Augmentation for Visuomotor Policies",
  "abstract": "Visuomotor policies for manipulation have demonstrated remarkable potential in modeling complex robotic behaviors, yet minor alterations in the robot's initial configuration and unseen obstacles easily lead to out-of-distribution observations. Without extensive data collection effort, these result in catastrophic execution failures. In this work, we introduce an effective data augmentation framework that generates visually realistic fisheye image sequences and corresponding physically feasible action trajectories from real-world eye-in-hand demonstrations, captured with a portable parallel gripper with a single fisheye camera. We introduce a novel Gaussian Splatting formulation, adapted to wide FoV fisheye cameras, to reconstruct and edit the 3D scene with unseen objects. We utilize trajectory optimization to generate smooth, collision-free, view-rendering-friendly action trajectories and render visual observations from corresponding novel views. Comprehensive experiments in simulation and the real world show that our augmentation framework improves the success rate for various manipulation tasks in both the same scene and the augmented scene with obstacles requiring collision avoidance.",
  "published": "2026-06-17",
  "updated": "2026-06-17",
  "year": "2026",
  "authors": [
   "Chuer Pan",
   "Litian Liang",
   "Dominik Bauer",
   "Eric Cousineau",
   "Benjamin Burchfiel",
   "Siyuan Feng",
   "Shuran Song"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL 2025",
  "venue_source": "arxiv-comment",
  "citations": 8,
  "influential_citations": 2,
  "tldr": "An effective data augmentation framework that generates visually realistic fisheye image sequences and corresponding physically feasible action trajectories from real-world eye-in-hand demonstrations, captured with a portable parallel gripper with a single fisheye camera is introduced.",
  "doi": "10.48550/arXiv.2606.19586",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Chuer Pan",
    "id": "2253801737",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Litian Liang",
    "id": "2274757850",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Dominik Bauer",
    "id": "2297668171",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Eric Cousineau",
    "id": "2090529",
    "h_index": 12,
    "papers": 16
   },
   {
    "name": "Benjamin Burchfiel",
    "id": "2319412766",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Siyuan Feng",
    "id": "2284620540",
    "h_index": 8,
    "papers": 10
   },
   {
    "name": "Shuran Song",
    "id": "2348261633",
    "h_index": 3,
    "papers": 4
   }
  ],
  "comment": "Project website: https://chuerpan.com/1001-demos.github.io/. Published at CoRL 2025",
  "topics": [
   "dexterous-manipulation",
   "spatial-3d",
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.19586v1",
  "pdf_url": "https://arxiv.org/pdf/2606.19586v1",
  "html_url": "https://arxiv.org/html/2606.19586v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.45
 },
 {
  "id": "2606.19340",
  "slug": "zerodex-zero-shot-long-horizon-dexterous-manipulation-via-multi-view-3",
  "title": "ZeroDex: Zero-Shot Long-Horizon Dexterous Manipulation via Multi-View 3D-Grounded VLM Reasoning",
  "abstract": "We present ZeroDex, a zero-shot framework for long-horizon dexterous manipulation that grounds language instructions into executable 3D task plans from calibrated multi-view RGB images. Rather than training an end-to-end policy, our system uses a vision-language model (VLM) to produce reference-frame task grounding and primitive-level 2D keypoints, then lifts them into 3D via multi-view fusion. This lifting combines triangulation of view-wise VLM groundings with reference-view ray voting, which searches along a semantic camera ray for geometrically consistent candidates across neighboring views. The resulting 3D keypoints support both pick-and-place and tool-use: for tool-use, we retrieve an object-centric atomic action corresponding to the inferred skill category and align its stored 6D tool trajectory to the scene; for dexterous execution, we expand the lifted grasp keypoint into a task-conditioned grasp affordance region and generate feasible grasp-motion pairs with an arm-hand motion generator. Real-world experiments show improved 3D grounding accuracy and execution reliability over single-view RGB-D grounding and fine-tuned VLA baselines. We further demonstrate long-horizon manipulation through closed-loop status verification and replan, enabling zero-shot execution on unseen objects and tool-use tasks in novel scenes.",
  "published": "2026-06-17",
  "updated": "2026-06-19",
  "year": "2026",
  "authors": [
   "Jisoo Kim",
   "Sangwon Baik",
   "Taeksoo Kim",
   "Sungjoo Kim",
   "Junyoung Lee",
   "Mingi Choi",
   "Hanbyul Joo"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "ZeroDex is presented, a zero-shot framework for long-horizon dexterous manipulation that grounds language instructions into executable 3D task plans from calibrated multi-view RGB images, enabling zero-shot execution on unseen objects and tool-use tasks in novel scenes.",
  "doi": "10.48550/arXiv.2606.19340",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jisoo Kim",
    "id": "2279809276",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "S. Baik",
    "id": "2394172877",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Taeksoo Kim",
    "id": "2280709385",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Sungjoo Kim",
    "id": "2261424514",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Junyoung Lee",
    "id": "2394238494",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Mingi Choi",
    "id": "2262455307",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Hanbyul Joo",
    "id": "2277246764",
    "h_index": 8,
    "papers": 27
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "dexterous-manipulation",
   "video-generation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.19340v2",
  "pdf_url": "https://arxiv.org/pdf/2606.19340v2",
  "html_url": "https://arxiv.org/html/2606.19340v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.19333",
  "slug": "do-as-i-do-dexterous-manipulation-data-from-everyday-human-videos",
  "title": "Do as I Do: Dexterous Manipulation Data from Everyday Human Videos",
  "abstract": "How can we scalably generate data for robotic manipulation, especially on human-like platforms such as dexterous multi-fingered hands? Learning from human videos has recently emerged as a likely answer to this question. However, difficulties in estimating hand-object interaction and crossing the human-to-robot embodiment gap have hindered the adoption of abundant monocular RGB-only human videos as the primary source of robot manipulation data. In this work, we present DO AS I DO, an algorithm to reconstruct and retarget monocular RGB human videos to multi-fingered dexterous robotic hands. DO AS I DO reconstructs hand-object interactions from various egocentric and exocentric in-the-wild video sources. The algorithm then retargets these hand-object interaction estimates into a sequence of actions executable in the real world, yielding robot-complete manipulation data from disparate human videos. Overall, DO AS I DO outperforms previous state of the art in estimating hand-object interactions and extracting dexterous manipulation trajectories from RGB videos, as we show in experiments on datasets with ground truths and on a dataset of video clips collected online. Our experiments enable us to propose an efficacy playbook for practitioners collecting human data for manipulation.",
  "published": "2026-06-17",
  "updated": "2026-06-17",
  "year": "2026",
  "authors": [
   "Bhawna Paliwal",
   "Haritheja Etukuru",
   "William Liang",
   "Pieter Abbeel",
   "Nur Muhammad Mahi Shafiullah",
   "Jitendra Malik"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 5,
  "influential_citations": 2,
  "tldr": "An algorithm to reconstruct and retarget monocular RGB human videos to multi-fingered dexterous robotic hands, yielding robot-complete manipulation data from disparate human videos and an efficacy playbook for practitioners collecting human data for manipulation is proposed.",
  "doi": "10.48550/arXiv.2606.19333",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Bhawna Paliwal",
    "id": "2127989133",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Haritheja Etukuru",
    "id": "2268398574",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "William Liang",
    "id": "2410899324",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Pieter Abbeel",
    "id": "2352943248",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Nur Muhammad Shafiullah",
    "id": "2253752930",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Jitendra Malik",
    "id": "2242761335",
    "h_index": 15,
    "papers": 32
   }
  ],
  "comment": "Project website: https://do-as-i-do.com/",
  "topics": [
   "dexterous-manipulation",
   "egocentric-data"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.19333v1",
  "pdf_url": "https://arxiv.org/pdf/2606.19333v1",
  "html_url": "https://arxiv.org/html/2606.19333v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.78
 },
 {
  "id": "2606.19161",
  "slug": "ht-bench-benchmarking-and-learning-dexterous-full-hand-tactile-represe",
  "title": "HT-Bench: Benchmarking and Learning Dexterous Full-Hand Tactile Representations with Egocentric Vision",
  "abstract": "Establishing a universal benchmark for tactile representation learning in robotic manipulation remains challenging due to the diversity of tactile sensor designs, data formats, and robot embodiments. Rather than seeking to establish such, we explore a scalable and promising direction for future development: egocentric vision paired with full-hand tactile data. To this end, we introduce \\textbf{HT-Bench}, a large-scale multi-task benchmark for dexterous full-hand tactile sensing, comprising 10M RGB frames and 7.8M tactile frames collected across 226 tasks. HT-Bench evaluates tactile representations from three key perspectives: whether they encode meaningful contact geometry, whether they can align tactile observations with visual information, and whether they generalize to unseen tasks. To assess these capabilities, HT-Bench includes four tasks: fine-grained tactile similarity retrieval, masked tactile inpainting, vision-to-tactile synthesis, and multimodal tactile frame prediction. We further propose \\textbf{HandTouch}, a vector-quantized vision--tactile encoder that learns tactile representations through progressive spatial, cross-modal, and temporal training. Across HT-Bench, HandTouch consistently outperforms representative tactile encoder baselines, improving Recall@5 on fine-grained tactile similarity retrieval from 74.65\\% to 85.23\\%, reducing RMSE on masked tactile inpainting from 0.022 to 0.010, and increasing OOD cIoU on vision-to-tactile synthesis from 0.628 to 0.705. These results demonstrate the effectiveness of HandTouch and suggest that large-scale egocentric full-hand tactile data provides a scalable basis for evaluating and advancing tactile representation learning in dexterous manipulation.",
  "published": "2026-06-17",
  "updated": "2026-08-20",
  "year": "2026",
  "authors": [
   "Yuzhe Huang",
   "Jiaping Wu",
   "Jiaming Jiang",
   "Hezhe Lin",
   "Aikebaier Aierken",
   "Yunlong Wang",
   "Kun Cheng",
   "Wanlin Li",
   "Chenxi Xiao",
   "Ziyuan Jiao",
   "Yuanxin Zhong"
  ],
  "author_count": 11,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 3,
  "influential_citations": 0,
  "tldr": "The proposed HandTouch is a vector-quantized vision--tactile encoder that learns tactile representations through progressive spatial, cross-modal, and temporal training and is suggested to provide a scalable basis for evaluating and advancing tactile representation learning in dexterous manipulation.",
  "doi": "10.48550/arXiv.2606.19161",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yuzhe Huang",
    "id": "2356910381",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Jiaping Wu",
    "id": "2445223101",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Jiaming Jiang",
    "id": "2167229335",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Hezhe Lin",
    "id": "2445244214",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Aikebaier Aierken",
    "id": "2444847230",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yunlong Wang",
    "id": "2445285932",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "K. Cheng",
    "id": "2343780747",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Ziyuan Jiao",
    "id": "2356946141",
    "h_index": 3,
    "papers": 13
   },
   {
    "name": "Yuanxin Zhong",
    "id": "2488390",
    "h_index": 8,
    "papers": 19
   }
  ],
  "comment": "9pages, 4figures",
  "topics": [
   "dexterous-manipulation",
   "egocentric-data",
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.19161v2",
  "pdf_url": "https://arxiv.org/pdf/2606.19161v2",
  "html_url": "https://arxiv.org/html/2606.19161v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.6
 },
 {
  "id": "2606.18704",
  "slug": "selective-unit-cell-actuation-in-lattice-structures-for-distributed-mo",
  "title": "Selective Unit-Cell Actuation in Lattice Structures for Distributed Morphology in Soft Robots",
  "abstract": "Soft lattice structures are increasingly used in robotics to tailor compliance and guide deformation; however, actuation is typically introduced at the device or module level, with actuators inserted into otherwise passive architectures. In this work, we move actuator-lattice co-design to the unit-cell scale. We present an embedded pneumatic unit cell that integrates curved-strut lattice geometry with a bidirectional bellow actuator within a single monolithic element. When tessellated, the lattice functions as a distributed actuation field in which global morphology is governed by spatial actuation patterns rather than uniform pressurization. Experimental characterization of 1x1, 2x2, and 3x3 tessellations demonstrates scalable displacement and force generation with repeatable cyclic performance. Selective actuation of unit cells in a 3x3x3 array produces distinct global deformation modes, including bending and directional grasping, without altering hardware configuration. Additionally, coupling active and passive unit cells enables bending-driven crawling locomotion, demonstrating that heterogeneous tessellations can translate through asymmetric deformation. These results establish unit-cell-level actuation as a strategy for distributed morphing in lattice-based soft robots and provide a foundation for scalable, monolithic robotic architectures.",
  "published": "2026-06-17",
  "updated": "2026-06-17",
  "year": "2026",
  "authors": [
   "Trevor Exley",
   "Altair Coutinho",
   "Lucia Beccai"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "10.48550/arXiv.2606.18704",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Trevor Exley",
    "id": "2140869986",
    "h_index": 4,
    "papers": 19
   },
   {
    "name": "A. Coutinho",
    "id": "2159440488",
    "h_index": 7,
    "papers": 18
   },
   {
    "name": "Lucia Beccai",
    "id": "2248656862",
    "h_index": 6,
    "papers": 24
   }
  ],
  "comment": "Accepted to IROS 2026, 8 pages, 5 figures",
  "topics": [
   "dexterous-manipulation",
   "humanoids",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.18704v1",
  "pdf_url": "https://arxiv.org/pdf/2606.18704v1",
  "html_url": "https://arxiv.org/html/2606.18704v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2606.18097",
  "slug": "wirecraft-a-simulation-benchmark-for-industrial-dlo-manipulation",
  "title": "WireCraft: A Simulation Benchmark for Industrial DLO Manipulation",
  "abstract": "Deformable Linear Objects (DLOs), such as wires and cables, are central to industrial assembly. Unlike rigid objects, whose state is captured by a 6-DoF pose, DLOs have an infinite-dimensional configuration space and deform continuously under contact with grippers, fixtures, and the workspace, making them a demanding benchmark for general dexterous manipulation. Despite their importance, policy development and comparison remain difficult: existing benchmarks are often tied to specific hardware setups, lack modular and customizable task assets, or study generic deformable-object tasks without the fixtures relevant to real-world industrial wire manipulation. Few benchmarks align simulation, real-world data, and shared evaluation protocols. To bridge this gap, we introduce WireCraft, a simulation benchmark for industrial DLO manipulation with configurable difficulty and assets, spanning three task families: connector insertion, clip routing, and channel seating. It supports two complementary DLO physics models, articulated and deformable, and the trajectories come from both simulation and a physical UR5. We benchmark reinforcement learning (RL), imitation learning (IL), and vision-language-action (VLA) policies under shared metrics. Privileged state-based RL solves a representative setting in each task family with over 82\\% success, confirming the tasks are well-posed. For connector insertion, however, the transition from reaching the socket to contact-rich alignment remains a key bottleneck for vision RL, IL, and VLA policies. These results indicate that industrial DLO manipulation, though tractable under privileged state, remains an open challenge for current vision-based learning. The benchmark, data, and tools will be open-sourced upon acceptance.",
  "published": "2026-06-16",
  "updated": "2026-06-16",
  "year": "2026",
  "authors": [
   "Chongyu Zhu",
   "Ramy ElMallah",
   "Hyegang Kim",
   "Zachary Tang",
   "Jiachen Rao",
   "Artem Arutyunov",
   "Seungyeon Ha",
   "Chi-Guhn Lee"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "WireCraft is introduced, a simulation benchmark for industrial DLO manipulation with configurable difficulty and assets, spanning three task families: connector insertion, clip routing, and channel seating, and it supports two complementary DLO physics models, articulated and deformable, and the trajectories come from both simulation and a physical UR5.",
  "doi": "10.48550/arXiv.2606.18097",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Chongyu Zhu",
    "id": "2434861675",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Ramy Elmallah",
    "id": "2170418523",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Hyegang Kim",
    "id": "2445344750",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Zachary Tang",
    "id": "2445229501",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Jiacheng Rao",
    "id": "2376748614",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Artem Arutyunov",
    "id": "2386339899",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Seung-Yeop Ha",
    "id": "2369682860",
    "h_index": 0,
    "papers": 5
   },
   {
    "name": "Chi-Guhn Lee",
    "id": "2325961587",
    "h_index": 4,
    "papers": 18
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "dexterous-manipulation",
   "tactile",
   "sim2real",
   "imitation-diffusion",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.18097v1",
  "pdf_url": "https://arxiv.org/pdf/2606.18097v1",
  "html_url": "https://arxiv.org/html/2606.18097v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.18092",
  "slug": "eagg-embodiment-aligned-grasp-generation-via-geometry-aware-graph-cond",
  "title": "EAGG: Embodiment-Aligned Grasp Generation via Geometry-Aware Graph Conditioning",
  "abstract": "Cross-end-effector grasp generation seeks a unified model that generalizes across objects and across embodiments ranging from parallel grippers to dexterous end effectors. Existing grasp generators are typically designed for a fixed embodiment or encode embodiment identity with a static descriptor, which weakens transfer when topology, actuation coupling, and contact geometry differ substantially. We present EAGG, an embodiment-aligned grasp generator that represents each embodiment with a topology-aware end-effector graph and an embodiment-specific low-dimensional end-effector control space. A frozen end-effector-cognition backbone converts the current articulated state into geometry-aware tokens that act as a reusable morphology prior, and iterative geometry injection refreshes these tokens throughout sampling so that conditioning remains synchronized with the evolving end-effector geometry. On the MultiGripperGrasp benchmark, EAGG reaches 56.17% average success across six training end effectors, remaining within 1.10 percentage points of specialized training while preserving transfer to finetuning and zero-shot end effectors. Iterative geometry injection further reduces the pooled median contact distance from 0.239 cm to 0.189 cm. These results show that cross-end-effector grasp generation is strengthened by aligning embodiment structure inside a shared generator rather than suppressing embodiment differences. Code is available at https://github.com/wanhaoniu/EAGG.",
  "published": "2026-06-16",
  "updated": "2026-06-16",
  "year": "2026",
  "authors": [
   "Wanhao Niu",
   "Qiyan Ke",
   "Yuan Sun",
   "Hao Sun",
   "Jie Xu",
   "Muyuan Ma",
   "Ruiqi Hu",
   "Fuchun Sun"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "EAGG is presented, an embodiment-aligned grasp generator that represents each embodiment with a topology-aware end-effector graph and an embodiment-specific low-dimensional end-effector control space and is strengthened by aligning embodiment structure inside a shared generator rather than suppressing embodiment differences.",
  "doi": "10.48550/arXiv.2606.18092",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "W. Niu",
    "id": "2215407350",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Qiyan Ke",
    "id": "2339479580",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Yuan Sun",
    "id": "2306170013",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Hao Sun",
    "id": "2321328731",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Jie Xu",
    "id": "2273556820",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Muyuan Ma",
    "id": "2290129636",
    "h_index": 2,
    "papers": 11
   },
   {
    "name": "Ruiqi Hu",
    "id": "2336958789",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Fuchun Sun",
    "id": "2344586529",
    "h_index": 2,
    "papers": 7
   }
  ],
  "comment": "16 pages, 8 figures. Code is available at https://github.com/wanhaoniu/EAGG",
  "topics": [
   "dexterous-manipulation",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.18092v1",
  "pdf_url": "https://arxiv.org/pdf/2606.18092v1",
  "html_url": "https://arxiv.org/html/2606.18092v1",
  "code_url": "https://github.com/wanhaoniu/EAGG",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2606.17598",
  "slug": "musevla-an-adaptive-multimodal-sensing-vision-language-action-model-fo",
  "title": "MuseVLA: An Adaptive Multimodal Sensing Vision-Language-Action Model for Robotic Manipulation",
  "abstract": "Humans naturally leverage diverse sensing modalities to interact with the physical world, while most Vision-Language-Action (VLA) models for robotics rely solely on RGB observations. This limits their ability to perceive physical properties that are difficult or impossible to infer from RGB cameras, such as temperature, sound, or radar response. We present MuseVLA, an adaptive multimodal sensing VLA model that integrates novel sensors as on-demand tools for robotic manipulation. Given a task instruction and visual context, MuseVLA first generates a sensor token and target description that select the sensing modality to invoke and what to attend to, analogous to a tool call with arguments. It then converts the selected sensor measurement into a grounded sensor image, a unified intermediate representation that encodes heterogeneous readings for multimodal fusion and action generation. This design decouples sensor-specific processing from the VLA backbone, enabling efficient integration of diverse modalities. To reduce the need for expensive multisensory robot datasets, we further introduce a data synthesis pipeline that augments existing RGB video datasets with grounded sensor images, enabling generalization to unseen sensor-guided tasks. We evaluate MuseVLA on a real-world robot across challenging dexterous hand manipulation tasks that require multimodal sensing inputs, including temperature-guided pick-and-place, audio-driven object search, and radar-assisted hidden object retrieval. MuseVLA achieves 80.6% success rate on average, outperforming RGB-only and multisensory VLA baselines significantly, and exhibits strong zero-shot capabilities on unseen tasks. Code, model and dataset are available at https://github.com/microsoft/MuseVLA.",
  "published": "2026-06-16",
  "updated": "2026-08-12",
  "year": "2026",
  "authors": [
   "Xingyuming Liu",
   "Ruichun Ma",
   "Heyu Guo",
   "Qixiu Li",
   "Qingwen Yang",
   "Lin Luo",
   "Shiqi Jiang",
   "Chenren Xu",
   "Jiaolong Yang",
   "Baining Guo"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "MuseVLA is presented, an adaptive multimodal sensing VLA model that integrates novel sensors as on-demand tools for robotic manipulation that outperforms RGB-only and multisensory VLA baselines significantly, and exhibits strong zero-shot capabilities on unseen tasks.",
  "doi": "10.48550/arXiv.2606.17598",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xingyuming Liu",
    "id": "2261495696",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Rui Ma",
    "id": "2210435658",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Heyuan Guo",
    "id": "1470694215",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Qixiu Li",
    "id": "2282095267",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Qingwen Yang",
    "id": "2445571701",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Lin Luo",
    "id": "2333975449",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Shiqi Jiang",
    "id": "2328014585",
    "h_index": 5,
    "papers": 19
   },
   {
    "name": "Chenren Xu",
    "id": "2248770422",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Jiaolong Yang",
    "id": "2237946707",
    "h_index": 10,
    "papers": 22
   },
   {
    "name": "Baining Guo",
    "id": "2400505652",
    "h_index": 3,
    "papers": 6
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "dexterous-manipulation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.17598v2",
  "pdf_url": "https://arxiv.org/pdf/2606.17598v2",
  "html_url": "https://arxiv.org/html/2606.17598v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2606.17456",
  "slug": "embodiment-shapes-rolling-behavior-in-a-multimodal-infant-model",
  "title": "Embodiment Shapes Rolling Behavior in a Multimodal Infant Model",
  "abstract": "Rolling over is one of the earliest milestones in infant motor development, reflecting the emergence of coordinated, whole-body sensorimotor control. Here, we conduct a computational study of infant rolling using MIMo, a virtual infant embodiment equipped with proprioception and vestibular sensation. MIMo learns supine-to-prone rolls with reinforcement learning. Interestingly, the learned behaviors capture developmental trends and coordination patterns consistent with those reported in real infants, including improved performance and faster execution with age. Our results explain how infant capabilities and constraints can give rise to realistic behaviors in artificial agents, with a particular emphasis on how motor development is shaped by the changing body morphology. This work highlights the role of embodied computational models as a powerful tool for studying sensorimotor development.",
  "published": "2026-06-16",
  "updated": "2026-06-16",
  "year": "2026",
  "authors": [
   "Leon Philipp",
   "Francisco M. L\u00f3pez",
   "Jochen Triesch"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "q-bio.NC"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "A computational study of infant rolling using MIMo, a virtual infant embodiment equipped with proprioception and vestibular sensation that learns supine-to-prone rolls with reinforcement learning, highlighting the role of embodied computational models as a powerful tool for studying sensorimotor development.",
  "doi": "10.48550/arXiv.2606.17456",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Leon Philipp",
    "id": "2393746516",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Francisco M. L\u00f3pez",
    "id": "2276485966",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Jochen Triesch",
    "id": "2241690604",
    "h_index": 3,
    "papers": 9
   }
  ],
  "comment": "7 pages, 7 figures. Accepted at the 2026 IEEE ICDL Conference. Cite as: L. Philipp, F. M. L\u00f3pez, and J. Triesch, \"Embodiment Shapes Rolling Behavior in a Multimodal Infant Model\", in 2026 IEEE International Conference on Development and Learning (ICDL). IEEE, 2026, pp. 1-7",
  "topics": [
   "humanoids",
   "rl-control",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.17456v1",
  "pdf_url": "https://arxiv.org/pdf/2606.17456v1",
  "html_url": "https://arxiv.org/html/2606.17456v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2606.17418",
  "slug": "dexlink-hand-a-compact-affordable-16-dof-linkage-driven-hand-with-huma",
  "title": "DexLink Hand: A Compact, Affordable, 16-DOF Linkage-Driven Hand with Human-Like Dexterity",
  "abstract": "Dexterous robotic hands face a longstanding trade-off among dexterity, compactness, and affordability. Particularly, high-degree-of-freedom designs typically demand complex actuation and transmission, hindering integration into human-scale forms. To address these challenges, this work presents a compact, low-cost linkage-driven anthropomorphic hand that achieves high dexterity, structural integration, and human-hand-like functionality. The hand integrates 20 joints driven by 16 independent actuators, with all actuation, sensing, and transmission components compactly embedded within a human-hand-sized structure. The resulting prototype weighs only 320g at a total cost below USD 400. To meet these objectives, a hybrid mechanical architecture combining planar and spatial linkage mechanisms is proposed, enabling decoupled multidirectional motion, biomimetic joint synergies, and high passive load-bearing capability. The thumb further incorporates biomimetic features supporting human-like reconfiguration and opposition movements. Through the coordinated integration of these mechanisms and structural layout, the prototype achieves a highly integrated design with anthropomorphic dexterity. Experimental evaluations demonstrate that the hand achieves the maximum Kapandji score, reproduces all 33 Feix grasp types, and performs stable grasping and dexterous manipulation across a wide variety of daily objects and tools. These results validate the proposed hand as an affordable, compact, and mechanically efficient platform for dexterous manipulation, teleoperation, and robot learning in human-centered environments.",
  "published": "2026-06-16",
  "updated": "2026-06-16",
  "year": "2026",
  "authors": [
   "Hao Wu",
   "Yanzhe Wang",
   "Yu Feng",
   "Jian Liu",
   "Jihao Li",
   "Jianshu Zhou",
   "Huixu Dong"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "10.48550/arXiv.2606.17418",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hao Wu",
    "id": "2333346231",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Yanzhe Wang",
    "id": "2333310469",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Yu Feng",
    "id": "2150671265",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Jian Liu",
    "id": "2150169967",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Jihao Li",
    "id": "2243294660",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Jianshu Zhou",
    "id": "20476309",
    "h_index": 20,
    "papers": 53
   },
   {
    "name": "Huixu Dong",
    "id": "2290329568",
    "h_index": 3,
    "papers": 12
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "data-teleop",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.17418v1",
  "pdf_url": "https://arxiv.org/pdf/2606.17418v1",
  "html_url": "https://arxiv.org/html/2606.17418v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.17394",
  "slug": "damage-adaptation-in-seconds-for-architected-materials",
  "title": "Damage Adaptation in Seconds for Architected Materials",
  "abstract": "Adaptation to damages and in-situ physical repairs is essential for long-term robot autonomy, yet challenging outside of narrowly defined and well-anticipated bounds. In this work we proprioceptively adapt to catastrophic damage in soft-actuated systems in under one minute. Architected materials are well equipped for adaptation: actuator failure occurs gradually rather than acutely, and damage can be described in a low-dimensional, discrete coordinate space. Surprisingly, latent damage representations plus a simple yet robust ensemble method is sufficient for adapting to unseen damage in real-time. Moreover, we identify conditions under which exponential sample complexity collapses to linear sample complexity for learned representations of architected materials, a concrete advantage over rigid components or continuum soft mechanisms. We demonstrate LEAP, our method for adaptive proprioception, via a tracing task for a 6DoF soft wrist based on Handed Shearing Auxetic (HSA) actuators. Our algorithm is able to adapt to cuts, burns, and actuator repairs, enabling simulation-free real-time adaptation that is critical for realizing the promise of soft robots outside the lab. Videos and more information are available at https://murpheylab.github.io/leap.",
  "published": "2026-06-16",
  "updated": "2026-06-16",
  "year": "2026",
  "authors": [
   "James Avtges",
   "Jake Ketchum",
   "Helena Young",
   "Taekyoung Kim",
   "Ryan Truby",
   "Todd Murphey"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "10.48550/arXiv.2606.17394",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "James Avtges",
    "id": "2183451526",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Jake Ketchum",
    "id": "2249763393",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Helena Young",
    "id": "2355875088",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Taekyoung Kim",
    "id": "2310490478",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Ryan L. Truby",
    "id": "2365119694",
    "h_index": 3,
    "papers": 13
   },
   {
    "name": "Todd Murphey",
    "id": "2392721307",
    "h_index": 1,
    "papers": 6
   }
  ],
  "comment": "Proceedings of Robotics: Science and Systems",
  "topics": [
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.17394v1",
  "pdf_url": "https://arxiv.org/pdf/2606.17394v1",
  "html_url": "https://arxiv.org/html/2606.17394v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.17386",
  "slug": "terratransfer-learning-end-to-end-driving-policies-without-expert-demo",
  "title": "TerraTransfer: Learning End-to-End Driving Policies Without Expert Demonstrations",
  "abstract": "End-to-end autonomous driving has achieved state-of-the-art performance on benchmarks and real-world deployments. Its standard training recipe, however, is expensive across all stages: collecting and labeling millions of driving frames is costly, and closed-loop RL on images is bottlenecked by the per-step cost of photorealistic rendering plus a forward pass through a large vision backbone. Self-play in vectorized simulators changes the economics: millions of rollout steps per second, and a state distribution naturally rich in collisions, near-misses, and recoveries that no driving log contains. Our approach exploits this asymmetry by decoupling learning to drive from learning to see. We pretrain a single policy by self-play, then align its latent space with a pretrained vision backbone, through the action KL divergence and a batch-relational low-rank structural loss. The action target comes from the self-play policy, so alignment never supervises against a logged trajectory: a paired dataset of (image, scene-state) frames suffices, with no need for the curated expert demonstrations that imitation pretraining is built on. On photorealistic 3D Gaussian splatting closed-loop scenarios, the resulting end-to-end policy matches or exceeds prior end-to-end methods.",
  "published": "2026-06-16",
  "updated": "2026-07-15",
  "year": "2026",
  "authors": [
   "Zikang Xiong",
   "Weixin Li",
   "Zhouchonghao Wu",
   "Akshay Rangesh",
   "Saarth Bonde",
   "Grantland Hall",
   "Chen Tang",
   "Yihan Hu",
   "Wei Zhan"
  ],
  "author_count": 9,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "This work pretrain a single policy by self-play, then align its latent space with a pretrained vision backbone, through the action KL divergence and a batch-relational low-rank structural loss, and the resulting end-to-end policy matches or exceeds prior end-to-end methods.",
  "doi": "10.48550/arXiv.2606.17386",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zikang Xiong",
    "id": "2005629679",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Weixin Li",
    "id": "2377235423",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Zhouchonghao Wu",
    "id": "2350302302",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Akshay Rangesh",
    "id": "3394813",
    "h_index": 18,
    "papers": 33
   },
   {
    "name": "Saarth Bonde",
    "id": "2444878089",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Grantland Hall",
    "id": "2444879550",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Chen Tang",
    "id": "1491105028",
    "h_index": 16,
    "papers": 38
   },
   {
    "name": "Yihan Hu",
    "id": "2325677836",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Wei Zhan",
    "id": "2366010456",
    "h_index": 2,
    "papers": 10
   }
  ],
  "comment": "",
  "topics": [
   "sim2real",
   "spatial-3d",
   "navigation",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.17386v2",
  "pdf_url": "https://arxiv.org/pdf/2606.17386v2",
  "html_url": "https://arxiv.org/html/2606.17386v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2606.17385",
  "slug": "egoinfinity-a-web-scale-4d-hand-object-interaction-data-engine-for-any",
  "title": "EgoInfinity: A Web-Scale 4D Hand-Object Interaction Data Engine for Any-View Robot Retargeting and Video-to-Action Robot Learning",
  "abstract": "Internet videos constitute the largest reservoir of embodied human manipulation knowledge, yet converting arbitrary RGB footage into actionable robot training data remains a major bottleneck. Existing lab- or factory-collected datasets are narrow in scale and diversity, limiting open-world robot learning. Instead of proposing a static dataset, we introduce EgoInfinity, a universal 4D hand-object interaction data engine that enables web-scale data generation for robot retargeting and learning. EgoInfinity is a modular engine integrating perception, segmentation, reconstruction, interaction-aware refinement, and retargeting to automate this traditionally unscalable video-to-action problem without human-in-the-loop annotation. Its modular design lets the engine continuously benefit from advances in any incorporated component. With EgoInfinity, in-the-wild human manipulation videos are lifted into agent-agnostic, metric 4D hand-object representations, including hand trajectories, 6-DoF object poses, and contact-relevant states. Rather than naively connecting standalone components, EgoInfinity combines cross-module metric calibration with interaction-aware refinement to improve physical reliability, reducing drift and contact inconsistencies common in pure visual reconstruction. We further propose a novel motion retargeter that compiles the recovered 3D hand motions into executable joint trajectories for diverse robot morphologies, enabling video-to-action retargeting on any robot from arbitrary viewpoints and shot sizes (e.g., the human body is only partially visible). We validate EgoInfinity across perception fidelity, kinematic feasibility, contact consistency, cross-embodiment generalization, and real-robot skill acquisition (e.g., grasping, cutting, wiping, and pouring), demonstrating a scalable bridge from internet videos to executable robot behavior for open-world robot learning.",
  "published": "2026-06-16",
  "updated": "2026-06-19",
  "year": "2026",
  "authors": [
   "Gaotian Wang",
   "Kejia Ren",
   "Andrew Morgan",
   "Yiting Chen",
   "Howard H. Qian",
   "Podshara Chanrungmaneekul",
   "Kaiyu Hang"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 3,
  "influential_citations": 0,
  "tldr": "EgoInfinity is a modular engine integrating perception, segmentation, reconstruction, interaction-aware refinement, and retargeting to automate this traditionally unscalable video-to-action problem without human-in-the-loop annotation, and validate EgoInfinity across perception fidelity, kinematic feasibility, contact consistency, cross-embodiment generalization, and real-robot skill acquisition.",
  "doi": "10.48550/arXiv.2606.17385",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Gaotian Wang",
    "id": "2290064206",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Kejia Ren",
    "id": "147247100",
    "h_index": 6,
    "papers": 21
   },
   {
    "name": "Andrew S. Morgan",
    "id": "2323753707",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Yiting Chen",
    "id": "2350835780",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Howard H. Qian",
    "id": "2290044807",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Podshara Chanrungmaneekul",
    "id": "2204954159",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Kaiyu Hang",
    "id": "3179381",
    "h_index": 24,
    "papers": 67
   }
  ],
  "comment": "24 pages. Project page: https://huggingface.co/spaces/Rice-RobotPI-Lab/EgoInfinity",
  "topics": [
   "dexterous-manipulation",
   "foundation-pretraining",
   "data-teleop",
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.17385v2",
  "pdf_url": "https://arxiv.org/pdf/2606.17385v2",
  "html_url": "https://arxiv.org/html/2606.17385v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.6
 },
 {
  "id": "2606.19383",
  "slug": "3d-scene-graphs-open-challenges-and-future-directions",
  "title": "3D Scene Graphs: Open Challenges and Future Directions",
  "abstract": "3D Scene Graphs (3DSGs) have emerged as a powerful representation for spatial AI by combining geometric grounding with semantic and relational abstractions of the environment. Their expressiveness has made them relevant to a broad range of problems in robotics and computer vision, including manipulation, navigation, task planning, scene understanding, and many others. However, the field remains fragmented: different communities adopt distinct formulations, construction pipelines, and evaluation protocols, making it difficult to compare methods, identify common assumptions, and assess remaining challenges for robust real-world deployment. This survey provides a unified and critical review of 3DSGs, with particular emphasis on open challenges and future directions. We first formalize 3DSGs under a common definition and analyze the principal modeling choices that characterize existing formulations, including node and edge attributes, hierarchical structure, dynamic scene representations, and affordance-aware extensions. We then review how 3DSGs are built from raw sensory observations, discussing the most common terminologies, conventions, and techniques. Finally, we examine downstream applications and evaluation strategies, from intrinsic graph quality to task-level performance. To support the community, we also provide a dedicated website that organizes and extends the surveyed content, accessible at https://3dscenegraphs.com/.",
  "published": "2026-06-15",
  "updated": "2026-06-15",
  "year": "2026",
  "authors": [
   "Dennis Rotondi",
   "Francesco Argenziano",
   "Sebastian Koch",
   "Nathan Hughes",
   "Martin Buechner",
   "Johanna Wald",
   "Lukas Rosenberger Schmid",
   "Daniele Nardi",
   "Abhinav Valada",
   "Liam Paull",
   "Federico Tombari",
   "Luca Carlone",
   "Kai O. Arras"
  ],
  "author_count": 13,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 3,
  "influential_citations": 0,
  "tldr": "This survey provides a unified and critical review of 3DSGs, with particular emphasis on open challenges and future directions, and formalize 3DSGs under a common definition and analyze the principal modeling choices that characterize existing formulations, including node and edge attributes, hierarchical structure, dynamic scene representations, and affordance-aware extensions.",
  "doi": "10.48550/arXiv.2606.19383",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Dennis Rotondi",
    "id": "2349541575",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "F. Argenziano",
    "id": "2201327954",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Sebastian Koch",
    "id": "2247902742",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Nathan Hughes",
    "id": "1816742313",
    "h_index": 11,
    "papers": 20
   },
   {
    "name": "Martin Buechner",
    "id": "2444862906",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Johanna Wald",
    "id": "51149869",
    "h_index": 11,
    "papers": 16
   },
   {
    "name": "Lukas Schmid",
    "id": "2284872045",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Daniele Nardi",
    "id": "2243467680",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "A. Valada",
    "id": "2131132945",
    "h_index": 17,
    "papers": 86
   },
   {
    "name": "Liam Paull",
    "id": "2333356827",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Federico Tombari",
    "id": "2406837275",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Luca Carlone",
    "id": "2239485876",
    "h_index": 11,
    "papers": 37
   },
   {
    "name": "Kai O. Arras",
    "id": "2390641736",
    "h_index": 1,
    "papers": 5
   }
  ],
  "comment": "Invited article for the Annual Review of Control, Robotics, and Autonomous Systems Volume 10",
  "topics": [
   "sim2real",
   "spatial-3d",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.19383v1",
  "pdf_url": "https://arxiv.org/pdf/2606.19383v1",
  "html_url": "https://arxiv.org/html/2606.19383v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.6
 },
 {
  "id": "2606.17256",
  "slug": "contrastive-action-image-pre-training-for-visuomotor-control",
  "title": "Contrastive Action-Image Pre-training for Visuomotor Control",
  "abstract": "Existing vision encoders for robotics face a fundamental bottleneck: robotic datasets lack the scale necessary for large-scale pre-training. Prior work circumvents this data scarcity by turning to internet-scale image and language data or egocentric human video. While these models show promise, neither paradigm learns from paired vision and action data, which downstream visuomotor control policies require. However, robot trajectories, the most direct source of this paired signal, are not available at pre-training scale, motivating us to extract action signals from abundant human video instead. To this end, we introduce CAIP (Contrastive Action-Image Pre-training), a vision encoder that treats human hand poses from large-scale egocentric video as a proxy for end-effector actions. By extracting 3D hand keypoints, a representation that aligns naturally with downstream robot action spaces, CAIP learns a unified action-image representation through a contrastive objective. Leveraging 32,041 hours of egocentric human video and only 88 hours of robotic manipulation data, CAIP outperforms state-of-the-art vision encoders including DINOv2, SigLIP, MVP, and R3M. Evaluated on a challenging real-world dexterous manipulation setup using Dexmate Vega and Sharpa Wave hands, CAIP yields performance gains of more than 30% on tasks involving folding, pouring, and fine-grained manipulation. Our results show that our method of contrastive action-centric pre-training yields a scalable path to achieving robust visual representations better suited for physical interaction.",
  "published": "2026-06-15",
  "updated": "2026-06-15",
  "year": "2026",
  "authors": [
   "Yuvan Sharma",
   "Dantong Niu",
   "Anirudh Pai",
   "Zekai Wang",
   "Zhuoyang Liu",
   "Baifeng Shi",
   "Stefano Saravalle",
   "Boning Shao",
   "Ruijie Zheng",
   "Jing Wang",
   "Konstantinos Kallidromitis",
   "Yusuke Kato",
   "Fabio Galasso",
   "Yuke Zhu",
   "Danfei Xu",
   "Linxi \"Jim\" Fan",
   "Jitendra Malik",
   "Trevor Darrell",
   "Roei Herzig"
  ],
  "author_count": 19,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work introduces CAIP (Contrastive Action-Image Pre-training), a vision encoder that treats human hand poses from large-scale egocentric video as a proxy for end-effector actions and learns a unified action-image representation through a contrastive objective.",
  "doi": "10.48550/arXiv.2606.17256",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yuvan Sharma",
    "id": "2307007378",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Dantong Niu",
    "id": "2268757542",
    "h_index": 8,
    "papers": 17
   },
   {
    "name": "Anirudh Pai",
    "id": "2385785302",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Zekai Wang",
    "id": "2326300753",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Zhuoyang Liu",
    "id": "2349957939",
    "h_index": 8,
    "papers": 18
   },
   {
    "name": "Baifeng Shi",
    "id": "1596823732",
    "h_index": 17,
    "papers": 24
   },
   {
    "name": "Stefano Saravalle",
    "id": "2366073966",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Boning Shao",
    "id": "2322198049",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Ruijie Zheng",
    "id": "2345931905",
    "h_index": 9,
    "papers": 15
   },
   {
    "name": "Jing Wang",
    "id": "2350827994",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Konstantinos Kallidromitis",
    "id": "2134886789",
    "h_index": 11,
    "papers": 17
   },
   {
    "name": "Yusuke Kato",
    "id": "2295675577",
    "h_index": 7,
    "papers": 8
   },
   {
    "name": "Fabio Galasso",
    "id": "2294568447",
    "h_index": 5,
    "papers": 20
   },
   {
    "name": "Yuke Zhu",
    "id": "2258068214",
    "h_index": 13,
    "papers": 21
   },
   {
    "name": "Danfei Xu",
    "id": "2264393671",
    "h_index": 4,
    "papers": 15
   },
   {
    "name": "LinxiJimFan",
    "id": "2350861618",
    "h_index": 10,
    "papers": 15
   },
   {
    "name": "Jitendra Malik",
    "id": "2242761335",
    "h_index": 15,
    "papers": 32
   },
   {
    "name": "Trevor Darrell",
    "id": "2355862630",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Roei Herzig",
    "id": "46796686",
    "h_index": 22,
    "papers": 56
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "egocentric-data",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.17256v1",
  "pdf_url": "https://arxiv.org/pdf/2606.17256v1",
  "html_url": "https://arxiv.org/html/2606.17256v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.17200",
  "slug": "ace-ego-0-unifying-egocentric-human-and-robotic-data-for-vla-pretraini",
  "title": "ACE-Ego-0: Unifying Egocentric Human and Robotic Data for VLA Pretraining",
  "abstract": "Vision-Language-Action (VLA) models benefit from large-scale and diverse embodied data, yet scaling robot trajectory collection is costly and labor-intensive. Recent advances show that large-scale egocentric human videos provide complementary real-world supervision in pretraining. However, joint training on human and robot data remains challenging due to divergences in action spaces, embodiment structures, temporal dynamics, and supervision quality. We introduce ACE-EGO-0, a unified VLA pretraining framework jointly leveraging heterogeneous data sources. To extract large-scale pretraining supervision from egocentric human videos, we build a scalable egocentric video-to-action pipeline that converts raw human videos into robot-format pseudo-action trajectories. To make these labels comparable with robot demonstrations, ACE-EGO-0 uses a unified action representation based on camera-space actions, morphology conditioning, and time-aligned action chunking. To robustly leverage noisy pseudo-action supervision from egocentric human videos, we formulate a reliability-aware training objective with a human auxiliary loss that concentrates supervision on reliable signals. We instantiate ACE-EGO-0 on 4.53K hours of robot and simulation data, together with 1.48K hours of pseudo-action-labeled egocentric human data. Experiments show that incorporating large-scale human supervision under reliability-aware weighting consistently improves both unified joint pretraining and supervised fine-tuning. ACE-EGO-0 achieves state-of-the-art performance on RoboCasa GR1 TableTop and RoboTwin 2.0, while demonstrating strong transfer to real-world bimanual manipulation.",
  "published": "2026-06-15",
  "updated": "2026-06-15",
  "year": "2026",
  "authors": [
   "Hao Li",
   "Ganlong Zhao",
   "Yufei Liu",
   "Haotian Hou",
   "Guoquan Ye",
   "Tongyan Fang",
   "Chunxiao Liu",
   "Siyuan Huang",
   "Jianbo Liu",
   "Xiaogang Wang",
   "Hongsheng Li"
  ],
  "author_count": 11,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "ACE-EGO-0, a unified VLA pretraining framework jointly leveraging heterogeneous data sources, achieves state-of-the-art performance on RoboCasa GR1 TableTop and RoboTwin 2.0, while demonstrating strong transfer to real-world bimanual manipulation.",
  "doi": "10.48550/arXiv.2606.17200",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hao Li",
    "id": "2380082230",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Ganlong Zhao",
    "id": "1946923537",
    "h_index": 7,
    "papers": 17
   },
   {
    "name": "Yufei Liu",
    "id": "2445469285",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Haotian Hou",
    "id": "2359255417",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Guo Ye",
    "id": "2054255862",
    "h_index": 5,
    "papers": 15
   },
   {
    "name": "Tongyan Fang",
    "id": "2442801577",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Chunxiao Liu",
    "id": "2107926923",
    "h_index": 9,
    "papers": 17
   },
   {
    "name": "Siyuan Huang",
    "id": "2359208891",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Jianbo Liu",
    "id": "2124809722",
    "h_index": 12,
    "papers": 20
   },
   {
    "name": "Xiaogang Wang",
    "id": "2303617445",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Hongsheng Li",
    "id": "2282094661",
    "h_index": 2,
    "papers": 6
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "egocentric-data",
   "imitation-diffusion",
   "foundation-pretraining",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.17200v1",
  "pdf_url": "https://arxiv.org/pdf/2606.17200v1",
  "html_url": "https://arxiv.org/html/2606.17200v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2606.16436",
  "slug": "v2p-manip-learning-dexterous-manipulation-from-monocular-human-videos",
  "title": "V2P-Manip: Learning Dexterous Manipulation from Monocular Human Videos",
  "abstract": "Achieving autonomous robotic dexterous manipulation requires precise, human-like action sequences at scale. As a scalable supplement to costly teleoperation data, extracting trajectories with both visual fidelity and physical plausibility from monocular videos represents a promising frontier in embodied AI. To this end, we introduce V2P-Manip, an efficient framework designed to learn dexterous manipulation policies directly from human demonstration videos. We establish an efficient, integrated pipeline encompassing 3D asset acquisition, trajectory estimation, and dexterous policy learning. To bridge the gap between visual perception and physical constraints, we introduce a two-stage refinement process to enforce spatial alignment and physical consistency. Evaluations on the TACO and OakInk benchmarks demonstrate that our approach significantly outperforms previous methods in pose accuracy, adaptability to unstructured environments, and training efficiency. Ultimately, experimental results confirm an average success rate of over 75% across multiple synthetic manipulation tasks and validate the adaptability of the extracted manipulation priors across diverse dexterous hand embodiments.",
  "published": "2026-06-15",
  "updated": "2026-06-15",
  "year": "2026",
  "authors": [
   "Kaihan Chen",
   "Yanming Shao",
   "Haifeng Ji",
   "Xiaokang Yang",
   "Yao Mu"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "V2P-Manip is introduced, an efficient framework designed to learn dexterous manipulation policies directly from human demonstration videos, and an efficient, integrated pipeline encompassing 3D asset acquisition, trajectory estimation, and dexterous policy learning is established.",
  "doi": "10.48550/arXiv.2606.16436",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Kai Chen",
    "id": "2384713413",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Yanming Shao",
    "id": "2329058615",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Haifeng Ji",
    "id": "2280295298",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Xiaokang Yang",
    "id": "2362515842",
    "h_index": 6,
    "papers": 21
   },
   {
    "name": "Yao Mu",
    "id": "2348161293",
    "h_index": 2,
    "papers": 18
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "egocentric-data",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.16436v1",
  "pdf_url": "https://arxiv.org/pdf/2606.16436v1",
  "html_url": "https://arxiv.org/html/2606.16436v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.16370",
  "slug": "art-glove-articulated-tactile-glove-for-contact-grounded-dexterous-int",
  "title": "ART-Glove: Articulated Tactile Glove for Contact-Grounded Dexterous Interaction Capture",
  "abstract": "We present ART-Glove, an articulated tactile glove designed to capture contact-grounded dexterous demonstrations while preserving human dexterity. ART-Glove makes hand-side contact geometry explicit with 16 rigid functional surfaces covering the fingers, thumb, and palm. Twenty-two anatomically aligned joints connect these surfaces and allow them to follow human hand motion during dexterous manipulation. Encoder-based sensing tracks surface motion, while dense piezoresistive tactile sensing records contact over the same surfaces. The complete system captures synchronized 22-DoF joint measurements and 2048-taxel tactile measurements at 120 Hz. We evaluate ART-Glove across experiments on motion freedom, joint sensing, tactile sensing, and contact-rich interaction capture, demonstrating its ability to preserve human dexterity while recording contact-grounded information that can support downstream dexterous robot learning.",
  "published": "2026-06-15",
  "updated": "2026-06-15",
  "year": "2026",
  "authors": [
   "Changyi Lin",
   "Ding Zhao"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "ART-Glove is evaluated across experiments on motion freedom, joint sensing, tactile sensing, and contact-rich interaction capture, demonstrating its ability to preserve human dexterity while recording contact-grounded information that can support downstream dexterous robot learning.",
  "doi": "10.48550/arXiv.2606.16370",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Changyi Lin",
    "id": "2293625960",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Ding Zhao",
    "id": "47783130",
    "h_index": 30,
    "papers": 81
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.16370v1",
  "pdf_url": "https://arxiv.org/pdf/2606.16370v1",
  "html_url": "https://arxiv.org/html/2606.16370v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.16272",
  "slug": "toporetarget-interaction-preserving-retargeting-for-dexterous-manipula",
  "title": "TopoRetarget: Interaction-Preserving Retargeting for Dexterous Manipulation",
  "abstract": "Human hand-object demonstrations provide dense reference motions for training dexterous manipulation reinforcement learning (RL) policies through reference tracking. However, to use such demonstrations for RL policy learning, retargeting must preserve hand pose and task-relevant hand-object contact structure. Otherwise, contact and feasibility artifacts can degrade downstream RL policy performance. We introduce TopoRetarget, an interaction-preserving retargeting framework that uses a single set of parameters across diverse retargeting conditions while maintaining task-relevant hand-object interaction and adapting human demonstrations to dexterous robot hands. The method constructs a sparse interaction graph over hand and object keypoints and optimizes distance-weighted Laplacian deformation with directional consistency, kinematic constraints, and penetration handling. Evaluations show that the generated references improve both interaction fidelity and policy learning: TopoRetarget achieves the best contact precision and alignment over all baselines on the ContactPose Dataset, improves Pen-Spin training success by 40.6 percentage points over the existing baseline methods, and enables zero-shot transfer to Wuji Hand hardware on cube reorientation and pen spinning.",
  "published": "2026-06-15",
  "updated": "2026-06-22",
  "year": "2026",
  "authors": [
   "Jielin Wu",
   "Shenzhe Yao",
   "Guanqi He",
   "Xiaohan Liu",
   "Zhaoqing Zeng",
   "Xiangrui Jiang",
   "Han Yang",
   "Wentao Zhang",
   "Hang Zhao"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "TopoRetarget is introduced, an interaction-preserving retargeting framework that uses a single set of parameters across diverse retargeting conditions while maintaining task-relevant hand-object interaction and adapting human demonstrations to dexterous robot hands.",
  "doi": "10.48550/arXiv.2606.16272",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "J. Wu",
    "id": "2327660118",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Shenzhe Yao",
    "id": "2307047859",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Guanqi He",
    "id": "2279862620",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "Xiaohang Liu",
    "id": "2337428482",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Zhaoqing Zeng",
    "id": "2357232182",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Xiangrui Jiang",
    "id": "2375085439",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Han Yang",
    "id": "2109714402",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Wentao Zhang",
    "id": "2385502332",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Hang Zhao",
    "id": "2239158612",
    "h_index": 5,
    "papers": 9
   }
  ],
  "comment": "Project page: https://toporetarget2026.github.io/TopoRetarget/",
  "topics": [
   "dexterous-manipulation",
   "egocentric-data",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.16272v2",
  "pdf_url": "https://arxiv.org/pdf/2606.16272v2",
  "html_url": "https://arxiv.org/html/2606.16272v2",
  "code_url": "https://toporetarget2026.github.io/TopoRetarget/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.3
 },
 {
  "id": "2606.16232",
  "slug": "polymerge-compressing-3d-gaussian-splats-with-polytope-coverings-for-p",
  "title": "PolyMerge: Compressing 3D Gaussian Splats with Polytope Coverings for Provably Safe Resource-Constrained Navigation",
  "abstract": "Obstacle avoidance is essential for safe navigation and motion planning. Recent radiance field reconstruction methods enable object detection and modeling with high fidelity, but remain too memory- and compute-intensive for on-board perception-based path planning. To address these limitations, we propose PolyMerge to convert a large, photorealistic 3D Gaussian Splatting (3DGS) model of a scene into a lightweight representation of convex polytopes whose union provably over-approximates all obstacles in the original 3DGS model. PolyMerge tunes the polytope count to trade off conservativeness and compute cost, and integrates with control barrier functions (CBFs) to plan collision-free paths. We showcase PolyMerge in simulation and hardware experiments on a Crazyflie drone, which uses PolyMerge to compute and follow safe trajectories in real time under severe onboard compute constraints, outperforming baselines in speed while guaranteeing safety. For our code and videos, visit https://athlon76.github.io/PolyMerge-website/.",
  "published": "2026-06-15",
  "updated": "2026-06-15",
  "year": "2026",
  "authors": [
   "Jihoon Hong",
   "Chih-Yuan Chiu",
   "Sara Fridovich-Keil",
   "Glen Chou"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "RA-L",
  "venue_source": "semantic-scholar",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "This work proposes PolyMerge to convert a large, photorealistic 3D Gaussian Splatting model of a scene into a lightweight representation of convex polytopes whose union provably over-approximates all obstacles in the original 3DGS model.",
  "doi": "10.1109/LRA.2026.3692083",
  "oa_pdf": "https://arxiv.org/pdf/2606.16232",
  "s2_authors": [
   {
    "name": "Jihoon Hong",
    "id": "2338051408",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Chih-Yuan Chiu",
    "id": "2378154437",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Sara Fridovich-Keil",
    "id": "1405260934",
    "h_index": 10,
    "papers": 32
   },
   {
    "name": "Glen Chou",
    "id": "2377555998",
    "h_index": 2,
    "papers": 12
   }
  ],
  "comment": "",
  "topics": [
   "rl-control",
   "spatial-3d",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.16232v1",
  "pdf_url": "https://arxiv.org/pdf/2606.16232v1",
  "html_url": "https://arxiv.org/html/2606.16232v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.8
 },
 {
  "id": "2606.15645",
  "slug": "to-sofit-topology-optimization-of-hydraulic-soft-fish-tail-design-for",
  "title": "TO-SoFiT: Topology Optimization of Hydraulic Soft Fish Tail Design for programmable undulating locomotion",
  "abstract": "Soft robots leverage compliant materials to generate motion through controlled elastic deformation, making them ideal for delicate tasks such as underwater exploration and biomimetic marine systems. Although hydraulic/pneumatic actuation remains pivotal for such systems, the lack of systematic design frameworks has hindered the development of robots capable of complex 3D motion, such as fish-like swimming. This work introduces a topology optimization method to automate the design of a hydraulic soft fish tail, explicitly addressing the design-dependent coupling between fluidic actuation and structural deformation. We use a Darcy law-based model augmented with a drainage term to simulate spatially varying hydraulic pressure loads, translating these into consistent nodal forces via finite element analysis. The employed robust multi-criteria optimization formulation balances deformation efficiency, fluid-structure interaction, geometric manufacturability, and required stiffness for optimizing a bioinspired soft fish tail for 3D swimming kinematics. The optimized tail topology is incorporated into a pneumatic network actuator and computationally validated under various hydraulic loads, achieving tunable undulatory amplitudes and multiaxis bending for depth adjustment. The optimized 2D tail outperforms its rectangular counterpart. By cascading optimized tail segments, we demonstrate programmable swimming patterns in soft robotic fish tails at different hydraulic loads. This work advances the systematic codesign of hydraulic actuators and soft structures, offering a pathway to automate underwater robots with optimized design and vertebrate-like agility in confined aquatic environments. Our implementations and simulations are publicly available at 'https://github.com/PrabhatIn/TO-SoFiT'.",
  "published": "2026-06-14",
  "updated": "2026-06-14",
  "year": "2026",
  "authors": [
   "A Padmaprabhan",
   "Amal Shaji",
   "Prabhat Kumar"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "Advanced Robotics",
  "venue_source": "semantic-scholar",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "10.1145/3787370.3787419",
  "oa_pdf": "https://doi.org/10.1145/3787370.3787419",
  "s2_authors": [
   {
    "name": "A. Padmaprabhan",
    "id": "2343633065",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Amal Shaji",
    "id": "2079511871",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Prabhat Kumar",
    "id": "2258988568",
    "h_index": 2,
    "papers": 7
   }
  ],
  "comment": "Accepted for publication at the Advances in Robotics (AIR), 2025, IIT Jodhpur",
  "topics": [
   "humanoids",
   "navigation",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.15645v1",
  "pdf_url": "https://arxiv.org/pdf/2606.15645v1",
  "html_url": "https://arxiv.org/html/2606.15645v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2606.15434",
  "slug": "a-bilateral-teleoperation-framework-for-dexterous-manipulation",
  "title": "A Bilateral Teleoperation Framework for Dexterous Manipulation",
  "abstract": "Dexterous teleoperation requires precise arm-hand coordination, low-latency feedback, and robust interaction in real-world contact-rich environments. This paper presents a modular bilateral teleoperation framework that integrates operator-side input interfaces with a robot-side dexterous hand and compliant robotic arm in a unified control architecture. The system supports position-based hand retargeting, differential arm control, multi-scale haptic feedback, and shared control for stable manipulation. We validate the framework through a real-world dexterous manipulation task, highlighting coordinated arm-hand control and contact-aware interaction. Beyond feasibility, we identify key design insights related to cross-embodiment mismatch, haptic feedback granularity, and shared control. The proposed platform provides a practical teleoperation system and a foundation for collecting high-quality demonstrations for future learning-from-demonstration research.",
  "published": "2026-06-13",
  "updated": "2026-06-13",
  "year": "2026",
  "authors": [
   "Stefano Dalla Gasperina",
   "Dong Ho Kang",
   "Haiyun Zhang",
   "Aldo Galvan",
   "Job D. Ramirez",
   "Aaron Kim",
   "Mark Helwig",
   "Kazuto Yokoyama",
   "Takahisa Ueno",
   "Tetsuya Narita",
   "Ann Majewicz-Fey",
   "Ashish D. Deshpande",
   "Luis Sentis"
  ],
  "author_count": 13,
  "categories": [
   "cs.RO",
   "cs.HC",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper presents a modular bilateral teleoperation framework that integrates operator-side input interfaces with a robot-side dexterous hand and compliant robotic arm in a unified control architecture and identifies key design insights related to cross-embodiment mismatch, haptic feedback granularity, and shared control.",
  "doi": "10.48550/arXiv.2606.15434",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "S. D. Gasperina",
    "id": "117020413",
    "h_index": 8,
    "papers": 21
   },
   {
    "name": "Dong Ho Kang",
    "id": "2267864164",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Haiyun Zhang",
    "id": "2326549926",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "A. Galvan",
    "id": "2124418645",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Job D. Ramirez",
    "id": "2233735707",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "A. Kim",
    "id": "2439655609",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "M. Helwig",
    "id": "123513690",
    "h_index": 0,
    "papers": 12
   },
   {
    "name": "Kazuto Yokoyama",
    "id": "2361053858",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Takahisa Ueno",
    "id": "2442805583",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Tetsuya Narita",
    "id": "2361053366",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "A. M. Fey",
    "id": "40958969",
    "h_index": 10,
    "papers": 70
   },
   {
    "name": "Ashish D. Deshpande",
    "id": "2326369598",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Luis Sentis",
    "id": "2237810419",
    "h_index": 6,
    "papers": 18
   }
  ],
  "comment": "4 pages, 7 figures, 1 appendix,",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "foundation-pretraining",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.15434v1",
  "pdf_url": "https://arxiv.org/pdf/2606.15434v1",
  "html_url": "https://arxiv.org/html/2606.15434v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.14606",
  "slug": "interaction-dynamics-for-dexterous-manipulation",
  "title": "Interaction Dynamics for Dexterous Manipulation",
  "abstract": "Dexterous manipulation is fundamentally a problem of interaction dynamics: the hand must track precise finger trajectories, regulate the contact force exchanged with grasped objects, respect actuation and safety limits, and remain predictable when contact persists -- objectives in tension for any fixed-gain controller. A sustained contact torque $\u03c4_{\\text{ext}}$ through a joint stiffness $K_d$ produces the structural bias $e_\\infty=\u03c4_{\\text{ext}}/K_d$, so stiffening for accuracy sacrifices contact safety while softening yields by design. We make these interaction dynamics explicit and actuator-agnostic through a constant-$A_d$ double-integrator backbone, instantiating the offset-free architecture established for physical human-robot interaction (pHRI) and preserving its modeling assumptions on the reduced residual dynamics. An algebraic feedforward reduces the tendon transmission -- hydraulic, cable, pneumatic, twisted-string, or series-elastic -- to a constant-coefficient double integrator, so the QP cost inverse is precomputed offline and a 10-step receding-horizon QP runs at 500\\,Hz under contact-force (ISO/TS 15066), actuation, and jerk constraints. An encoder-only augmented-Kalman disturbance state drives steady-state error to zero under constant contact loads in the nominal detectable case. In simulation, a hydraulically actuated finger -- the worked example, adding pressure and cavitation constraints -- attains 0.6\\,mrad RMS, 0.1\\,mrad steady-state, and 7.3\\,mrad peak deflection under 1.5\\,Nm contact: 153$\\times$, 1500$\\times$, and 21$\\times$ better than classical impedance. The realized first-move stiffness (18$\\to$323\\,Nm/rad with update rate) is independently verified, and the architecture scales to a 16-DOF LEAP Hand MuJoCo model, recovering from 2.5\\,N grasp disturbances within 0.7\\,s.",
  "published": "2026-06-12",
  "updated": "2026-08-02",
  "year": "2026",
  "authors": [
   "Yongyan Cao"
  ],
  "author_count": 1,
  "categories": [
   "cs.RO",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work instantiating the offset-free architecture established for physical human-robot interaction (pHRI) and preserving its modeling assumptions on the reduced residual dynamics through a constant-$A_d$ double-integrator backbone.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yongyan Cao",
    "id": "9310770",
    "h_index": 5,
    "papers": 19
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "hardware-codesign",
   "hri",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.14606v3",
  "pdf_url": "https://arxiv.org/pdf/2606.14606v3",
  "html_url": "https://arxiv.org/html/2606.14606v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.14561",
  "slug": "orca-a-platform-for-open-source-dexterity-research",
  "title": "ORCA: A Platform for Open-Source Dexterity Research",
  "abstract": "Robotics manipulation research increasingly focuses on two-finger parallel grippers for their effectiveness, affordability, and ease of teleoperation. Grippers are nonetheless limited by their form factor, often requiring bimanual setups even for simple reorientation tasks. Anthropomorphic hands are a more natural platform for dexterous robot learning -- closer to the human hand, and capable of learning from human video -- yet they remain hard to use in learning research: even where open and accessible hand hardware exists, the software for control, simulation, teleoperation, and retargeting is scattered in one-off code bases, and largely disconnected from the robot-learning ecosystem. In this work, we introduce the \\orca~learning stack, an open-source research stack for dexterity as a first-class robot learning domain. Our \\orca~stack unifies low-level control, simulation, teleoperation from a range of consumer platforms, and hand retargeting, behind a single interface, and integrates natively with popular robot-learning frameworks such as \\lerobot, so dexterous hand researchers can leverage the same data, training, and evaluation pipelines used for non-dexterous robot learning. We demonstrate a complete end-to-end workflow, collecting expert demonstrations of an in-hand reorientation task by teleoperation with a consumer-grade VR headset, training an autonomous policy with \\lerobot, and evaluating the learned policy in a fully reproducible and observable setup. We open-source the entire stack as a shared, reproducible foundation for dexterous-manipulation research.",
  "published": "2026-06-12",
  "updated": "2026-06-12",
  "year": "2026",
  "authors": [
   "Francesco Capuano",
   "Maximilian Eberlein",
   "Fabrice Bourquin",
   "Clemens Claudio Christoph"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "The \\orca~learning stack is introduced, an open-source research stack for dexterity as a first-class robot learning domain that unifies low-level control, simulation, teleoperation from a range of consumer platforms, and hand retargeting, behind a single interface and integrates natively with popular robot-learning frameworks such as \\lerobot.",
  "doi": "10.48550/arXiv.2606.14561",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Francesco Capuano",
    "id": "2365036806",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Maximilian Eberlein",
    "id": "2295513835",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Fabrice Bourquin",
    "id": "2295514463",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Clemens C. Christoph",
    "id": "2295514671",
    "h_index": 4,
    "papers": 5
   }
  ],
  "comment": "15 pages",
  "topics": [
   "dexterous-manipulation",
   "egocentric-data",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.14561v1",
  "pdf_url": "https://arxiv.org/pdf/2606.14561v1",
  "html_url": "https://arxiv.org/html/2606.14561v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.14433",
  "slug": "kine2go-kinematic-dataset-for-the-unitree-go2-robot-with-diverse-gaits",
  "title": "Kine2Go: Kinematic dataset for the Unitree Go2 robot with diverse gaits and motions",
  "abstract": "The recent popularity of robotics, combined with the steadily decreasing cost of robotic hardware, has lowered the entry barrier to robotics research and enabled rapid advancements in the field. One of the primary examples is the Unitree Go2 quadruped robot, which is often used by researchers in the areas of locomotion, navigation, control, and others. Many researchers use the Go2 robot in combination with techniques like imitation learning, reinforcement learning, and behavioral cloning to allow machine learning systems to take full control of the robot. At the same time, many of those techniques require demonstration data consisting of the robot's kinematics information and actions applied to the motors. Obtaining such data is difficult, requires building complex pipelines, and can take significant time. To aid in those kinds of efforts, we present Kine2Go - a dataset with 800 diverse gait kinematics trajectory motion data for the Unitree Go2 robot, derived from 40 distinct policies. Our pipeline accepts data from various quadruped morphologies and translates them to a Go2-compatible format. Then we use Reinforcement Learning to train policies following a given motion, and finally we gather data from those policies, which grants robust, perturbed kinematic data with corresponding motor-level actions.",
  "published": "2026-06-12",
  "updated": "2026-06-12",
  "year": "2026",
  "authors": [
   "W\u0142adys\u0142aw Pa\u0142ucki",
   "Pawe\u0142 Siwak",
   "Krzysztof Ciebiera",
   "Marek Cygan"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Kine2Go presents Kine2Go - a dataset with 800 diverse gait kinematics trajectory motion data for the Unitree Go2 robot, derived from 40 distinct policies, derived from 40 distinct policies.",
  "doi": "10.48550/arXiv.2606.14433",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Wladyslaw Palucki",
    "id": "2360308695",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "P. Siwak",
    "id": "88678672",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Krzysztof Ciebiera",
    "id": "2463169",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Marek Cygan",
    "id": "2264354287",
    "h_index": 5,
    "papers": 12
   }
  ],
  "comment": "9 pages, 6 figures",
  "topics": [
   "humanoids",
   "imitation-diffusion",
   "rl-control",
   "navigation"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2606.14433v1",
  "pdf_url": "https://arxiv.org/pdf/2606.14433v1",
  "html_url": "https://arxiv.org/html/2606.14433v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2606.13990",
  "slug": "splatlessdf-continuous-distance-field-mapping-with-non-splatting-gauss",
  "title": "SplatlessDF: Continuous Distance Field Mapping with Non-Splatting Gaussians",
  "abstract": "Recent Gaussian splatting (GS) methods have shown that scenes can be represented efficiently with optimisable Gaussians for high-quality reconstruction and rendering. In this paper, building on this principle, we introduce SplatlessDF, a continuous distance field (DF) mapping framework that uses anisotropic Gaussian elements from a spatial rather than photometric perspective. SplatlessDF directly parameterises the Gaussians and optimises to recover a differentiable DF, enabling distances and gradients to be queried in the spatial domain for downstream robotic tasks such as navigation. Furthermore, SplatlessDF can be coupled with 2D Gaussian splatting (2DGS), providing a unified framework based solely on Gaussian primitives that can learn continuous DF and surface models and supports photometric rendering. We consider two settings: a standalone DF-only formulation and a joint DF-rendering formulation coupled with 2DGS. Experiments show that the standalone formulation provides efficient and accurate distance and gradient queries, while the joint formulation improves rendering geometry and simultaneously models a continuous DF. These results highlight the potential of GS-style representations not only for surface modelling and rendering but also for mapping representations suited to robotic navigation.",
  "published": "2026-06-12",
  "updated": "2026-06-12",
  "year": "2026",
  "authors": [
   "Monisha Mushtary Uttsha",
   "Lan Wu",
   "Teresa Vidal-Calleja"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "SplatlessDF is introduced, a continuous distance field (DF) mapping framework that uses anisotropic Gaussian elements from a spatial rather than photometric perspective, enabling distances and gradients to be queried in the spatial domain for downstream robotic tasks such as navigation.",
  "doi": "10.48550/arXiv.2606.13990",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Monisha Mushtary Uttsha",
    "id": "2122188045",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Lan Wu",
    "id": "2112297577",
    "h_index": 7,
    "papers": 19
   },
   {
    "name": "Teresa Vidal-Calleja",
    "id": "1401939307",
    "h_index": 26,
    "papers": 132
   }
  ],
  "comment": "",
  "topics": [
   "spatial-3d",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.13990v1",
  "pdf_url": "https://arxiv.org/pdf/2606.13990v1",
  "html_url": "https://arxiv.org/html/2606.13990v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.13677",
  "slug": "mana-dexterous-manipulation-of-articulated-tools",
  "title": "Mana: Dexterous Manipulation of Articulated Tools",
  "abstract": "Articulated tool manipulation remains a major challenge in dexterous robotics due to the need to coordinate internal degrees of freedom and contact-rich interactions. While prior work has largely focused on rigid objects, articulated tool use remains underexplored because of its physical complexity and the difficulty of learning functional grasping and manipulation policies. We present Mana (Manipulation Animator), a general sim-to-real framework that reinterprets dexterous manipulation as an animation problem. Inspired by computer animation, Mana employs a coarse-to-fine pipeline that transforms procedurally-generated grasp keyframes into manipulation trajectories through motion planning and reinforcement learning. The data generation process is largely automatic, requiring only a few mouse clicks to specify functional affordances (<1 minute per tool). Across four articulated tools spanning different scales and joint types, Mana achieves zero-shot sim-to-real transfer for both grasping and in-hand manipulation, demonstrating a scalable approach to dexterous articulated tool use.",
  "published": "2026-06-11",
  "updated": "2026-06-11",
  "year": "2026",
  "authors": [
   "Zhao-Heng Yin",
   "Guanya Shi",
   "Pieter Abbeel",
   "C. Karen Liu"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Mana (Manipulation Animator), a general sim-to-real framework that reinterprets dexterous manipulation as an animation problem, employs a coarse-to-fine pipeline that transforms procedurally-generated grasp keyframes into manipulation trajectories through motion planning and reinforcement learning.",
  "doi": "10.48550/arXiv.2606.13677",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhao-Heng Yin",
    "id": "2290035529",
    "h_index": 7,
    "papers": 12
   },
   {
    "name": "Guanya Shi",
    "id": "2384824402",
    "h_index": 9,
    "papers": 24
   },
   {
    "name": "Pieter Abbeel",
    "id": "2363582403",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "C. K. Liu",
    "id": "2242943522",
    "h_index": 4,
    "papers": 7
   }
  ],
  "comment": "Project Page: https://zhaohengyin.github.io/mana",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "sim2real",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.13677v1",
  "pdf_url": "https://arxiv.org/pdf/2606.13677v1",
  "html_url": "https://arxiv.org/html/2606.13677v1",
  "code_url": "https://zhaohengyin.github.io/mana",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2606.12604",
  "slug": "egoengine-from-egocentric-human-videos-to-high-fidelity-dexterous-robo",
  "title": "EgoEngine: From Egocentric Human Videos to High-Fidelity Dexterous Robot Demonstrations",
  "abstract": "Dexterous manipulation is limited by the cost of collecting large-scale robot demonstrations. Egocentric human videos offer a scalable source of diverse manipulation behaviors, but directly using them for robot learning requires bridging two gaps: the visual gap between human and robot observations, and the action gap between human motion and robot-executable action. We propose EgoEngine, a scalable framework for transforming egocentric human manipulation videos into high-fidelity robot data. Given an egocentric RGB video, EgoEngine produces: (i) a high-fidelity robot observation video replacing human with robot while preserving scene context and temporal alignment, and (ii) a task-aligned, executable robot action trajectory under feasibility constraints. Experiments in simulation and on real robots show that EgoEngine enables scalable conversion of human videos into robot data and, to our knowledge, demonstrates the first zero-shot visuomotor dexterous policy learning from egocentric human videos without real-robot demonstrations. Project website: https://egoengine.github.io.",
  "published": "2026-06-10",
  "updated": "2026-06-10",
  "year": "2026",
  "authors": [
   "Yangcen Liu",
   "Shuo Cheng",
   "Xinchen Yin",
   "Woo Chul Shin",
   "Alfred Cueva",
   "Yiran Yang",
   "Zhenyang Chen",
   "Chuye Zhang",
   "Danfei Xu"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 5,
  "influential_citations": 1,
  "tldr": "Experiments show that EgoEngine enables scalable conversion of human videos into robot data and, to the knowledge, demonstrates the first zero-shot visuomotor dexterous policy learning from egocentric human videos without real-robot demonstrations.",
  "doi": "10.48550/arXiv.2606.12604",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yangcen Liu",
    "id": "2297343604",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Shuo Cheng",
    "id": "2232588215",
    "h_index": 6,
    "papers": 16
   },
   {
    "name": "Xi Yin",
    "id": "2315784082",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Woo-Chul Shin",
    "id": "2360757506",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "A. Cueva",
    "id": "104623039",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Yiran Yang",
    "id": "2292213029",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Zhenyang Chen",
    "id": "2321516735",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Chuye Zhang",
    "id": "2215500204",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Danfei Xu",
    "id": "2260291195",
    "h_index": 7,
    "papers": 14
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "egocentric-data"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.12604v1",
  "pdf_url": "https://arxiv.org/pdf/2606.12604v1",
  "html_url": "https://arxiv.org/html/2606.12604v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.78
 },
 {
  "id": "2606.12109",
  "slug": "index-empowering-vla-models-with-intent-conditioned-arm-hand-coordinat",
  "title": "InDex: Empowering VLA Models with Intent-Conditioned Arm-Hand Coordination for Dexterous Manipulation",
  "abstract": "Pre-trained Vision-Language-Action (VLA) models provide useful semantic and spatial priors, yet their parallel-gripper action interfaces do not specify how those priors should be realized by a dexterous hand. Directly appending finger joints conflates two decisions with different structure: when contact should be established and how a morphology-specific hand trajectory should establish it. We introduce InDex, an intent-conditioned adaptation framework that separates these decisions without discarding full hand supervision. InDex derives a normalized grasp intent from retargeted demonstrations. A first stage predicts synchronized end-effector--intent chunks; conditioned on these predictions, VLA context, and proprioception, a diffusion decoder generates multi-joint hand actions. The scalar intent is therefore a temporal coordination interface rather than a compressed hand pose. Across four simulated tasks, three VLA backbones, and a physical arm--hand platform, InDex preserves the VLA's reaching competence while markedly improving conversion from approach to stable grasp and task completion. Ablations isolate complementary roles: intent aligns the contact transition, whereas diffusion represents the multiple hand trajectories compatible with the same task-space plan. These results identify post-reach arm--hand coordination, rather than object localization alone, as the principal bottleneck in adapting parallel-gripper VLAs to dexterous manipulation.",
  "published": "2026-06-10",
  "updated": "2026-07-28",
  "year": "2026",
  "authors": [
   "Chuanke Pang",
   "Junyi Huang",
   "Zhijun Zhao",
   "Yaobing Wang",
   "Kun Xu",
   "Xilun Ding"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "InDex is introduced, an intent-conditioned adaptation framework that preserves the VLA's reaching competence while markedly improving conversion from approach to stable grasp and task completion and identifies post-reach arm--hand coordination as the principal bottleneck in adapting parallel-gripper VLAs to dexterous manipulation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Chuanke Pang",
    "id": "2191636712",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Jun Huang",
    "id": "2159105144",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Zhijun Zhao",
    "id": "2266814432",
    "h_index": 2,
    "papers": 15
   },
   {
    "name": "Yaobing Wang",
    "id": "2266548891",
    "h_index": 2,
    "papers": 14
   },
   {
    "name": "Kun Xu",
    "id": "2285164917",
    "h_index": 2,
    "papers": 17
   },
   {
    "name": "Xilun Ding",
    "id": "2269700499",
    "h_index": 6,
    "papers": 50
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "dexterous-manipulation",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.12109v2",
  "pdf_url": "https://arxiv.org/pdf/2606.12109v2",
  "html_url": "https://arxiv.org/html/2606.12109v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.11952",
  "slug": "deformable-in-hand-slip-aware-tactile-sensor-with-integrated-velocity",
  "title": "Deformable In-Hand Slip-Aware Tactile Sensor with Integrated Velocity, Force/Torque, and Pressure Map Sensing",
  "abstract": "This paper introduces a novel tactile sensor for in-hand manipulation with slip-aware control that integrates velocity, force/torque, and pressure map sensing into a single device with a deformable contact pad. To the best of our knowledge, this is the first sensor to combine these sensing modalities within a single compliant structure. The sensor features a deformable contact surface and can robustly track both flat and curved surfaces across a wide range of diffuse surface materials. Its performance is evaluated through a comprehensive set of experiments that highlight both its capabilities and limitations. The sensor is designed for rapid and low-cost fabrication using a combination of standard PCB manufacturing and rapid prototyping techniques.",
  "published": "2026-06-10",
  "updated": "2026-08-12",
  "year": "2026",
  "authors": [
   "Gabriel Arslan Waltersson",
   "Yiannis Karayiannidis"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper introduces a novel tactile sensor for in-hand manipulation with slip-aware control that integrates velocity, force/torque, and pressure map sensing into a single device with a deformable contact pad, the first sensor to combine these sensing modalities within a single compliant structure.",
  "doi": "10.48550/arXiv.2606.11952",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Gabriel Arslan Waltersson",
    "id": "2176097494",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Y. Karayiannidis",
    "id": "1778250",
    "h_index": 22,
    "papers": 114
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.11952v2",
  "pdf_url": "https://arxiv.org/pdf/2606.11952v2",
  "html_url": "https://arxiv.org/html/2606.11952v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.11396",
  "slug": "plume-probabilistic-latent-unified-world-modeling-and-parameter-estima",
  "title": "PLUME: Probabilistic Latent Unified World Modeling and Parameter Estimation for Multi-Finger Manipulation",
  "abstract": "Dexterous manipulation with multi-finger hands can be sensitive to physical parameters such as object shape, pose, and friction coefficients. While simulation enables large-scale data collection with known parameter values, simulation-trained policies must still handle uncertainty at deployment, where the true parameters and therefore the true dynamics are unknown. Standard domain randomization strategies may be insufficient for precise tasks like screwdriver turning, as manipulation strategies may need to change depending on specific parameter values. To address this, we propose Probabilistic Latent Unified world Modeling and parameter Estimation (PLUME), a world model that jointly learns to evolve a belief over parameter values as well as the system dynamics conditioned on those parameters. We learn a latent space to jointly represent multiple qualitatively different physical parameters along with rewards, themselves functions of partially-observable variables, to inform planning. Our novel learning framework leads to efficient alignment of the world model to true dynamics through online parameter inference as opposed to re-training or fine-tuning. We evaluate our method on simulated screwdriver turning, valve turning, bucket lifting, and disk flicking tasks, as well as a hardware screwdriver turning task, where we achieve successful zero-shot transfer of our simulation-trained policy and outperform state-of-the-art offline reinforcement learning and world-model-augmented behavior cloning baselines. Please see our website at https://plume-world-model.github.io for videos.",
  "published": "2026-06-09",
  "updated": "2026-06-09",
  "year": "2026",
  "authors": [
   "Abhinav Kumar",
   "Soshi Iba",
   "Rana Soltani Zarrin",
   "Dmitry Berenson"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Probabilistic Latent Unified world Modeling and parameter Estimation (PLUME), a world model that jointly learns to evolve a belief over parameter values as well as the system dynamics conditioned on those parameters, is proposed.",
  "doi": "10.48550/arXiv.2606.11396",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Abhinav Kumar",
    "id": "2323730977",
    "h_index": 1,
    "papers": 10
   },
   {
    "name": "Soshi Iba",
    "id": "31126209",
    "h_index": 17,
    "papers": 38
   },
   {
    "name": "Rana Soltani-Zarrin",
    "id": "2256380110",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Dmitry Berenson",
    "id": "2256380388",
    "h_index": 4,
    "papers": 10
   }
  ],
  "comment": "16 pages, 5 figures",
  "topics": [
   "world-models",
   "dexterous-manipulation",
   "sim2real",
   "imitation-diffusion",
   "rl-control",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.11396v1",
  "pdf_url": "https://arxiv.org/pdf/2606.11396v1",
  "html_url": "https://arxiv.org/html/2606.11396v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.10614",
  "slug": "dexterous-point-policy-learning-point-based-dexterous-hand-policies-fr",
  "title": "Dexterous Point Policy: Learning Point-based Dexterous Hand Policies from Human Demonstrations",
  "abstract": "Robotic foundation models pre-trained on human demonstration videos have shown promise, but a significant embodiment gap remains when the resulting policies are deployed on real robots. A common remedy is to fine-tune these models on robot-specific demonstrations. However, robot data collection can be prohibitively expensive and time-consuming, which is particularly acute in dexterous manipulation, e.g., teleoperating a multi-fingered hand for even a single atomic task can take days. To address this, we introduce Dexterous Point Policy, a framework that learns dexterous manipulation policies directly from human videos and requires no robot demonstrations. Our core insight is that a unified 3D keypoint representation can bridge human and robot embodiments when used for both observations and actions. Specifically, we extract 3D keypoints of task-relevant objects and human hands from raw videos, and train an autoregressive transformer over these keypoints. We observe that at the keypoint level, specifically the wrist and fingertips, human and robot behaviors closely align, enabling direct policy transfer. On a suite of real-robot tasks spanning pick-and-place and tool use, Dexterous Point Policy attains 75.0% success, whereas a state-of-the-art VLA baseline reaches only 1.0%. Furthermore, our method generalizes strongly to unseen scenarios, including multi-object environments and novel object categories.",
  "published": "2026-06-09",
  "updated": "2026-06-09",
  "year": "2026",
  "authors": [
   "Beomjun Kim",
   "Seong Hyeon Park",
   "Seunghoon Sim",
   "Seungjun Moon",
   "Sanghyeok Lee",
   "Jinwoo Shin"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "10.48550/arXiv.2606.10614",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Beomjun Kim",
    "id": "2344031284",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "S. Park",
    "id": "2316650993",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Seunghoon Sim",
    "id": "2441648604",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Seungjun Moon",
    "id": "2277885705",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Sanghyeok Lee",
    "id": "2291986383",
    "h_index": 4,
    "papers": 16
   },
   {
    "name": "Jinwoo Shin",
    "id": "2273036795",
    "h_index": 10,
    "papers": 20
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "dexterous-manipulation",
   "egocentric-data",
   "foundation-pretraining",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.10614v1",
  "pdf_url": "https://arxiv.org/pdf/2606.10614v1",
  "html_url": "https://arxiv.org/html/2606.10614v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.10568",
  "slug": "verispace-spatially-grounded-action-verification-for-vision-language-a",
  "title": "VeriSpace: Spatially Grounded Action Verification for Vision-Language-Action Models",
  "abstract": "Vision-language-action (VLA) models have shown strong promise for robotic manipulation, but their reliability at test time remains limited by one-shot action prediction, where even small action errors can cause grasp failure, collision, or incorrect task progression. A natural alternative is to equip VLA systems with test-time verification, allowing multiple candidate actions to be proposed and evaluated before execution. However, reliable action verification is challenging because it requires not only distinguishing subtle geometric differences between candidate actions, but also assessing whether an action makes meaningful progress toward the task goal. We present VeriSpace, a 3D-aware action verifier for test-time action selection in VLA systems. VeriSpace evaluates candidate actions through two key components: Dual-Path 3D-Injected Scene Encoding, which constructs a scene representation that jointly preserves visual semantics and explicit 3D geometry, and Spatially-Grounded Action Reasoning, which evaluates each action by reasoning over task-relevant spatial relations, geometric validity, and expected goal progress. Together, these components enable more reliable discrimination between subtle yet outcome-critical action candidates while remaining fully compatible with existing VLA policies. Experiments on public benchmarks and real-world robotic manipulation tasks show that VeriSpace consistently improves decision reliability over both underlying VLA policies and prior verification-based methods, yielding substantial gains in both in-distribution and out-of-distribution settings.",
  "published": "2026-06-09",
  "updated": "2026-06-09",
  "year": "2026",
  "authors": [
   "Guiyu Zhao",
   "Longteng Guo",
   "Junyou Zhu",
   "Jun Fu",
   "Yanghong Mei",
   "Bin Cao",
   "Jie Jiang",
   "Xingjian He",
   "Jing Liu"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 1,
  "tldr": "VeriSpace is presented, a 3D-aware action verifier for test-time action selection in VLA systems that consistently improves decision reliability over both underlying VLA policies and prior verification-based methods, yielding substantial gains in both in-distribution and out-of-distribution settings.",
  "doi": "10.48550/arXiv.2606.10568",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Guiyu Zhao",
    "id": "2260857503",
    "h_index": 5,
    "papers": 16
   },
   {
    "name": "Longteng Guo",
    "id": "26982950",
    "h_index": 18,
    "papers": 73
   },
   {
    "name": "Junyou Zhu",
    "id": "2323532476",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Jun Fu",
    "id": "2119735653",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Yanghong Mei",
    "id": "2397618849",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Bin Cao",
    "id": "2307455673",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Jie Jiang",
    "id": "2297824922",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Xingjian He",
    "id": "153003010",
    "h_index": 11,
    "papers": 41
   },
   {
    "name": "Jing Liu",
    "id": "2258562505",
    "h_index": 6,
    "papers": 40
   }
  ],
  "comment": "Submit to ACM MM",
  "topics": [
   "vla",
   "dexterous-manipulation",
   "spatial-3d",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.10568v1",
  "pdf_url": "https://arxiv.org/pdf/2606.10568v1",
  "html_url": "https://arxiv.org/html/2606.10568v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2606.10244",
  "slug": "yubi-yielding-universal-bidigital-interface-for-bimanual-dexterous-man",
  "title": "YUBI: Yielding Universal Bidigital Interface for Bimanual Dexterous Manipulation at Scale",
  "abstract": "We introduce Yielding Universal Bidigital Interface (YUBI), a finger-aligned gripper designed to enable intuitive, ergonomic, and scalable data collection for bimanual dexterous manipulation. While handheld data collection systems such as Universal Manipulation Interface (UMI) enable affordable data collection, their bulky pistol-grip designs can pose ergonomic and usability challenges for fine-grained, dexterous manipulation tasks. To address this, YUBI presents a distinct design principle: yielding, finger-driven actuation that directly maps human finger movements to gripper jaw motion. Using the YUBI devices, we set up a data collection system with integrated VR-based 6 DoF tracking of the gripper, ensuring high-fidelity trajectory data acquisition. We curate a UMI-based dataset of unprecedented scale: 8,434 hours across 1.20M episodes and 119 tasks. Experiments show that YUBI offers advantages over the UMI gripper in versatility for complex bimanual tasks, dexterity, and operational efficiency. A single policy trained on the YUBI dataset transfers across multiple bimanual robots (UR, Franka, and ELEY) simply by mounting the gripper on each platform, confirming that the collected data are directly executable as policy supervision. We release the gripper hardware, data-collection software, and dataset as one integrated stack, offering the open community a reproducible path to large-scale data acquisition for advancing robotic foundation models.",
  "published": "2026-06-08",
  "updated": "2026-06-08",
  "year": "2026",
  "authors": [
   "Takehiko Ohkawa",
   "Jumpei Arima",
   "Yuki Noguchi",
   "Masatoshi Tateno",
   "Makoto Sugiura",
   "Takuya Okubo",
   "Kengo Ikeuchi",
   "Yuma Shin",
   "Hiroki Nishizawa",
   "Naoaki Kanazawa",
   "Yuki Wakayama",
   "Daiki Fukunaga",
   "Koshi Makihara",
   "Tomohiro Motoda",
   "Floris Erich",
   "Yukiyasu Domae",
   "Tatsuya Matsushima",
   "Yohishiro Okumatsu",
   "Kei Ota"
  ],
  "author_count": 19,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 2,
  "influential_citations": 0,
  "tldr": "YUBI is introduced, a finger-aligned gripper designed to enable intuitive, ergonomic, and scalable data collection for bimanual dexterous manipulation and is released as one integrated stack, offering the open community a reproducible path to large-scale data acquisition for advancing robotic foundation models.",
  "doi": "10.48550/arXiv.2606.10244",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Takehiko Ohkawa",
    "id": "1515548135",
    "h_index": 8,
    "papers": 21
   },
   {
    "name": "Jumpei Arima",
    "id": "151420321",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Yuki Noguchi",
    "id": "2292782886",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Masatoshi Tateno",
    "id": "2254899134",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Makoto Sugiura",
    "id": "2258175943",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Takuya Okubo",
    "id": "2241471380",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "K. Ikeuchi",
    "id": "35612014",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Y. Shin",
    "id": "2421868478",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Hiroki Nishizawa",
    "id": "2334737718",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Naoaki Kanazawa",
    "id": "2173085172",
    "h_index": 8,
    "papers": 29
   },
   {
    "name": "Yuki Wakayama",
    "id": "2281169978",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Daiki Fukunaga",
    "id": "1477953706",
    "h_index": 0,
    "papers": 7
   },
   {
    "name": "Koshi Makihara",
    "id": "1491238018",
    "h_index": 2,
    "papers": 22
   },
   {
    "name": "Tomohiro Motoda",
    "id": "2328411976",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Floris Erich",
    "id": "3211515",
    "h_index": 7,
    "papers": 35
   },
   {
    "name": "Y. Domae",
    "id": "2512607",
    "h_index": 13,
    "papers": 128
   },
   {
    "name": "T. Matsushima",
    "id": "145930468",
    "h_index": 10,
    "papers": 35
   },
   {
    "name": "Yohishiro Okumatsu",
    "id": "2441653227",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Keita Ota",
    "id": "104236766",
    "h_index": 10,
    "papers": 36
   }
  ],
  "comment": "Project page: https://yubi.airoa.io/",
  "topics": [
   "dexterous-manipulation",
   "foundation-pretraining",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.10244v1",
  "pdf_url": "https://arxiv.org/pdf/2606.10244v1",
  "html_url": "https://arxiv.org/html/2606.10244v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.48
 },
 {
  "id": "2606.09615",
  "slug": "dexpie-stable-dexterous-policy-improvement-from-real-world-experience",
  "title": "DexPIE: Stable Dexterous Policy Improvement from Real-World Experience",
  "abstract": "Dexterous manipulation presents substantial challenges for imitation learning due to its high-dimensional action space and complex contact-rich dynamics. Policies trained purely from demonstrations often suffer from compounding errors during deployment and require large amounts of expert data to achieve reliable performance. To move beyond the limitations of demonstration data, in this work, we propose DexPIE, a post-training framework for dexterous policy improvement from experience collected through real-world deployment. First, DexPIE enables effective exploration coverage through a dexterous-hand-adapted intervention system and multi-stage DAgger-style data collection across initial and intermediate task stages, providing reliable supervision for accurate policy evaluation. To reduce temporal noise between post-training rollouts and demonstration data, we introduce asynchronous inference in the relative action space, which better aligns rollout data with demonstrated behavior and allows the critic to learn a value function induced by a more consistent underlying policy. Finally, DexPIE improves the policy through conditioning on a continuous optimality indicator, allowing the policy to leverage the quality of data in a more fine-grained manner. Across three challenging real-world dexterous manipulation tasks, DexPIE achieves a 37% improvement in success rate over the demonstration-based reference policy, outperforming all baseline methods and demonstrating stronger robustness. The source code and dataset will be made publicly available.",
  "published": "2026-06-08",
  "updated": "2026-06-08",
  "year": "2026",
  "authors": [
   "Ruizhe Liao",
   "Wenrui Chen",
   "Liangji Zeng",
   "Haoran Lin",
   "Fan Yang",
   "Kailun Yang",
   "Yaonan Wang"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes DexPIE, a post-training framework for dexterous policy improvement from experience collected through real-world deployment that achieves a 37% improvement in success rate over the demonstration-based reference policy, outperforming all baseline methods and demonstrating stronger robustness.",
  "doi": "10.48550/arXiv.2606.09615",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ruizhe Liao",
    "id": "2441464868",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Wenrui Chen",
    "id": "2292300370",
    "h_index": 4,
    "papers": 21
   },
   {
    "name": "Liangjing Zeng",
    "id": "2362079005",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Haoran Lin",
    "id": "2309308759",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Fan Yang",
    "id": "2307191608",
    "h_index": 4,
    "papers": 15
   },
   {
    "name": "Kailun Yang",
    "id": "8689702",
    "h_index": 40,
    "papers": 284
   },
   {
    "name": "Yaonan Wang",
    "id": "2319918506",
    "h_index": 4,
    "papers": 47
   }
  ],
  "comment": "Project website: https://siiuuuuuu.github.io/DexPIE",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "imitation-diffusion",
   "navigation",
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.09615v1",
  "pdf_url": "https://arxiv.org/pdf/2606.09615v1",
  "html_url": "https://arxiv.org/html/2606.09615v1",
  "code_url": "https://siiuuuuuu.github.io/DexPIE",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2606.09451",
  "slug": "dense-force-estimation-with-an-event-based-optical-tactile-sensor",
  "title": "Dense Force Estimation with an Event-based Optical Tactile Sensor",
  "abstract": "Humans rely on spatially dense, geometry and force-aware tactile feedback at high temporal resolution for dexterous manipulation. While vision-based tactile sensors enable dense force estimation, they are limited by camera frame rates, motion blur, and data bandwidth. Event-based optical tactile sensors offer an attractive alternative with microsecond temporal resolution and low motion blur, but existing methods are restricted to predicting only net forces. We introduce the first framework for dense 3D force field reconstruction using event-based optical tactile sensors. Our approach estimates 3D surface displacements from event data and maps them to forces via the inverse Finite Elements Method (iFEM). Shear displacements are recovered through the proposed event-based marker tracking algorithm, while normal displacements are predicted by a convolutional neural network trained on a collected dataset of synchronized force-displacement-event data. Experiments demonstrate accurate reconstruction of physically grounded forces, achieving a mean absolute error of (0.14 N, 0.10 N, 0.93 N) over force ranges up to (4 N, 4 N, 20 N), while operating at an average of 100 Hz. This work constitutes a first step toward enabling dense force feedback for high-frequency control in robotic grasping and dexterous manipulation.",
  "published": "2026-06-08",
  "updated": "2026-06-08",
  "year": "2026",
  "authors": [
   "Agis Politis",
   "Ren\u00e9 Zurbr\u00fcgg",
   "Valentina Cavinato"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work introduces the first framework for dense 3D force field reconstruction using event-based optical tactile sensors and estimates 3D surface displacements from event data and maps them to forces via the inverse Finite Elements Method (iFEM).",
  "doi": "10.48550/arXiv.2606.09451",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Agis Politis",
    "id": "2115014366",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Ren\u00e9 Zurbr\u00fcgg",
    "id": "2376189695",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Valentina Cavinato",
    "id": "2343503424",
    "h_index": 3,
    "papers": 7
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.09451v1",
  "pdf_url": "https://arxiv.org/pdf/2606.09451v1",
  "html_url": "https://arxiv.org/html/2606.09451v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.09215",
  "slug": "motionwam-towards-foundation-world-action-models-for-real-time-humanoi",
  "title": "MotionWAM: Towards Foundation World Action Models for Real-Time Humanoid Loco-Manipulation",
  "abstract": "World Action Models (WAMs) couple a video dynamics prior to the policy and have shown encouraging results on tabletop manipulation, but iterative denoising over high-dimensional video-action latents leaves them too slow for real-time humanoid loco-manipulation. The problem is compounded by the dominant hierarchical paradigm, in which a high-level manipulation policy controls only the upper body while a low-level controller tracks coarse base commands -- placing upper and lower body in inconsistent action spaces and reducing the legs to balance-preserving locomotion. We present MotionWAM, a real-time WAM that drives autonomous humanoid loco-manipulation from a single egocentric camera by conditioning the policy on the intermediate denoising features of a video world model. MotionWAM replaces the upper-lower split with a unified motion latent and predicts whole-body motion tokens that jointly cover locomotion, torso motion, height regulation, foot interaction, and hand manipulation in a single action space. A three-stage learning framework progressively adapts the video world model to egocentric visual dynamics and to the target humanoid embodiment. On nine real-world Unitree G1 tasks, MotionWAM runs in real time, substantially outperforms Vision-Language-Action (VLA) baselines fine-tuned on the same demonstrations by over 30% in overall success rate, and executes task-driven foot interaction that decoupled upper-lower policies cannot reach. Our results suggest that video-pretrained WAMs can be lifted from tabletop manipulation to coordinated, human-like whole-body humanoid control.",
  "published": "2026-06-08",
  "updated": "2026-06-08",
  "year": "2026",
  "authors": [
   "Jia Zheng",
   "Teli Ma",
   "Yudong Fan",
   "Zifan Wang",
   "Shuo Yang",
   "Junwei Liang"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 2,
  "influential_citations": 0,
  "tldr": "MotionWAM is presented, a real-time WAM that drives autonomous humanoid loco-manipulation from a single egocentric camera by conditioning the policy on the intermediate denoising features of a video world model, and suggests that video-pretrained WAMs can be lifted from tabletop manipulation to coordinated, human-like whole-body humanoid control.",
  "doi": "10.48550/arXiv.2606.09215",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jiakun Zheng",
    "id": "2366110000",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Teli Ma",
    "id": "2308628765",
    "h_index": 8,
    "papers": 16
   },
   {
    "name": "Yudong Fan",
    "id": "2441293013",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Zifan Wang",
    "id": "2306832903",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Shuo Yang",
    "id": "2344197064",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Junwei Liang",
    "id": "2268726427",
    "h_index": 10,
    "papers": 19
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "vla",
   "humanoids",
   "egocentric-data"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2606.09215v1",
  "pdf_url": "https://arxiv.org/pdf/2606.09215v1",
  "html_url": "https://arxiv.org/html/2606.09215v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.98
 },
 {
  "id": "2606.08765",
  "slug": "rgb-s-image-aligned-tactile-saliency-for-robust-dexterous-manipulation",
  "title": "RGB-S: Image-Aligned Tactile Saliency for Robust Dexterous Manipulation",
  "abstract": "Effective visuo-tactile integration is critical for robotic dexterous manipulation, especially when visual observations are unreliable or occluded. However, robustly aligning sparse, heterogeneous tactile measurements with dense visual representations remains a fundamental challenge. Most existing approaches require policies to learn cross-modal correspondences implicitly from limited demonstrations, without leveraging geometric priors. As a result, they are often data-inefficient and generalize poorly when visual observations are degraded. To address this limitation, we propose a framework that explicitly grounds physical contacts in the image domain. Using robot forward kinematics and camera calibration, we project tactile sensor locations directly onto the RGB image plane. We then render force-modulated Gaussian saliency maps to model spatial uncertainty arising from kinematic and calibration errors. By integrating these 2D spatial anchors through a zero-initialized conditioning architecture, our method injects physical contact priors into standard visual backbones while preserving pre-trained visual representations. We evaluate our method on six dexterous manipulation tasks in both simulation and the real world under severe visual occlusions. Real-world experiments show that explicit RGB-S grounding in the image domain improves real-world occluded manipulation success rates by $26.7$ percentage points over the strongest implicit visuo-tactile baseline, suggesting its improved spatial reasoning and robustness to occlusion. Project page: touch-as-saliency.github.io",
  "published": "2026-06-07",
  "updated": "2026-06-11",
  "year": "2026",
  "authors": [
   "Shengcheng Luo",
   "Kefei Wu",
   "Xiaoying Zhou",
   "Wanlin Li",
   "Ziyuan Jiao",
   "Chenxi Xiao"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "Real-world experiments show that explicit RGB-S grounding in the image domain improves real-world occluded manipulation success rates by $26.7$ percentage points over the strongest implicit visuo-tactile baseline, suggesting its improved spatial reasoning and robustness to occlusion.",
  "doi": "10.48550/arXiv.2606.08765",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Shengcheng Luo",
    "id": "2309215693",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Kefei Wu",
    "id": "2441477591",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Xiaoying Zhou",
    "id": "2323590966",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Wanlin Li",
    "id": "2357100311",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Ziyuan Jiao",
    "id": "2356946141",
    "h_index": 3,
    "papers": 13
   },
   {
    "name": "Chenxi Xiao",
    "id": "2356920855",
    "h_index": 3,
    "papers": 14
   }
  ],
  "comment": "20 pages, 7 figures",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "spatial-3d",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.08765v2",
  "pdf_url": "https://arxiv.org/pdf/2606.08765v2",
  "html_url": "https://arxiv.org/html/2606.08765v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2606.08057",
  "slug": "egoaero-learning-dexterous-manipulation-from-a-single-egocentric-video",
  "title": "EgoAERO: Learning Dexterous Manipulation from a Single Egocentric Video without Object Assets",
  "abstract": "Egocentric RGB-D videos offer a natural source of human dexterous manipulation demonstrations, but existing data is difficult to use for robot learning because object pose, geometry, and contact information are often missing or require pre-scanned object assets. We present EgoAERO, the first framework that learns dexterous manipulation from a single egocentric RGB-D human demonstration without object assets. EgoAERO reconstructs contact-consistent hand-object trajectories through asset-free object tracking and reconstruction, ego motion compensation, and adaptive contact optimization, then converts them into robot policies using two-stage residual learning. We further introduce an online quality assessment mechanism and construct EgoDex-R, a large-scale egocentric dataset with 4.3M RGB-D frames for dexterous policy learning. Simulation and real-world experiments show that EgoAERO enables single-demonstration dexterous manipulation and achieves downstream performance close to CAD-based reconstructions on HOI4D.",
  "published": "2026-06-06",
  "updated": "2026-06-06",
  "year": "2026",
  "authors": [
   "Yichen Niu",
   "Haoran Lv",
   "Xinrui Zhang",
   "Xueyao Wan",
   "Shiyu Gao",
   "Ying Ai",
   "Hui Xu",
   "Yongqi Hu",
   "Hengyi Zhang",
   "Yang Xie",
   " Zhaxizhuoma",
   "Yue Zhao",
   "Zhenshan Bing",
   "Yan Ding",
   "Jianxing Liu"
  ],
  "author_count": 15,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 2,
  "influential_citations": 0,
  "tldr": "Simulation and real-world experiments show that EgoAERO enables single-demonstration dexterous manipulation and achieves downstream performance close to CAD-based reconstructions on HOI4D.",
  "doi": "10.48550/arXiv.2606.08057",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yichen Niu",
    "id": "2267068090",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Haoran Lv",
    "id": "2374079942",
    "h_index": 7,
    "papers": 17
   },
   {
    "name": "Xinrui Zhang",
    "id": "2445787651",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Xueyao Wan",
    "id": "2375202031",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Shiyu Gao",
    "id": "2363527722",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Ying Ai",
    "id": "2365265494",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "H. Xu",
    "id": "2149190069",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Yong Hu",
    "id": "2381559166",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Heng Zhang",
    "id": "2345810079",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Yang Xie",
    "id": "2110338186",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Zhaxizhuoma",
    "id": "2321550933",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Yue Zhao",
    "id": "2364545088",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Zhenshan Bing",
    "id": "2301348492",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Yan Ding",
    "id": "2316892567",
    "h_index": 11,
    "papers": 25
   },
   {
    "name": "Jianxing Liu",
    "id": "2365259335",
    "h_index": 2,
    "papers": 5
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "egocentric-data"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.08057v1",
  "pdf_url": "https://arxiv.org/pdf/2606.08057v1",
  "html_url": "https://arxiv.org/html/2606.08057v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.48
 },
 {
  "id": "2606.07386",
  "slug": "spline-policy-a-structured-representation-for-robot-policies",
  "title": "Spline Policy: A Structured Representation for Robot Policies",
  "abstract": "Modern imitation-learning policies for robot manipulation often represent actions as fixed-resolution action chunks, which are simple and effective but expose limited geometric and temporal structure before execution. This paper studies Spline Policy (SP), a structured representation that replaces action chunks with spline parameters while keeping the policy backbone unchanged. The predicted spline can be decoded as a compact continuous trajectory, queried at different temporal resolutions, constrained or edited in parameter space, and passed to downstream controllers. For quadratic spline outputs, the same representation can also be converted into a state-dependent vector field through an analytical distance-field construction. Under the regularity and projection assumptions of this construction, the induced dynamics do not increase the distance to the generated spline, yielding a principled local corrective mechanism around the predicted motion. The spline output further supports uncertainty propagation from observations to spline parameters, trajectories, and flow fields, and can be combined with classical control mechanisms such as null-space collision avoidance without retraining the policy backbone. We instantiate SP with diffusion, flow-matching, transformer-based, and vision-language-action backbones. Experiments in low-dimensional motion learning, simulated manipulation under matched backbones, dexterous manipulation, and real-robot case studies show that SP remains compatible with modern policy learners while exposing useful motion-structure properties, including compact decoding, temporal resampling, local correction around predicted motions, uncertainty evaluation, and controller compatibility.",
  "published": "2026-06-05",
  "updated": "2026-08-02",
  "year": "2026",
  "authors": [
   "Mengze Tian",
   "Yiming Li",
   "Sichao Liu",
   "Auke Ijspeert",
   "Sylvain Calinon"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "Experiments in low-dimensional motion learning, simulated manipulation under matched backbones, dexterous manipulation, and real-robot case studies show that SP remains compatible with modern policy learners while exposing useful motion-structure properties, including compact decoding, temporal resampling, local correction around predicted motions, uncertainty evaluation, and controller compatibility.",
  "doi": "10.48550/arXiv.2606.07386",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Mengze Tian",
    "id": "2374100957",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Yiming Li",
    "id": "2243012008",
    "h_index": 4,
    "papers": 18
   },
   {
    "name": "Sichao Liu",
    "id": "2383017453",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "A. Ijspeert",
    "id": "8038419",
    "h_index": 76,
    "papers": 645
   },
   {
    "name": "Sylvain Calinon",
    "id": "2243187964",
    "h_index": 7,
    "papers": 44
   }
  ],
  "comment": "This work has been submitted to the IEEE for possible publication",
  "topics": [
   "vla",
   "dexterous-manipulation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.07386v2",
  "pdf_url": "https://arxiv.org/pdf/2606.07386v2",
  "html_url": "https://arxiv.org/html/2606.07386v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2606.07118",
  "slug": "quadverse-an-integrated-framework-aligning-visual-physical-reality-for",
  "title": "QuadVerse: An Integrated Framework Aligning Visual-Physical Reality for Quadruped Simulation",
  "abstract": "Simulation is central to robot learning, yet the sim-to-real gap remains a major bottleneck. Existing approaches often tackle visual or dynamic gaps separately, overlooking how these individual mismatches accumulate and propagate throughout the robot's state evolution. In this paper, we introduce QuadVerse, an integrated framework that uses reconstructed scenes as a calibration substrate for aligning visual perception, physical interaction, and actuator dynamics. From captured RGB videos, we reconstruct geometry-constrained 3D Gaussian Splatting (3DGS) scenes that support batched photorealistic ego-view rendering and collision-ready semantic mesh extraction. The meshes further enable contact calibration by initializing spatially varying friction priors and refining them through trajectory-based posterior search. To address remaining actuator discrepancies, QuadVerse trains a residual dynamics compensator by replaying real-world trajectories on the contact-calibrated terrain, reducing the entanglement between terrain-induced contact errors and actuator non-idealities. Experiments show that QuadVerse improves reconstruction quality and locomotion tracking over relevant baselines. Leveraging this foundation, we demonstrate robust zero-shot visual-navigation policy deployment without task-specific real-world rollouts.",
  "published": "2026-06-05",
  "updated": "2026-06-08",
  "year": "2026",
  "authors": [
   "Yuxiang Chen",
   "Yuanhao Wang",
   "Ziheng Zhang",
   "Meng Zhang",
   "Yu Liu",
   "Yufei Jia",
   "Tiancai Wang",
   "Erjin Zhou",
   "Jin Xie"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This paper introduces QuadVerse, an integrated framework that uses reconstructed scenes as a calibration substrate for aligning visual perception, physical interaction, and actuator dynamics, and demonstrates robust zero-shot visual-navigation policy deployment without task-specific real-world rollouts.",
  "doi": "10.48550/arXiv.2606.07118",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yuxiang Chen",
    "id": "2374473890",
    "h_index": 4,
    "papers": 14
   },
   {
    "name": "Yuanhao Wang",
    "id": "2324001209",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Ziheng Zhang",
    "id": "2303885890",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Meng Zhang",
    "id": "2153208684",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Yu Liu",
    "id": "2327496961",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Yufei Jia",
    "id": "2293662614",
    "h_index": 6,
    "papers": 18
   },
   {
    "name": "Tiancai Wang",
    "id": "2325923837",
    "h_index": 10,
    "papers": 30
   },
   {
    "name": "Erjin Zhou",
    "id": "2277599907",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Jin Xie",
    "id": "2222785487",
    "h_index": 2,
    "papers": 4
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "sim2real",
   "spatial-3d",
   "navigation",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.07118v2",
  "pdf_url": "https://arxiv.org/pdf/2606.07118v2",
  "html_url": "https://arxiv.org/html/2606.07118v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.06721",
  "slug": "scout-semantic-scene-coverage-via-uncertainty-guided-traversal",
  "title": "SCOUT: Semantic scene COverage via Uncertainty-guided Traversal",
  "abstract": "Robots that operate over extended periods should not merely visit space; they should progressively understand it. Yet most 3D scene graph pipelines treat perception as a post-processing stage over a fixed dataset, decoupling scene representation from the decisions that determine what is observed in the first place. We present SCOUT, an online semantic exploration framework that closes this loop by coupling active traversal with probabilistic scene graph construction. Given a prior 2D occupancy map and posed RGB-D observations, SCOUT incrementally builds an uncertainty-aware 3D scene graph whose nodes maintain fused geometry and posterior beliefs over open-vocabulary object labels, while edges encode structural relations such as on, inside, belong, and next to. These beliefs are fed back to an uncertainty-guided traversal planner, which selects viewpoints by balancing expected semantic certainty gain, geometric coverage gain, and travel cost. In this way, the robot revisits ambiguous objects when additional evidence matters and expands into unseen free space when the scene remains incomplete. The resulting system treats semantic scene completeness as an operational objective rather than a passive by-product of semantic mapping, moving toward autonomous agents that can patrol, update, and reason about evolving indoor environments with minimal human intervention.",
  "published": "2026-06-04",
  "updated": "2026-06-04",
  "year": "2026",
  "authors": [
   "Junyu Mao",
   "Sara Ayoubi",
   "Vishnu D. Sharma",
   "Ilija Had\u017ei\u0107",
   "Matthew Andrews"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "SCOUT is an online semantic exploration framework that closes this loop by coupling active traversal with probabilistic scene graph construction, and treats semantic scene completeness as an operational objective rather than a passive by-product of semantic mapping.",
  "doi": "10.48550/arXiv.2606.06721",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jun Mao",
    "id": "2330376855",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Sara Ayoubi",
    "id": "144517577",
    "h_index": 14,
    "papers": 46
   },
   {
    "name": "V. Sharma",
    "id": "153403222",
    "h_index": 10,
    "papers": 22
   },
   {
    "name": "Ilija Hadzic",
    "id": "2315514594",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Matthew Andrews",
    "id": "2322095522",
    "h_index": 1,
    "papers": 4
   }
  ],
  "comment": "2026 ICRA Workshop on Uncertainty in Open World Robotics",
  "topics": [
   "spatial-3d",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.06721v1",
  "pdf_url": "https://arxiv.org/pdf/2606.06721v1",
  "html_url": "https://arxiv.org/html/2606.06721v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2606.06033",
  "slug": "realdexumi-a-wearable-universal-manipulation-interface-for-dexterous-r",
  "title": "RealDexUMI: A Wearable Universal Manipulation Interface for Dexterous Robot Learning",
  "abstract": "Learning dexterous manipulation requires demonstrations that preserve fine hand-object interactions while remaining executable at deployment. Existing pipelines either lose deployable dexterity through retargeting or embodiment conversion, or rely on robot-specific teleoperation that is costly to scale and often lacks intuitive, contact-aware control for dexterous data collection. We present RealDexUMI, a wearable universal manipulation interface built around a shared dexterous end-effector module that integrates a lightweight dexterous hand, in-hand vision, and fingertip tactile sensing. A palm-side isomorphic teleoperation glove maps human finger inputs to robot-hand joint commands, enabling real-time, retargeting-free, intuitive, and precise hand control. The shared hand and sensing modules yield zero-gap end-effector data, with matched in-hand observations, tactile signals, contacts, and hand actions between collection and deployment. Across eight real-robot tasks spanning fine-grained, contact-rich, long-horizon, and bimanual manipulation, policies trained on RealDexUMI data achieve an average success rate of 88.75%, generalize to unseen initial poses, and transfer across three embodiments. Website: https://research.beingbeyond.com/realdexumi",
  "published": "2026-06-04",
  "updated": "2026-06-06",
  "year": "2026",
  "authors": [
   "Chaoyi Xu",
   "Yixuan Jiang",
   "Jiahui Huan",
   "Yuhui Fu",
   "Haoyu Zhou",
   "Weitian Yuan",
   "Jiayi Yu",
   "Wanpeng Zhang",
   "Haoqi Yuan",
   "Zongqing Lu"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "RealDexUMI is presented, a wearable universal manipulation interface built around a shared dexterous end-effector module that integrates a lightweight dexterous hand, in-hand vision, and fingertip tactile sensing that enables real-time, retargeting-free, intuitive, and precise hand control.",
  "doi": "10.48550/arXiv.2606.06033",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Chaoyi Xu",
    "id": "2373397887",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Yixuan Jiang",
    "id": "2266816194",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Jiahui Huan",
    "id": "2440607312",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Yuhui Fu",
    "id": "2269755398",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "Haoyu Zhou",
    "id": "2232928326",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Weitian Yuan",
    "id": "2440631221",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Jiayi Yu",
    "id": "2374828292",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Wanpeng Zhang",
    "id": "2324490975",
    "h_index": 8,
    "papers": 23
   },
   {
    "name": "Haoqi Yuan",
    "id": "1429192914",
    "h_index": 12,
    "papers": 37
   },
   {
    "name": "Zongqing Lu",
    "id": "2258676670",
    "h_index": 29,
    "papers": 144
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.06033v2",
  "pdf_url": "https://arxiv.org/pdf/2606.06033v2",
  "html_url": "https://arxiv.org/html/2606.06033v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.05880",
  "slug": "taga-terrain-aware-active-gaze-learning-for-generalizable-agile-humano",
  "title": "TAGA: Terrain-aware Active Gaze Learning for Generalizable Agile Humanoid Locomotion",
  "abstract": "Agile humanoid locomotion across diverse challenging terrain demands both wide perceptual coverage and precise local geometry understanding. Motivated by the way humans selectively look at relevant terrain during locomotion, we introduce TAGA, a Terrain-aware Active Gaze learning framework for Attention-based humanoid control. By fusing vision, proprioception, and motion commands, our framework guides the model to learn anticipatory cues and actively attend to specific areas of the height scan, selectively using these informative regions for the downstream network. This adaptively increases the information density of observations under tight onboard computational constraints, thus enabling fine-grained perceptive locomotion over larger-scale terrains. We find that such gaze behaviors can naturally emerge through reinforcement learning alone, without requiring additional supervision or explicit guidance, significantly improve training efficiency. As a result, the trained policy demonstrates robust and generalizable locomotion in simulation and on hardware, including reliable terrain-aware foothold selection, elevated-platform traversal, competitive sparse-foothold traversal, and the largest reported real-world gap traversal distance of 1.2m among perceptive humanoid locomotion systems, while maintaining stability under severe perceptual disturbances and environmental interference.",
  "published": "2026-06-04",
  "updated": "2026-06-04",
  "year": "2026",
  "authors": [
   "Peizhuo Li",
   "Hongyi Li",
   "Mingfeng Fan",
   "Fangzhou Xu",
   "Shuhao Liao",
   "Yuxuan Ma",
   "Zicheng Zeng",
   "Ze Wang",
   "Yongbin Jin",
   "Yuhong Cao",
   "Hongtao Wang",
   "Guillaume Sartoretti"
  ],
  "author_count": 12,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 2,
  "influential_citations": 0,
  "tldr": "",
  "doi": "10.48550/arXiv.2606.05880",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Peizhuo Li",
    "id": "2257130998",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Hongyi Li",
    "id": "2269792976",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Mingfeng Fan",
    "id": "2299534746",
    "h_index": 4,
    "papers": 19
   },
   {
    "name": "Fang Xu",
    "id": "2432957250",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Shuhao Liao",
    "id": "2273992414",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Yuxuan Ma",
    "id": "2325817436",
    "h_index": 0,
    "papers": 5
   },
   {
    "name": "Zicheng Zeng",
    "id": "2406449860",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Zehao Wang",
    "id": "2118402851",
    "h_index": 9,
    "papers": 28
   },
   {
    "name": "Yongbin Jin",
    "id": "2110559147",
    "h_index": 5,
    "papers": 22
   },
   {
    "name": "Yuhong Cao",
    "id": "2146175818",
    "h_index": 12,
    "papers": 39
   },
   {
    "name": "Hongtao Wang",
    "id": "2289370527",
    "h_index": 2,
    "papers": 14
   },
   {
    "name": "G. Sartoretti",
    "id": "2292917033",
    "h_index": 11,
    "papers": 68
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.05880v1",
  "pdf_url": "https://arxiv.org/pdf/2606.05880v1",
  "html_url": "https://arxiv.org/html/2606.05880v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.48
 },
 {
  "id": "2606.05491",
  "slug": "unpaired-rgb-thermal-gaussian-splatting-using-visual-geometric-transfo",
  "title": "Unpaired RGB-Thermal Gaussian-Splatting Using Visual Geometric Transformers",
  "abstract": "Multi-modal novel view synthesis (NVS) combining RGB and thermal imagery enables precise 3D scene reconstruction with visual and thermal information. However, existing methods typically rely on precisely calibrated RGB-thermal image pairs or stereo setups, limiting scalability and practical deployment. To address this, we introduce a framework for unpaired RGB-thermal NVS that leverages VGGT, a 3D feed-forward transformer architecture, to independently estimate camera poses for each modality. The pose sets are then aligned using the Procrustes algorithm with a cross-modal feature matcher, enabling joint registration without paired calibration. Building on this alignment, we further propose a multi-modal 3D Gaussian Splatting approach that learns directly from unpaired RGB and thermal images. Experiments on diverse scenes demonstrate that our method achieves competitive performance in thermal view synthesis while maintaining RGB fidelity. Moreover, we show that existing reconstruction approaches can produce modality-specific reconstructions that lack cross-modal consistency. We thus introduce a benchmarking framework to rigorously evaluate both per-modality image synthesis and the multi-modal coherence of reconstructed scenes.",
  "published": "2026-06-03",
  "updated": "2026-06-03",
  "year": "2026",
  "authors": [
   "Jean Cordonnier",
   "Chenghao Xu",
   "Olga Fink",
   "Malcolm Mielle"
  ],
  "author_count": 4,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "ICRA 2026",
  "venue_source": "arxiv-comment",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "A framework for unpaired RGB-thermal NVS that leverages VGGT, a 3D feed-forward transformer architecture, to independently estimate camera poses for each modality, and a multi-modal 3D Gaussian Splatting approach that learns directly from unpaired RGB and thermal images.",
  "doi": "10.48550/arXiv.2606.05491",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jean-Baptiste Cordonnier",
    "id": "51440515",
    "h_index": 9,
    "papers": 15
   },
   {
    "name": "Chenghao Xu",
    "id": "2153077574",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Olga Fink",
    "id": "2319401213",
    "h_index": 4,
    "papers": 17
   },
   {
    "name": "Malcolm Mielle",
    "id": "8316213",
    "h_index": 8,
    "papers": 24
   }
  ],
  "comment": "Accepted at ICRA 2026's Workshop MM-SpatialAI: Multi-Modal Spatial AI for Robust Navigation and Open-World Understanding",
  "topics": [
   "spatial-3d",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.05491v1",
  "pdf_url": "https://arxiv.org/pdf/2606.05491v1",
  "html_url": "https://arxiv.org/html/2606.05491v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.8
 },
 {
  "id": "2606.04788",
  "slug": "z-floc-zero-shot-floorplan-localization-via-geometric-primitives",
  "title": "Z-FLoc: Zero-Shot Floorplan Localization via Geometric Primitives",
  "abstract": "Visual localization -- estimating a camera pose within a pre-existing map -- is a fundamental problem in computer vision. Floorplans are an attractive map representation: they are readily available for most buildings, compact, and inherently invariant to visual appearance changes. However, bridging the severe domain gap between camera observations and floorplan geometry remains challenging. Existing methods address this gap through data-driven learning, yet they require large-scale training data and environment-specific retraining, limiting their practical deployment. We propose a zero-shot floorplan localization method that generalizes to novel environments without any retraining. Our key insight is that dominant geometric primitives -- lines and circles -- are ubiquitous in human-made environments and provide appearance-invariant structural constraints. We extract these primitives from a bird's-eye-view (BEV) projection of monocular 3D reconstructions and match them to the floorplan via dedicated minimal solvers within a robust estimation framework. Experiments on both simulated and real-world datasets show that our approach outperforms state-of-the-art learning-based methods on unseen environments, while using a single fixed set of hyperparameters across all experiments. The source code will be made publicly available.",
  "published": "2026-06-03",
  "updated": "2026-06-03",
  "year": "2026",
  "authors": [
   "Ayumi Umemura",
   "Toshinori Kuwahara",
   "Marc Pollefeys",
   "Daniel Barath"
  ],
  "author_count": 4,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "This work proposes a zero-shot floorplan localization method that generalizes to novel environments without any retraining, and outperforms state-of-the-art learning-based methods on unseen environments, while using a single fixed set of hyperparameters across all experiments.",
  "doi": "10.48550/arXiv.2606.04788",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ayumi Umemura",
    "id": "108559212",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Toshinori Kuwahara",
    "id": "2282853567",
    "h_index": 2,
    "papers": 17
   },
   {
    "name": "Marc Pollefeys",
    "id": "2300182737",
    "h_index": 11,
    "papers": 37
   },
   {
    "name": "D\u00e1niel Bar\u00e1th",
    "id": "2326113501",
    "h_index": 2,
    "papers": 6
   }
  ],
  "comment": "",
  "topics": [
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.04788v1",
  "pdf_url": "https://arxiv.org/pdf/2606.04788v1",
  "html_url": "https://arxiv.org/html/2606.04788v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2606.04226",
  "slug": "percepttwin-semantic-scene-reconstruction-for-iterative-llm-planning-a",
  "title": "PerceptTwin: Semantic Scene Reconstruction for Iterative LLM Planning and Verification",
  "abstract": "Simulation environments are useful for both robot policy learning and planning verification and validation. Traditionally, the process of creating a simulation was onerous. Creating a bespoke simulation environment for each individual environment that a robot would operate in was simply infeasible. In this work, we introduce PerceptTwin, a fully automatic pipeline that constructs interactive simulations directly from semantic scene representations produced by a robot's perception stack. PerceptTwin combines open-vocabulary object maps with 3D asset generation, affordance prediction, and commonsense condition checking. These interactive simulations can be used to validate and refine plans before they are executed on the robot hardware. Borrowing from the AI alignment literature, we also introduce an LLM judge that verifies plan correctness and alignment with human preferences. Experiments show that PerceptTwin feedback allows LLM planners to refine plans, enhance safety, and resist harmful black-box prompting attacks. In our suite of tasks, PerceptTwin improves plan success by an average of approximately 39% for GPT5, GPT5Mini, and GPT5Nano planners. Additionally, PerceptTwin also improves human plan verification by up to 18% on average for plans that fail due to unfilled skill preconditions. Our results demonstrate the potential of open-vocabulary scene simulation from robot perception as a foundation for safer, more reliable robot planning.",
  "published": "2026-06-02",
  "updated": "2026-06-02",
  "year": "2026",
  "authors": [
   "Charlie Gauthier",
   "Sacha Morin",
   "Liam Paull"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "PerceptTwin is introduced, a fully automatic pipeline that constructs interactive simulations directly from semantic scene representations produced by a robot's perception stack that combines open-vocabulary object maps with 3D asset generation, affordance prediction, and commonsense condition checking.",
  "doi": "10.48550/arXiv.2606.04226",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Charlie Gauthier",
    "id": "2347536252",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Sacha Morin",
    "id": "2333357020",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Liam Paull",
    "id": "2333356827",
    "h_index": 2,
    "papers": 10
   }
  ],
  "comment": "Accepted at ICRA 2026 (Vienna); published on arxiv for archival purposes. See also https://percept-twin.github.io/",
  "topics": [
   "spatial-3d",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.04226v1",
  "pdf_url": "https://arxiv.org/pdf/2606.04226v1",
  "html_url": "https://arxiv.org/html/2606.04226v1",
  "code_url": "https://percept-twin.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.5
 },
 {
  "id": "2606.04158",
  "slug": "multi-agent-next-best-view-optimization-for-risk-averse-planning",
  "title": "Multi-Agent Next-Best-View Optimization for Risk-Averse Planning",
  "abstract": "Multi-agent Next-Best-View (NBV) selection for safe path planning in uncertain and unknown environments requires informative, safety-aware, and efficient coordination. Centralized approaches rely on sharing raw sensor data or significant communication overhead, resulting in limited scalability. We propose a distributed, risk-aware multi-agent NBV framework in which each robot maintains a private local 3D Gaussian Splatting map and the team jointly maximizes expected information gain (EIG) restricted to masked zones along planned trajectories. The resulting distributed objective is solved by Consensus ADMM (C-ADMM) over a communication graph, with each robot exchanging only candidate viewpoints, planned trajectory descriptors, and scalar EIG contributions. Collision risk along each trajectory is modeled via Average Value-at-Risk (AV@R) over the local 3DGS map and used both to shape the masking radius and to score planned paths. Experiments in Gibson environments at multiple team sizes show that the distributed formulation approaches the centralized baseline in mapping quality and trajectory safety while reducing communication by orders of magnitude.",
  "published": "2026-06-02",
  "updated": "2026-06-02",
  "year": "2026",
  "authors": [
   "Amirhossein Mollaei Khass",
   "Vivek Pandey",
   "Guangyi Liu",
   "Athanasios Cosse",
   "Emrah Bayrak",
   "Nader Motee"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A distributed, risk-aware multi-agent NBV framework in which each robot maintains a private local 3D Gaussian Splatting map and the team jointly maximizes expected information gain restricted to masked zones along planned trajectories is proposed.",
  "doi": "10.48550/arXiv.2606.04158",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Amirhossein Mollaei Khass",
    "id": "2384449418",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Vivek Pandey",
    "id": "2238951029",
    "h_index": 3,
    "papers": 13
   },
   {
    "name": "Guangyi Liu",
    "id": "2259930602",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Athanasios Cosse",
    "id": "2438611168",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Emrah Bayrak",
    "id": "89174933",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "N. Motee",
    "id": "3329695",
    "h_index": 22,
    "papers": 142
   }
  ],
  "comment": "8 pages, 5 figures. Submitted to IROS 2026",
  "topics": [
   "spatial-3d",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.04158v1",
  "pdf_url": "https://arxiv.org/pdf/2606.04158v1",
  "html_url": "https://arxiv.org/html/2606.04158v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2606.03994",
  "slug": "simuscene-simulation-ready-compositional-3d-scene-reconstruction-from",
  "title": "SimuScene: Simulation-Ready Compositional 3D Scene Reconstruction from a Single Image",
  "abstract": "Reconstructing interactive, simulation-ready 3D scenes from a single image is a critical bottleneck for robotic manipulation. While recent single-image lifters recover plausible per-object shapes, composing them yields scenes that collapse under physical simulation due to interpenetrating, hovering, or sinking objects. Existing physics-aware methods address this strictly as a post-hoc layout correction, leaving the underlying geometric errors unresolved. To address this, we introduce SimuScene, a compositional 3D reconstruction pipeline that puts physics in the loop of shape and layout estimation. Rather than using physics merely for layout cleanup, we utilize the physics engine as a diagnostic measurement tool during the generative process itself. By diagnostically simulating reconstructed objects under gravity, we convert penetration and support failures into quantitative correction signals that drive gravity-axis stretching and amodal shape resampling. This physics-informed feedback loop mitigates accumulated reconstruction errors and produces a stable, simulation-ready compositional 3D scene. Extensive experiments demonstrate state-of-the-art performance on physical stability and geometric alignment benchmarks. We further highlight SimuScene's utility by deploying reconstructed environments in humanoid control and robot-arm manipulation tasks.",
  "published": "2026-06-02",
  "updated": "2026-06-02",
  "year": "2026",
  "authors": [
   "Inhee Lee",
   "Sangwon Baik",
   "Sungjoo Kim",
   "Hyeonwoo Kim",
   "Hyunsoo Cha",
   "Hanbyul Joo"
  ],
  "author_count": 6,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "SimuScene is introduced, a compositional 3D reconstruction pipeline that puts physics in the loop of shape and layout estimation during the generative process itself and converts penetration and support failures into quantitative correction signals that drive gravity-axis stretching and amodal shape resampling.",
  "doi": "10.48550/arXiv.2606.03994",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Inhee Lee",
    "id": "2297840579",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "S. Baik",
    "id": "2394172877",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Sungjoo Kim",
    "id": "2157112177",
    "h_index": 0,
    "papers": 7
   },
   {
    "name": "Hyeonwoo Kim",
    "id": "2281297992",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Hyunsoo Cha",
    "id": "2284593116",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Hanbyul Joo",
    "id": "2277246764",
    "h_index": 8,
    "papers": 27
   }
  ],
  "comment": "Project Page: https://snuvclab.github.io/SimuScene/",
  "topics": [
   "humanoids",
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.03994v1",
  "pdf_url": "https://arxiv.org/pdf/2606.03994v1",
  "html_url": "https://arxiv.org/html/2606.03994v1",
  "code_url": "https://snuvclab.github.io/SimuScene/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2606.03682",
  "slug": "gn0-toward-a-unified-paradigm-for-generation-evaluation-and-policy-lea",
  "title": "GN0: Toward a Unified Paradigm for Generation, Evaluation, and Policy Learning in Visual-Language Navigation",
  "abstract": "Embodied navigation connects intelligent agents with the physical world and is fundamental for general robotic intelligence. Limited availability and quality of navigation data have constrained Vision-and-Language Navigation (VLN) systems' generalization and long-horizon capabilities. To address this, we curate diverse 3D scenes and develop an automated pipeline for large-scale navigation data, resulting in the GN-Matrix dataset. Building on a 3D Gaussian Splatting (3DGS) engine, we introduce a high-fidelity simulation platform supporting interactive roaming and collision-aware navigation. We further propose GN-Bench, the first BEV-based benchmark incorporating dynamic 3DGS avatars for human-robot interaction evaluation. To leverage the simulator, we develop an RL-driven navigation foundation model, Break and Establish (BAE). After supervised learning, DAgger exposes the model to rollout-induced states, breaking narrow expert-centric distributions and enabling downstream RL exploration. This unified VLN paradigm integrates map-based and map-free tasks, including instruction following, human following, and goal navigation. GN-BAE formalizes high-fidelity 3DGS-rendered Bird's Eye View representations as compact memory, unlocking latent spatial reasoning in VLMs. Extensive evaluations on GN-Bench and VLN-CE show that GN0 outperforms state-of-the-art VLN methods. Overall, GN-Matrix offers a unified framework spanning data, simulation, and learning, advancing embodied navigation in research and industrial applications.",
  "published": "2026-06-02",
  "updated": "2026-06-02",
  "year": "2026",
  "authors": [
   "Xinhai Li",
   "Xiaotao Zhang",
   "Yuehao Huang",
   "Jiankun Dong",
   "Tianhang Wang",
   "Sunyao Zhou",
   "Yunzi Wu",
   "Chengnuo Sun",
   "Yunfei Ge",
   "Qizhen Weng",
   "Chi Zhang",
   "Chenjia Bai",
   "Xuelong Li"
  ],
  "author_count": 13,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 5,
  "influential_citations": 0,
  "tldr": "This work introduces a high-fidelity simulation platform supporting interactive roaming and collision-aware navigation, and develops an RL-driven navigation foundation model, Break and Establish, which formalizes high-fidelity 3DGS-rendered Bird's Eye View representations as compact memory, unlocking latent spatial reasoning in VLMs.",
  "doi": "10.48550/arXiv.2606.03682",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xinhai Li",
    "id": "2267385564",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Xiaotao Zhang",
    "id": "2277983539",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Yuehao Huang",
    "id": "2282599261",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Jian Dong",
    "id": "2335320190",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Tianhang Wang",
    "id": "2266487903",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Sunyao Zhou",
    "id": "2391650727",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Yunzi Wu",
    "id": "2326255864",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Cheng-Huan Sun",
    "id": "2437861317",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Yunfei Ge",
    "id": "2347238466",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Qizhen Weng",
    "id": "2394464024",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Chi Zhang",
    "id": "2328621976",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Chenjia Bai",
    "id": "2320302313",
    "h_index": 9,
    "papers": 35
   },
   {
    "name": "Xuelong Li",
    "id": "2295686463",
    "h_index": 10,
    "papers": 30
   }
  ],
  "comment": "",
  "topics": [
   "sim2real",
   "spatial-3d",
   "navigation",
   "foundation-pretraining",
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.03682v1",
  "pdf_url": "https://arxiv.org/pdf/2606.03682v1",
  "html_url": "https://arxiv.org/html/2606.03682v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.78
 },
 {
  "id": "2606.03581",
  "slug": "unsocc-3d-semantic-occupancy-prediction-in-unstructured-scene-via-rend",
  "title": "UnsOcc: 3D Semantic Occupancy Prediction in Unstructured Scene via Rendering Fusion",
  "abstract": "Unstructured scenes present unique challenges for autonomous driving, as irregular obstacles and sparse scene layouts undermine the effectiveness of traditional perception methods such as 3D object detection. 3D semantic occupancy prediction has emerged as a prominent focus due to its ability to provide dense spatial representations by assigning semantic labels to individual voxels in 3D space. However, directly applying 3D semantic occupancy prediction to unstructured scenes remains challenging because scene sparsity hinders effective cross-modal fusion and the more severe long-tail distribution in these scenarios further degrades prediction performance. To validate the effectiveness of our approach, we construct a dedicated dataset of unstructured scenes collected from open-pit mines. Based on this, we propose UnsOcc, a multi-modal 3D semantic occupancy prediction framework that improves robustness in unstructured environments. At its core, we introduce a rendering-based fusion module, RenderFusion, which enhances cross-modal feature alignment through bidirectional rendering supervision. Furthermore, we propose GSRefinement, a detail-aware auxiliary supervision method based on Gaussian Splatting that projects sparse 3D occupancy predictions into dense 2D semantic segmentation maps, enabling effective supervision for long-tail categories. Extensive experiments on both the open-pit mine dataset and the nuScenes dataset demonstrate that our method significantly outperforms existing state-of-the-art approaches.",
  "published": "2026-06-02",
  "updated": "2026-06-02",
  "year": "2026",
  "authors": [
   "Ye Wu",
   "Ruiqi Song",
   "Baiyong Ding",
   "Nanxin Zeng",
   "Junjie Cheng",
   "Yunfeng Ai"
  ],
  "author_count": 6,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "This work proposes UnsOcc, a multi-modal 3D semantic occupancy prediction framework that improves robustness in unstructured environments and introduces a rendering-based fusion module, RenderFusion, which enhances cross-modal feature alignment through bidirectional rendering supervision.",
  "doi": "10.48550/arXiv.2606.03581",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ye Wu",
    "id": "2350681778",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Ruiqi Song",
    "id": "2284669619",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Baiyong Ding",
    "id": "2064493694",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Nanxin Zeng",
    "id": "2375390487",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Junjie Cheng",
    "id": "2345640991",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Yunfeng Ai",
    "id": "2284079914",
    "h_index": 2,
    "papers": 12
   }
  ],
  "comment": "8 pages",
  "topics": [
   "spatial-3d",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.03581v1",
  "pdf_url": "https://arxiv.org/pdf/2606.03581v1",
  "html_url": "https://arxiv.org/html/2606.03581v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2606.03268",
  "slug": "eadex-a-cross-embodiment-dexterous-manipulation-framework-from-low-cos",
  "title": "EaDex: A Cross-Embodiment Dexterous Manipulation Framework from Low-Cost Demonstrations",
  "abstract": "Dexterous manipulation learning has long been hindered by the high costs of data and training, as pure reinforcement learning typically requires large-scale interactive exploration and imitation learning depends on high-quality demonstrations that are expensive to collect. To address this problem, we propose EaDex, a multi-embodiment dexterous manipulation learning framework under low-cost demonstration conditions, which enables rapid generation of demonstration data and consequently reduces training time for efficient dexterous manipulation. At the data level, EaDex captures human hand motions using only a single RGB-D camera and constructs structured demonstration data through MANO-based hand modeling, data normalization, and motion retargeting. At the learning level, we introduce a contact-reward-based dynamic demonstration annealing mechanism, which guides early-stage exploration under demonstration and gradually transitions to autonomous optimization with accumulating contact rewards. Using our custom dataset, we evaluate EaDex on three dexterous hands and three articulated object-opening tasks, covering nine cross-embodiment manipulation settings, achieving a 55.3% relative improvement over the baseline without demonstration annealing. These results validate the effectiveness of the proposed low-cost demonstration pipeline and the dynamic demonstration annealing strategy for dexterous manipulation learning.",
  "published": "2026-06-02",
  "updated": "2026-06-02",
  "year": "2026",
  "authors": [
   "Qian Zhao",
   "Xin Tong",
   "Chengdong Wu",
   "Yang Yang",
   "Yingtian Li"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "EaDex, a multi-embodiment dexterous manipulation learning framework under low-cost demonstration conditions, which enables rapid generation of demonstration data and consequently reduces training time for efficient dexterous manipulation.",
  "doi": "10.48550/arXiv.2606.03268",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Qian Zhao",
    "id": "2226502272",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Xin Tong",
    "id": "2211990685",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Chengdong Wu",
    "id": "2244620518",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Yang Yang",
    "id": "2152369918",
    "h_index": 9,
    "papers": 13
   },
   {
    "name": "Yingtian Li",
    "id": "2273553340",
    "h_index": 3,
    "papers": 9
   }
  ],
  "comment": "11 pages, 5 figures, Conference: CoRL 2026, Submitted as Preprint",
  "topics": [
   "dexterous-manipulation",
   "imitation-diffusion",
   "rl-control",
   "navigation",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.03268v1",
  "pdf_url": "https://arxiv.org/pdf/2606.03268v1",
  "html_url": "https://arxiv.org/html/2606.03268v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2606.03177",
  "slug": "contrack-constrained-hand-motion-tracking-with-adaptive-trade-off-cont",
  "title": "ConTrack: Constrained Hand Motion Tracking with Adaptive Trade-off Control",
  "abstract": "Human demonstrations provide strong priors for robot manipulation, yet it is non-trivial to transfer them to execute on real robots due to the kinematic gap. In dexterous manipulation, it remains challenging to track long-horizon, contact-rich sequences even in simulators: a reference-tracking policy must keep objects on their target trajectories while preserving demonstrated joint motion and contact timing. Existing approaches often rely on hand-crafted reward tuning that require per-sequence tuning and break under limited interaction budgets. We introduce ConTrack, a reinforcement learning (RL) framework that scales with tracking data. ConTrack treats object tracking as a constraint and allocates remaining control authority to motion fidelity, which allows it to adapt task--style trade-offs online using a dual-variable update. In addition, ConTrack also stabilizes long-horizon learning with an adaptive mid-trajectory reset library that reuses policy-reachable simulator states. Our qualitative and quantitative results in simulation tracking and real robot demonstrate that ConTrack improves success and object pose accuracy significantly over prior arts while preserving joint and contact fidelity. Website: https://www.lyt0112.com/projects/ConTrack.",
  "published": "2026-06-02",
  "updated": "2026-06-15",
  "year": "2026",
  "authors": [
   "Yutong Liang",
   "Quanquan Peng",
   "Ri-Zhao Qiu",
   "Xiaolong Wang"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "ConTrack is introduced, a reinforcement learning (RL) framework that scales with tracking data and treats object tracking as a constraint and allocates remaining control authority to motion fidelity, which allows it to adapt task--style trade-offs online using a dual-variable update.",
  "doi": "10.48550/arXiv.2606.03177",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yutong Liang",
    "id": "2358611288",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Quanquan Peng",
    "id": "2409184845",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Ri-Zhao Qiu",
    "id": "2290904526",
    "h_index": 11,
    "papers": 19
   },
   {
    "name": "Xiaolong Wang",
    "id": "2294782536",
    "h_index": 12,
    "papers": 17
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "egocentric-data",
   "tactile",
   "sim2real",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.03177v2",
  "pdf_url": "https://arxiv.org/pdf/2606.03177v2",
  "html_url": "https://arxiv.org/html/2606.03177v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2606.02058",
  "slug": "tides-time-derivative-event-simulation-via-deformable-reconstruction",
  "title": "TIDES: Time-Derivative Event Simulation via Deformable Reconstruction",
  "abstract": "Event cameras emit asynchronous events in response to environmental appearance changes. The scarcity of real-world event datasets makes simulation essential. However, most simulators infer event timestamps from frame sequences, forcing many threshold crossings to share a small set of discrete times; a failure mode we term timestamp batching that worsens under fast motion and occlusion. We present TIDES, a continuous-time event simulator built on dynamic Gaussian splatting. Because TIDES operates on an explicit 3D scene representation with learnt geometry and motion, it can derive per-pixel intensity dynamics directly from the scene, rather than by differencing rendered frames. This enables accurate threshold-crossing prediction, including multiple crossings per rendering step, without temporal upsampling or frame interpolation. The same 3D scene model reveals where objects partially occlude one another; TIDES uses this to guide adaptive time stepping, concentrating computation only in regions where occlusion dynamics make simple models of brightness change unreliable. Finally, we model finite sensor bandwidth using a tile-level arbiter whose throughput, jitter, and event drops reproduce realistic sensor artifacts. Across paired RGB-event benchmarks, TIDES attains state-of-the-art event-stream fidelity. We also show that events simulated by TIDES transfer more effectively to real downstream tasks than competitors'.",
  "published": "2026-06-01",
  "updated": "2026-06-01",
  "year": "2026",
  "authors": [
   "Christopher Thirgood",
   "Dipon Kumar Ghosh",
   "Simon Hadfield"
  ],
  "author_count": 3,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work presents TIDES, a continuous-time event simulator built on dynamic Gaussian splatting that attains state-of-the-art event-stream fidelity and shows that events simulated by TIDES transfer more effectively to real downstream tasks than competitors'.",
  "doi": "10.48550/arXiv.2606.02058",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "C. Thirgood",
    "id": "2241205338",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Diponkar Ghosh",
    "id": "67062511",
    "h_index": 7,
    "papers": 9
   },
   {
    "name": "Simon Hadfield",
    "id": "2276432765",
    "h_index": 4,
    "papers": 13
   }
  ],
  "comment": "",
  "topics": [
   "sim2real",
   "spatial-3d",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.02058v1",
  "pdf_url": "https://arxiv.org/pdf/2606.02058v1",
  "html_url": "https://arxiv.org/html/2606.02058v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2606.01458",
  "slug": "legs-fine-tuning-teleop-free-vlas-for-humanoid-loco-manipulation-in-an",
  "title": "LEGS: Fine-Tuning Teleop-Free VLAs for Humanoid Loco-manipulation in an Embodied Gaussian Splatting World",
  "abstract": "Training vision-language-action (VLA) policies for humanoid loco-manipulation is constrained by the high cost and complexity of collecting human teleoperation demonstrations. VLA policies fine-tuned in simulators have, until now, failed to transfer effectively in humanoid loco-manipulation tasks. We present LEGS (Loco-manipulation via Embodied Gaussian Splatting), a hybrid simulator that composites a mesh foreground (robot, objects, props) over a photorealistic 3D Gaussian Splatting (3DGS) background reconstructed from a handheld scene capture. LEGS uses a procedural motion-primitive generator to synthesize labeled demonstrations at scale without human teleoperation, and a deterministic two-stage color calibration to align the rendered 3DGS image to the robot's deployment camera. On a Unitree G1 humanoid robot, across three pick-and-place tasks of increasing whole-body difficulty and three VLA backbones (psi_0, pi_0.5, GR00T N1.6), a policy trained purely on LEGS data matches or exceeds one trained on human teleoperation demos on every experiment. It also outperforms a mesh-only simulation baseline that ablates the effect of the 3DGS background, showing that photorealistic rendering is a key enabler for synthetic data transfer. Humanoid motion is recorded independently of scene appearance in LEGS, allowing the same auto-generated demonstrations to be re-rendered under new backgrounds and object meshes--covering a new scene at more than 15x lower cost than teleoperation--to augment training data for robustness to scene variations. Under combined object-and-scene appearance shift, the policy trained on re-rendered LEGS-AUG data maintains task success while the baseline trained on teleoperation data fails entirely. Our project page is located at https://legsvla.github.io/.",
  "published": "2026-05-31",
  "updated": "2026-05-31",
  "year": "2026",
  "authors": [
   "Hojune Kim",
   "Timothy Chen",
   "Jiankai Sun",
   "Lars W. Osterberg",
   "Qianzhong Chen",
   "Ke Wang",
   "Mac Schwager"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "10.48550/arXiv.2606.01458",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hojun Kim",
    "id": "2299893893",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Timothy Chen",
    "id": "2300331757",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Jiankai Sun",
    "id": "2247730541",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Lars W. Osterberg",
    "id": "2323996376",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Qian Chen",
    "id": "2257010473",
    "h_index": 12,
    "papers": 36
   },
   {
    "name": "Ke Wang",
    "id": "2446824359",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Mac Schwager",
    "id": "2360172520",
    "h_index": 6,
    "papers": 14
   }
  ],
  "comment": "https://legsvla.github.io/",
  "topics": [
   "vla",
   "humanoids",
   "sim2real",
   "spatial-3d",
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2606.01458v1",
  "pdf_url": "https://arxiv.org/pdf/2606.01458v1",
  "html_url": "https://arxiv.org/html/2606.01458v1",
  "code_url": "https://legsvla.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2606.00576",
  "slug": "dynamic-resilient-spatio-semantic-memory-with-hybrid-localization-for",
  "title": "Dynamic Resilient Spatio-Semantic Memory with Hybrid Localization for Mobile Manipulation",
  "abstract": "Reliable mobile manipulation in dynamic indoor environments requires a scene representation that remains geometrically consistent, semantically queryable, and computationally bounded as the environment changes. Existing systems often rely on pre-built maps, static-scene assumptions, or highly accurate camera poses, which can lead to stale or misaligned scene information when target objects are relocated or pose estimates are corrected. This paper presents DREAM, a real-robot mobile manipulation framework that integrates perception, memory, localization, navigation, and manipulation in previously unseen indoor environments without a pre-built map. DREAM constructs an online spatio-semantic voxel memory from RGB-D observations registered by a LiDAR-inertial-visual SLAM backend. It further introduces pose-graph-aware Redundancy-Aware Memory Pruning (RMP) to update historical observations after pose corrections while keeping long-horizon observation history bounded. For target localization and reacquisition, DREAM combines language-conditioned 3D retrieval, open-vocabulary image detection, and multimodal large language model based semantic verification. Real-robot experiments in four dynamic indoor laboratory scenes show that DREAM improves long-horizon task success rates from 40%-60% with DynaMem to 55%-70%, while maintaining a memory footprint of 0.37-0.63 GB and an online memory-update time of 0.43-0.53 s across scenes.",
  "published": "2026-05-30",
  "updated": "2026-05-30",
  "year": "2026",
  "authors": [
   "Zhijie Yan",
   "Shufei Li",
   "Ze Zhang",
   "Xin Liu",
   "Yuhang Zheng",
   "Zuoxu Wang"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "DREAM introduces pose-graph-aware Redundancy-Aware Memory Pruning (RMP) to update historical observations after pose corrections while keeping long-horizon observation history bounded and combines language-conditioned 3D retrieval, open-vocabulary image detection, and multimodal large language model based semantic verification.",
  "doi": "10.48550/arXiv.2606.00576",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhijie Yan",
    "id": "2299671405",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Shufei Li",
    "id": "2326248312",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Zeqing Zhang",
    "id": "2288259321",
    "h_index": 4,
    "papers": 19
   },
   {
    "name": "Xin Liu",
    "id": "2313479324",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Yuhang Zheng",
    "id": "2239159684",
    "h_index": 9,
    "papers": 26
   },
   {
    "name": "Zuoxu Wang",
    "id": "2248170540",
    "h_index": 9,
    "papers": 21
   }
  ],
  "comment": "Code, CAD model, and real-robot demonstrations are available at https://bjhyzj.github.io/dream-web",
  "topics": [
   "spatial-3d",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2606.00576v1",
  "pdf_url": "https://arxiv.org/pdf/2606.00576v1",
  "html_url": "https://arxiv.org/html/2606.00576v1",
  "code_url": "https://bjhyzj.github.io/dream-web",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2605.31419",
  "slug": "triangle-splatting-slam",
  "title": "Triangle Splatting SLAM",
  "abstract": "We present a dense RGB-D SLAM system using differentiable triangles as the 3D map representation. While 3D Gaussian Splatting has emerged as the leading method for novel-view synthesis, triangles remain the standard primitive for traditional rendering hardware, game engines, and downstream tasks requiring explicit geometry such as simulation, collision, and editing. Recent offline methods have demonstrated that an unstructured 'triangle soup' can be optimised into a photorealistic mesh via Delaunay triangulation across a set of posed images. Building upon this insight, we present the first dense SLAM system to employ Triangle Splatting to perform both tracking and mapping through online differentiable rendering of a triangle soup. The map can be converted into a connected mesh on-the-fly via restricted Delaunay triangulation, enabling new online capabilities such as mesh deformation and collision checking. On Replica and TUM-RGBD, our system outperforms baselines on 3D geometry, matches the camera-tracking accuracy, and enables online mesh-based scene editing.",
  "published": "2026-05-29",
  "updated": "2026-06-10",
  "year": "2026",
  "authors": [
   "Nicholas Fry",
   "Eric Dexheimer",
   "Kirill Mazur",
   "Paul H. J. Kelly",
   "Andrew J. Davison"
  ],
  "author_count": 5,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work presents the first dense RGB-D SLAM system to employ Triangle Splatting to perform both tracking and mapping through online differentiable rendering of a triangle soup, enabling new online capabilities such as mesh deformation and collision checking.",
  "doi": "10.48550/arXiv.2605.31419",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Nicholas Fry",
    "id": "103442292",
    "h_index": 1,
    "papers": 9
   },
   {
    "name": "Eric Dexheimer",
    "id": "2294874927",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Kirill Mazur",
    "id": "1828774527",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Paul H. J. Kelly",
    "id": "2273662522",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Andrew J. Davison",
    "id": "2279230572",
    "h_index": 2,
    "papers": 7
   }
  ],
  "comment": "26 pages, 11 figures",
  "topics": [
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2605.31419v2",
  "pdf_url": "https://arxiv.org/pdf/2605.31419v2",
  "html_url": "https://arxiv.org/html/2605.31419v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2605.31376",
  "slug": "liftnav-path-planning-via-semantic-lifting-in-tsdf-guided-gaussian-spl",
  "title": "LiftNav: Path Planning via Semantic Lifting in TSDF-Guided Gaussian Splatting",
  "abstract": "Autonomous robots in unknown indoor environments require both reliable collision avoidance and object-level understanding. Classical representations such as TSDF support safe planning but lack semantics, while photorealistic methods like Gaussian Splatting (GS) provide rich appearance yet suffer from soft geometry, limiting precise obstacle avoidance. We present LiftNav, a hybrid navigation framework built on GSFusion's TSDF+GS dual map, augmented with a real-time pipeline of YOLO-based detection, TSDF-based 3D lifting, and B-spline trajectory optimization. This design enables flexible semantic navigation without dense 3D embeddings. We further introduce a hinge-loss-based collision penalty that improves trajectory smoothness and safety. We evaluate our approach in a simulation using the Replica dataset. Compared against a state-of-the-art radiance field baseline we show a 100% feasibility rate and shorter trajectories.",
  "published": "2026-05-29",
  "updated": "2026-05-29",
  "year": "2026",
  "authors": [
   "Hannah Schieber",
   "Dominik Frischmann",
   "Victor Schaack",
   "Angela P. Schoellig",
   "Daniel Roth"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.CV",
   "cs.GR"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "LiftNav is presented, a hybrid navigation framework built on GSFusion's TSDF+GS dual map, augmented with a real-time pipeline of YOLO-based detection, TSDF-based 3D lifting, and B-spline trajectory optimization that enables flexible semantic navigation without dense 3D embeddings.",
  "doi": "10.48550/arXiv.2605.31376",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hannah Schieber",
    "id": "2162968883",
    "h_index": 9,
    "papers": 30
   },
   {
    "name": "Dominik Frischmann",
    "id": "2379545400",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Victor Schaack",
    "id": "2243456343",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Angela P. Schoellig",
    "id": "2321572233",
    "h_index": 4,
    "papers": 20
   },
   {
    "name": "Daniel Roth",
    "id": "2211731713",
    "h_index": 8,
    "papers": 17
   }
  ],
  "comment": "",
  "topics": [
   "spatial-3d",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2605.31376v1",
  "pdf_url": "https://arxiv.org/pdf/2605.31376v1",
  "html_url": "https://arxiv.org/html/2605.31376v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2605.31321",
  "slug": "surface-constraint-policy-for-learning-surface-constrained-and-dynamic",
  "title": "Surface Constraint Policy for Learning Surface-Constrained and Dynamically Feasible Robot Skills",
  "abstract": "Diffusion-based imitation learning methods have driven rapid progress in robot dexterous manipulation tasks. However, they have limitations when applied to tasks that involve complex free-form surface constraints because of their lack of explicit surface geometry constraint modeling and the dynamic feasibility issue, resulting in stochastic action generation that fails to achieve reliable surface alignment and maintain stable contact. To address these limitations, we propose a novel surface constraint policy (SCP) for generating robot actions that satisfy free-form surface constraints on the basis of human demonstrations and real-time visual observations. First, the surface geometry constraint is encoded using a two-dimensional weighted Gaussian kernel function that is derived from demonstrations. Building on the encoded surface geometry constraints, the diffusion-based policy is used to infer task-level action intentions from multimodal sensory inputs, including visual observations and robot state feedback. These intentions are further transformed into surface-constrained dynamic movement primitives (DMPs) through a similarity-based action mapping method, thereby enabling smooth and compliant motion execution. The SCP achieves generation of structured surface geometric intent and dynamically admissible actions. The proposed method is validated on multiple surface manipulation tasks and compared with existing techniques. The experimental results demonstrate superior task success rates and contact stability under surface constraints.",
  "published": "2026-05-29",
  "updated": "2026-05-29",
  "year": "2026",
  "authors": [
   "Shuai Ke",
   "Jiexin Zhang",
   "Huan Zhao",
   "Zhiao Wei",
   "Yikun Guo",
   "Jie Pan",
   "Han Ding"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A novel surface constraint policy for generating robot actions that satisfy free-form surface constraints on the basis of human demonstrations and real-time visual observations is proposed, achieving generation of structured surface geometric intent and dynamically admissible actions.",
  "doi": "10.1109/tii.2026.3694400",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Shuai Ke",
    "id": "2344648743",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Jiexin Zhang",
    "id": "2128689860",
    "h_index": 5,
    "papers": 23
   },
   {
    "name": "Huan Zhao",
    "id": "2323580263",
    "h_index": 5,
    "papers": 41
   },
   {
    "name": "Zhiao Wei",
    "id": "2344703738",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Yi Guo",
    "id": "2313309994",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Jie Pan",
    "id": "2346905267",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Han Ding",
    "id": "2315805186",
    "h_index": 5,
    "papers": 17
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "egocentric-data",
   "imitation-diffusion"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2605.31321v1",
  "pdf_url": "https://arxiv.org/pdf/2605.31321v1",
  "html_url": "https://arxiv.org/html/2605.31321v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2605.31067",
  "slug": "seeing-fast-and-slow-bimodal-3d-scene-graphs-for-open-set-tasks",
  "title": "Seeing Fast and Slow: Bimodal 3D Scene Graphs for Open-set Tasks",
  "abstract": "Open-set task execution can significantly benefit from seamlessly switching between coarse and fine scene representations depending on the context and the evolving information as the robot explores the environment. For example, it is often sufficient to start with a coarse scene representation initially and only employ a finer, more granular scene representation when the robot encounters regions which are likely to contain the task relevant objects. Hence, in this work, we propose BiMoSG, a bimodal 3D scene graph generation approach for open-set tasks. BiMoSG employs a \"fast\" mode by default to efficiently generate a coarse 3D scene graph and can switch to a \"slow\" mode for generating a finer open vocabulary 3D scene graph of task relevant objects. We demonstrate that our proposed 3D scene graph generation approach is significantly faster than the open-source state-of-the-art approaches. This allows us to integrate the scene graph generation process with task execution for real-time deployment.",
  "published": "2026-05-29",
  "updated": "2026-06-02",
  "year": "2026",
  "authors": [
   "Marcel Bartholomeus Prasetyo",
   "Shrutika Vishal Thengane",
   "A Manicka Praveen",
   "Yi Loo",
   "Malika Meghjani"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work proposes BiMoSG, a bimodal 3D scene graph generation approach for open-set tasks that allows the scene graph generation process with task execution for real-time deployment and demonstrates that this approach is significantly faster than the open-source state-of-the-art approaches.",
  "doi": "10.48550/arXiv.2605.31067",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Marcel Bartholomeus Prasetyo",
    "id": "2202583185",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "S. Vishal",
    "id": "2439659252",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Thengane",
    "id": "2439659250",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Manickavelu Praveen",
    "id": "2308392931",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Yi Loo",
    "id": "2366075088",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Malika Meghjani",
    "id": "2376514",
    "h_index": 15,
    "papers": 50
   }
  ],
  "comment": "Submission has not been cleared with funding agency",
  "topics": [
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2605.31067v2",
  "pdf_url": "https://arxiv.org/pdf/2605.31067v2",
  "html_url": "https://arxiv.org/html/2605.31067v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2605.30780",
  "slug": "two-degree-of-freedom-vibratory-transport-in-a-grasp",
  "title": "Two Degree-of-Freedom Vibratory Transport in a Grasp",
  "abstract": "In this paper, we use asymmetric vibrations to demonstrate two degree-of-freedom (DoF) in-hand manipulation of grasped parts. The asymmetric vibrations are achieved through closed-loop position control of a moving surface, which applies a periodic stick-slip waveform to the part to be manipulated. We show analytically how two vibratory waveform parameters, the sticking acceleration and the slipping acceleration, affect average part velocity when moving against gravity. The theoretical trends are then validated using an experimental setup where the squeeze force is controlled and part motion is recorded by a high-resolution encoder. We also develop a 2-DoF vibratory surface capable of translation in one direction and rotation about the surface normal. Using two of these 2-DoF surfaces in a parallel jaw gripper configuration, we bidirectionally translate and rotate a variety of grasped parts, as well as demonstrate that the same waveform trends for translation also persist for in-plane rotation.",
  "published": "2026-05-29",
  "updated": "2026-05-29",
  "year": "2026",
  "authors": [
   "C. L. Yako",
   "Shenli Yuan",
   "Kenneth Salisbury"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "10.48550/arXiv.2605.30780",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Connor L. Yako",
    "id": "1643691790",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Shenli Yuan",
    "id": "2315661821",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "John Kenneth Salisbury",
    "id": "2237433946",
    "h_index": 2,
    "papers": 4
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2605.30780v1",
  "pdf_url": "https://arxiv.org/pdf/2605.30780v1",
  "html_url": "https://arxiv.org/html/2605.30780v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2605.30488",
  "slug": "como3r-slam-collaborative-monocular-dense-slam-with-learned-3d-reconst",
  "title": "CoMo3R-SLAM: Collaborative Monocular Dense SLAM with Learned 3D Reconstruction Priors for Outdoor Multi-Agent Systems",
  "abstract": "Collaborative dense SLAM is essential for multi-robot teams to achieve scalable and consistent 3D perception across large-scale outdoor environments. Existing systems typically depend on depth sensors, incurring significant payload, power, and calibration costs. Monocular RGB cameras are a lightweight alternative, but collaborative monocular dense SLAM remains difficult due to scale ambiguity, unreliable inter-agent data association, especially in outdoor scenes where low overlap and repetitive structures make traditional feature matching unreliable, motivating robust geometric information. We propose CoMo3R-SLAM, the first collaborative monocular dense RGB SLAM system that leverages robust learned feed-forward 3D reconstruction priors for outdoor multi-agent mapping. Each agent runs a prior-guided front-end for real-time tracking and local dense fusion, while a coordinator performs dense pointmap matching for cross-agent verification, closed-form Sim(3) gauge synchronization, and GPU-accelerated global bundle adjustment with segment-level depth optimization. Requiring neither depth sensors nor parametric intrinsics, our system produces robust cross-agent constraints and globally consistent metric maps from monocular RGB alone. On Tanks and Temples and Waymo sequences, CoMo3R-SLAM achieves the best ATE on three of four Tanks and Temples scenes and competitive Waymo accuracy, matching or exceeding state-of-the-art RGB-D methods while running online at 8 FPS.",
  "published": "2026-05-28",
  "updated": "2026-05-28",
  "year": "2026",
  "authors": [
   "Zhihao Cao",
   "Qi Shao",
   "Shuhao Zhai",
   "Feng Tian",
   "Anh Nguyen",
   "Hesheng Wang",
   "Baoru Huang"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "CoMo3R-SLAM is proposed, the first collaborative monocular dense RGB SLAM system that leverages robust learned feed-forward 3D reconstruction priors for outdoor multi-agent mapping, and produces robust cross-agent constraints and globally consistent metric maps from monocular RGB alone.",
  "doi": "10.48550/arXiv.2605.30488",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhihao Cao",
    "id": "2287602874",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Qianyi Shao",
    "id": "2386142010",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Shuhao Zhai",
    "id": "2387930993",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Fenghao Tian",
    "id": "2355105431",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Anh Nguyen",
    "id": "2312004688",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Hesheng Wang",
    "id": "2326268463",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Baoru Huang",
    "id": "2243927673",
    "h_index": 10,
    "papers": 37
   }
  ],
  "comment": "",
  "topics": [
   "spatial-3d",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2605.30488v1",
  "pdf_url": "https://arxiv.org/pdf/2605.30488v1",
  "html_url": "https://arxiv.org/html/2605.30488v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2605.30342",
  "slug": "uncertainty-driven-3d-gaussian-splatting-active-mapping-via-anisotropi",
  "title": "Uncertainty-driven 3D Gaussian Splatting Active Mapping via Anisotropic Visibility Field",
  "abstract": "We present Gaussian Splatting Anisotropic Visibility Field (GAVIS), a novel framework for uncertainty quantification and active mapping in 3DGS. Our key insight is that regions unseen from the training views yield unreliable predictions from the 3DGS. To address this, we introduce a principled and efficient method for quantifying the visibility field in 3DGS, defined as the anisotropic visibility of each particle with respect to the training views, and represented using spherical harmonics. The resulting visibility field is integrated into a Bayesian Network-based uncertainty-aware 3DGS rasterizer, enabling real-time (200 FPS) uncertainty quantification for synthesized views. Active mapping is further performed within a maximum information gain framework building on this formulation. Extensive experiments across diverse environments demonstrate that GAVIS consistently and significantly outperforms prior approaches in both accuracy and efficiency. Moreover, beyond standalone use, our method can be applied post-hoc to improve the performance of existing approaches.",
  "published": "2026-05-28",
  "updated": "2026-05-28",
  "year": "2026",
  "authors": [
   "Shangjie Xue",
   "Jesse Dill",
   "Dhruv Ahuja",
   "Frank Dellaert",
   "Panagiotis Tsiotras",
   "Danfei Xu"
  ],
  "author_count": 6,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Gaussian Splatting Anisotropic Visibility Field (GAVIS), a novel framework for uncertainty quantification and active mapping in 3DGS, is presented, demonstrating that GAVIS consistently and significantly outperforms prior approaches in both accuracy and efficiency.",
  "doi": "10.48550/arXiv.2605.30342",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Shangjie Xue",
    "id": "2264135621",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "J. Dill",
    "id": "2251315029",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Dhruv Ahuja",
    "id": "2321557051",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "F. Dellaert",
    "id": "2038264",
    "h_index": 71,
    "papers": 311
   },
   {
    "name": "Panagiotis Tsiotras",
    "id": "2272597413",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Danfei Xu",
    "id": "2265464552",
    "h_index": 8,
    "papers": 12
   }
  ],
  "comment": "Accepted to CVPR 2026. Project page https://gatech-rl2.github.io/GAVIS/",
  "topics": [
   "spatial-3d",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2605.30342v1",
  "pdf_url": "https://arxiv.org/pdf/2605.30342v1",
  "html_url": "https://arxiv.org/html/2605.30342v1",
  "code_url": "https://gatech-rl2.github.io/GAVIS/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.5
 },
 {
  "id": "2605.30226",
  "slug": "bora-bridging-offline-reinforcement-learning-and-online-residual-adapt",
  "title": "BORA: Bridging Offline Reinforcement Learning and Online Residual Adaptation for Real-World Dexterous VLA Models",
  "abstract": "Vision-Language-Action (VLA) models have emerged as a promising paradigm for grounding visual-language understanding into real-world robotic manipulation. However, dexterous manipulation remains challenging for VLA policies due to high-dimensional hand control and compounding execution errors, which makes real-world RL post-training essential for bridging the gap between visually grounded action generation and physically reliable dexterous execution. However, high-dimensional dexterous exploration often triggers temporal inconsistency, sample inefficiency and hardware risks in the real world. To address these challenges, we propose BORA, an offline-to-online RL post-training framework designed for real-world dexterous VLA models. In the offline phase, BORA constructs a critic that takes both the VLM's cognition tokens and action chunks as inputs. This design enables action-conditioned value guidance, allowing the critic to evaluate dexterous hand motions beyond visual context alone. During the subsequent online phase, BORA freezes the VLA base and introduces a lightweight, Human-in-the-Loop (HiL) chunk-wise residual adaptation mechanism to mitigate real-world execution errors and further correct the offline-learned intents within the actual physical environment. By inheriting the offline critic and employing intervention-driven rewards, BORA effectively corrects execution discrepancies and adapts to real-world physical variances while preserving the pretrained policy as a stable prior. Extensive evaluations across five complex real-world dexterous tasks demonstrate that BORA significantly outperforms pure imitation learning and traditional decoupled RL baselines, achieving a 33% absolute increase in average success rate under standard settings and up to a 43% improvement in unseen object generalization.",
  "published": "2026-05-28",
  "updated": "2026-06-06",
  "year": "2026",
  "authors": [
   "Zhongxi Chen",
   "Yifan Han",
   "Yanming Shao",
   "Huanming Liu",
   "Congsheng Xu",
   "Xiaoyu Chen",
   "Yao Mu",
   "Wenzhao Lian"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Extensive evaluations across five complex real-world dexterous tasks demonstrate that BORA significantly outperforms pure imitation learning and traditional decoupled RL baselines, achieving a 33% absolute increase in average success rate under standard settings and up to a 43% improvement in unseen object generalization.",
  "doi": "10.48550/arXiv.2605.30226",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhongxia Chen",
    "id": "2331628195",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Yifan Han",
    "id": "2323619357",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Yanming Shao",
    "id": "2329058615",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Huan Liu",
    "id": "2349337112",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Congsheng Xu",
    "id": "2276446441",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Xiaoyu Chen",
    "id": "2325107465",
    "h_index": 6,
    "papers": 20
   },
   {
    "name": "Yao Mu",
    "id": "2328078915",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Wenzhao Lian",
    "id": "2323497285",
    "h_index": 3,
    "papers": 16
   }
  ],
  "comment": "24 pages,11 figures",
  "topics": [
   "vla",
   "dexterous-manipulation",
   "imitation-diffusion",
   "rl-control",
   "navigation",
   "hri",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2605.30226v2",
  "pdf_url": "https://arxiv.org/pdf/2605.30226v2",
  "html_url": "https://arxiv.org/html/2605.30226v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2605.29879",
  "slug": "dgsg-mind-dynamic-3d-gaussian-scene-graphs-for-long-term-scene-underst",
  "title": "DGSG-Mind: Dynamic 3D Gaussian Scene Graphs for Long-Term Scene Understanding and Grounding",
  "abstract": "Integrating open-vocabulary semantic information into dynamic 3D scene representations is essential for long-term embodied scene understanding. However, existing methods often suffer from fragile instance association due to incomplete cross-view cues, while their limited ability to handle object-level topological changes restricts long-term robotic task execution. Moreover, current 3D scene understanding methods either rely on simple feature matching without explicit spatial reasoning or assume offline ground-truth 3D geometry. To address these challenges, we present DGSG-Mind, a hybrid instance-aware 3D Gaussian dynamic scene graph system with an embodied reasoning agent. Our system couples a probabilistic voxel grid with explicit 3D Gaussians to enable robust cross-modal instance fusion and incremental semantic mapping. It handles dynamic changes through Gaussian-based visual relocalization and localized masked refinement guided by geometric-semantic consistency. Built on the instance Gaussian map, DGSG-Mind further constructs a hierarchical scene graph and develops the 3D Gaussian Mind, which integrates structural relations, spatial-semantic information, and visually annotated RoI Gaussian renderings for multimodal reasoning. Extensive experiments show that DGSG-Mind achieves the best zero-shot 3DVG performance among methods operating on self-reconstructed maps, while also delivering strong performance in 3D open-vocabulary semantic segmentation and scene reconstruction. We further deploy DGSG-Mind on real-world robots to demonstrate its target-oriented reasoning and dynamic update capabilities. The project page of DGSG-Mind is available at https://icr-lab.github.io/DGSG-Mind",
  "published": "2026-05-28",
  "updated": "2026-05-29",
  "year": "2026",
  "authors": [
   "Luzhou Ge",
   "Xiangyu Zhu",
   "Jinyan Liu",
   "Xuesong Li"
  ],
  "author_count": 4,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "DGSG-Mind is presented, a hybrid instance-aware 3D Gaussian dynamic scene graph system with an embodied reasoning agent that achieves the best zero-shot 3DVG performance among methods operating on self-reconstructed maps, while also delivering strong performance in 3D open-vocabulary semantic segmentation and scene reconstruction.",
  "doi": "10.48550/arXiv.2605.29879",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Luzhou Ge",
    "id": "2346836006",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Xiangyu Zhu",
    "id": "2144103981",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Jinyan Liu",
    "id": "2352006318",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Xuesong Li",
    "id": "2346896847",
    "h_index": 1,
    "papers": 10
   }
  ],
  "comment": "9 pages, 6 figures",
  "topics": [
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2605.29879v2",
  "pdf_url": "https://arxiv.org/pdf/2605.29879v2",
  "html_url": "https://arxiv.org/html/2605.29879v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2605.28237",
  "slug": "poinav-benchmarking-and-enhancing-final-meters-arrival-in-real-world-v",
  "title": "POINav: Benchmarking and Enhancing Final-Meters Arrival in Real-World Vision-Language Navigation",
  "abstract": "Real-world navigation is fundamentally driven by Points of Interest (POIs), yet reaching a precise POI remains a critical \"final-meters\" challenge. Existing Vision-Language Navigation (VLN) benchmarks of POI-goal navigation often suffer from coarse granularity or significant sim-to-real gaps due to generated scene. To bridge this gap, we present POINav-Bench, the first benchmark designed for closed-loop evaluation of real-world POI-goal navigation. It comprises 11 commercial areas reconstructed from real-world captures using 3D Gaussian Splatting (3DGS), covering 126,398 $m^{2}$ in total and spanning 163 distinct POIs. With traversability-aware annotations and reference trajectories, POINav-Bench enables high-fidelity evaluation of navigation agents in realistic, POI-rich real-world environments. Building on this, we propose the POINav Brain-Action Framework where a Brain module performs POI-grounded reasoning to guide an Action module in predicting continuous waypoints for real-world execution. We further curate the POINav-Dataset, containing 70K real-world signage-entrance pairs. Experiments show that our framework provides a viable path toward refining real-world POI-goal navigation.",
  "published": "2026-05-27",
  "updated": "2026-05-27",
  "year": "2026",
  "authors": [
   "Ruiyan Gong",
   "Meisheng Zhang",
   "Yuxiang Zhao",
   "Mingchao Sun",
   "Yanfen Shen",
   "Zedong Chu",
   "Zhining Gu",
   "Wei Guo",
   "Xiaolong Cheng",
   "Qiming Li",
   "Kangning Niu",
   "Yanqing Zhu",
   "Xiaolong Wu",
   "Tianlun Li",
   "Mu Xu"
  ],
  "author_count": 15,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 4,
  "influential_citations": 0,
  "tldr": "The POINav-Bench is presented, the first benchmark designed for closed-loop evaluation of real-world POI-goal navigation and the POINav Brain-Action Framework is proposed, where a Brain module performs POI-grounded reasoning to guide an Action module in predicting continuous waypoints for real-world execution.",
  "doi": "10.48550/arXiv.2605.28237",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ruiyan Gong",
    "id": "2380827364",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Meisheng Zhang",
    "id": "2410955127",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Yuxiang Zhao",
    "id": "2409594876",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Mingchao Sun",
    "id": "2386126343",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Yanfen Shen",
    "id": "2394804148",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Zedong Chu",
    "id": "2333428585",
    "h_index": 7,
    "papers": 14
   },
   {
    "name": "Zhining Gu",
    "id": "2395846668",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Wei Guo",
    "id": "2271589639",
    "h_index": 15,
    "papers": 42
   },
   {
    "name": "Xiaolong Cheng",
    "id": "2149479457",
    "h_index": 3,
    "papers": 26
   },
   {
    "name": "Qiming Li",
    "id": "2309202481",
    "h_index": 3,
    "papers": 14
   },
   {
    "name": "Kangning Niu",
    "id": "2096743820",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Yanqing Zhu",
    "id": "2410224299",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Xiaolong Wu",
    "id": "2382838519",
    "h_index": 6,
    "papers": 20
   },
   {
    "name": "Tianlun Li",
    "id": "2118910472",
    "h_index": 2,
    "papers": 12
   },
   {
    "name": "Mu Xu",
    "id": "2382940270",
    "h_index": 6,
    "papers": 22
   }
  ],
  "comment": "25 pages, 9 figures",
  "topics": [
   "sim2real",
   "spatial-3d",
   "navigation",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2605.28237v1",
  "pdf_url": "https://arxiv.org/pdf/2605.28237v1",
  "html_url": "https://arxiv.org/html/2605.28237v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.7
 },
 {
  "id": "2605.27114",
  "slug": "vr-dagger-immersive-vr-for-dexterous-data-collection-and-uncertainty-g",
  "title": "VR-DAgger: Immersive VR for Dexterous Data Collection and Uncertainty-Guided On-Policy Correction",
  "abstract": "Learning from demonstrations is effective for robotic manipulation, but collecting sufficient task-specific data remains a major bottleneck. Under distribution shift, small errors compound, performance degrades, and expert time is often spent on redundant, low-value corrections instead of the few critical failure cases. We present VR-DAgger, a human-in-the-loop framework centered on an immersive VR application for dexterous teleoperation, demonstration collection, and selective policy correction. The VR client provides intuitive hand control with synchronized scene visualization, while a backend workstation runs simulation and learning, enabling autonomous rollouts without continuous operator oversight. We use Monte Carlo (MC) dropout to score uncertainty during Isaac Lab rollouts of a diffusion policy and select informative failure segments for correction. These segments are replayed in VR as clips, where the operator selectively labels and corrects the policy's behavior, concentrating supervision where uncertainty is highest without full-rollout monitoring or a separate intervention classifier. We evaluate on three dexterous manipulation tasks (Pan pick-and-place, Drawer opening, Valve turning) with a 10-DoF XHand under standard and challenging initial configurations. Active labeling consistently improves over behavioral cloning across all tasks, with gains of up to 23 percentage points. Compared to unguided human-in-the-loop inspection, VR-DAgger reduces per-sample collection time by approximately 40% by focusing review on selected segments rather than full rollouts.",
  "published": "2026-05-26",
  "updated": "2026-05-28",
  "year": "2026",
  "authors": [
   "Ren\u00e9 Zurbr\u00fcgg",
   "Tifanny Portela",
   "Arjun Bhardwaj",
   "Aravind Elanjimattathil Vijayan",
   "Maximum Wilder-Smith",
   "Marco Hutter"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "VR-DAgger, a human-in-the-loop framework centered on an immersive VR application for dexterous teleoperation, demonstration collection, and selective policy correction, uses Monte Carlo (MC) dropout to score uncertainty during Isaac Lab rollouts of a diffusion policy and select informative failure segments for correction.",
  "doi": "10.48550/arXiv.2605.27114",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ren\u00e9 Zurbr\u00fcgg",
    "id": "2376189695",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Tifanny Portela",
    "id": "2299325952",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Arjun Bhardwaj",
    "id": "2061543783",
    "h_index": 3,
    "papers": 16
   },
   {
    "name": "A. E. Vijayan",
    "id": "51301522",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Maximum Wilder-Smith",
    "id": "2288687972",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Marco Hutter",
    "id": "2381057021",
    "h_index": 2,
    "papers": 10
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "imitation-diffusion",
   "data-teleop",
   "hri"
  ],
  "orgs": [
   "NVIDIA"
  ],
  "abs_url": "https://arxiv.org/abs/2605.27114v2",
  "pdf_url": "https://arxiv.org/pdf/2605.27114v2",
  "html_url": "https://arxiv.org/html/2605.27114v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.8
 },
 {
  "id": "2605.26831",
  "slug": "osma-bench-toward-open-ended-benchmarking-of-semantic-mapping-for-mani",
  "title": "OSMa-Bench++: Toward Open-Ended Benchmarking of Semantic Mapping for Manipulation with Prompt-Generated Synthetic Scenes",
  "abstract": "Semantic mapping methods are increasingly used as intermediate scene representations for downstream robotic reasoning and manipulation, yet their evaluation is still largely tied to fixed benchmark datasets with limited coverage of manipulation-relevant corner cases. In this work, we extend OSMa-Bench toward controllable benchmarking with prompt-generated synthetic indoor scenes. Our pipeline automatically generates scene descriptions, synthesizes corresponding environments with SceneSmith, and adapts the resulting assets into an OSMa-Bench-compatible simulation format. This adaptation requires a nontrivial intermediate layer, including semantic normalization, material and texture repair, shader fallback policies, floor handling, navigation setup, and controlled lighting configuration. A key advantage of the proposed setup is that the original scene-generation prompt is known in advance and can therefore serve as an auxiliary semantic specification of the intended scene. We use this property to extend the VQA component of OSMa-Bench with a prompt-grounded question category. The resulting framework supports targeted stress-testing of semantic scene representations under conditions such as clutter, small objects, partial occlusions, and lighting variation, and makes benchmarking more extensible and better aligned with downstream manipulation requirements. Our code is available at https://github.com/be2rlab/OSMa-Bench-v2.",
  "published": "2026-05-26",
  "updated": "2026-05-26",
  "year": "2026",
  "authors": [
   "Regina Kurkova",
   "Maxim Popov",
   "Sergey Kolyubin"
  ],
  "author_count": 3,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work extends OSMa-Bench toward controllable benchmarking with prompt-generated synthetic indoor scenes with a key advantage of the proposed setup is that the original scene-generation prompt is known in advance and can therefore serve as an auxiliary semantic specification of the intended scene.",
  "doi": "10.48550/arXiv.2605.26831",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "R. Kurkova",
    "id": "48154499",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "M. Popov",
    "id": "2349801615",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "S. Kolyubin",
    "id": "2273991155",
    "h_index": 3,
    "papers": 17
   }
  ],
  "comment": "Code: https://github.com/be2rlab/OSMa-Bench-v2",
  "topics": [
   "spatial-3d",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2605.26831v1",
  "pdf_url": "https://arxiv.org/pdf/2605.26831v1",
  "html_url": "https://arxiv.org/html/2605.26831v1",
  "code_url": "https://github.com/be2rlab/OSMa-Bench-v2",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2605.26478",
  "slug": "efficient-on-policy-visual-rl-via-stochastic-decoupled-policy-gradient",
  "title": "Efficient On-policy Visual-RL via Stochastic Decoupled Policy Gradient",
  "abstract": "We present the stochastic decoupled policy gradient (SDPG), a lightweight visual reinforcement learning (RL) method that trains diverse visuomotor control policies end-to-end within a few hours on a single NVIDIA RTX 4080 GPU. SDPG estimates policy gradients via random perturbations of trajectory rollouts, requiring orders of magnitude fewer batch-rendered environments and substantially reducing compute and memory overhead. On visual MuJoCo benchmarks, SDPG consistently outperforms baseline methods in training time, memory usage, and rewards. Finally, to support future research, we introduce a suite of realistic visual robotics benchmarks spanning dexterous manipulation, challenging locomotion, and demonstrate effective sim-to-real transfer on physical hardware.",
  "published": "2026-05-26",
  "updated": "2026-05-26",
  "year": "2026",
  "authors": [
   "Haoxiang You",
   "Yilang Liu",
   "Davis Zong",
   "Qian Wang",
   "Teeratham Vitchutripop",
   "Qi Wang",
   "Daniel Rakita",
   "Ian Abraham"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "10.48550/arXiv.2605.26478",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Haoxiang You",
    "id": "2348475726",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Yilang Liu",
    "id": "3437745",
    "h_index": 17,
    "papers": 51
   },
   {
    "name": "Davis Zong",
    "id": "2438737849",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Qian Wang",
    "id": "2392359414",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Teeratham Vitchutripop",
    "id": "2402701123",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Qi Wang",
    "id": "2315272896",
    "h_index": 2,
    "papers": 11
   },
   {
    "name": "Daniel Rakita",
    "id": "2356784657",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Ian Abraham",
    "id": "2348477039",
    "h_index": 1,
    "papers": 5
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "humanoids",
   "sim2real",
   "rl-control"
  ],
  "orgs": [
   "NVIDIA"
  ],
  "abs_url": "https://arxiv.org/abs/2605.26478v1",
  "pdf_url": "https://arxiv.org/pdf/2605.26478v1",
  "html_url": "https://arxiv.org/html/2605.26478v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2605.26113",
  "slug": "anyscene-towards-highly-controllable-driving-scene-generation-at-anywh",
  "title": "AnyScene: Towards Highly Controllable Driving Scene Generation at Anywhere and Beyond",
  "abstract": "Generating high-fidelity and controllable synthetic data is critical for advancing end-to-end autonomous driving, particularly for addressing the long tail of rare safety-critical scenarios. Existing occupancy-guided methods typically rely on shallow conditioning mechanisms and reference-frame-dependent video synthesis, which limits fine-grained controllability from arbitrary BEV layouts and restricts their applicability for scalable simulation. In this paper, we propose AnyScene, a unified occupancy-centric framework for driving scene generation. AnyScene generates semantic occupancy sequences from BEV layouts through a Spatial-Temporal Occupancy Diffusion Transformer that jointly tokenizes BEV and occupancy features in an autoregressive manner. This design enables precise controllability from cross-dataset and user-defined BEV inputs while naturally supporting long-horizon generation. Building upon the generated occupancy, a Geometry-Grounded View Expansion module treats occupancy as the canonical spatial representation and synthesizes temporally consistent multi-view driving videos in a reference-free and autoregressive fashion, supporting flexible camera configurations at inference time. Extensive experiments demonstrate that AnyScene achieves state-of-the-art performance in both occupancy and video generation. It exhibits strong generalization to unseen and customized layouts, and provides measurable benefits for downstream tasks such as sparse-view 3D reconstruction.",
  "published": "2026-05-25",
  "updated": "2026-05-25",
  "year": "2026",
  "authors": [
   "Haiming Zhang",
   "Junfei Zhou",
   "Feng Jiang",
   "Jingzhong Li",
   "Zhenglong Guo",
   "Penglin Dai",
   "Jifeng Dai",
   "Yan Xie",
   "Benjin Zhu"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "AnyScene, a unified occupancy-centric framework for driving scene generation, is proposed that achieves state-of-the-art performance in both occupancy and video generation, exhibits strong generalization to unseen and customized layouts, and provides measurable benefits for downstream tasks such as sparse-view 3D reconstruction.",
  "doi": "10.48550/arXiv.2605.26113",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Haiming Zhang",
    "id": "2261887052",
    "h_index": 7,
    "papers": 17
   },
   {
    "name": "Junfei Zhou",
    "id": "2351078910",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Fengze Jiang",
    "id": "2310794591",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Jingzhong Li",
    "id": "2254820155",
    "h_index": 6,
    "papers": 18
   },
   {
    "name": "Zhenglong Guo",
    "id": "2395869147",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Peng Dai",
    "id": "2256630164",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Jifeng Dai",
    "id": "2325954167",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Yanzhe Xie",
    "id": "2438801639",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Benjin Zhu",
    "id": "32109772",
    "h_index": 8,
    "papers": 13
   }
  ],
  "comment": "Work in progress. Project page: https://mind-omni.github.io/",
  "topics": [
   "spatial-3d",
   "navigation",
   "video-generation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2605.26113v1",
  "pdf_url": "https://arxiv.org/pdf/2605.26113v1",
  "html_url": "https://arxiv.org/html/2605.26113v1",
  "code_url": "https://mind-omni.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2605.25029",
  "slug": "parkingworld-end-to-end-autonomous-parking-reinforcement-learning-from",
  "title": "ParkingWorld: End-to-End Autonomous Parking Reinforcement Learning from Corrective Experience in 3DGS Simulation",
  "abstract": "Autonomous parking demands precise low-speed maneuvering within narrow, cluttered, and highly constrained environments, where vehicles must navigate tight spaces while avoiding static obstacles and complex geometric boundaries. Unlike imitation learning, which typically requires massive volumes of high-quality expert demonstrations to converge to a stable policy and often suffers from limited generalization to unseen scenarios, traditional reinforcement learning (RL) methods face persistent challenges including excessive training overhead, inefficient exploration, and even failure to learn viable parking strategies in challenging settings. To address these limitations, this paper presents a correction-in-the-loop sample-efficient reinforcement learning (CIL-SERL) framework for end-to-end autonomous parking, which is entirely trained in a photorealistic 3D Gaussian Splatting (3DGS) parking simulator that enables high-fidelity digital reconstruction of real-world scenes. Inspired by error-correction notebooks used in learning practice, we design a novel multi-level replay buffer mechanism. These buffers hierarchically organize and store standard RL rollouts, human corrective interventions, failed exploration trajectories, and rollback-based correction segments in separate yet interconnected memory regions, facilitating structured sampling and targeted learning during training. The proposed framework is systematically evaluated in both the 3DGS simulation environment and a physical vehicle platform. Extensive experimental results demonstrate that our method achieves substantial improvements in parking success rate, operational efficiency, and safety performance across diverse scenarios, validating the effectiveness and practical applicability of the proposed CIL-SERL-based end-to-end autonomous parking solution.",
  "published": "2026-05-24",
  "updated": "2026-05-26",
  "year": "2026",
  "authors": [
   "Zhengcheng Yu",
   "Changze Li",
   "Haoran Liu",
   "Tong Qin"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A correction-in-the-loop sample-efficient reinforcement learning (CIL-SERL) framework for end-to-end autonomous parking, which is entirely trained in a photorealistic 3D Gaussian Splatting (3DGS) parking simulator that enables high-fidelity digital reconstruction of real-world scenes.",
  "doi": "10.48550/arXiv.2605.25029",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhengcheng Yu",
    "id": "2449447387",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Changze Li",
    "id": "2308415874",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Haoran Liu",
    "id": "2333402095",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Tong Qin",
    "id": "2310826338",
    "h_index": 7,
    "papers": 26
   }
  ],
  "comment": "9 pages(including 1 page of Appendix), 6 figures. Will be submitted to RA-L 2026",
  "topics": [
   "sim2real",
   "imitation-diffusion",
   "rl-control",
   "spatial-3d",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2605.25029v2",
  "pdf_url": "https://arxiv.org/pdf/2605.25029v2",
  "html_url": "https://arxiv.org/html/2605.25029v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2605.24934",
  "slug": "humanego-zero-shot-robot-learning-from-minutes-of-human-egocentric-vid",
  "title": "HumanEgo: Zero-Shot Robot Learning from Minutes of Human Egocentric Videos",
  "abstract": "Human egocentric video captures rich manipulation demonstrations without any robot hardware, yet transferring these skills to robots remains challenging due to the embodiment gap between human and robot in both visual appearance and kinematics. We present HumanEgo, a framework that bridges the embodiment gap by lifting each human demonstration to an entity-level representation of hand-object interaction, and training a flow matching policy with dense auxiliary objectives that amplify supervision from every trajectory. HumanEgo is robot-data-free, hardware-agnostic, data-efficient, and zero-shot human-to-robot transferable. With only 30 minutes of human videos per task, HumanEgo achieves 92.5% average success across four real-world tasks (75% with just 15 minutes), outperforms matched-time robot teleoperation by 41%, and robustly transfers zero-shot across novel robots, cameras, and environments. We release HumanEgo as an easy-to-use, open-source framework for learning robot policies directly from human data: https://github.com/TX-Leo/HumanEgo",
  "published": "2026-05-24",
  "updated": "2026-05-28",
  "year": "2026",
  "authors": [
   "Zhi Wang",
   "Botao He",
   "Kelin Yu",
   "Seungjae Lee",
   "Ruohan Gao",
   "Furong Huang",
   "Yiannis Aloimonos"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 9,
  "influential_citations": 1,
  "tldr": "This work presents HumanEgo, a framework that bridges the embodiment gap by lifting each human demonstration to an entity-level representation of hand-object interaction, and training a flow matching policy with dense auxiliary objectives that amplify supervision from every trajectory.",
  "doi": "10.48550/arXiv.2605.24934",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhi Wang",
    "id": "2282565866",
    "h_index": 3,
    "papers": 15
   },
   {
    "name": "Botao He",
    "id": "2296438839",
    "h_index": 5,
    "papers": 16
   },
   {
    "name": "Kelin Yu",
    "id": "2376516296",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Seungjae Lee",
    "id": "2153321718",
    "h_index": 6,
    "papers": 19
   },
   {
    "name": "Ruohan Gao",
    "id": "2376472371",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Furong Huang",
    "id": "2347721825",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Y. Aloimonos",
    "id": "1697493",
    "h_index": 55,
    "papers": 418
   }
  ],
  "comment": "Project page: https://humanego-ai.github.io",
  "topics": [
   "egocentric-data",
   "imitation-diffusion",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2605.24934v2",
  "pdf_url": "https://arxiv.org/pdf/2605.24934v2",
  "html_url": "https://arxiv.org/html/2605.24934v2",
  "code_url": "https://humanego-ai.github.io",
  "presented": [
   {
    "club": "saturday-robotics",
    "club_name": "Saturday Robotics",
    "session_number": 13,
    "session_title": "Robotics & World Models Reading Club 13: HumanEgo: Train Robot Policy from 30 min Egocentric Videos \u2014 SF 0620",
    "date_text": "Saturday, June 20, 2026 | 2:00 PM \u2013 5:00 PM",
    "city": "San Francisco, CA",
    "url": "https://lu.ma/6vkhxnum",
    "listed_as": ""
   }
  ],
  "club_note": "Code: https://github.com/TX-Leo/HumanEgo",
  "featured": true,
  "signal": 5.0
 },
 {
  "id": "2605.24642",
  "slug": "understanding-the-impact-of-geometric-foundation-models-on-vision-lang",
  "title": "Understanding the Impact of Geometric Foundation Models on Vision-Language-Action Models",
  "abstract": "Recent work explores new opportunities at the intersection of vision-language-action models (VLAs) and geometric foundation models (GFMs) for 3D reconstruction, such as VGGT. While the resulting geometric VLAs often show improved performance, it remains unclear (i) if modern VLAs already have sufficient geometric understanding to start with, (ii) what is the best architecture to inject geometric understanding into a VLA, and (iii) what is the effect of other design choices that affect geometric VLAs. In this paper we provide a rigorous experimental analysis to shed light on these questions, for a specific choice of VLA (GR00T-N1.5) and GFM (VGGT). Our first contribution is to formalize prior work's intuition that current VLAs lack geometric understanding, by providing a rigorous analysis based on linear probing. The analysis quantifies, for the first time, the \"geometric gap\" between VLAs and GFMs. Our second contribution is to identify and compare different strategies to bridge GFMs with VLAs. We implement three different architectures, which differ in the way they inject geometry in the VLA, while keeping low-level implementation details as similar as possible, to ensure a fair comparison. Finally, we analyze the impact of non-architectural choices (e.g., training data, number of cameras, reconstruction quality) on the performance of the geometric VLAs.",
  "published": "2026-05-23",
  "updated": "2026-05-23",
  "year": "2026",
  "authors": [
   "Yurou Yang",
   "Muyuan Lin",
   "Roberto Martin-Martin",
   "Martin Labrie",
   "Shreekant Gayaka",
   "Cheng-Hao Kuo",
   "Luca Carlone"
  ],
  "author_count": 7,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 2,
  "influential_citations": 1,
  "tldr": "A rigorous experimental analysis quantifies the \"geometric gap\" between VLAs and GFMs, and implements three different architectures, which differ in the way they inject geometry in the VLA, while keeping low-level implementation details as similar as possible, to ensure a fair comparison.",
  "doi": "10.48550/arXiv.2605.24642",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yurou Yang",
    "id": "2300422915",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Muyuan Lin",
    "id": "2148634482",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Roberto Mart\u00edn-Mart\u00edn",
    "id": "2334889156",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Martin Labrie",
    "id": "2315509822",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Shreekant Gayaka",
    "id": "3171086",
    "h_index": 12,
    "papers": 25
   },
   {
    "name": "Cheng-Hao Kuo",
    "id": "2259972955",
    "h_index": 7,
    "papers": 20
   },
   {
    "name": "Luca Carlone",
    "id": "2239485876",
    "h_index": 11,
    "papers": 37
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "spatial-3d",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2605.24642v1",
  "pdf_url": "https://arxiv.org/pdf/2605.24642v1",
  "html_url": "https://arxiv.org/html/2605.24642v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.48
 },
 {
  "id": "2605.21811",
  "slug": "safe-and-steerable-geometric-motion-policies-for-robotic-dexterous-man",
  "title": "Safe and Steerable Geometric Motion Policies for Robotic Dexterous Manipulation",
  "abstract": "Robotic dexterous manipulation requires continuously reconciling objectives and constraints defined on heterogeneous geometric spaces: a robot controlled on a $\\mathbb{R}^7$ configuration manifold may need to track end effector poses on $\\mathrm{SE}(3)$ while satisfying obstacle avoidance margins in $\\mathbb{R}$. We present Safe Pullback Bundle Dynamical Systems (SafePBDS), a geometrically consistent framework that computes optimal, certifiably safe configuration manifold accelerations from objectives and safety requirements on arbitrary task manifolds. SafePBDS builds on prior work that combines predefined task manifold dynamical systems to produce autonomous motion. Its first innovation is a pullback control barrier function construction, which converts task manifold safety conditions into linear constraints on configuration manifold accelerations. The second innovation is a task manifold action interface that allows a high-level policy to inject low dimensional residual motions; zero input recovers the autonomous behavior, while safety is preserved under arbitrary inputs. This lets high-level policies efficiently steer exploration while leaving precise motion to the autonomous behavior. We validate SafePBDS in simulation and on a 23-DOF Franka Panda-Allegro Hand platform. On dexterous grasping, SafePBDS achieves a $92.5\\%$ success rate across 20 household objects and 120 trials. Using the action interface, the method can exclude any one of the four fingers during grasping via a one-dimensional action, achieving $94.4\\%$ 3-finger grasp success across 3 objects and 36 trials. The efficient planning and safety guarantee of SafePBDS also enables the first model-based, fully actuated palm-down in-hand reorientation, exceeding $360^\\circ$ of yaw rotation in both directions under varying object weight and wrist motion. Demo video and details: https://tml.stanford.edu/safe-pbds",
  "published": "2026-05-20",
  "updated": "2026-05-20",
  "year": "2026",
  "authors": [
   "Albert Wu",
   "Riccardo Bonalli",
   "Thomas Lew",
   "C. Karen Liu"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "A geometrically consistent framework that computes optimal, certifiably safe configuration manifold accelerations from objectives and safety requirements on arbitrary task manifolds, and enables the first model-based, fully actuated palm-down in-hand reorientation.",
  "doi": "10.48550/arXiv.2605.21811",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Albert Wu",
    "id": "2296000404",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Riccardo Bonalli",
    "id": "2301204487",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "T. Lew",
    "id": "49989997",
    "h_index": 17,
    "papers": 42
   },
   {
    "name": "C. K. Liu",
    "id": "2278583770",
    "h_index": 5,
    "papers": 7
   }
  ],
  "comment": "24 pages, 10 figures, 5 tables. Project page and demo video: https://tml.stanford.edu/safe-pbds",
  "topics": [
   "dexterous-manipulation",
   "rl-control",
   "navigation",
   "safety-eval"
  ],
  "orgs": [
   "Stanford"
  ],
  "abs_url": "https://arxiv.org/abs/2605.21811v1",
  "pdf_url": "https://arxiv.org/pdf/2605.21811v1",
  "html_url": "https://arxiv.org/html/2605.21811v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.5
 },
 {
  "id": "2605.21330",
  "slug": "learning-robust-dexterous-in-hand-manipulation-from-joint-sensors-with",
  "title": "Learning Robust Dexterous In-Hand Manipulation from Joint Sensors with Proprioceptive Transformer",
  "abstract": "In-hand object manipulation is a fundamental yet challenging capability for dexterous robots. Despite significant progress in dexterous manipulation, existing approaches rely heavily on vision or tactile sensing to track object states, while joint sensing -- the most readily available modality on any robotic hand -- remains largely overlooked, particularly for tendon-driven hands. In this paper, we study how far joint sensing alone can go by asking: (i) whether motor encoders or direct joint sensing provides better proprioceptive feedback, (ii) how to extract environment information from joint measurements, and (iii) whether joint-only control can achieve competitive real-world performance without external perception. We present the Proprioceptive Transformer (PT), an exteroceptive-free approach for continuous cube rotation on a tendon-driven dexterous hand that uses only joint sensing feedback. A teacher policy is first trained via reinforcement learning with privileged object information, then distilled into PT, which operates solely on joint position and velocity histories. The Transformer architecture effectively extracts implicit object state information from temporal patterns in joint sensor readings. Experiments on the real ORCA hand show that our approach achieves 3.1x higher rotation speed than baselines. We also demonstrate that our PT achieves a 23.4% lower RMSE for cube position estimation than the MLP baseline, indicating superior extraction of exteroceptive information from proprioceptive sources.",
  "published": "2026-05-20",
  "updated": "2026-05-20",
  "year": "2026",
  "authors": [
   "Senlan Yao",
   "Chenyu Yang",
   "Jaehoon Kim",
   "Aristotelis Sympetheros",
   "Robert K. Katzschmann"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "How far joint sensing alone can go is studied by asking whether motor encoders or direct joint sensing provides better proprioceptive feedback, how to extract environment information from joint measurements, and whether joint-only control can achieve competitive real-world performance without external perception.",
  "doi": "10.48550/arXiv.2605.21330",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Senlan Yao",
    "id": "2438694786",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "Chenyu Yang",
    "id": "1641363976",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Jaehoon Kim",
    "id": "2218470898",
    "h_index": 6,
    "papers": 21
   },
   {
    "name": "Aristotelis Sympetheros",
    "id": "2354180004",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Robert K. Katzschmann",
    "id": "50191333",
    "h_index": 26,
    "papers": 87
   }
  ],
  "comment": "8 pages, 6 figures, 3 tables",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2605.21330v1",
  "pdf_url": "https://arxiv.org/pdf/2605.21330v1",
  "html_url": "https://arxiv.org/html/2605.21330v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2605.19104",
  "slug": "neural-operators-for-design-space-surrogate-modeling-of-tendon-actuate",
  "title": "Neural Operators for Design-Space Surrogate Modeling of Tendon-Actuated Continuum Robots",
  "abstract": "Continuum robots enable dexterous manipulation in constrained environments, but require accurate and efficient models for real-time manipulation and control. Traditional physics-based models can be computationally expensive and may suffer from inaccuracies due to unmodeled effects, while current learning-based methods often generalize poorly beyond the specific robot on which they are trained. We present a formulation of surrogate modeling for tendon-driven continuum robots as an operator learning problem that maps robot design parameters and tendon actuation inputs to resulting configurations. This formulation enables a single trained model to generalize across a large class of robot designs. We develop four novel neural operator architectures--two based on Deep Operator Networks (DeepONets) and two based on Fourier Neural Operators (FNOs)--and train them on simulation data to predict robot configurations. All architectures achieve good accuracy while allowing for fast and accurate generalization across designs. Our results demonstrate that operator learning provides an effective and generalizable surrogate for continuum robot mechanics in the design space, enabling fast modeling for control, planning, and design optimization in surgical and industrial applications.",
  "published": "2026-05-18",
  "updated": "2026-05-18",
  "year": "2026",
  "authors": [
   "Branden Frieden",
   "James M. Ferguson",
   "Alan Kuntz",
   "Varun Shankar"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work develops four novel neural operator architectures--two based on Deep Operator Networks (DeepONets) and two based on Fourier Neural Operators (FNOs)--and train them on simulation data to predict robot configurations, allowing for fast and accurate generalization across designs.",
  "doi": "10.48550/arXiv.2605.19104",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "B. Frieden",
    "id": "2259927010",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "James M. Ferguson",
    "id": "2391814581",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Alan Kuntz",
    "id": "2301247698",
    "h_index": 2,
    "papers": 13
   },
   {
    "name": "Varun Shankar",
    "id": "2304554737",
    "h_index": 3,
    "papers": 10
   }
  ],
  "comment": "Accepted to ICRA 2026",
  "topics": [
   "dexterous-manipulation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2605.19104v1",
  "pdf_url": "https://arxiv.org/pdf/2605.19104v1",
  "html_url": "https://arxiv.org/html/2605.19104v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2605.18727",
  "slug": "dexholdem-playing-texas-hold-em-with-dexterous-embodied-system",
  "title": "DexHoldem: Playing Texas Hold'em with Dexterous Embodied System",
  "abstract": "Evaluating embodied systems on real dexterous hardware requires more than isolated primitive skills: an agent must perceive a changing tabletop scene, choose a context-appropriate action, execute it with a dexterous hand, and leave the scene usable for later decisions. We introduce DexHoldem, a real-world system-level benchmark built around Texas Hold'em dexterous manipulation with a ShadowHand. DexHoldem provides 1,470 teleoperated demonstrations across 14 Texas Hold'em manipulation primitives, a standardized physical policy benchmark, and an agentic perception benchmark that tests whether agents can recover the structured game state needed for embodied decision making. On primitive execution, $\u03c0_{0.5}$ obtains the highest task completion rate ($61.2\\%$), while $\u03c0_{0.5}$ and $\u03c0_0$ tie on scene-preserving success rate ($47.5\\%$). On agentic perception, Opus 4.7 obtains the best strict problem-level accuracy ($34.3\\%$), while GPT 5.5 obtains the best average field-wise accuracy ($66.8\\%$), exposing a gap between isolated visual sub-capabilities and complete routing-relevant state recovery. Finally, we instantiate the full embodied-agent loop in three case studies, where waiting, recovery dispatches, human-help requests, and repeated primitive execution reveal how perception and policy errors accumulate during closed-loop deployment. DexHoldem therefore evaluates dexterous tabletop execution, agentic perception, and embodied decision routing in a shared physical setting. Project page: https://dexholdem.github.io/Dexholdem/.",
  "published": "2026-05-18",
  "updated": "2026-05-18",
  "year": "2026",
  "authors": [
   "Feng Chen",
   "Tianzhe Chu",
   "Li Sun",
   "Pei Zhou",
   "Zhuxiu Xu",
   "Shenghua Gao",
   "Yuexiang Zhai",
   "Yanchao Yang",
   "Yi Ma"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "DexHoldem, a real-world system-level benchmark built around Texas Hold'em dexterous manipulation with a ShadowHand, is introduced, which evaluates dexterous tabletop execution, agentic perception, and embodied decision routing in a shared physical setting.",
  "doi": "10.48550/arXiv.2605.18727",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Feng Chen",
    "id": "2377276163",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Tianzhe Chu",
    "id": "2218725693",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Li Sun",
    "id": "2377530432",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Pei Zhou",
    "id": "2313269953",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Zhuxiu Xu",
    "id": "2308983686",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Shenghua Gao",
    "id": "2285702784",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Yuexiang Zhai",
    "id": "119692515",
    "h_index": 19,
    "papers": 30
   },
   {
    "name": "Yanchao Yang",
    "id": "2321228523",
    "h_index": 2,
    "papers": 12
   },
   {
    "name": "Yi Ma",
    "id": "2255716385",
    "h_index": 7,
    "papers": 14
   }
  ],
  "comment": "30 Pages",
  "topics": [
   "dexterous-manipulation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2605.18727v1",
  "pdf_url": "https://arxiv.org/pdf/2605.18727v1",
  "html_url": "https://arxiv.org/html/2605.18727v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2605.18722",
  "slug": "dexora-open-source-vla-for-high-dof-bimanual-dexterity",
  "title": "Dexora: Open-source VLA for High-DoF Bimanual Dexterity",
  "abstract": "Vision-Language-Action (VLA) models have recently become a central direction in embodied AI, but current systems are restricted to either dual-gripper control or single-arm dexterous hand manipulation. While low-dimensional gripper control can often be handled with simpler methods, high-dimensional dexterous hand control benefits greatly from full end-to-end VLA learning. In this work, we introduce Dexora, the first open-source VLA system that natively targets dual-arm, dual-hand high-DoF manipulation. We design a hybrid teleoperation pipeline that decouples gross arm kinematics (captured with a custom exoskeleton backpack) from fine finger motion (markerless hand tracking via Apple Vision Pro), and that drives both a physical dual-arm dual-hand platform and an identical MuJoCo digital twin. Using that interface, we assemble a large training corpus: an embodiment-matched synthetic corpus (100K simulated trajectories, 6.5M frames) and a real-world dataset of 10K teleoperated episodes (2.92M frames). To mitigate noisy teleoperation demonstrations, we propose a data-quality-aware training recipe: an offline discriminator provides clip-level weights for diffusion-transformer policy training, down-weighting low-quality demonstrations. Empirically, Dexora outperforms competitive VLA baselines on both basic and dexterous benchmarks (e.g., average dexterous success 66.7% vs. 51.7%), attains 90% success on basic tasks, and shows robust out-of-distribution and cross-embodiment generalization. Ablations confirm the importance of real data and the discriminator for dexterity.",
  "published": "2026-05-18",
  "updated": "2026-05-18",
  "year": "2026",
  "authors": [
   "Zongzheng Zhang",
   "Jingrui Pang",
   "Zhuo Yang",
   "Kun Li",
   "Minwen Liao",
   "Saining Zhang",
   "Guoxuan Chi",
   "Jinbang Guo",
   "Huan-ang Gao",
   "Modi Shi",
   "Dongyun Ge",
   "Yao Mu",
   "Jiayuan Gu",
   "Rui Chen",
   "Hao Dong",
   "Huazhe Xu",
   "Li Yi",
   "Yixin Zhu",
   "Hang Zhao",
   "Pengwei Wang",
   "Shanghang Zhang",
   "Guocai Yao",
   "Jianyu Chen",
   "Hongyang Li",
   "Hao Zhao"
  ],
  "author_count": 25,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA 2026",
  "venue_source": "arxiv-comment",
  "citations": 6,
  "influential_citations": 0,
  "tldr": "",
  "doi": "10.48550/arXiv.2605.18722",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zongzheng Zhang",
    "id": "2294931371",
    "h_index": 7,
    "papers": 22
   },
   {
    "name": "Jin-Li Pang",
    "id": "2328913612",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Zhuo Yang",
    "id": "2300137968",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Kunyang Li",
    "id": "2332577862",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Minwen Liao",
    "id": "2438719979",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Saining Zhang",
    "id": "2233844460",
    "h_index": 7,
    "papers": 28
   },
   {
    "name": "Guoxuan Chi",
    "id": "2349493736",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Jinbang Guo",
    "id": "2438710168",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Huan-ang Gao",
    "id": "2221148510",
    "h_index": 14,
    "papers": 33
   },
   {
    "name": "Modi Shi",
    "id": "2292215295",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Dongyun Ge",
    "id": "2312967128",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Yao Mu",
    "id": "2348161293",
    "h_index": 2,
    "papers": 18
   },
   {
    "name": "Jiayuan Gu",
    "id": "2347843301",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Rui Chen",
    "id": "2148926393",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Hao Dong",
    "id": "2292234604",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Huazhe Xu",
    "id": "2283877819",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Li Yi",
    "id": "2286064445",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Yixin Zhu",
    "id": "2290017516",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Hang Zhao",
    "id": "2284724822",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Pengwei Wang",
    "id": "2338357829",
    "h_index": 17,
    "papers": 44
   },
   {
    "name": "Shanghang Zhang",
    "id": "2376781333",
    "h_index": 3,
    "papers": 17
   },
   {
    "name": "Guocai Yao",
    "id": "2376597393",
    "h_index": 6,
    "papers": 23
   },
   {
    "name": "Jianyu Chen",
    "id": "2243372539",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Hongyang Li",
    "id": "2303918926",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Hao Zhao",
    "id": "2365487717",
    "h_index": 5,
    "papers": 21
   }
  ],
  "comment": "Accpeted by ICRA 2026",
  "topics": [
   "vla",
   "dexterous-manipulation",
   "foundation-pretraining",
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2605.18722v1",
  "pdf_url": "https://arxiv.org/pdf/2605.18722v1",
  "html_url": "https://arxiv.org/html/2605.18722v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.35
 },
 {
  "id": "2605.17929",
  "slug": "tacse3-equivariant-se-3-motion-estimation-from-low-texture-visuotactil",
  "title": "TacSE3: Equivariant SE(3) Motion Estimation from Low-Texture Visuotactile Images for In-Gripper Tracking and Compensation",
  "abstract": "Robotic in-hand manipulation requires reliable object-motion tracking under frequent visual occlusion, yet low-texture visuotactile images provide few stable correspondences for conventional image- or geometry-matching methods. This paper presents TacSE3, a tactile motion-estimation pipeline that converts low-texture visuotactile observations into a decoupled three-dimensional force field and estimates incremental rigid-body motion on SE(3). The method derives planar translation from contact-centroid motion and estimates rotation primarily from shear-related tactile responses, yielding a physically interpretable signal for in-gripper tracking and compensation. Experiments with paired DM-Tac fingertip sensors show that dual-sensor sensing reduces translation-rotation ambiguity, supports rotation tracking across axes and object geometries, and provides a lightweight compensation signal that improves disturbance tolerance in downstream manipulation tasks without retraining the base policy.",
  "published": "2026-05-18",
  "updated": "2026-05-27",
  "year": "2026",
  "authors": [
   "Zhongyuan Liao",
   "Junzhe Wang",
   "Qingyang Liu",
   "Zhenmin Huang",
   "Jun Ma",
   "Yi Cai",
   "Fei Meng",
   "Haobo Liang",
   "Michael Yu Wang"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Experiments with paired DM-Tac fingertip sensors show that dual-sensor sensing reduces translation-rotation ambiguity, supports rotation tracking across axes and object geometries, and provides a lightweight compensation signal that improves disturbance tolerance in downstream manipulation tasks without retraining the base policy.",
  "doi": "10.48550/arXiv.2605.17929",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhongyuan Liao",
    "id": "2303291657",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Junzhe Wang",
    "id": "2110116233",
    "h_index": 0,
    "papers": 2
   },
   {
    "name": "Qingyang Liu",
    "id": "2349226228",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Zhen Huang",
    "id": "2151326313",
    "h_index": 6,
    "papers": 22
   },
   {
    "name": "Jun Ma",
    "id": "2278831230",
    "h_index": 6,
    "papers": 22
   },
   {
    "name": "Yi Cai",
    "id": "2303342982",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "F. Meng",
    "id": "2420425365",
    "h_index": 0,
    "papers": 3
   },
   {
    "name": "Haobo Liang",
    "id": "2371109040",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "M. Wang",
    "id": "2283080404",
    "h_index": 1,
    "papers": 4
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2605.17929v2",
  "pdf_url": "https://arxiv.org/pdf/2605.17929v2",
  "html_url": "https://arxiv.org/html/2605.17929v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2605.17851",
  "slug": "a-dexterous-and-compliant-gripper-with-soft-hydraulic-actuation-for-mi",
  "title": "A Dexterous and Compliant Gripper With Soft Hydraulic Actuation for Microgravity Manipulation",
  "abstract": "Astrobee's existing one-degree-of-freedom (DOF) underactuated compliant claw gripper enables perching on the International Space Station (ISS), but provides limited capability for continuous dexterous manipulation. More complex microgravity tasks require an end-effector that can maintain stable contact while limiting disturbance to the free-flying base, since contact forces directly couple into base motion. This article presents the integration of DexCoHand, a dexterous and compliant two-finger, 6-DOF gripper, with the Astrobee free-flying robot for microgravity manipulation. The system is evaluated in MuJoCo using Astrobee's standard handrail perching sequence, including approach, perching, and subsequent pan and tilt motions. Compared with Astrobee's existing gripper, DexCoHand preserves the commanded pan and tilt motions while reducing unintended cross-axis base motion. Hardware experiments on Earth further demonstrate DexCoHand's dexterous manipulation capabilities and its potential for more adaptable intelligent manipulation tasks.",
  "published": "2026-05-18",
  "updated": "2026-05-18",
  "year": "2026",
  "authors": [
   "William Su",
   "Jordan Kam",
   "Yixiao Wang",
   "Jianshu Zhou"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA 2026",
  "venue_source": "arxiv-comment",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "",
  "doi": "10.48550/arXiv.2605.17851",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "W. Su",
    "id": "2113767555",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Jordan Kam",
    "id": "2343442916",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Yixiao Wang",
    "id": "2324896626",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Jianshu Zhou",
    "id": "2299188652",
    "h_index": 3,
    "papers": 24
   }
  ],
  "comment": "Accepted to the IEEE ICRA 2026 Space Robotics Workshop (SRW). 4 pages, 3 figures",
  "topics": [
   "dexterous-manipulation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2605.17851v1",
  "pdf_url": "https://arxiv.org/pdf/2605.17851v1",
  "html_url": "https://arxiv.org/html/2605.17851v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2605.17556",
  "slug": "visual-sculpting-visually-aligned-planning-representations-for-long-ho",
  "title": "Visual Sculpting: Visually-Aligned Planning Representations for Long-Horizon Robot Clay Sculpting",
  "abstract": "Clay sculpting is a nuanced, artistic task involving dexterous manipulation with long-horizon planning to achieve high-level goals. As a robotics problem, we formulate clay sculpting as a shape-to-shape matching challenge. Prior deformable object manipulation work either requires retraining a policy per goal or relies on dynamics models which represent state as sparse point clouds which do not capture important clay features, such as textures, well. We present a method for modeling the dynamics of deformable materials and planning for robotic sculpting in a representation that is visually-aligned, capturing lighting and texture features. With three different deformable materials and various end-effectors, we demonstrate that our dynamics model is comparable in performance to the state-of-the-art with the added benefit of being compatible with visual planning. Our actions are represented as parametrized pushes into clay with a single end-effector, which proved to be suitable for long-horizon (>100 actions) clay relief sculptures. Lastly, we show the benefits of planning in a visually-aligned representation, but also provide analysis providing evidence as to why this representation is challenging to plan in compared to 3D representations.",
  "published": "2026-05-17",
  "updated": "2026-05-17",
  "year": "2026",
  "authors": [
   "Peter Schaldenbrand",
   "Jean Oh"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "RA-L",
  "venue_source": "semantic-scholar",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "This work presents a method for modeling the dynamics of deformable materials and planning for robotic sculpting in a representation that is visually-aligned, capturing lighting and texture features and provides analysis providing evidence as to why this representation is challenging to plan in compared to 3D representations.",
  "doi": "10.1109/LRA.2026.3673896",
  "oa_pdf": "https://arxiv.org/pdf/2605.17556",
  "s2_authors": [
   {
    "name": "Peter Schaldenbrand",
    "id": "51254537",
    "h_index": 9,
    "papers": 22
   },
   {
    "name": "Jean Oh",
    "id": "2305653745",
    "h_index": 3,
    "papers": 6
   }
  ],
  "comment": "8 pages, 14 figures. Accepted for publication in IEEE Robotics and Automation Letters (RA-L)",
  "topics": [
   "dexterous-manipulation",
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2605.17556v1",
  "pdf_url": "https://arxiv.org/pdf/2605.17556v1",
  "html_url": "https://arxiv.org/html/2605.17556v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.5
 },
 {
  "id": "2605.16257",
  "slug": "dexjoco-a-benchmark-and-toolkit-for-task-oriented-dexterous-manipulati",
  "title": "DexJoCo: A Benchmark and Toolkit for Task-Oriented Dexterous Manipulation on MuJoCo",
  "abstract": "Achieving human-level manipulation requires dexterous robotic hands capable of complex object interactions. Advancing such capabilities further demands standardized benchmarks for systematic evaluation. However, existing dexterous benchmarks lack tasks that reflect the unique manipulation capabilities of dexterous hands over parallel grippers, as well as comprehensive evaluation pipelines. In this paper, we present DexJoCo, a benchmark and toolkit for task-oriented dexterous manipulation, comprising 11 functionally grounded tasks that evaluate tool-use, bimanual coordination, long-horizon execution, and reasoning. We develop a low-cost data collection system and collect 1.1K trajectories across these tasks, with support for domain randomization to assess robustness. We benchmark modern models under diverse settings, including visual and dynamics randomization, multi-task training, and action-head adaptation. Through extensive empirical analysis, we identify several important insights and common limitations of current policies in dexterous manipulation, highlighting key challenges for future research in dexterous hand robot learning. Project page available at: https://dexjoco.github.io",
  "published": "2026-05-15",
  "updated": "2026-05-15",
  "year": "2026",
  "authors": [
   "Hanwen Wang",
   "Weizhi Zhao",
   "Xiangyu Wang",
   "Siyuan Huang",
   "He Lin",
   "Boyuan Zheng",
   "Rongtao Xu",
   "Gang Wang",
   "Yao Mu",
   "He Wang",
   "Lue Fan",
   "Hongsheng Li",
   "Zhaoxiang Zhang",
   "Tieniu Tan"
  ],
  "author_count": 14,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 6,
  "influential_citations": 0,
  "tldr": "",
  "doi": "10.48550/arXiv.2605.16257",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hanwen Wang",
    "id": "1561545899",
    "h_index": 8,
    "papers": 27
   },
   {
    "name": "Weizhi Zhao",
    "id": "2295935140",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Xiangyu Wang",
    "id": "2290855113",
    "h_index": 2,
    "papers": 15
   },
   {
    "name": "Siyuan Huang",
    "id": "2243292807",
    "h_index": 14,
    "papers": 20
   },
   {
    "name": "Henry C. Lin",
    "id": "144587221",
    "h_index": 10,
    "papers": 13
   },
   {
    "name": "Boyuan Zheng",
    "id": "2114719042",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Rongtao Xu",
    "id": "2376110671",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Gang Wang",
    "id": "2394311702",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Yao Mu",
    "id": "2348161293",
    "h_index": 2,
    "papers": 18
   },
   {
    "name": "He Wang",
    "id": "2330238483",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Lue Fan",
    "id": "2087083213",
    "h_index": 21,
    "papers": 45
   },
   {
    "name": "Hongsheng Li",
    "id": "2305732815",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Zhaoxiang Zhang",
    "id": "2374145316",
    "h_index": 2,
    "papers": 12
   },
   {
    "name": "Tieniu Tan",
    "id": "2269788130",
    "h_index": 3,
    "papers": 6
   }
  ],
  "comment": "8 pages, 6 figures, project page is available at: https://dexjoco.github.io",
  "topics": [
   "dexterous-manipulation",
   "sim2real",
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2605.16257v1",
  "pdf_url": "https://arxiv.org/pdf/2605.16257v1",
  "html_url": "https://arxiv.org/html/2605.16257v1",
  "code_url": "https://dexjoco.github.io",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.85
 },
 {
  "id": "2605.15157",
  "slug": "hand-in-the-loop-improving-vla-policies-for-dexterous-manipulation-via",
  "title": "Hand-in-the-Loop: Improving VLA Policies for Dexterous Manipulation via Seamless Hand-Arm Intervention",
  "abstract": "Vision-Language-Action (VLA) models are prone to compounding errors in dexterous manipulation, where high-dimensional action spaces and contact-rich dynamics amplify small policy deviations over long horizons. While Interactive Imitation Learning (IIL) can refine policies through human correction data, applying it to high-degree-of-freedom (DoF) robotic hands remains challenging due to a command mismatch between human teleoperation and policy execution at the intervention moment, which causes abrupt robot-hand configuration changes, or \"gesture jumps\". We present Hand-in-the-Loop (HandITL), a seamless human-in-the-loop intervention method that blends human corrective intent with autonomous policy execution to avoid gesture jumps during bimanual dexterous manipulation. Compared with taking over control using direct teleoperation, HandITL reduces intervention jitter by 99.8% and preserves robust post-intervention manipulation, reducing grasp failures by 87.5% and mean completion time by 19.1%. We validate HandITL on tasks requiring bimanual coordination, tool use, and fine-grained long-horizon manipulation. When used to collect correction data for policy refinement, HandITL yields policies that outperform those trained with standard teleoperation data by 19% on average across three long-horizon dexterous tasks.",
  "published": "2026-05-14",
  "updated": "2026-05-20",
  "year": "2026",
  "authors": [
   "Zhuohang Li",
   "Liqun Huang",
   "Wei Xu",
   "Zhengming Zhu",
   "Nie Lin",
   "Xiao Ma",
   "Xinjun Sheng",
   "Ruoshi Wen"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 0,
  "influential_citations": 0,
  "tldr": "Hand-in-the-Loop (HandITL), a seamless human-in-the-loop intervention method that blends human corrective intent with autonomous policy execution to avoid gesture jumps during bimanual dexterous manipulation is presented.",
  "doi": "10.48550/arXiv.2605.15157",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhuohang Li",
    "id": "2290740132",
    "h_index": 4,
    "papers": 17
   },
   {
    "name": "Liqun Huang",
    "id": "2373416739",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Wei Xu",
    "id": "2373713682",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Zhengming Zhu",
    "id": "2374086349",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Nie Lin",
    "id": "2293324964",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Xiao Ma",
    "id": "2125110703",
    "h_index": 13,
    "papers": 45
   },
   {
    "name": "Xinjun Sheng",
    "id": "2258959092",
    "h_index": 5,
    "papers": 33
   },
   {
    "name": "Ruoshi Wen",
    "id": "30932282",
    "h_index": 8,
    "papers": 19
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "dexterous-manipulation",
   "tactile",
   "imitation-diffusion",
   "data-teleop",
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2605.15157v2",
  "pdf_url": "https://arxiv.org/pdf/2605.15157v2",
  "html_url": "https://arxiv.org/html/2605.15157v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.0
 },
 {
  "id": "2604.25554",
  "slug": "egocentric-tactile-and-proximity-sensors-as-observation-priors-for-hum",
  "title": "Egocentric Tactile and Proximity Sensors as Observation Priors for Humanoid Collision Avoidance",
  "abstract": "Collision-free motion is often aided by tactile and proximity sensors distributed on the body of the robot due to their resistance to occlusion as opposed to external cameras. However, how to shape the sensor's properties, such as sensing coverage; type; and range, to enable avoidant behavior remains unclear. In this work, we present a reinforcement learning framework for whole-body collision avoidance on a humanoid H1-2 robot and use it to characterize how sensor properties shape learned avoidance behavior. Using dodgeball as a benchmark task, we ablate the properties of sensors distributed across the upper body of the robot and find that raw proximity measurements can substitute for explicit object localization provided the sensing range is sufficient and that sparse non-directional proximity signals outpace dense directional alternatives in sample efficiency.",
  "published": "2026-04-28",
  "updated": "2026-04-28",
  "year": "2026",
  "authors": [
   "Carson Kohlbrenner",
   "Niraj Pudasaini",
   "William Xie",
   "Naren Sivagnanadasan",
   "Nikolaus Correll",
   "Alessandro Roncone"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA 2026",
  "venue_source": "arxiv-comment",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "This work ablate the properties of sensors distributed across the upper body of the robot and finds that raw proximity measurements can substitute for explicit object localization provided the sensing range is sufficient and that sparse non-directional proximity signals outpace dense directional alternatives in sample efficiency.",
  "doi": "10.48550/arXiv.2604.25554",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Carson Kohlbrenner",
    "id": "2333354233",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Niraj Pudasaini",
    "id": "2092477521",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "W. Xie",
    "id": "2291098851",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Naren Sivagnanadasan",
    "id": "3264609",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "N. Correll",
    "id": "2886493",
    "h_index": 36,
    "papers": 174
   },
   {
    "name": "Alessandro Roncone",
    "id": "2365740226",
    "h_index": 5,
    "papers": 19
   }
  ],
  "comment": "This work was accepted at the 8th RoboTac Workshop at the International Conference on Robotics and Automation (ICRA) 2026",
  "topics": [
   "humanoids",
   "egocentric-data",
   "tactile",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2604.25554v1",
  "pdf_url": "https://arxiv.org/pdf/2604.25554v1",
  "html_url": "https://arxiv.org/html/2604.25554v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.8
 },
 {
  "id": "2604.17335",
  "slug": "learning-whole-body-humanoid-locomotion-via-motion-generation-and-moti",
  "title": "Learning Whole-Body Humanoid Locomotion via Motion Generation and Motion Tracking",
  "abstract": "Whole-body humanoid locomotion is challenging due to high-dimensional control, morphological instability, and the need for real-time adaptation to various terrains using onboard perception. Directly applying reinforcement learning (RL) with reward shaping to humanoid locomotion often leads to lower-body-dominated behaviors, whereas imitation-based RL can learn more coordinated whole-body skills but is typically limited to replaying reference motions without a mechanism to adapt them online from perception for terrain-aware locomotion. To address this gap, we propose a whole-body humanoid locomotion framework that combines skills learned from reference motions with terrain-aware adaptation. We first train a diffusion model on retargeted human motions for real-time prediction of terrain-aware reference motions. Concurrently, we train a whole-body reference tracker with RL using this motion data. To improve robustness under imperfectly generated references, we further fine-tune the tracker with a frozen motion generator in a closed-loop setting. The resulting system supports directional goal-reaching control with terrain-aware whole-body adaptation, and can be deployed on a Unitree G1 humanoid robot with onboard perception and computation. The hardware experiments demonstrate successful traversal over boxes, hurdles, stairs, and mixed terrain combinations. Quantitative results further show the benefits of incorporating online motion generation and fine-tuning the motion tracker for improved generalization and robustness.",
  "published": "2026-04-19",
  "updated": "2026-07-12",
  "year": "2026",
  "authors": [
   "Zewei Zhang",
   "Kehan Wen",
   "Michael Xu",
   "Junzhe He",
   "Chenhao Li",
   "Takahiro Miki",
   "Clemens Schwarke",
   "Chong Zhang",
   "Xue Bin Peng",
   "Marco Hutter"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "RA-L",
  "venue_source": "semantic-scholar",
  "citations": 12,
  "influential_citations": 1,
  "tldr": "A whole-body humanoid locomotion framework that combines skills learned from reference motions with terrain-aware adaptation and the benefits of incorporating online motion generation and fine-tuning the motion tracker for improved generalization and robustness is proposed.",
  "doi": "10.1109/LRA.2026.3710365",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zewei Zhang",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Kehan Wen",
    "id": "2292399697",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Michael Xu",
    "id": "2361864986",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Junzhe He",
    "id": "2249076206",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Chenhao Li",
    "id": "2306055264",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Takahiro Miki",
    "id": "2305758231",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Clemens Schwarke",
    "id": "2284772421",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Chong Zhang",
    "id": "2254275727",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Xue Bin Peng",
    "id": "2300358388",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Marco Hutter",
    "id": "2340685198",
    "h_index": 6,
    "papers": 20
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "rl-control",
   "video-generation",
   "safety-eval"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2604.17335v2",
  "pdf_url": "https://arxiv.org/pdf/2604.17335v2",
  "html_url": "https://arxiv.org/html/2604.17335v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.11
 },
 {
  "id": "2604.08534",
  "slug": "activeglasses-learning-manipulation-with-active-vision-from-ego-centri",
  "title": "ActiveGlasses: Learning Manipulation with Active Vision from Ego-centric Human Demonstration",
  "abstract": "Large-scale real-world robot data collection is a prerequisite for bringing robots into everyday deployment. However, existing pipelines often rely on specialized handheld devices to bridge the embodiment gap, which not only increases operator burden and limits scalability, but also makes it difficult to capture the naturally coordinated perception-manipulation behaviors of human daily interaction. This challenge calls for a more natural system that can faithfully capture human manipulation and perception behaviors while enabling zero-shot transfer to robotic platforms. We introduce ActiveGlasses, a system for learning robot manipulation from ego-centric human demonstrations with active vision. A stereo camera mounted on smart glasses serves as the sole perception device for both data collection and policy inference: the operator wears it during bare-hand demonstrations, and the same camera is mounted on a 6-DoF perception arm during deployment to reproduce human active vision. To enable zero-transfer, we extract object trajectories from demonstrations and use an object-centric point-cloud policy to jointly predict manipulation and head movement. Across several challenging tasks involving occlusion and precise interaction, ActiveGlasses achieves zero-shot transfer with active vision, consistently outperforms strong baselines under the same hardware setup, and generalizes across two robot platforms.",
  "published": "2026-04-09",
  "updated": "2026-04-09",
  "year": "2026",
  "authors": [
   "Yanwen Zou",
   "Chenyang Shi",
   "Wenye Yu",
   "Han Xue",
   "Jun Lv",
   "Ye Pan",
   "Chuan Wen",
   "Cewu Lu"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 6,
  "influential_citations": 0,
  "tldr": "This work introduces ActiveGlasses, a system for learning robot manipulation from ego-centric human demonstrations with active vision that achieves zero-shot transfer with active vision, consistently outperforms strong baselines under the same hardware setup, and generalizes across two robot platforms.",
  "doi": "10.48550/arXiv.2604.08534",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yanwen Zou",
    "id": "2374371333",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Chen Shi",
    "id": "2307225089",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Wenye Yu",
    "id": "2298210236",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Han Xue",
    "id": "2053313606",
    "h_index": 10,
    "papers": 15
   },
   {
    "name": "Jun Lv",
    "id": "2054671126",
    "h_index": 11,
    "papers": 26
   },
   {
    "name": "Ye Pan",
    "id": "2279659002",
    "h_index": 8,
    "papers": 19
   },
   {
    "name": "Chuan Wen",
    "id": "2381956964",
    "h_index": 5,
    "papers": 20
   },
   {
    "name": "Cewu Lu",
    "id": "2301174899",
    "h_index": 7,
    "papers": 29
   }
  ],
  "comment": "",
  "topics": [
   "egocentric-data",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2604.08534v1",
  "pdf_url": "https://arxiv.org/pdf/2604.08534v1",
  "html_url": "https://arxiv.org/html/2604.08534v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.85
 },
 {
  "id": "2604.07607",
  "slug": "egoverse-an-egocentric-human-dataset-for-robot-learning-from-around-th",
  "title": "EgoVerse: An Egocentric Human Dataset for Robot Learning from Around the World",
  "abstract": "Robot learning increasingly depends on large and diverse data, yet robot data collection remains expensive and difficult to scale. Egocentric human data offer a promising alternative by capturing rich manipulation behavior across everyday environments. However, existing human datasets are often limited in scope, difficult to extend, and fragmented across institutions. We introduce EgoVerse, a collaborative platform for human data-driven robot learning that unifies data collection, processing, and access under a shared framework, enabling contributions from individual researchers, academic labs, and industry partners. The current release includes 1,362 hours (80k episodes) of human demonstrations spanning 1,965 tasks, 240 scenes, and 2,087 unique demonstrators, with standardized formats, manipulation-relevant annotations, and tooling for downstream learning. Beyond the dataset, we conduct a large-scale study of human-to-robot transfer with experiments replicated across multiple labs, tasks, and robot embodiments under shared protocols. We find that policy performance generally improves with increased human data, but that effective scaling depends on alignment between human data and robot learning objectives. Together, the dataset, platform, and study establish a foundation for reproducible progress in human data-driven robot learning. Videos and additional information can be found at https://egoverse.ai/",
  "published": "2026-04-08",
  "updated": "2026-07-07",
  "year": "2026",
  "authors": [
   "Ryan Punamiya",
   "Simar Kareer",
   "Zeyi Liu",
   "Josh Citron",
   "Ri-Zhao Qiu",
   "Xiongyi Cai",
   "Alexey Gavryushin",
   "Jiaqi Chen",
   "Davide Liconti",
   "Lawrence Y. Zhu",
   "Patcharapong Aphiwetsa",
   "Baoyu Li",
   "Aniketh Cheluva",
   "Pranav Kuppili",
   "Yangcen Liu",
   "Dhruv Patel",
   "Aidan Gao",
   "Hye-Young Chung",
   "Ryan Co",
   "Renee Zbizika",
   "Jeff Liu",
   "Xiaomeng Xu",
   "Haoyu Xiong",
   "Geng Chen",
   "Sebastiano Oliani",
   "Wenkai Xuan",
   "Chenyu Yang",
   "Xi Wang",
   "James Fort",
   "Richard Newcombe",
   "Josh Gao",
   "Jason Chong",
   "Garrett Matsuda",
   "Aseem Doriwala",
   "Marc Pollefeys",
   "Robert Katzschmann",
   "Xiaolong Wang",
   "Shuran Song",
   "Judy Hoffman",
   "Danfei Xu"
  ],
  "author_count": 40,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 34,
  "influential_citations": 3,
  "tldr": "EgoVerse is introduced, a collaborative platform for human data-driven robot learning that unifies data collection, processing, and access under a shared framework, enabling contributions from individual researchers, academic labs, and industry partners.",
  "doi": "10.48550/arXiv.2604.07607",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ryan Punamiya",
    "id": "2328411560",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Simar Kareer",
    "id": "2188833033",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Zeyi Liu",
    "id": "2176845464",
    "h_index": 9,
    "papers": 11
   },
   {
    "name": "Joshua Citron",
    "id": "2284223401",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Ri-Zhao Qiu",
    "id": "2290904526",
    "h_index": 11,
    "papers": 19
   },
   {
    "name": "Xiongyi Cai",
    "id": "2393079892",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Alexey Gavryushin",
    "id": "2202228606",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Jiaqi Chen",
    "id": "2253918255",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Davide Liconti",
    "id": "2298269450",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Lawrence Y. Zhu",
    "id": "2379184973",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Patcharapong Aphiwetsa",
    "id": "2310435059",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Baoyu Li",
    "id": "2294776955",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Aniketh Cheluva",
    "id": "2428625859",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Pranav Kuppili",
    "id": "2378954879",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Yangcen Liu",
    "id": "2297343604",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Dhruv Patel",
    "id": "2328566408",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Aidan Gao",
    "id": "2349471003",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Hyejin Chung",
    "id": "2378011911",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "R. Co.",
    "id": "1450581655",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Renee Zbizika",
    "id": "2378980381",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Jeff Liu",
    "id": "2108346123",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Xiaomeng Xu",
    "id": "2286521452",
    "h_index": 7,
    "papers": 9
   },
   {
    "name": "Haoyu Xiong",
    "id": "2281036863",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Geng Chen",
    "id": "2339970647",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Sebastiano Oliani",
    "id": "2333603569",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Chenyu Yang",
    "id": "2329514965",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Xi Wang",
    "id": "2320327993",
    "h_index": 1,
    "papers": 11
   },
   {
    "name": "James Fort",
    "id": "2355082896",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Richard A. Newcombe",
    "id": "2292257340",
    "h_index": 9,
    "papers": 20
   },
   {
    "name": "Joshua Gao",
    "id": "2334598706",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Jason Chong",
    "id": "47656473",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Garrett Matsuda",
    "id": "2428626141",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Aseem Doriwala",
    "id": "2428626130",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Marc Pollefeys",
    "id": "2385471198",
    "h_index": 3,
    "papers": 17
   },
   {
    "name": "Robert K. Katzschmann",
    "id": "2139596377",
    "h_index": 16,
    "papers": 58
   },
   {
    "name": "Xiaolong Wang",
    "id": "2327466874",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Shuran Song",
    "id": "2289085682",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Judy Hoffman",
    "id": "2328413304",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Dan Xu",
    "id": "2394311551",
    "h_index": 1,
    "papers": 5
   }
  ],
  "comment": "",
  "topics": [
   "egocentric-data",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2604.07607v2",
  "pdf_url": "https://arxiv.org/pdf/2604.07607v2",
  "html_url": "https://arxiv.org/html/2604.07607v2",
  "code_url": null,
  "presented": [
   {
    "club": "saturday-robotics",
    "club_name": "Saturday Robotics",
    "session_number": 8,
    "session_title": "Robotics & World Models Reading Club 08: Embodied Human Data as the \u201cInternet of Motion and Behavior\u201d \u2014 San Francisco 0516",
    "date_text": "Saturday, May 16, 2026 | 2:00 PM \u2013 5:00 PM",
    "city": "San Francisco, CA",
    "url": "https://lu.ma/qoxioge7",
    "listed_as": "Internet-Scale Human Motion Pretraining for Robotics"
   },
   {
    "club": "saturday-robotics",
    "club_name": "Saturday Robotics",
    "session_number": 13,
    "session_title": "Robotics & World Models Reading Club 13: HumanEgo: Train Robot Policy from 30 min Egocentric Videos \u2014 SF 0620",
    "date_text": "Saturday, June 20, 2026 | 2:00 PM \u2013 5:00 PM",
    "city": "San Francisco, CA",
    "url": "https://lu.ma/6vkhxnum",
    "listed_as": "X: https://x.com/TX\\Leo\\Wang/status/2059320921228546220 (Highly recommend)"
   }
  ],
  "club_note": "Pretrains behavior foundation models on massive internet-scale human motion datasets for downstream robotic policy transfer.",
  "featured": true,
  "signal": 4.54
 },
 {
  "id": "2604.01064",
  "slug": "bat-balancing-agility-and-stability-via-online-policy-switching-for-lo",
  "title": "BAT: Balancing Agility and Stability via Online Policy Switching for Long-Horizon Whole-Body Humanoid Control",
  "abstract": "Despite recent advances in control, reinforcement learning, and imitation learning, developing a unified framework that can achieve agile, precise, and robust whole-body behaviors, particularly in long-horizon tasks, remains challenging. Existing approaches typically follow two paradigms: coupled whole-body policies for global coordination and decoupled policies for modular precision. However, without a systematic method to integrate both, this trade-off between agility, robustness, and precision remains unresolved. In this work, we propose BAT, an online policy-switching framework that dynamically selects between two complementary whole-body RL controllers to balance agility and stability across different motion contexts. Our framework consists of two complementary modules: a switching policy learned via hierarchical RL with an expert guidance from sliding-horizon policy pre-evaluation, and an option-aware VQ-VAE that predicts option preference from discrete motion token sequences for improved generalization. The final decision is obtained via confidence-weighted fusion of two modules. Extensive simulations and real-world experiments on the Unitree G1 humanoid robot demonstrate that BAT enables versatile long-horizon loco-manipulation and outperforms prior methods across diverse tasks.",
  "published": "2026-04-01",
  "updated": "2026-04-01",
  "year": "2026",
  "authors": [
   "Donghoon Baek",
   "Sang-Hun Kim",
   "Sehoon Ha"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 2,
  "influential_citations": 0,
  "tldr": "BAT is an online policy-switching framework that dynamically selects between two complementary whole-body RL controllers to balance agility and stability across different motion contexts and enables versatile long-horizon loco-manipulation and outperforms prior methods across diverse tasks.",
  "doi": "10.48550/arXiv.2604.01064",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "D. Baek",
    "id": "51232393",
    "h_index": 8,
    "papers": 26
   },
   {
    "name": "Sanghyun Kim",
    "id": "2314327043",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Sehoon Ha",
    "id": "2258707295",
    "h_index": 2,
    "papers": 5
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "imitation-diffusion",
   "rl-control",
   "safety-eval"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2604.01064v1",
  "pdf_url": "https://arxiv.org/pdf/2604.01064v1",
  "html_url": "https://arxiv.org/html/2604.01064v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.98
 },
 {
  "id": "2603.29452",
  "slug": "cref-cross-modal-and-recurrent-fusion-for-depth-conditioned-humanoid-l",
  "title": "CReF: Cross-modal and Recurrent Fusion for Depth-conditioned Humanoid Locomotion",
  "abstract": "Stable traversal over geometrically complex terrain increasingly requires exteroceptive perception, yet prior perceptive humanoid locomotion methods often remain tied to explicit geometric abstractions, either by mediating control through robot-centric 2.5D terrain representations or by shaping depth learning with auxiliary geometry-related targets. While effective, these approaches introduce additional map-construction procedures or multi-stage skill-transfer processes beyond direct depth-to-control learning. We propose CReF (Cross-modal and Recurrent Fusion), a single-stage depth-conditioned humanoid locomotion framework that learns locomotion-relevant features directly from raw forward-facing depth without explicit geometric intermediates. CReF couples proprioception and depth tokens through proprioception-queried cross-modal attention, fuses the resulting representation with a gated residual fusion block, and performs temporal integration with a Gated Recurrent Unit (GRU) regulated by a highway-style output gate for state-dependent blending of recurrent and feedforward features. To further improve terrain interaction, we introduce a terrain-aware foothold placement reward that extracts supportable foothold candidates from foot-end point-cloud samples and rewards touchdown locations that lie close to the nearest supportable candidate. Experiments in simulation and on a physical humanoid demonstrate robust traversal over diverse terrains and effective zero-shot transfer to real-world scenes containing handrails, hollow pallet assemblies, severe reflective interference, and visually cluttered outdoor surroundings.",
  "published": "2026-03-31",
  "updated": "2026-07-27",
  "year": "2026",
  "authors": [
   "Yuan Hao",
   "Ruiqi Yu",
   "Shixin Luo",
   "Guoteng Zhang",
   "Jun Wu",
   "Qiuguo Zhu"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "RA-L",
  "venue_source": "semantic-scholar",
  "citations": 3,
  "influential_citations": 0,
  "tldr": "This work proposes CReF (Cross-modal and Recurrent Fusion), a single-stage depth-conditioned humanoid locomotion framework that learns locomotion-relevant features directly from raw forward-facing depth without explicit geometric intermediates and demonstrates robust traversal over diverse terrains and effective zero-shot transfer to real-world scenes.",
  "doi": "10.1109/LRA.2026.3723339",
  "oa_pdf": "https://doi.org/10.48550/arxiv.2603.29452",
  "s2_authors": [
   {
    "name": "Yuansu Hao",
    "id": "2391889075",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Ruiqi Yu",
    "id": "2290969193",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Shixin Luo",
    "id": "2319972230",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Guoteng Zhang",
    "id": "2374002304",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Junchen Wu",
    "id": "2426533927",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Qi Zhu",
    "id": "2372253712",
    "h_index": 2,
    "papers": 6
   }
  ],
  "comment": "",
  "topics": [
   "humanoids"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2603.29452v3",
  "pdf_url": "https://arxiv.org/pdf/2603.29452v3",
  "html_url": "https://arxiv.org/html/2603.29452v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.1
 },
 {
  "id": "2603.23481",
  "slug": "vtam-video-tactile-action-models-for-complex-physical-interaction-beyo",
  "title": "VTAM: Video-Tactile-Action Models for Complex Physical Interaction Beyond VLAs",
  "abstract": "Video-Action Models (VAMs) have emerged as a promising framework for embodied intelligence, learning implicit world dynamics from raw video streams to produce temporally consistent action predictions. Although such models demonstrate strong performance on long-horizon tasks through visual reasoning, they remain limited in contact-rich scenarios where critical interaction states are only partially observable from vision alone. In particular, fine-grained force modulation and contact transitions are not reliably encoded in visual tokens, leading to unstable or imprecise behaviors. To bridge this gap, we introduce the Video-Tactile Action Model (VTAM), a multimodal world modeling framework that incorporates tactile perception as a complementary grounding signal. VTAM augments a pretrained video transformer with tactile streams via a lightweight modality transfer finetuning, enabling efficient cross-modal representation learning without tactile-language paired data or independent tactile pretraining. To stabilize multimodal fusion, we introduce a tactile regularization loss that enforces balanced cross-modal attention, preventing visual latent dominance in the action model. VTAM demonstrates superior performance in contact-rich manipulation, maintaining a robust success rate of 90 percent on average. In challenging scenarios such as potato chip pick-and-place requiring high-fidelity force awareness, VTAM outperforms the pi 0.5 baseline by 80 percent. Our findings demonstrate that integrating tactile feedback is essential for correcting visual estimation errors in world action models, providing a scalable approach to physically grounded embodied foundation models.",
  "published": "2026-03-24",
  "updated": "2026-03-24",
  "year": "2026",
  "authors": [
   "Haoran Yuan",
   "Weigang Yi",
   "Zhenyu Zhang",
   "Wendi Chen",
   "Yuchen Mo",
   "Jiashi Yin",
   "Xinzhuo Li",
   "Xiangyu Zeng",
   "Chuan Wen",
   "Cewu Lu",
   "Katherine Driggs-Campbell",
   "Ismini Lourentzou"
  ],
  "author_count": 12,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 12,
  "influential_citations": 0,
  "tldr": "This work introduces the Video-Tactile Action Model (VTAM), a multimodal world modeling framework that incorporates tactile perception as a complementary grounding signal, and demonstrates superior performance in contact-rich manipulation.",
  "doi": "10.48550/arXiv.2603.23481",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Haoran Yuan",
    "id": "2308821931",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Weigang Yi",
    "id": "29853822",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Zhenyu Zhang",
    "id": "2382934176",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Wendi Chen",
    "id": "2326063487",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Yuchen Mo",
    "id": "2343834053",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Jiashi Yin",
    "id": "2425493165",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Xinzhuo Li",
    "id": "2336119887",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Xiangyu Zeng",
    "id": "2375296549",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Chuan Wen",
    "id": "2381956964",
    "h_index": 5,
    "papers": 20
   },
   {
    "name": "Cewu Lu",
    "id": "2301174899",
    "h_index": 7,
    "papers": 29
   },
   {
    "name": "K. Driggs-Campbell",
    "id": "1404112858",
    "h_index": 26,
    "papers": 71
   },
   {
    "name": "Ismini Lourentzou",
    "id": "2099420",
    "h_index": 19,
    "papers": 96
   }
  ],
  "comment": "https://plan-lab.github.io/projects/vtam/",
  "topics": [
   "world-models",
   "tactile",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2603.23481v1",
  "pdf_url": "https://arxiv.org/pdf/2603.23481v1",
  "html_url": "https://arxiv.org/html/2603.23481v1",
  "code_url": "https://plan-lab.github.io/projects/vtam/",
  "presented": [
   {
    "club": "saturday-robotics",
    "club_name": "Saturday Robotics",
    "session_number": 26,
    "session_title": "Robotics & World Models Reading Club 26: Video Generation to Robot Manipulation: Bridging Embodiment Gap+Video-Tactile-Action Model. SF 8/29",
    "date_text": "Saturday, August 29, 2026 | 2:00 PM \u2013 5:00 PM",
    "city": "San Francisco, CA",
    "url": "https://lu.ma/00z3oxw6",
    "listed_as": "VTAM: Video-Tactile-Action Models for Complex Physical Interaction Beyond VLAs"
   }
  ],
  "club_note": "",
  "featured": true,
  "signal": 5.11
 },
 {
  "id": "2603.22264",
  "slug": "unidex-a-robot-foundation-suite-for-universal-dexterous-hand-control-f",
  "title": "UniDex: A Robot Foundation Suite for Universal Dexterous Hand Control from Egocentric Human Videos",
  "abstract": "Dexterous manipulation remains challenging due to the cost of collecting real-robot teleoperation data, the heterogeneity of hand embodiments, and the high dimensionality of control. We present UniDex, a robot foundation suite that couples a large-scale robot-centric dataset with a unified vision-language-action (VLA) policy and a practical human-data capture setup for universal dexterous hand control. First, we construct UniDex-Dataset, a robot-centric dataset over 50K trajectories across eight dexterous hands (6--24 DoFs), derived from egocentric human video datasets. To transform human data into robot-executable trajectories, we employ a human-in-the-loop retargeting procedure to align fingertip trajectories while preserving plausible hand-object contacts, and we operate on explicit 3D pointclouds with human hands masked to narrow kinematic and visual gaps. Second, we introduce the Function-Actuator-Aligned Space (FAAS), a unified action space that maps functionally similar actuators to shared coordinates, enabling cross-hand transfer. Leveraging FAAS as the action parameterization, we train UniDex-VLA, a 3D VLA policy pretrained on UniDex-Dataset and finetuned with task demonstrations. In addition, we build UniDex-Cap, a simple portable capture setup that records synchronized RGB-D streams and human hand poses and converts them into robot-executable trajectories to enable human-robot data co-training that reduces reliance on costly robot demonstrations. On challenging tool-use tasks across two different hands, UniDex-VLA achieves 81% average task progress and outperforms prior VLA baselines by a large margin, while exhibiting strong spatial, object, and zero-shot cross-hand generalization. Together, UniDex-Dataset, UniDex-VLA, and UniDex-Cap provide a scalable foundation suite for universal dexterous manipulation.",
  "published": "2026-03-23",
  "updated": "2026-03-23",
  "year": "2026",
  "authors": [
   "Gu Zhang",
   "Qicheng Xu",
   "Haozhe Zhang",
   "Jianhan Ma",
   "Long He",
   "Yiming Bao",
   "Zeyu Ping",
   "Zhecheng Yuan",
   "Chenhao Lu",
   "Chengbo Yuan",
   "Tianhai Liang",
   "Xiaoyu Tian",
   "Maanping Shao",
   "Feihong Zhang",
   "Mingyu Ding",
   "Yang Gao",
   "Hao Zhao",
   "Hang Zhao",
   "Huazhe Xu"
  ],
  "author_count": 19,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "CVPR 2026",
  "venue_source": "arxiv-comment",
  "citations": 17,
  "influential_citations": 0,
  "tldr": "UniDex is a robot foundation suite that couples a large-scale robot-centric dataset with a unified vision-language-action (VLA) policy and a practical human-data capture setup for universal dexterous hand control to enable human-robot data co-training that reduces reliance on costly robot demonstrations.",
  "doi": "10.48550/arXiv.2603.22264",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Gu Zhang",
    "id": "2220861421",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Qicheng Xu",
    "id": "2219000237",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Haozhe Zhang",
    "id": "2135719838",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Jianhang Ma",
    "id": "2146393199",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Long He",
    "id": "2216878930",
    "h_index": 5,
    "papers": 15
   },
   {
    "name": "Yiming Bao",
    "id": "2027671817",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Zeyu Ping",
    "id": "2345019784",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Zhecheng Yuan",
    "id": "2156151359",
    "h_index": 15,
    "papers": 24
   },
   {
    "name": "Chenhao Lu",
    "id": "2265619607",
    "h_index": 8,
    "papers": 20
   },
   {
    "name": "Chengbo Yuan",
    "id": "2280187985",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Tianhai Liang",
    "id": "2232782812",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Xiaoyu Tian",
    "id": "2149326880",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "Maanping Shao",
    "id": "2391581521",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Feihong Zhang",
    "id": "2391564961",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Mingyu Ding",
    "id": "2346837065",
    "h_index": 4,
    "papers": 15
   },
   {
    "name": "Yang Gao",
    "id": "2330756947",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Hao Zhao",
    "id": "2326064308",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Hang Zhao",
    "id": "2362754008",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Huazhe Xu",
    "id": "2345912040",
    "h_index": 2,
    "papers": 3
   }
  ],
  "comment": "Accepted by CVPR 2026",
  "topics": [
   "vla",
   "dexterous-manipulation",
   "egocentric-data",
   "data-teleop",
   "hardware-codesign",
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2603.22264v1",
  "pdf_url": "https://arxiv.org/pdf/2603.22264v1",
  "html_url": "https://arxiv.org/html/2603.22264v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.76
 },
 {
  "id": "2603.20850",
  "slug": "glove2hand-synthesizing-natural-hand-object-interaction-from-multi-mod",
  "title": "Glove2Hand: Synthesizing Natural Hand-Object Interaction from Multi-Modal Sensing Gloves",
  "abstract": "Understanding hand-object interaction (HOI) is fundamental to computer vision, robotics, and AR/VR. However, conventional hand videos often lack essential physical information such as contact forces and motion signals, and are prone to frequent occlusions. To address the challenges, we present Glove2Hand, a framework that translates multi-modal sensing glove HOI videos into photorealistic bare hands, while faithfully preserving the underlying physical interaction dynamics. We introduce a novel 3D Gaussian hand model that ensures temporal rendering consistency. The rendered hand is seamlessly integrated into the scene using a diffusion-based hand restorer, which effectively handles complex hand-object interactions and non-rigid deformations. Leveraging Glove2Hand, we create HandSense, the first multi-modal HOI dataset featuring glove-to-hand videos with synchronized tactile and IMU signals. We demonstrate that HandSense significantly enhances downstream bare-hand applications, including video-based contact estimation and hand tracking under severe occlusion.",
  "published": "2026-03-21",
  "updated": "2026-06-08",
  "year": "2026",
  "authors": [
   "Xinyu Zhang",
   "Ziyi Kou",
   "Chuan Qin",
   "Mia Huang",
   "Ergys Ristani",
   "Ankit Kumar",
   "Lele Chen",
   "Kun He",
   "Abdeslam Boularias",
   "Li Guan"
  ],
  "author_count": 10,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR 2026",
  "venue_source": "arxiv-comment",
  "citations": 5,
  "influential_citations": 0,
  "tldr": "Glove2Hand is presented, a framework that translates multi-modal sensing glove HOI videos into photorealistic bare hands, while faithfully preserving the underlying physical interaction dynamics, and a novel 3D Gaussian hand model is introduced that ensures temporal rendering consistency.",
  "doi": "10.48550/arXiv.2603.20850",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xinyu Zhang",
    "id": "2244773454",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Ziyi Kou",
    "id": "2399163710",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Chuan Qin",
    "id": "2409120838",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Miao Huang",
    "id": "2196817543",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Ergys Ristani",
    "id": "2788204",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Ankit Kumar",
    "id": "2274928101",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Lele Chen",
    "id": "152875073",
    "h_index": 16,
    "papers": 32
   },
   {
    "name": "Kun He",
    "id": "2293315762",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Abdeslam Boularias",
    "id": "2288336431",
    "h_index": 6,
    "papers": 22
   },
   {
    "name": "Li Guan",
    "id": "2287035186",
    "h_index": 2,
    "papers": 7
   }
  ],
  "comment": "CVPR 2026 Highlight. This version includes the motion retarget process in the appendix",
  "topics": [
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2603.20850v2",
  "pdf_url": "https://arxiv.org/pdf/2603.20850v2",
  "html_url": "https://arxiv.org/html/2603.20850v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.28
 },
 {
  "id": "2603.12263",
  "slug": "0-an-open-foundation-model-towards-universal-humanoid-loco-manipulatio",
  "title": "$\u03a8_0$: An Open Foundation Model Towards Universal Humanoid Loco-Manipulation",
  "abstract": "We introduce $\u03a8_0$ (Psi-Zero), an open foundation model to address challenging humanoid loco-manipulation tasks. While existing approaches often attempt to address this fundamental problem by co-training on large and diverse human and humanoid data, we argue that this strategy is suboptimal due to the fundamental kinematic and motion disparities between humans and humanoid robots. Therefore, data efficiency and model performance remain unsatisfactory despite the considerable data volume. To address this challenge, \\ours\\;decouples the learning process to maximize the utility of heterogeneous data sources. Specifically, we propose a staged training paradigm with different learning objectives: First, we autoregressively pre-train a VLM backbone on large-scale egocentric human videos to acquire generalizable visual-action representations. Then, we post-train a flow-based action expert on high-quality humanoid robot data to learn precise robot joint control. Our research further identifies a critical yet often overlooked data recipe: in contrast to approaches that scale with noisy Internet clips or heterogeneous cross-embodiment robot datasets, we demonstrate that pre-training on high-quality egocentric human manipulation data followed by post-training on domain-specific real-world humanoid trajectories yields superior performance. Extensive real-world experiments demonstrate that \\ours\\ achieves the best performance using only about 800 hours of human video data and 30 hours of real-world robot data, outperforming baselines pre-trained on more than 10$\\times$ as much data by over 40\\% in overall success rate across multiple tasks. We will open-source the entire ecosystem to the community, including a data processing and training pipeline, a humanoid foundation model, and a real-time action inference engine.",
  "published": "2026-03-12",
  "updated": "2026-03-12",
  "year": "2026",
  "authors": [
   "Songlin Wei",
   "Hongyi Jing",
   "Boqian Li",
   "Zhenyu Zhao",
   "Jiageng Mao",
   "Zhenhao Ni",
   "Sicheng He",
   "Jie Liu",
   "Xiawei Liu",
   "Kaidi Kang",
   "Sheng Zang",
   "Weiduo Yuan",
   "Marco Pavone",
   "Di Huang",
   "Yue Wang"
  ],
  "author_count": 15,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 26,
  "influential_citations": 4,
  "tldr": "It is demonstrated that pre-training on high-quality egocentric human manipulation data followed by post-training on domain-specific real-world humanoid trajectories yields superior performance, in contrast to approaches that scale with noisy Internet clips or heterogeneous cross-embodiment robot datasets.",
  "doi": "10.48550/arXiv.2603.12263",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Songlin Wei",
    "id": "2249895162",
    "h_index": 8,
    "papers": 21
   },
   {
    "name": "Hongyi Jing",
    "id": "2204802732",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Boqian Li",
    "id": "2302479703",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Zhenyu Zhao",
    "id": "2326952666",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Jiageng Mao",
    "id": "2253462944",
    "h_index": 12,
    "papers": 26
   },
   {
    "name": "Zhe-Yuan Ni",
    "id": "2358514842",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Sicheng He",
    "id": "2310814135",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Jie Liu",
    "id": "2348727920",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Xiawei Liu",
    "id": "2310182253",
    "h_index": 6,
    "papers": 24
   },
   {
    "name": "Kai Kang",
    "id": "2312734644",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Sheng Zang",
    "id": "2373717740",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Weiduo Yuan",
    "id": "2422725272",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Marco Pavone",
    "id": "2328979520",
    "h_index": 10,
    "papers": 34
   },
   {
    "name": "Di Huang",
    "id": "2292899023",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Yue Wang",
    "id": "2385729806",
    "h_index": 5,
    "papers": 10
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "egocentric-data",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2603.12263v1",
  "pdf_url": "https://arxiv.org/pdf/2603.12263v1",
  "html_url": "https://arxiv.org/html/2603.12263v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.43
 },
 {
  "id": "2602.16710",
  "slug": "egoscale-scaling-dexterous-manipulation-with-diverse-egocentric-human",
  "title": "EgoScale: Scaling Dexterous Manipulation with Diverse Egocentric Human Data",
  "abstract": "Human behavior is among the most scalable sources of data for learning physical intelligence, yet how to effectively leverage it for dexterous manipulation remains unclear. While prior work demonstrates human to robot transfer in constrained settings, it is unclear whether large scale human data can support fine grained, high degree of freedom dexterous manipulation. We present EgoScale, a human to dexterous manipulation transfer framework built on large scale egocentric human data. We train a Vision Language Action (VLA) model on over 20,854 hours of action labeled egocentric human video, more than 20 times larger than prior efforts, and uncover a log linear scaling law between human data scale and validation loss. This validation loss strongly correlates with downstream real robot performance, establishing large scale human data as a predictable supervision source. Beyond scale, we introduce a simple two stage transfer recipe: large scale human pretraining followed by lightweight aligned human robot mid training. This enables strong long horizon dexterous manipulation and one shot task adaptation with minimal robot supervision. Our final policy improves average success rate by 54% over a no pretraining baseline using a 22 DoF dexterous robotic hand, and transfers effectively to robots with lower DoF hands, indicating that large scale human motion provides a reusable, embodiment agnostic motor prior.",
  "published": "2026-02-18",
  "updated": "2026-02-18",
  "year": "2026",
  "authors": [
   "Ruijie Zheng",
   "Dantong Niu",
   "Yuqi Xie",
   "Jing Wang",
   "Mengda Xu",
   "Yunfan Jiang",
   "Fernando Casta\u00f1eda",
   "Fengyuan Hu",
   "You Liang Tan",
   "Letian Fu",
   "Trevor Darrell",
   "Furong Huang",
   "Yuke Zhu",
   "Danfei Xu",
   "Linxi Fan"
  ],
  "author_count": 15,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 71,
  "influential_citations": 5,
  "tldr": "The final policy improves average success rate by 54% over a no pretraining baseline using a 22 DoF dexterous robotic hand, and transfers effectively to robots with lower DoF hands, indicating that large scale human motion provides a reusable, embodiment agnostic motor prior.",
  "doi": "10.48550/arXiv.2602.16710",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ruijie Zheng",
    "id": "2345931905",
    "h_index": 9,
    "papers": 15
   },
   {
    "name": "Dantong Niu",
    "id": "2268757542",
    "h_index": 8,
    "papers": 17
   },
   {
    "name": "Yuqi Xie",
    "id": "2218866691",
    "h_index": 11,
    "papers": 17
   },
   {
    "name": "Jing Wang",
    "id": "2350827994",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Meng Xu",
    "id": "2200077525",
    "h_index": 5,
    "papers": 19
   },
   {
    "name": "Yunfan Jiang",
    "id": "2171112793",
    "h_index": 11,
    "papers": 16
   },
   {
    "name": "Fernando Casta\u00f1eda",
    "id": "2350859158",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Fengyuan Hu",
    "id": "2352992614",
    "h_index": 8,
    "papers": 11
   },
   {
    "name": "Y. Tan",
    "id": "2350869215",
    "h_index": 7,
    "papers": 9
   },
   {
    "name": "Letian Fu",
    "id": "2087112499",
    "h_index": 12,
    "papers": 28
   },
   {
    "name": "Trevor Darrell",
    "id": "2257973285",
    "h_index": 11,
    "papers": 27
   },
   {
    "name": "Furong Huang",
    "id": "2408754565",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Yuke Zhu",
    "id": "2258068214",
    "h_index": 13,
    "papers": 21
   },
   {
    "name": "Dan Xu",
    "id": "2299305501",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "L. Fan",
    "id": "2257381161",
    "h_index": 18,
    "papers": 25
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "dexterous-manipulation",
   "egocentric-data",
   "foundation-pretraining"
  ],
  "orgs": [
   "Physical Intelligence"
  ],
  "abs_url": "https://arxiv.org/abs/2602.16710v1",
  "pdf_url": "https://arxiv.org/pdf/2602.16710v1",
  "html_url": "https://arxiv.org/html/2602.16710v1",
  "code_url": null,
  "presented": [
   {
    "club": "saturday-robotics",
    "club_name": "Saturday Robotics",
    "session_number": 8,
    "session_title": "Robotics & World Models Reading Club 08: Embodied Human Data as the \u201cInternet of Motion and Behavior\u201d \u2014 San Francisco 0516",
    "date_text": "Saturday, May 16, 2026 | 2:00 PM \u2013 5:00 PM",
    "city": "San Francisco, CA",
    "url": "https://lu.ma/qoxioge7",
    "listed_as": "World Models from Human Experience"
   },
   {
    "club": "saturday-robotics",
    "club_name": "Saturday Robotics",
    "session_number": 13,
    "session_title": "Robotics & World Models Reading Club 13: HumanEgo: Train Robot Policy from 30 min Egocentric Videos \u2014 SF 0620",
    "date_text": "Saturday, June 20, 2026 | 2:00 PM \u2013 5:00 PM",
    "city": "San Francisco, CA",
    "url": "https://lu.ma/6vkhxnum",
    "listed_as": ""
   }
  ],
  "club_note": "Constructs action-conditioned latent world models from human experience data for prediction, planning, and imagination-based control.",
  "featured": true,
  "signal": 5.36
 },
 {
  "id": "2602.15922",
  "slug": "world-action-models-are-zero-shot-policies",
  "title": "World Action Models are Zero-shot Policies",
  "abstract": "State-of-the-art Vision-Language-Action (VLA) models excel at semantic generalization but struggle to generalize to unseen physical motions in novel environments. We introduce DreamZero, a World Action Model (WAM) built upon a pretrained video diffusion backbone. Unlike VLAs, WAMs learn physical dynamics by predicting future world states and actions, using video as a dense representation of how the world evolves. By jointly modeling video and action, DreamZero learns diverse skills effectively from heterogeneous robot data without relying on repetitive demonstrations. This results in over 2x improvement in generalization to new tasks and environments compared to state-of-the-art VLAs in real robot experiments. Crucially, through model and system optimizations, we enable a 14B autoregressive video diffusion model to perform real-time closed-loop control at 7Hz. Finally, we demonstrate two forms of cross-embodiment transfer: video-only demonstrations from other robots or humans yield a relative improvement of over 42% on unseen task performance with just 10-20 minutes of data. More surprisingly, DreamZero enables few-shot embodiment adaptation, transferring to a new embodiment with only 30 minutes of play data while retaining zero-shot generalization.",
  "published": "2026-02-17",
  "updated": "2026-02-17",
  "year": "2026",
  "authors": [
   "Seonghyeon Ye",
   "Yunhao Ge",
   "Kaiyuan Zheng",
   "Shenyuan Gao",
   "Sihyun Yu",
   "George Kurian",
   "Suneel Indupuru",
   "You Liang Tan",
   "Chuning Zhu",
   "Jiannan Xiang",
   "Ayaan Malik",
   "Kyungmin Lee",
   "William Liang",
   "Nadun Ranawaka",
   "Jiasheng Gu",
   "Yinzhen Xu",
   "Guanzhi Wang",
   "Fengyuan Hu",
   "Avnish Narayan",
   "Johan Bjorck",
   "Jing Wang",
   "Gwanghyun Kim",
   "Dantong Niu",
   "Ruijie Zheng",
   "Yuqi Xie",
   "Jimmy Wu",
   "Qi Wang",
   "Ryan Julian",
   "Danfei Xu",
   "Yilun Du",
   "Yevgen Chebotar",
   "Scott Reed",
   "Jan Kautz",
   "Yuke Zhu",
   "Linxi \"Jim\" Fan",
   "Joel Jang"
  ],
  "author_count": 36,
  "categories": [
   "cs.RO",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 276,
  "influential_citations": 31,
  "tldr": "DreamZero, a World Action Model (WAM) built upon a pretrained video diffusion backbone, is introduced, a World Action Model (WAM) built upon a pretrained video diffusion backbone that enables few-shot embodiment adaptation and enables few-shot embodiment adaptation.",
  "doi": "10.48550/arXiv.2602.15922",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Seonghyeon Ye",
    "id": "2152111477",
    "h_index": 23,
    "papers": 30
   },
   {
    "name": "Yunhao Ge",
    "id": "2299104362",
    "h_index": 11,
    "papers": 20
   },
   {
    "name": "Kaiyuan Zheng",
    "id": "2191345759",
    "h_index": 4,
    "papers": 22
   },
   {
    "name": "Shenyuan Gao",
    "id": "2177775616",
    "h_index": 8,
    "papers": 9
   },
   {
    "name": "Sihyun Yu",
    "id": "2052088734",
    "h_index": 13,
    "papers": 24
   },
   {
    "name": "George Kurian",
    "id": "2409497610",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Suneel Indupuru",
    "id": "73308232",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Y. Tan",
    "id": "2350869215",
    "h_index": 7,
    "papers": 9
   },
   {
    "name": "Chuning Zhu",
    "id": "2118160513",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "Jiannan Xiang",
    "id": "2362827946",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "A. Malik",
    "id": "2276271702",
    "h_index": 4,
    "papers": 24
   },
   {
    "name": "Kyungmin Lee",
    "id": "2305845947",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "William Liang",
    "id": "2410899324",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Nadun Ranawaka",
    "id": "2412072627",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Jiasheng Gu",
    "id": "2275750026",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Yinzhen Xu",
    "id": "2351665438",
    "h_index": 7,
    "papers": 8
   },
   {
    "name": "Guanzhi Wang",
    "id": "2316513666",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Fengyuan Hu",
    "id": "2352992614",
    "h_index": 8,
    "papers": 11
   },
   {
    "name": "Avnish Narayan",
    "id": "2014184266",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Johan Bjorck",
    "id": "2362299097",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Jing Wang",
    "id": "2350827994",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Gwanghyun Kim",
    "id": "2109334279",
    "h_index": 11,
    "papers": 23
   },
   {
    "name": "Dantong Niu",
    "id": "2268757542",
    "h_index": 8,
    "papers": 17
   },
   {
    "name": "Ruijie Zheng",
    "id": "2345931905",
    "h_index": 9,
    "papers": 15
   },
   {
    "name": "Yuqi Xie",
    "id": "2218866691",
    "h_index": 11,
    "papers": 17
   },
   {
    "name": "Jimmy Wu",
    "id": "2390404352",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Qi Wang",
    "id": "2358236936",
    "h_index": 4,
    "papers": 16
   },
   {
    "name": "Ryan C. Julian",
    "id": "144885996",
    "h_index": 19,
    "papers": 34
   },
   {
    "name": "Dan Xu",
    "id": "2299305501",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Yilun Du",
    "id": "2378997853",
    "h_index": 12,
    "papers": 26
   },
   {
    "name": "Yevgen Chebotar",
    "id": "2527420",
    "h_index": 33,
    "papers": 57
   },
   {
    "name": "Scott Reed",
    "id": "2344615904",
    "h_index": 8,
    "papers": 9
   },
   {
    "name": "Jan Kautz",
    "id": "2364684748",
    "h_index": 21,
    "papers": 30
   },
   {
    "name": "Yuke Zhu",
    "id": "2258068214",
    "h_index": 13,
    "papers": 21
   },
   {
    "name": "LinxiJimFan",
    "id": "2350861618",
    "h_index": 10,
    "papers": 15
   },
   {
    "name": "J. Jang",
    "id": "2333419032",
    "h_index": 11,
    "papers": 13
   }
  ],
  "comment": "Project page: https://dreamzero0.github.io/",
  "topics": [
   "vla",
   "foundation-pretraining",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2602.15922v1",
  "pdf_url": "https://arxiv.org/pdf/2602.15922v1",
  "html_url": "https://arxiv.org/html/2602.15922v1",
  "code_url": "https://dreamzero0.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.44
 },
 {
  "id": "2602.15827",
  "slug": "perceptive-humanoid-parkour-chaining-dynamic-human-skills-via-motion-m",
  "title": "Perceptive Humanoid Parkour: Chaining Dynamic Human Skills via Motion Matching",
  "abstract": "While recent advances in humanoid locomotion have achieved stable walking on varied terrains, capturing the agility and adaptivity of highly dynamic human motions remains an open challenge. In particular, agile parkour in complex environments demands not only low-level robustness, but also human-like motion expressiveness, long-horizon skill composition, and perception-driven decision-making. In this paper, we present Perceptive Humanoid Parkour (PHP), a modular framework that enables humanoid robots to autonomously perform long-horizon, vision-based parkour across challenging obstacle courses. Our approach first leverages motion matching, formulated as nearest-neighbor search in a feature space, to compose retargeted atomic human skills into long-horizon kinematic trajectories. This framework enables the flexible composition and smooth transition of complex skill chains while preserving the elegance and fluidity of dynamic human motions. Next, we train motion-tracking reinforcement learning (RL) expert policies for these composed motions, and distill them into a single depth-based, multi-skill student policy, using a combination of DAgger and RL. Crucially, the combination of perception and skill composition enables autonomous, context-aware decision-making: using only onboard depth sensing and a discrete 2D velocity command, the robot selects and executes whether to step over, climb onto, vault or roll off obstacles of varying geometries and heights. We validate our framework with extensive real-world experiments on a Unitree G1 humanoid robot, demonstrating highly dynamic parkour skills such as climbing tall obstacles up to 1.25m (96% robot height), as well as long-horizon multi-obstacle traversal with closed-loop adaptation to real-time obstacle perturbations.",
  "published": "2026-02-17",
  "updated": "2026-05-06",
  "year": "2026",
  "authors": [
   "Zhen Wu",
   "Xiaoyu Huang",
   "Lujie Yang",
   "Yuanhang Zhang",
   "Xi Chen",
   "Pieter Abbeel",
   "Rocky Duan",
   "Angjoo Kanazawa",
   "Carmelo Sferrazza",
   "Guanya Shi",
   "C. Karen Liu"
  ],
  "author_count": 11,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 29,
  "influential_citations": 2,
  "tldr": "Perceptive Humanoid Parkour (PHP) is presented, a modular framework that enables humanoid robots to autonomously perform long-horizon, vision-based parkour across challenging obstacle courses and enables autonomous, context-aware decision-making.",
  "doi": "10.48550/arXiv.2602.15827",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhen Wu",
    "id": "2308574851",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Xiaoyu Huang",
    "id": "2383100448",
    "h_index": 6,
    "papers": 6
   },
   {
    "name": "Lujie Yang",
    "id": "2383205600",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Yuanhang Zhang",
    "id": "2343726233",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "K. Sreenath",
    "id": "144116765",
    "h_index": 55,
    "papers": 231
   },
   {
    "name": "Xi Chen",
    "id": "2254208802",
    "h_index": 9,
    "papers": 15
   },
   {
    "name": "Pieter Abbeel",
    "id": "2381724317",
    "h_index": 9,
    "papers": 17
   },
   {
    "name": "Rocky Duan",
    "id": "2381724728",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Angjoo Kanazawa",
    "id": "20615377",
    "h_index": 60,
    "papers": 126
   },
   {
    "name": "Carmelo Sferrazza",
    "id": "47218071",
    "h_index": 21,
    "papers": 44
   },
   {
    "name": "Guanya Shi",
    "id": "2384824402",
    "h_index": 9,
    "papers": 24
   },
   {
    "name": "C. K. Liu",
    "id": "2376138723",
    "h_index": 9,
    "papers": 11
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "rl-control",
   "safety-eval"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2602.15827v2",
  "pdf_url": "https://arxiv.org/pdf/2602.15827v2",
  "html_url": "https://arxiv.org/html/2602.15827v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.98
 },
 {
  "id": "2602.09013",
  "slug": "dexterous-manipulation-policies-from-rgb-human-videos-via-3d-hand-obje",
  "title": "Dexterous Manipulation Policies from RGB Human Videos via 3D Hand-Object Trajectory Reconstruction",
  "abstract": "Multi-finger robotic hand manipulation and grasping are challenging due to the high-dimensional action space and the difficulty of acquiring large-scale training data. Existing approaches largely rely on human teleoperation with wearable devices or specialized sensing equipment to capture hand-object interactions, which limits scalability. In this work, we propose VIDEOMANIP, a device-free framework that learns dexterous manipulation directly from RGB human videos. Leveraging recent advances in computer vision, VIDEOMANIP reconstructs explicit 3D robot-object trajectories from monocular videos by estimating human hand poses, object meshes, and retargets the reconstructed human motions to robotic hands for manipulation learning. To make the reconstructed robot data suitable for dexterous manipulation training, we introduce hand-object contact optimization with interaction-centric grasp modeling, as well as a demonstration synthesis strategy that generates diverse training trajectories from a single video, enabling generalizable policy learning without additional robot demonstrations. In simulation, the learned grasping model achieves a 70.25% success rate across 20 diverse objects using the Inspire Hand. In the real world, manipulation policies trained from RGB videos achieve an average 62.86% success rate across seven tasks using the LEAP Hand, outperforming retargeting-based methods by 15.87%. Project videos are available at videomanip.github.io.",
  "published": "2026-02-09",
  "updated": "2026-02-11",
  "year": "2026",
  "authors": [
   "Hongyi Chen",
   "Tony Dong",
   "Tiancheng Wu",
   "Liquan Wang",
   "Yash Jangir",
   "Yaru Niu",
   "Yufei Ye",
   "Homanga Bharadhwaj",
   "Zackory Erickson",
   "Jeffrey Ichnowski"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 9,
  "influential_citations": 0,
  "tldr": "V VIDEOMANIP is a device-free framework that learns dexterous manipulation directly from RGB human videos that reconstructs explicit 3D robot-object trajectories from monocular videos by estimating human hand poses, object meshes, and retargets the reconstructed human motions to robotic hands for manipulation learning.",
  "doi": "10.48550/arXiv.2602.09013",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hongyi Chen",
    "id": "2309203483",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Tony Dong",
    "id": "2369339728",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Tiancheng Wu",
    "id": "2381070716",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Liquang Wang",
    "id": "2108908463",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Yash Jangir",
    "id": "2130181270",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Yaru Niu",
    "id": "2253400416",
    "h_index": 7,
    "papers": 16
   },
   {
    "name": "Yufei Ye",
    "id": "9653518",
    "h_index": 10,
    "papers": 18
   },
   {
    "name": "Homanga Bharadhwaj",
    "id": "51113848",
    "h_index": 23,
    "papers": 59
   },
   {
    "name": "Zackory Erickson",
    "id": "2360174150",
    "h_index": 4,
    "papers": 20
   },
   {
    "name": "Jeffrey Ichnowski",
    "id": "2269146110",
    "h_index": 9,
    "papers": 29
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "egocentric-data",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2602.09013v2",
  "pdf_url": "https://arxiv.org/pdf/2602.09013v2",
  "html_url": "https://arxiv.org/html/2602.09013v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.0
 },
 {
  "id": "2602.06949",
  "slug": "dreamdojo-a-generalist-robot-world-model-from-large-scale-human-videos",
  "title": "DreamDojo: A Generalist Robot World Model from Large-Scale Human Videos",
  "abstract": "Being able to simulate the outcomes of actions in varied environments will revolutionize the development of generalist agents at scale. However, modeling these world dynamics, especially for dexterous robotics tasks, poses significant challenges due to limited data coverage and scarce action labels. As an endeavor towards this end, we introduce DreamDojo, a foundation world model that learns diverse interactions and dexterous controls from 44k hours of egocentric human videos. Our data mixture represents the largest video dataset to date for world model pretraining, spanning a wide range of daily scenarios with diverse objects and skills. To address the scarcity of action labels, we introduce continuous latent actions as unified proxy actions, enhancing interaction knowledge transfer from unlabeled videos. After post-training on small-scale target robot data, DreamDojo demonstrates a strong understanding of physics and precise action controllability. We also devise a distillation pipeline that accelerates DreamDojo to a real-time speed of 10.81 FPS and further improves context consistency. Our work enables several important applications based on generative world models, including live teleoperation, policy evaluation, and model-based planning. Systematic evaluation on multiple challenging out-of-distribution (OOD) benchmarks verifies the significance of our method for simulating open-world, contact-rich tasks, paving the way for general-purpose robot world models.",
  "published": "2026-02-06",
  "updated": "2026-02-06",
  "year": "2026",
  "authors": [
   "Shenyuan Gao",
   "William Liang",
   "Kaiyuan Zheng",
   "Ayaan Malik",
   "Seonghyeon Ye",
   "Sihyun Yu",
   "Wei-Cheng Tseng",
   "Yuzhu Dong",
   "Kaichun Mo",
   "Chen-Hsuan Lin",
   "Qianli Ma",
   "Seungjun Nah",
   "Loic Magne",
   "Jiannan Xiang",
   "Yuqi Xie",
   "Ruijie Zheng",
   "Dantong Niu",
   "You Liang Tan",
   "K. R. Zentner",
   "George Kurian",
   "Suneel Indupuru",
   "Pooya Jannaty",
   "Jinwei Gu",
   "Jun Zhang",
   "Jitendra Malik",
   "Pieter Abbeel",
   "Ming-Yu Liu",
   "Yuke Zhu",
   "Joel Jang",
   "Linxi \"Jim\" Fan"
  ],
  "author_count": 30,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 96,
  "influential_citations": 11,
  "tldr": "Systematic evaluation on multiple challenging out-of-distribution (OOD) benchmarks verifies the significance of the method for simulating open-world, contact-rich tasks, paving the way for general-purpose robot world models.",
  "doi": "10.48550/arXiv.2602.06949",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Shenyuan Gao",
    "id": "2177775616",
    "h_index": 8,
    "papers": 9
   },
   {
    "name": "William Liang",
    "id": "2410899324",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Kaiyuan Zheng",
    "id": "2333236365",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "A. Malik",
    "id": "2276271702",
    "h_index": 4,
    "papers": 24
   },
   {
    "name": "Seonghyeon Ye",
    "id": "2152111477",
    "h_index": 23,
    "papers": 30
   },
   {
    "name": "Sihyun Yu",
    "id": "2052088734",
    "h_index": 13,
    "papers": 24
   },
   {
    "name": "Wei-Cheng Tseng",
    "id": "2321873035",
    "h_index": 8,
    "papers": 17
   },
   {
    "name": "Yuzhu Dong",
    "id": "2115460097",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Kaichun Mo",
    "id": "2261042152",
    "h_index": 8,
    "papers": 14
   },
   {
    "name": "Chen-Hsuan Lin",
    "id": "2313497801",
    "h_index": 8,
    "papers": 11
   },
   {
    "name": "Qianli Ma",
    "id": "2346907867",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Seungjun Nah",
    "id": "40648435",
    "h_index": 21,
    "papers": 30
   },
   {
    "name": "Loic Magne",
    "id": "2350862986",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Jiannan Xiang",
    "id": "2362827946",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Yuqi Xie",
    "id": "2218866691",
    "h_index": 11,
    "papers": 17
   },
   {
    "name": "Ruijie Zheng",
    "id": "2345931905",
    "h_index": 9,
    "papers": 15
   },
   {
    "name": "Dantong Niu",
    "id": "2268757542",
    "h_index": 8,
    "papers": 17
   },
   {
    "name": "Y. Tan",
    "id": "2350869215",
    "h_index": 7,
    "papers": 9
   },
   {
    "name": "K. Zentner",
    "id": "2101709172",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "George Kurian",
    "id": "2409497610",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Suneel Indupuru",
    "id": "73308232",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Pooya Jannaty",
    "id": "2313203519",
    "h_index": 8,
    "papers": 9
   },
   {
    "name": "Jinwei Gu",
    "id": "2338980231",
    "h_index": 9,
    "papers": 13
   },
   {
    "name": "Jun Zhang",
    "id": "2291155419",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Jitendra Malik",
    "id": "2362090500",
    "h_index": 8,
    "papers": 18
   },
   {
    "name": "Pieter Abbeel",
    "id": "2350871358",
    "h_index": 8,
    "papers": 22
   },
   {
    "name": "Ming-Yu Liu",
    "id": "2385747822",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "Yuke Zhu",
    "id": "2253507326",
    "h_index": 16,
    "papers": 20
   },
   {
    "name": "J. Jang",
    "id": "2333419032",
    "h_index": 11,
    "papers": 13
   },
   {
    "name": "LinxiJimFan",
    "id": "2350861618",
    "h_index": 10,
    "papers": 15
   }
  ],
  "comment": "Project page: https://dreamdojo-world.github.io/",
  "topics": [
   "world-models",
   "dexterous-manipulation",
   "egocentric-data",
   "tactile",
   "foundation-pretraining",
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2602.06949v1",
  "pdf_url": "https://arxiv.org/pdf/2602.06949v1",
  "html_url": "https://arxiv.org/html/2602.06949v1",
  "code_url": "https://dreamdojo-world.github.io/",
  "presented": [
   {
    "club": "saturday-robotics",
    "club_name": "Saturday Robotics",
    "session_number": 8,
    "session_title": "Robotics & World Models Reading Club 08: Embodied Human Data as the \u201cInternet of Motion and Behavior\u201d \u2014 San Francisco 0516",
    "date_text": "Saturday, May 16, 2026 | 2:00 PM \u2013 5:00 PM",
    "city": "San Francisco, CA",
    "url": "https://lu.ma/qoxioge7",
    "listed_as": "Learning Generalist Robot Policies from Human Demonstrations"
   }
  ],
  "club_note": "Trains multitask generalist robot policies through sequence modeling over heterogeneous human demonstrations.",
  "featured": true,
  "signal": 5.99
 },
 {
  "id": "2601.22074",
  "slug": "mjlab-a-lightweight-framework-for-gpu-accelerated-robot-learning",
  "title": "mjlab: A Lightweight Framework for GPU-Accelerated Robot Learning",
  "abstract": "We present mjlab, a lightweight, open-source framework for robot learning that combines GPU-accelerated simulation with composable environments and minimal setup friction. mjlab adopts the manager-based API introduced by Isaac Lab, where users compose modular building blocks for observations, rewards, and events, and pairs it with MuJoCo Warp for GPU-accelerated physics. The result is a framework installable with a single command, requiring minimal dependencies, and providing direct access to native MuJoCo data structures. mjlab ships with reference implementations of velocity tracking, motion imitation, and manipulation tasks.",
  "published": "2026-01-29",
  "updated": "2026-02-25",
  "year": "2026",
  "authors": [
   "Kevin Zakka",
   "Qiayuan Liao",
   "Brent Yi",
   "Louis Le Lay",
   "Koushil Sreenath",
   "Pieter Abbeel"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 34,
  "influential_citations": 2,
  "tldr": "",
  "doi": "10.48550/arXiv.2601.22074",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Kevin Zakka",
    "id": "1390101014",
    "h_index": 9,
    "papers": 10
   },
   {
    "name": "Qiayuan Liao",
    "id": "1713616371",
    "h_index": 13,
    "papers": 20
   },
   {
    "name": "Brent Yi",
    "id": "2242880086",
    "h_index": 18,
    "papers": 27
   },
   {
    "name": "L. Lay",
    "id": "2321599348",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "K. Sreenath",
    "id": "144116765",
    "h_index": 55,
    "papers": 231
   },
   {
    "name": "Pieter Abbeel",
    "id": "2381724317",
    "h_index": 9,
    "papers": 17
   }
  ],
  "comment": "Comments: 11 pages; Code is available at https://github.com/mujocolab/mjlab ; Expanded sensor and domain randomization sections, added references, minor edits",
  "topics": [
   "sim2real"
  ],
  "orgs": [
   "NVIDIA"
  ],
  "abs_url": "https://arxiv.org/abs/2601.22074v2",
  "pdf_url": "https://arxiv.org/pdf/2601.22074v2",
  "html_url": "https://arxiv.org/html/2601.22074v2",
  "code_url": "https://github.com/mujocolab/mjlab",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.04
 },
 {
  "id": "2601.20540",
  "slug": "advancing-open-source-world-models",
  "title": "Advancing Open-source World Models",
  "abstract": "We present LingBot-World, an open-sourced world simulator stemming from video generation. Positioned as a top-tier world model, LingBot-World offers the following features. (1) It maintains high fidelity and robust dynamics in a broad spectrum of environments, including realism, scientific contexts, cartoon styles, and beyond. (2) It enables a minute-level horizon while preserving contextual consistency over time, which is also known as \"long-term memory\". (3) It supports real-time interactivity, achieving a latency of under 1 second when producing 16 frames per second. We provide public access to the code and model in an effort to narrow the divide between open-source and closed-source technologies. We believe our release will empower the community with practical applications across areas like content creation, gaming, and robot learning.",
  "published": "2026-01-28",
  "updated": "2026-01-28",
  "year": "2026",
  "authors": [
   " Robbyant Team",
   "Zelin Gao",
   "Qiuyu Wang",
   "Yanhong Zeng",
   "Jiapeng Zhu",
   "Ka Leong Cheng",
   "Yixuan Li",
   "Hanlin Wang",
   "Yinghao Xu",
   "Shuailei Ma",
   "Yihang Chen",
   "Jie Liu",
   "Yansong Cheng",
   "Yao Yao",
   "Jiayi Zhu",
   "Yihao Meng",
   "Kecheng Zheng",
   "Qingyan Bai",
   "Jingye Chen",
   "Zehong Shen",
   "Yue Yu",
   "Xing Zhu",
   "Yujun Shen",
   "Hao Ouyang"
  ],
  "author_count": 24,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 98,
  "influential_citations": 23,
  "tldr": "LingBot-World is presented, an open-sourced world simulator stemming from video generation that maintains high fidelity and robust dynamics in a broad spectrum of environments, including realism, scientific contexts, cartoon styles, and beyond.",
  "doi": "10.48550/arXiv.2601.20540",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "R. Gao",
    "id": "2393888688",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Qiuyu Wang",
    "id": "2220375698",
    "h_index": 12,
    "papers": 29
   },
   {
    "name": "Yanhong Zeng",
    "id": "2386790447",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Jiapeng Zhu",
    "id": "47054925",
    "h_index": 13,
    "papers": 32
   },
   {
    "name": "K. Cheng",
    "id": "2297462874",
    "h_index": 10,
    "papers": 23
   },
   {
    "name": "Yixuan Li",
    "id": "2387345930",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Hanlin Wang",
    "id": "2335893095",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Yinghao Xu",
    "id": "121983635",
    "h_index": 35,
    "papers": 74
   },
   {
    "name": "Shuailei Ma",
    "id": "2293403497",
    "h_index": 9,
    "papers": 17
   },
   {
    "name": "Yihan Chen",
    "id": "2277275491",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Jie Liu",
    "id": "2380078693",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Yansong Cheng",
    "id": "2407648046",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Yao Yao",
    "id": "2267861475",
    "h_index": 11,
    "papers": 17
   },
   {
    "name": "Jiayi Zhu",
    "id": "2224768981",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Yihao Meng",
    "id": "2297635473",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Kecheng Zheng",
    "id": "2295668874",
    "h_index": 7,
    "papers": 7
   },
   {
    "name": "Qingyan Bai",
    "id": "2083548117",
    "h_index": 10,
    "papers": 19
   },
   {
    "name": "Jingye Chen",
    "id": "2244136105",
    "h_index": 15,
    "papers": 27
   },
   {
    "name": "Zehong Shen",
    "id": "1384523019",
    "h_index": 10,
    "papers": 19
   },
   {
    "name": "Yue Yu",
    "id": "2331287555",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "Xing Zhu",
    "id": "2399027419",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Yujun Shen",
    "id": "2392945842",
    "h_index": 6,
    "papers": 21
   },
   {
    "name": "Ouyang Hao",
    "id": "2247557433",
    "h_index": 14,
    "papers": 29
   }
  ],
  "comment": "Project page: https://technology.robbyant.com/lingbot-world; Code: https://github.com/robbyant/lingbot-world",
  "topics": [
   "world-models",
   "sim2real",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2601.20540v1",
  "pdf_url": "https://arxiv.org/pdf/2601.20540v1",
  "html_url": "https://arxiv.org/html/2601.20540v1",
  "code_url": "https://github.com/robbyant/lingbot-world",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.0
 },
 {
  "id": "2601.20334",
  "slug": "demonstration-free-robotic-control-via-llm-agents",
  "title": "Demonstration-Free Robotic Control via LLM Agents",
  "abstract": "Robotic manipulation has increasingly adopted vision-language-action (VLA) models, which achieve strong performance but typically require task-specific demonstrations and fine-tuning, and often generalize poorly under domain shift. We investigate whether general-purpose large language model (LLM) agent frameworks, originally developed for software engineering, can serve as an alternative control paradigm for embodied manipulation. We introduce FAEA (Frontier Agent as Embodied Agent), which applies an LLM agent framework directly to embodied manipulation without modification. Using the same iterative reasoning that enables software agents to debug code, FAEA enables embodied agents to reason through manipulation strategies. We evaluate an unmodified frontier agent, Claude Agent SDK, across the LIBERO, ManiSkill3, and MetaWorld benchmarks. With privileged environment state access, FAEA achieves success rates of 84.9%, 85.7%, and 96%, respectively. This level of task success approaches that of VLA models trained with less than 100 demonstrations per task, without requiring demonstrations or fine-tuning. With one round of human feedback as an optional optimization, performance increases to 88.2% on LIBERO. This demonstration-free capability has immediate practical value: FAEA can autonomously explore novel scenarios in simulation and generate successful trajectories for training data augmentation in embodied learning. Our results indicate that general-purpose agents are sufficient for a class of manipulation tasks dominated by deliberative, task-level planning. This opens a path for robotics systems to leverage actively maintained agent infrastructure and benefit directly from ongoing advances in frontier models. Code is available at https://github.com/robiemusketeer/faea-sim",
  "published": "2026-01-28",
  "updated": "2026-06-28",
  "year": "2026",
  "authors": [
   "Brian Y. Tsui",
   "Alan Y. Fang",
   "Tiffany J. Hwu"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 2,
  "influential_citations": 0,
  "tldr": "This work introduces FAEA (Frontier Agent as Embodied Agent), which applies an LLM agent framework directly to embodied manipulation without modification, and indicates that general-purpose agents are sufficient for a class of manipulation tasks dominated by deliberative, task-level planning.",
  "doi": "10.48550/arXiv.2601.20334",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Brian Y. Tsui",
    "id": "2407204585",
    "h_index": 0,
    "papers": 1
   },
   {
    "name": "A. Fang",
    "id": "49394781",
    "h_index": 0,
    "papers": 4
   },
   {
    "name": "Tiffany J. Hwu",
    "id": "2407204604",
    "h_index": 0,
    "papers": 1
   }
  ],
  "comment": "Accepted to IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS) 2026",
  "topics": [
   "vla"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2601.20334v2",
  "pdf_url": "https://arxiv.org/pdf/2601.20334v2",
  "html_url": "https://arxiv.org/html/2601.20334v2",
  "code_url": null,
  "presented": [
   {
    "club": "saturday-robotics",
    "club_name": "Saturday Robotics",
    "session_number": 16,
    "session_title": "Robotics & World Models Reading Club 16: The Embodied AI Hardware Stack \u2014 Supply Chain, Sensors, and the Data Flywheel \u2014 SF 07/04",
    "date_text": "Saturday, July 4, 2026 | 2:00 PM \u2013 5:00 PM",
    "city": "San Francisco, CA",
    "url": "https://lu.ma/cgzyfpeb",
    "listed_as": "Pre-reading:"
   }
  ],
  "club_note": "Github: https://github.com/robiemusketeer/faea-sim",
  "featured": true,
  "signal": 4.98
 },
 {
  "id": "2601.12799",
  "slug": "from-w1-towards-general-humanoid-whole-body-control-with-language-inst",
  "title": "FRoM-W1: Towards General Humanoid Whole-Body Control with Language Instructions",
  "abstract": "Humanoid robots are capable of performing various actions such as greeting, dancing and even backflipping. However, these motions are often hard-coded or specifically trained, which limits their versatility. In this work, we present FRoM-W1, an open-source framework designed to achieve general humanoid whole-body motion control using natural language. To universally understand natural language and generate corresponding motions, as well as enable various humanoid robots to stably execute these motions in the physical world under gravity, FRoM-W1 operates in two stages: (a) H-GPT: utilizing massive human data, a large-scale language-driven human whole-body motion generation model is trained to generate diverse natural behaviors. We further leverage the Chain-of-Thought technique to improve the model's generalization in instruction understanding. (b) H-ACT: After retargeting generated human whole-body motions into robot-specific actions, a motion controller that is pretrained and further fine-tuned through reinforcement learning in physical simulation enables humanoid robots to accurately and stably perform corresponding actions. It is then deployed on real robots via a modular simulation-to-reality module. We extensively evaluate FRoM-W1 on Unitree H1 and G1 robots. Results demonstrate superior performance on the HumanML3D-X benchmark for human whole-body motion generation, and our introduced reinforcement learning fine-tuning consistently improves both motion tracking accuracy and task success rates of these humanoid robots. We open-source the entire FRoM-W1 framework and hope it will advance the development of humanoid intelligence.",
  "published": "2026-01-19",
  "updated": "2026-01-19",
  "year": "2026",
  "authors": [
   "Peng Li",
   "Zihan Zhuang",
   "Yangfan Gao",
   "Yi Dong",
   "Sixian Li",
   "Changhao Jiang",
   "Shihan Dou",
   "Zhiheng Xi",
   "Enyu Zhou",
   "Jixuan Huang",
   "Hui Li",
   "Jingjing Gong",
   "Xingjun Ma",
   "Tao Gui",
   "Zuxuan Wu",
   "Qi Zhang",
   "Xuanjing Huang",
   "Yu-Gang Jiang",
   "Xipeng Qiu"
  ],
  "author_count": 19,
  "categories": [
   "cs.RO",
   "cs.CL",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 11,
  "influential_citations": 1,
  "tldr": "This work presents FRoM-W1, an open-source framework designed to achieve general humanoid whole-body motion control using natural language, and introduced reinforcement learning fine-tuning consistently improves both motion tracking accuracy and task success rates of these humanoid robots.",
  "doi": "10.48550/arXiv.2601.12799",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Peng Li",
    "id": "2062390146",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Zihan Zhuang",
    "id": "2322181108",
    "h_index": 4,
    "papers": 17
   },
   {
    "name": "Yangfan Gao",
    "id": "2354088281",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Yi Dong",
    "id": "2344198993",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Sixian Li",
    "id": "2277430485",
    "h_index": 6,
    "papers": 14
   },
   {
    "name": "Changhao Jiang",
    "id": "2240482661",
    "h_index": 6,
    "papers": 21
   },
   {
    "name": "Shihan Dou",
    "id": "2042683163",
    "h_index": 26,
    "papers": 116
   },
   {
    "name": "Zhiheng Xi",
    "id": "2218237934",
    "h_index": 17,
    "papers": 83
   },
   {
    "name": "Enyu Zhou",
    "id": "2240446306",
    "h_index": 13,
    "papers": 30
   },
   {
    "name": "Jixuan Huang",
    "id": "2332490660",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Hui Li",
    "id": "2374636352",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Jingjing Gong",
    "id": "49294702",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "X. Ma",
    "id": "2239782048",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Tao Gui",
    "id": "2067331064",
    "h_index": 42,
    "papers": 206
   },
   {
    "name": "Zuxuan Wu",
    "id": "3099139",
    "h_index": 59,
    "papers": 193
   },
   {
    "name": "Qi Zhang",
    "id": "2403599333",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Xuanjing Huang",
    "id": "2257129989",
    "h_index": 33,
    "papers": 185
   },
   {
    "name": "Yu-Gang Jiang",
    "id": "1717861",
    "h_index": 47,
    "papers": 142
   },
   {
    "name": "Xipeng Qiu",
    "id": "2282972251",
    "h_index": 15,
    "papers": 39
   }
  ],
  "comment": "Project Page: https://openmoss.github.io/FRoM-W1",
  "topics": [
   "humanoids",
   "rl-control"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2601.12799v1",
  "pdf_url": "https://arxiv.org/pdf/2601.12799v1",
  "html_url": "https://arxiv.org/html/2601.12799v1",
  "code_url": "https://openmoss.github.io/FRoM-W1",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.58
 },
 {
  "id": "2601.07701",
  "slug": "deep-whole-body-parkour",
  "title": "Deep Whole-body Parkour",
  "abstract": "Current approaches to humanoid control generally fall into two paradigms: perceptive locomotion, which handles terrain well but is limited to pedal gaits, and general motion tracking, which reproduces complex skills but ignores environmental capabilities. This work unites these paradigms to achieve perceptive general motion control. We present a framework where exteroceptive sensing is integrated into whole-body motion tracking, permitting a humanoid to perform highly dynamic, non-locomotion tasks on uneven terrain. By training a single policy to perform multiple distinct motions across varied terrestrial features, we demonstrate the non-trivial benefit of integrating perception into the control loop. Our results show that this framework enables robust, highly dynamic multi-contact motions, such as vaulting and dive-rolling, on unstructured terrain, significantly expanding the robot's traversability beyond simple walking or running. https://project-instinct.github.io/deep-whole-body-parkour",
  "published": "2026-01-12",
  "updated": "2026-01-12",
  "year": "2026",
  "authors": [
   "Ziwen Zhuang",
   "Shaoting Zhu",
   "Mengjie Zhao",
   "Hang Zhao"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 12,
  "influential_citations": 0,
  "tldr": "This work presents a framework where exteroceptive sensing is integrated into whole-body motion tracking, permitting a humanoid to perform highly dynamic, non-locomotion tasks on uneven terrain, and demonstrates the non-trivial benefit of integrating perception into the control loop.",
  "doi": "10.48550/arXiv.2601.07701",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ziwen Zhuang",
    "id": "1972362408",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Shaoting Zhu",
    "id": "2216251632",
    "h_index": 8,
    "papers": 16
   },
   {
    "name": "Mengjie Zhao",
    "id": "2404323436",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Hang Zhao",
    "id": "2239158612",
    "h_index": 5,
    "papers": 9
   }
  ],
  "comment": "",
  "topics": [
   "humanoids"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2601.07701v1",
  "pdf_url": "https://arxiv.org/pdf/2601.07701v1",
  "html_url": "https://arxiv.org/html/2601.07701v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.11
 },
 {
  "id": "2601.06286",
  "slug": "walk-the-planc-physics-guided-rl-for-agile-humanoid-locomotion-on-cons",
  "title": "Walk the PLANC: Physics-Guided RL for Agile Humanoid Locomotion on Constrained Footholds",
  "abstract": "Bipedal humanoid robots must precisely coordinate balance, timing, and contact decisions when locomoting on constrained footholds such as stepping stones, beams, and planks -- even minor errors can lead to catastrophic failure. Classical optimization and control pipelines handle these constraints well but depend on highly accurate mathematical representations of terrain geometry, making them prone to error when perception is noisy or incomplete. Meanwhile, reinforcement learning has shown strong resilience to disturbances and modeling errors, yet end-to-end policies rarely discover the precise foothold placement and step sequencing required for discontinuous terrain. These contrasting limitations motivate approaches that guide learning with physics-based structure rather than relying purely on reward shaping. In this work, we introduce a locomotion framework in which a reduced-order stepping planner supplies dynamically consistent motion targets that steer the RL training process via Control Lyapunov Function (CLF) rewards. This combination of structured footstep planning and data-driven adaptation produces accurate, agile, and hardware-validated stepping-stone locomotion on a humanoid robot, substantially improving reliability compared to conventional model-free reinforcement-learning baselines.",
  "published": "2026-01-09",
  "updated": "2026-01-09",
  "year": "2026",
  "authors": [
   "Min Dai",
   "William D. Compton",
   "Junheng Li",
   "Lizhi Yang",
   "Aaron D. Ames"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 8,
  "influential_citations": 1,
  "tldr": "A reduced-order stepping planner supplies dynamically consistent motion targets that steer the RL training process via Control Lyapunov Function (CLF) rewards, which produces accurate, agile, and hardware-validated stepping-stone locomotion on a humanoid robot, substantially improving reliability compared to conventional model-free reinforcement-learning baselines.",
  "doi": "10.48550/arXiv.2601.06286",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Min Dai",
    "id": "1633564264",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "William D. Compton",
    "id": "2317009268",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Junheng Li",
    "id": "2108988997",
    "h_index": 7,
    "papers": 25
   },
   {
    "name": "Lizhi Yang",
    "id": "2362153740",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Aaron D. Ames",
    "id": "2338277217",
    "h_index": 4,
    "papers": 24
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2601.06286v1",
  "pdf_url": "https://arxiv.org/pdf/2601.06286v1",
  "html_url": "https://arxiv.org/html/2601.06286v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.95
 },
 {
  "id": "2601.05573",
  "slug": "orient-anything-v2-unifying-orientation-and-rotation-understanding",
  "title": "Orient Anything V2: Unifying Orientation and Rotation Understanding",
  "abstract": "This work presents Orient Anything V2, an enhanced foundation model for unified understanding of object 3D orientation and rotation from single or paired images. Building upon Orient Anything V1, which defines orientation via a single unique front face, V2 extends this capability to handle objects with diverse rotational symmetries and directly estimate relative rotations. These improvements are enabled by four key innovations: 1) Scalable 3D assets synthesized by generative models, ensuring broad category coverage and balanced data distribution; 2) An efficient, model-in-the-loop annotation system that robustly identifies 0 to N valid front faces for each object; 3) A symmetry-aware, periodic distribution fitting objective that captures all plausible front-facing orientations, effectively modeling object rotational symmetry; 4) A multi-frame architecture that directly predicts relative object rotations. Extensive experiments show that Orient Anything V2 achieves state-of-the-art zero-shot performance on orientation estimation, 6DoF pose estimation, and object symmetry recognition across 11 widely used benchmarks. The model demonstrates strong generalization, significantly broadening the applicability of orientation estimation in diverse downstream tasks.",
  "published": "2026-01-09",
  "updated": "2026-01-09",
  "year": "2026",
  "authors": [
   "Zehan Wang",
   "Ziang Zhang",
   "Jiayang Xu",
   "Jialei Wang",
   "Tianyu Pang",
   "Chao Du",
   "HengShuang Zhao",
   "Zhou Zhao"
  ],
  "author_count": 8,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 18,
  "influential_citations": 3,
  "tldr": "This work presents Orient Anything V2, an enhanced foundation model for unified understanding of object 3D orientation and rotation from single or paired images, which demonstrates strong generalization, significantly broadening the applicability of orientation estimation in diverse downstream tasks.",
  "doi": "10.48550/arXiv.2601.05573",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zehan Wang",
    "id": "2258561621",
    "h_index": 20,
    "papers": 52
   },
   {
    "name": "Ziang Zhang",
    "id": "2116461847",
    "h_index": 12,
    "papers": 23
   },
   {
    "name": "Jiayang Xu",
    "id": "2311999126",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Jialei Wang",
    "id": "2304513799",
    "h_index": 8,
    "papers": 11
   },
   {
    "name": "Tianyu Pang",
    "id": "2348451895",
    "h_index": 13,
    "papers": 44
   },
   {
    "name": "Chao Du",
    "id": "2364870166",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Hengshuang Zhao",
    "id": "2311545304",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Zhou Zhao",
    "id": "2358285525",
    "h_index": 6,
    "papers": 15
   }
  ],
  "comment": "NeurIPS 2025 Spotlight, Repo: https://github.com/SpatialVision/Orient-Anything-V2",
  "topics": [
   "foundation-pretraining",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2601.05573v1",
  "pdf_url": "https://arxiv.org/pdf/2601.05573v1",
  "html_url": "https://arxiv.org/html/2601.05573v1",
  "code_url": "https://github.com/SpatialVision/Orient-Anything-V2",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.78
 },
 {
  "id": "2601.05230",
  "slug": "learning-latent-action-world-models-in-the-wild",
  "title": "Learning Latent Action World Models In The Wild",
  "abstract": "Agents capable of reasoning and planning in the real world require the ability of predicting the consequences of their actions. While world models possess this capability, they most often require action labels, that can be complex to obtain at scale. This motivates the learning of latent action models, that can learn an action space from videos alone. Our work addresses the problem of learning latent actions world models on in-the-wild videos, expanding the scope of existing works that focus on simple robotics simulations, video games, or manipulation data. While this allows us to capture richer actions, it also introduces challenges stemming from the video diversity, such as environmental noise, or the lack of a common embodiment across videos. To address some of the challenges, we discuss properties that actions should follow as well as relevant architectural choices and evaluations. We find that continuous, but constrained, latent actions are able to capture the complexity of actions from in-the-wild videos, something that the common vector quantization does not. We for example find that changes in the environment coming from agents, such as humans entering the room, can be transferred across videos. This highlights the capability of learning actions that are specific to in-the-wild videos. In the absence of a common embodiment across videos, we are mainly able to learn latent actions that become localized in space, relative to the camera. Nonetheless, we are able to train a controller that maps known actions to latent ones, allowing us to use latent actions as a universal interface and solve planning tasks with our world model with similar performance as action-conditioned baselines. Our analyses and experiments provide a step towards scaling latent action models to the real world.",
  "published": "2026-01-08",
  "updated": "2026-01-20",
  "year": "2026",
  "authors": [
   "Quentin Garrido",
   "Tushar Nagarajan",
   "Basile Terver",
   "Nicolas Ballas",
   "Yann LeCun",
   "Michael Rabbat"
  ],
  "author_count": 6,
  "categories": [
   "cs.AI",
   "cs.CV"
  ],
  "primary_category": "cs.AI",
  "venue": "",
  "venue_source": "",
  "citations": 41,
  "influential_citations": 5,
  "tldr": "This work is able to train a controller that maps known actions to latent ones, allowing us to use latent actions as a universal interface and solve planning tasks with the authors' world model with similar performance as action-conditioned baselines.",
  "doi": "10.48550/arXiv.2601.05230",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Q. Garrido",
    "id": "2048163343",
    "h_index": 16,
    "papers": 22
   },
   {
    "name": "Tushar Nagarajan",
    "id": "2324578816",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Basile Terver",
    "id": "2216065931",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Nicolas Ballas",
    "id": "2289844757",
    "h_index": 13,
    "papers": 19
   },
   {
    "name": "Yann LeCun",
    "id": "2265899558",
    "h_index": 22,
    "papers": 47
   },
   {
    "name": "Michael Rabbat",
    "id": "2284991448",
    "h_index": 12,
    "papers": 18
   }
  ],
  "comment": "37 pages, 25 figures; updated references and experimental details",
  "topics": [
   "world-models"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2601.05230v2",
  "pdf_url": "https://arxiv.org/pdf/2601.05230v2",
  "html_url": "https://arxiv.org/html/2601.05230v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.62
 },
 {
  "id": "2601.04194",
  "slug": "choreographing-a-world-of-dynamic-objects",
  "title": "Choreographing a World of Dynamic Objects",
  "abstract": "Dynamic objects in our physical 4D (3D + time) world are constantly evolving, deforming, and interacting with other objects, leading to diverse 4D scene dynamics. In this paper, we present a universal generative pipeline, CHORD, for CHOReographing Dynamic objects and scenes and synthesizing this type of phenomena. Traditional rule-based graphics pipelines to create these dynamics are based on category-specific heuristics, yet are labor-intensive and not scalable. Recent learning-based methods typically demand large-scale datasets, which may not cover all object categories in interest. Our approach instead inherits the universality from the video generative models by proposing a distillation-based pipeline to extract the rich Lagrangian motion information hidden in the Eulerian representations of 2D videos. Our method is universal, versatile, and category-agnostic. We demonstrate its effectiveness by conducting experiments to generate a diverse range of multi-body 4D dynamics, show its advantage compared to existing methods, and demonstrate its applicability in generating robotics manipulation policies. Project page: https://yanzhelyu.github.io/chord",
  "published": "2026-01-07",
  "updated": "2026-01-07",
  "year": "2026",
  "authors": [
   "Yanzhe Lyu",
   "Chen Geng",
   "Karthik Dharmarajan",
   "Yunzhi Zhang",
   "Hadi Alzayer",
   "Shangzhe Wu",
   "Jiajun Wu"
  ],
  "author_count": 7,
  "categories": [
   "cs.CV",
   "cs.GR",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 4,
  "influential_citations": 1,
  "tldr": "This paper presents a universal generative pipeline, CHORD, for CHOReographing Dynamic objects and scenes and synthesizing this type of phenomena, and demonstrates its effectiveness by conducting experiments, and shows its advantage compared to existing methods.",
  "doi": "10.48550/arXiv.2601.04194",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yanzhe Lyu",
    "id": "2403070725",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Chen Geng",
    "id": "2334353317",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Karthik Dharmarajan",
    "id": "2401833053",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Yunzhi Zhang",
    "id": "2261420360",
    "h_index": 14,
    "papers": 30
   },
   {
    "name": "Hadi Alzayer",
    "id": "2127598598",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Shangzhe Wu",
    "id": "2112538311",
    "h_index": 11,
    "papers": 21
   },
   {
    "name": "Jiajun Wu",
    "id": "2326444500",
    "h_index": 35,
    "papers": 214
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2601.04194v1",
  "pdf_url": "https://arxiv.org/pdf/2601.04194v1",
  "html_url": "https://arxiv.org/html/2601.04194v1",
  "code_url": null,
  "presented": [
   {
    "club": "saturday-robotics",
    "club_name": "Saturday Robotics",
    "session_number": 26,
    "session_title": "Robotics & World Models Reading Club 26: Video Generation to Robot Manipulation: Bridging Embodiment Gap+Video-Tactile-Action Model. SF 8/29",
    "date_text": "Saturday, August 29, 2026 | 2:00 PM \u2013 5:00 PM",
    "city": "San Francisco, CA",
    "url": "https://lu.ma/00z3oxw6",
    "listed_as": ""
   }
  ],
  "club_note": "Keynote 2: VTAM: Video-Tactile-Action Models for Complex Physical Interaction Beyond VLAs",
  "featured": true,
  "signal": 3.7
 },
 {
  "id": "2601.04153",
  "slug": "diffusion-drf-free-rich-and-differentiable-reward-for-video-diffusion",
  "title": "Diffusion-DRF: Free, Rich, and Differentiable Reward for Video Diffusion Fine-Tuning",
  "abstract": "Video diffusion alignment has been heavily relied on scalar rewards. These rewards are typically derived from learned reward models in human preference datasets, requiring additional training and extensive collection. Moreover, scalar rewards provide coarse, global supervision, offering limited prompt-generation mismatch credit assignment and making models prone to reward exploitation and unstable optimization. We propose Diffusion-DRF, a free, rich, and differentiable reward framework for video diffusion fine-tuning. Diffusion-DRF employs a frozen, off-the-shelf Vision-Language Model (VLM) as the critic, eliminating the need for reward model training. Instead of relying on a single scalar reward, it decomposes each user prompt into multi-dimensional questions with freeform dense VQA explanation queries, yielding information-rich feedback. By direct differentiable optimization over this rich feedback, Diffusion-DRF achieves stable reward-based tuning without preference datasets collection. Diffusion-DRF achieves significant gains both quantitatively and qualitatively, outperforming state-of-the-art Flow-GRPO by 4.74% in overall performance on unseen VBench-2.0.",
  "published": "2026-01-07",
  "updated": "2026-03-17",
  "year": "2026",
  "authors": [
   "Yifan Wang",
   "Yanyu Li",
   "Gordon Guocheng Qian",
   "Sergey Tulyakov",
   "Yun Fu",
   "Anil Kag"
  ],
  "author_count": 6,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 2,
  "influential_citations": 0,
  "tldr": "Diffusion-DRF is proposed, a free, rich, and differentiable reward framework for video diffusion fine-tuning that decomposes each user prompt into multi-dimensional questions with freeform dense VQA explanation queries, yielding information-rich feedback.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yifan Wang",
    "id": "2353313821",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Yanyu Li",
    "id": "2257369142",
    "h_index": 8,
    "papers": 23
   },
   {
    "name": "G. Qian",
    "id": "2424172560",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Sergey Tulyakov",
    "id": "2292401534",
    "h_index": 17,
    "papers": 63
   },
   {
    "name": "Yun Fu",
    "id": "2261898910",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Anil Kag",
    "id": "2284982329",
    "h_index": 9,
    "papers": 23
   }
  ],
  "comment": "Webpage: https://snap-research.github.io/diffusion-drf/",
  "topics": [
   "rl-control",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2601.04153v2",
  "pdf_url": "https://arxiv.org/pdf/2601.04153v2",
  "html_url": "https://arxiv.org/html/2601.04153v2",
  "code_url": "https://snap-research.github.io/diffusion-drf/",
  "presented": [
   {
    "club": "saturday-robotics",
    "club_name": "Saturday Robotics",
    "session_number": 11,
    "session_title": "\ud83e\udd16\ud83e\udd58 Saturday Robotics x Manycore Tech x Neural Motion | CVPR 2026 Denver Research Night | Robotics & World Models Reading Club 11",
    "date_text": "",
    "city": "Denver, CO",
    "url": "https://lu.ma/zamm9g2g",
    "listed_as": ""
   }
  ],
  "club_note": "We'd also love to hear your hot takes on:",
  "featured": true,
  "signal": 4.48
 },
 {
  "id": "2512.24766",
  "slug": "dream2flow-bridging-video-generation-and-open-world-manipulation-with",
  "title": "Dream2Flow: Bridging Video Generation and Open-World Manipulation with 3D Object Flow",
  "abstract": "Generative video modeling has emerged as a compelling tool to zero-shot reason about plausible physical interactions for open-world manipulation. Yet, it remains a challenge to translate such human-led motions into the low-level actions demanded by robotic systems. We observe that given an initial image and task instruction, these models excel at synthesizing sensible object motions. Thus, we introduce Dream2Flow, a framework that bridges video generation and robotic control through 3D object flow as an intermediate representation. Our method reconstructs 3D object motions from generated videos and formulates manipulation as object trajectory tracking. By separating the state changes from the actuators that realize those changes, Dream2Flow overcomes the embodiment gap and enables zero-shot guidance from pre-trained video models to manipulate objects of diverse categories-including rigid, articulated, deformable, and granular. Through trajectory optimization or reinforcement learning, Dream2Flow converts reconstructed 3D object flow into executable low-level commands without task-specific demonstrations. Simulation and real-world experiments highlight 3D object flow as a general and scalable interface for adapting video generation models to open-world robotic manipulation. Videos and visualizations are available at https://dream2flow.github.io/.",
  "published": "2025-12-31",
  "updated": "2025-12-31",
  "year": "2025",
  "authors": [
   "Karthik Dharmarajan",
   "Wenlong Huang",
   "Jiajun Wu",
   "Li Fei-Fei",
   "Ruohan Zhang"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 30,
  "influential_citations": 6,
  "tldr": "This work introduces Dream2Flow, a framework that bridges video generation and robotic control through 3D object flow as an intermediate representation and enables zero-shot guidance from pre-trained video models to manipulate objects of diverse categories-including rigid, articulated, deformable, and granular.",
  "doi": "10.48550/arXiv.2512.24766",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Karthik Dharmarajan",
    "id": "2401833053",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Wenlong Huang",
    "id": "2319777572",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Jiajun Wu",
    "id": "2326444500",
    "h_index": 35,
    "papers": 214
   },
   {
    "name": "Fei-Fei Li",
    "id": "2330589126",
    "h_index": 3,
    "papers": 15
   },
   {
    "name": "Ruohan Zhang",
    "id": "2328111135",
    "h_index": 2,
    "papers": 8
   }
  ],
  "comment": "Project website: https://dream2flow.github.io/",
  "topics": [
   "rl-control",
   "hardware-codesign",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2512.24766v1",
  "pdf_url": "https://arxiv.org/pdf/2512.24766v1",
  "html_url": "https://arxiv.org/html/2512.24766v1",
  "code_url": "https://dream2flow.github.io/",
  "presented": [
   {
    "club": "saturday-robotics",
    "club_name": "Saturday Robotics",
    "session_number": 26,
    "session_title": "Robotics & World Models Reading Club 26: Video Generation to Robot Manipulation: Bridging Embodiment Gap+Video-Tactile-Action Model. SF 8/29",
    "date_text": "Saturday, August 29, 2026 | 2:00 PM \u2013 5:00 PM",
    "city": "San Francisco, CA",
    "url": "https://lu.ma/00z3oxw6",
    "listed_as": ""
   }
  ],
  "club_note": "",
  "featured": true,
  "signal": 5.49
 },
 {
  "id": "2512.22414",
  "slug": "emergence-of-human-to-robot-transfer-in-vision-language-action-models",
  "title": "Emergence of Human to Robot Transfer in Vision-Language-Action Models",
  "abstract": "Vision-language-action (VLA) models can enable broad open world generalization, but require large and diverse datasets. It is appealing to consider whether some of this data can come from human videos, which cover diverse real-world situations and are easy to obtain. However, it is difficult to train VLAs with human videos alone, and establishing a mapping between humans and robots requires manual engineering and presents a major research challenge. Drawing inspiration from advances in large language models, where the ability to learn from diverse supervision emerges with scale, we ask whether a similar phenomenon holds for VLAs that incorporate human video data. We introduce a simple co-training recipe, and find that human-to-robot transfer emerges once the VLA is pre-trained on sufficient scenes, tasks, and embodiments. Our analysis suggests that this emergent capability arises because diverse pretraining produces embodiment-agnostic representations for human and robot data. We validate these findings through a series of experiments probing human to robot skill transfer and find that with sufficiently diverse robot pre-training our method can nearly double the performance on generalization settings seen only in human data.",
  "published": "2025-12-27",
  "updated": "2025-12-27",
  "year": "2025",
  "authors": [
   "Simar Kareer",
   "Karl Pertsch",
   "James Darpinian",
   "Judy Hoffman",
   "Danfei Xu",
   "Sergey Levine",
   "Chelsea Finn",
   "Suraj Nair"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 39,
  "influential_citations": 4,
  "tldr": "This work introduces a simple co-training recipe, and finds that human-to-robot transfer emerges once the VLA is pre-trained on sufficient scenes, tasks, and embodiments, and suggests that this emergent capability arises because diverse pretraining produces embodiment-agnostic representations for human and robot data.",
  "doi": "10.48550/arXiv.2512.22414",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Simar Kareer",
    "id": "2188833033",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Karl Pertsch",
    "id": "31719101",
    "h_index": 32,
    "papers": 56
   },
   {
    "name": "James Darpinian",
    "id": "2356784408",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Judy Hoffman",
    "id": "2328413304",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Danfei Xu",
    "id": "2312175432",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Sergey Levine",
    "id": "2249615151",
    "h_index": 32,
    "papers": 53
   },
   {
    "name": "Chelsea Finn",
    "id": "2347538211",
    "h_index": 9,
    "papers": 13
   },
   {
    "name": "Suraj Nair",
    "id": "2286638954",
    "h_index": 12,
    "papers": 19
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "egocentric-data",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2512.22414v1",
  "pdf_url": "https://arxiv.org/pdf/2512.22414v1",
  "html_url": "https://arxiv.org/html/2512.22414v1",
  "code_url": null,
  "presented": [
   {
    "club": "saturday-robotics",
    "club_name": "Saturday Robotics",
    "session_number": 8,
    "session_title": "Robotics & World Models Reading Club 08: Embodied Human Data as the \u201cInternet of Motion and Behavior\u201d \u2014 San Francisco 0516",
    "date_text": "Saturday, May 16, 2026 | 2:00 PM \u2013 5:00 PM",
    "city": "San Francisco, CA",
    "url": "https://lu.ma/qoxioge7",
    "listed_as": "Human-Robot Co-Design for Scalable Data Collection"
   }
  ],
  "club_note": "Jointly optimizes robot morphology, interfaces, and teleoperation pipelines to improve scalability and reduce human demonstration cost.",
  "featured": true,
  "signal": 4.6
 },
 {
  "id": "2512.15840",
  "slug": "large-video-planner-enables-generalizable-robot-control",
  "title": "Large Video Planner Enables Generalizable Robot Control",
  "abstract": "General-purpose robots require decision-making models that generalize across diverse tasks and environments. Recent works build robot foundation models by extending multimodal large language models (MLLMs) with action outputs, creating vision-language-action (VLA) systems. These efforts are motivated by the intuition that MLLMs' large-scale language and image pretraining can be effectively transferred to the action output modality. In this work, we explore an alternative paradigm of using large-scale video pretraining as a primary modality for building robot foundation models. Unlike static images and language, videos capture spatio-temporal sequences of states and actions in the physical world that are naturally aligned with robotic behavior. We curate an internet-scale video dataset of human activities and task demonstrations, and train, for the first time at a foundation-model scale, an open video model for generative robotics planning. The model produces zero-shot video plans for novel scenes and tasks, which we post-process to extract executable robot actions. We evaluate task-level generalization through third-party selected tasks in the wild and real-robot experiments, demonstrating successful physical execution. Together, these results show robust instruction following, strong generalization, and real-world feasibility. We release both the model and dataset to support open, reproducible video-based robot learning. Our website is available at https://www.boyuan.space/large-video-planner/.",
  "published": "2025-12-17",
  "updated": "2026-05-08",
  "year": "2025",
  "authors": [
   "Boyuan Chen",
   "Tianyuan Zhang",
   "Haoran Geng",
   "Caiyi Zhang",
   "Peihao Li",
   "Kiwhan Song",
   "William T. Freeman",
   "Jitendra Malik",
   "Pieter Abbeel",
   "Russ Tedrake",
   "Vincent Sitzmann",
   "Yilun Du"
  ],
  "author_count": 12,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 59,
  "influential_citations": 9,
  "tldr": "This work curate an internet-scale video dataset of human activities and task demonstrations, and train, for the first time at a foundation-model scale, an open video model for generative robotics planning, which produces zero-shot video plans for novel scenes and tasks, which are post-process to extract executable robot actions.",
  "doi": "10.48550/arXiv.2512.15840",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Boyuan Chen",
    "id": "8786274",
    "h_index": 10,
    "papers": 15
   },
   {
    "name": "Tianyuan Zhang",
    "id": "2297868523",
    "h_index": 10,
    "papers": 13
   },
   {
    "name": "Haoran Geng",
    "id": "2287929608",
    "h_index": 7,
    "papers": 15
   },
   {
    "name": "Kiwhan Song",
    "id": "2345339612",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Caiyi Zhang",
    "id": "2283747833",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Peihao Li",
    "id": "2358095761",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "William T. Freeman",
    "id": "2325098371",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Jitendra Malik",
    "id": "2362090500",
    "h_index": 8,
    "papers": 18
   },
   {
    "name": "Pieter Abbeel",
    "id": "2350871358",
    "h_index": 8,
    "papers": 22
   },
   {
    "name": "Russ Tedrake",
    "id": "2263905014",
    "h_index": 14,
    "papers": 36
   },
   {
    "name": "Vincent Sitzmann",
    "id": "2280906248",
    "h_index": 7,
    "papers": 8
   },
   {
    "name": "Yilun Du",
    "id": "2352201142",
    "h_index": 5,
    "papers": 8
   }
  ],
  "comment": "29 pages, 16 figures",
  "topics": [
   "vla",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2512.15840v2",
  "pdf_url": "https://arxiv.org/pdf/2512.15840v2",
  "html_url": "https://arxiv.org/html/2512.15840v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.78
 },
 {
  "id": "2512.14689",
  "slug": "chip-adaptive-compliance-for-humanoid-control-through-hindsight-pertur",
  "title": "CHIP: Adaptive Compliance for Humanoid Control through Hindsight Perturbation",
  "abstract": "Recent progress in humanoid robots has unlocked agile locomotion skills, including backflipping, running, and crawling. Yet it remains challenging for a humanoid robot to perform forceful manipulation tasks such as moving objects, wiping, and pushing a cart. We propose adaptive Compliance Humanoid control through hIsight Perturbation (CHIP), a plug-and-play module that enables controllable end-effector stiffness while preserving agile tracking of dynamic reference motions. CHIP is easy to implement and requires neither data augmentation nor additional reward tuning. We show that a generalist motion-tracking controller trained with CHIP can perform a diverse set of forceful manipulation tasks that require different end-effector compliance, such as multi-robot collaboration, wiping, box delivery, and door opening.",
  "published": "2025-12-16",
  "updated": "2026-02-09",
  "year": "2025",
  "authors": [
   "Sirui Chen",
   "Zi-ang Cao",
   "Zhengyi Luo",
   "Fernando Casta\u00f1eda",
   "Chenran Li",
   "Tingwu Wang",
   "Ye Yuan",
   "Linxi \"Jim\" Fan",
   "C. Karen Liu",
   "Yuke Zhu"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 10,
  "influential_citations": 1,
  "tldr": "It is shown that a generalist motion-tracking controller trained with CHIP can perform a diverse set of forceful manipulation tasks that require different end-effector compliance, such as multi-robot collaboration, wiping, box delivery, and door opening.",
  "doi": "10.48550/arXiv.2512.14689",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Sirui Chen",
    "id": "2209905328",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Zi-ang Cao",
    "id": "2359785826",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Zhengyi Luo",
    "id": "2329051397",
    "h_index": 9,
    "papers": 15
   },
   {
    "name": "Fernando Casta\u00f1eda",
    "id": "2350859158",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Chenran Li",
    "id": "2242186839",
    "h_index": 14,
    "papers": 70
   },
   {
    "name": "Tingwu Wang",
    "id": "2392415176",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Ye Yuan",
    "id": "2391922742",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "L. Fan",
    "id": "2257381161",
    "h_index": 18,
    "papers": 25
   },
   {
    "name": "C. K. Liu",
    "id": "2278583770",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Yuke Zhu",
    "id": "2338857250",
    "h_index": 7,
    "papers": 14
   }
  ],
  "comment": "The first two authors contributed equally. Project page: https://nvlabs.github.io/CHIP/",
  "topics": [
   "humanoids"
  ],
  "orgs": [
   "NVIDIA"
  ],
  "abs_url": "https://arxiv.org/abs/2512.14689v2",
  "pdf_url": "https://arxiv.org/pdf/2512.14689v2",
  "html_url": "https://arxiv.org/html/2512.14689v2",
  "code_url": "https://nvlabs.github.io/CHIP/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.54
 },
 {
  "id": "2512.14614",
  "slug": "worldplay-towards-long-term-geometric-consistency-for-real-time-intera",
  "title": "WorldPlay: Towards Long-Term Geometric Consistency for Real-Time Interactive World Modeling",
  "abstract": "This paper presents WorldPlay, a streaming video diffusion model that enables real-time, interactive world modeling with long-term geometric consistency, resolving the trade-off between speed and memory that limits current methods. WorldPlay draws power from three key ingredients. 1) We use a Dual Action Representation to enable robust action control in response to the user's keyboard and mouse inputs. 2) To enforce long-term consistency, our Reconstituted Context Memory dynamically rebuilds context from past frames and uses temporal reframing to keep geometrically important but long-past frames accessible, effectively alleviating memory attenuation. 3) We also propose Context Forcing, a novel distillation method designed for memory-aware model. Aligning memory context between the teacher and student preserves the student's capacity to use long-range information, enabling real-time speeds while preventing error drift. Taken together, WorldPlay generates long-horizon streaming 720p video at 24 FPS with superior consistency, comparing favorably with existing techniques and showing strong generalization across diverse scenes. Project page and online demo can be found: https://3d-models.hunyuan.tencent.com/world/ and https://3d.hunyuan.tencent.com/sceneTo3D.",
  "published": "2025-12-16",
  "updated": "2026-06-09",
  "year": "2025",
  "authors": [
   "Wenqiang Sun",
   "Haiyu Zhang",
   "Haoyuan Wang",
   "Junta Wu",
   "Zehan Wang",
   "Zhenwei Wang",
   "Yunhong Wang",
   "Jun Zhang",
   "Tengfei Wang",
   "Chunchao Guo"
  ],
  "author_count": 10,
  "categories": [
   "cs.CV",
   "cs.GR"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 112,
  "influential_citations": 30,
  "tldr": "",
  "doi": "10.48550/arXiv.2512.14614",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Wenqiang Sun",
    "id": "2215875229",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "Haiyu Zhang",
    "id": "2304360326",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Haoyuan Wang",
    "id": "2397402595",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Junta Wu",
    "id": "2341905605",
    "h_index": 8,
    "papers": 15
   },
   {
    "name": "Zehan Wang",
    "id": "2399748902",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Zhenwei Wang",
    "id": "2225113830",
    "h_index": 10,
    "papers": 45
   },
   {
    "name": "Yunhong Wang",
    "id": "2304481533",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Jun Zhang",
    "id": "2304515490",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Tengfei Wang",
    "id": "2365273942",
    "h_index": 6,
    "papers": 14
   },
   {
    "name": "Chunchao Guo",
    "id": "2387832828",
    "h_index": 5,
    "papers": 10
   }
  ],
  "comment": "project page: https://3d-models.hunyuan.tencent.com/world/, demo: https://3d.hunyuan.tencent.com/sceneTo3D, code: https://github.com/Tencent-Hunyuan/HY-WorldPlay",
  "topics": [
   "world-models",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2512.14614v2",
  "pdf_url": "https://arxiv.org/pdf/2512.14614v2",
  "html_url": "https://arxiv.org/html/2512.14614v2",
  "code_url": "https://github.com/Tencent-Hunyuan/HY-WorldPlay",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.05
 },
 {
  "id": "2512.13644",
  "slug": "world-models-for-learning-dexterous-hand-object-interactions-from-huma",
  "title": "World Models for Learning Dexterous Hand-Object Interactions from Human Videos",
  "abstract": "Modeling dexterous hand-object interactions is challenging as it requires understanding how subtle finger motions influence the environment through contact with objects. While recent world models address interaction modeling, they typically rely on coarse action spaces that fail to capture fine-grained dexterity. We, therefore, introduce DexWM, a Dexterous Interaction World Model that predicts future latent states of the environment conditioned on past states and dexterous actions. To overcome the scarcity of finely annotated dexterous datasets, DexWM represents actions using finger keypoints extracted from egocentric videos, enabling training on over 900 hours of human and non-dexterous robot data. Further, to accurately model dexterity, we find that predicting visual features alone is insufficient; therefore, we incorporate an auxiliary hand consistency loss that enforces accurate hand configurations. DexWM outperforms prior world models conditioned on text, navigation, or full-body actions in future-state prediction and demonstrates strong zero-shot transfer to unseen skills on a Franka Panda arm with an Allegro gripper, surpassing Diffusion Policy by over 50% on average across grasping, placing, and reaching tasks.",
  "published": "2025-12-15",
  "updated": "2026-03-16",
  "year": "2025",
  "authors": [
   "Raktim Gautam Goswami",
   "Amir Bar",
   "David Fan",
   "Tsung-Yen Yang",
   "Gaoyue Zhou",
   "Prashanth Krishnamurthy",
   "Michael Rabbat",
   "Farshad Khorrami",
   "Yann LeCun"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 16,
  "influential_citations": 1,
  "tldr": "DexWM, a Dexterous Interaction World Model that predicts future latent states of the environment conditioned on past states and dexterous actions, is introduced and demonstrates strong zero-shot transfer to unseen skills on a Franka Panda arm with an Allegro gripper.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Raktim Gautam Goswami",
    "id": "2003753110",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Amir Bar",
    "id": "2346979635",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "David Fan",
    "id": "2335871751",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Tsung-Yen Yang",
    "id": "2398853807",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Gaoyue Zhou",
    "id": "2257386929",
    "h_index": 9,
    "papers": 12
   },
   {
    "name": "P. Krishnamurthy",
    "id": "1920142",
    "h_index": 33,
    "papers": 295
   },
   {
    "name": "Michael Rabbat",
    "id": "2284991448",
    "h_index": 12,
    "papers": 18
   },
   {
    "name": "F. Khorrami",
    "id": "1880767",
    "h_index": 41,
    "papers": 462
   },
   {
    "name": "Yann LeCun",
    "id": "2252537770",
    "h_index": 8,
    "papers": 19
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "dexterous-manipulation",
   "egocentric-data",
   "imitation-diffusion",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2512.13644v2",
  "pdf_url": "https://arxiv.org/pdf/2512.13644v2",
  "html_url": "https://arxiv.org/html/2512.13644v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.23
 },
 {
  "id": "2512.12534",
  "slug": "animus3d-text-driven-3d-animation-via-motion-score-distillation",
  "title": "Animus3D: Text-driven 3D Animation via Motion Score Distillation",
  "abstract": "We present Animus3D, a text-driven 3D animation framework that generates motion field given a static 3D asset and text prompt. Previous methods mostly leverage the vanilla Score Distillation Sampling (SDS) objective to distill motion from pretrained text-to-video diffusion, leading to animations with minimal movement or noticeable jitter. To address this, our approach introduces a novel SDS alternative, Motion Score Distillation (MSD). Specifically, we introduce a LoRA-enhanced video diffusion model that defines a static source distribution rather than pure noise as in SDS, while another inversion-based noise estimation technique ensures appearance preservation when guiding motion. To further improve motion fidelity, we incorporate explicit temporal and spatial regularization terms that mitigate geometric distortions across time and space. Additionally, we propose a motion refinement module to upscale the temporal resolution and enhance fine-grained details, overcoming the fixed-resolution constraints of the underlying video model. Extensive experiments demonstrate that Animus3D successfully animates static 3D assets from diverse text prompts, generating significantly more substantial and detailed motion than state-of-the-art baselines while maintaining high visual integrity. Code will be released at https://qiisun.github.io/animus3d_page.",
  "published": "2025-12-14",
  "updated": "2025-12-14",
  "year": "2025",
  "authors": [
   "Qi Sun",
   "Can Wang",
   "Jiaxiang Shang",
   "Wensen Feng",
   "Jing Liao"
  ],
  "author_count": 5,
  "categories": [
   "cs.CV",
   "cs.GR",
   "cs.LG"
  ],
  "primary_category": "cs.CV",
  "venue": "SIGGRAPH",
  "venue_source": "semantic-scholar",
  "citations": 5,
  "influential_citations": 0,
  "tldr": "This work introduces a novel SDS alternative, Motion Score Distillation (MSD), a LoRA-enhanced video diffusion model that defines a static source distribution rather than pure noise as in SDS, while another inversion-based noise estimation technique ensures appearance preservation when guiding motion.",
  "doi": "10.1145/3757377.3763916",
  "oa_pdf": "https://arxiv.org/pdf/2512.12534",
  "s2_authors": [
   {
    "name": "Qi Sun",
    "id": "2365432914",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Can Wang",
    "id": "2276275184",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Jiaxiang Shang",
    "id": "2057079958",
    "h_index": 7,
    "papers": 16
   },
   {
    "name": "Wensen Feng",
    "id": "2267193716",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Jing Liao",
    "id": "2222581295",
    "h_index": 6,
    "papers": 13
   }
  ],
  "comment": "SIGGRAPH Asia 2025",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2512.12534v1",
  "pdf_url": "https://arxiv.org/pdf/2512.12534v1",
  "html_url": "https://arxiv.org/html/2512.12534v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.28
 },
 {
  "id": "2512.12430",
  "slug": "endless-world-real-time-3d-aware-long-video-generation",
  "title": "Endless World: Real-Time 3D-Aware Long Video Generation",
  "abstract": "Producing long, coherent video sequences with stable 3D structure remains a major challenge, particularly in streaming scenarios. Motivated by this, we introduce Endless World, a real-time framework for infinite, 3D-consistent video generation.To support infinite video generation, we introduce a conditional autoregressive training strategy that aligns newly generated content with existing video frames. This design preserves long-range dependencies while remaining computationally efficient, enabling real-time inference on a single GPU without additional training overhead.Moreover, our Endless World integrates global 3D-aware attention to provide continuous geometric guidance across time. Our 3D injection mechanism enforces physical plausibility and geometric consistency throughout extended sequences, addressing key challenges in long-horizon and dynamic scene synthesis.Extensive experiments demonstrate that Endless World produces long, stable, and visually coherent videos, achieving competitive or superior performance to existing methods in both visual fidelity and spatial consistency. Our project has been available on https://bwgzk-keke.github.io/EndlessWorld/.",
  "published": "2025-12-13",
  "updated": "2025-12-13",
  "year": "2025",
  "authors": [
   "Ke Zhang",
   "Yiqun Mei",
   "Jiacong Xu",
   "Vishal M. Patel"
  ],
  "author_count": 4,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 4,
  "influential_citations": 0,
  "tldr": "To support infinite video generation, a conditional autoregressive training strategy is introduced that aligns newly generated content with existing video frames, preserving long-range dependencies while remaining computationally efficient, enabling real-time inference on a single GPU without additional training overhead.",
  "doi": "10.48550/arXiv.2512.12430",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ke Zhang",
    "id": "2307452752",
    "h_index": 3,
    "papers": 15
   },
   {
    "name": "Yiqun Mei",
    "id": "1661057458",
    "h_index": 14,
    "papers": 25
   },
   {
    "name": "Jiacong Xu",
    "id": "2292419231",
    "h_index": 9,
    "papers": 22
   },
   {
    "name": "Vishal M. Patel",
    "id": "2292407560",
    "h_index": 5,
    "papers": 11
   }
  ],
  "comment": "10 pages,7 figures",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2512.12430v1",
  "pdf_url": "https://arxiv.org/pdf/2512.12430v1",
  "html_url": "https://arxiv.org/html/2512.12430v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.7
 },
 {
  "id": "2512.11047",
  "slug": "wholebodyvla-towards-unified-latent-vla-for-whole-body-loco-manipulati",
  "title": "WholeBodyVLA: Towards Unified Latent VLA for Whole-Body Loco-Manipulation Control",
  "abstract": "Humanoid robots require precise locomotion and dexterous manipulation to perform challenging loco-manipulation tasks. Yet existing approaches, modular or end-to-end, are deficient in manipulation-aware locomotion. This confines the robot to a limited workspace, preventing it from performing large-space loco-manipulation. We attribute this to: (1) the challenge of acquiring loco-manipulation knowledge due to the scarcity of humanoid teleoperation data, and (2) the difficulty of faithfully and reliably executing locomotion commands, stemming from the limited precision and stability of existing RL controllers. To acquire richer loco-manipulation knowledge, we propose a unified latent learning framework that enables Vision-Language-Action (VLA) system to learn from low-cost action-free egocentric videos. Moreover, an efficient human data collection pipeline is devised to augment the dataset and scale the benefits. To execute the desired locomotion commands more precisely, we present a loco-manipulation-oriented (LMO) RL policy specifically tailored for accurate and stable core loco-manipulation movements, such as advancing, turning, and squatting. Building on these components, we introduce WholeBodyVLA, a unified framework for humanoid loco-manipulation. To the best of our knowledge, WholeBodyVLA is one of its kind enabling large-space humanoid loco-manipulation. It is verified via comprehensive experiments on the AgiBot X2 humanoid, outperforming prior baseline by 21.3%. It also demonstrates strong generalization and high extensibility across a broad range of tasks.",
  "published": "2025-12-11",
  "updated": "2025-12-15",
  "year": "2025",
  "authors": [
   "Haoran Jiang",
   "Jin Chen",
   "Qingwen Bu",
   "Li Chen",
   "Modi Shi",
   "Yanjie Zhang",
   "Delong Li",
   "Chuanzhe Suo",
   "Chuang Wang",
   "Zhihui Peng",
   "Hongyang Li"
  ],
  "author_count": 11,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 42,
  "influential_citations": 2,
  "tldr": "W WholeBodyVLA is one of its kind enabling large-space humanoid loco-manipulation, verified via comprehensive experiments on the AgiBot X2 humanoid, outperforming prior baseline by 21.3%.",
  "doi": "10.48550/arXiv.2512.11047",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Haoran Jiang",
    "id": "2355446551",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Jin Chen",
    "id": "2373548997",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Qingwen Bu",
    "id": "2290184536",
    "h_index": 13,
    "papers": 19
   },
   {
    "name": "Li Chen",
    "id": "2357896168",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Modi Shi",
    "id": "2349410682",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Yanjie Zhang",
    "id": "2315119703",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "D. Li",
    "id": "46598891",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Chuanzhe Suo",
    "id": "52135088",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Chuang Wang",
    "id": "2295085904",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Zhihui Peng",
    "id": "2272978935",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Hongyang Li",
    "id": "2290243003",
    "h_index": 11,
    "papers": 19
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "dexterous-manipulation",
   "humanoids",
   "egocentric-data",
   "data-teleop"
  ],
  "orgs": [
   "AgiBot"
  ],
  "abs_url": "https://arxiv.org/abs/2512.11047v2",
  "pdf_url": "https://arxiv.org/pdf/2512.11047v2",
  "html_url": "https://arxiv.org/html/2512.11047v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.13
 },
 {
  "id": "2512.10946",
  "slug": "implicitrdp-an-end-to-end-visual-force-diffusion-policy-with-structura",
  "title": "ImplicitRDP: An End-to-End Visual-Force Diffusion Policy with Structural Slow-Fast Learning",
  "abstract": "Human-level contact-rich manipulation relies on the distinct roles of two key modalities: vision provides spatially rich but temporally slow global context, while force sensing captures rapid local contact dynamics. Integrating these signals is challenging due to their fundamental frequency and informational disparities. In this work, we propose ImplicitRDP, a unified end-to-end visual-force diffusion policy that integrates visual planning and reactive force control within a single network. We introduce Structural Slow-Fast Learning, a mechanism utilizing causal attention to simultaneously process asynchronous visual and force tokens, allowing the policy to perform rapid force control at the action rate while maintaining the temporal coherence of action chunks. Furthermore, to mitigate modality collapse where end-to-end models fail to adjust the weights across different modalities, we propose Virtual-target-based Representation Regularization. This auxiliary objective maps force feedback into the same space as the action, providing a stronger, physics-grounded learning signal than raw force prediction. Extensive experiments on contact-rich tasks demonstrate that ImplicitRDP significantly outperforms both vision-only and hierarchical baselines, achieving superior reactivity and success rates with a streamlined training pipeline. Code and videos are available at https://implicit-rdp.github.io.",
  "published": "2025-12-11",
  "updated": "2026-07-21",
  "year": "2025",
  "authors": [
   "Wendi Chen",
   "Han Xue",
   "Yi Wang",
   "Fangyuan Zhou",
   "Jun Lv",
   "Yang Jin",
   "Shirun Tang",
   "Chuan Wen",
   "Cewu Lu"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "RA-L",
  "venue_source": "semantic-scholar",
  "citations": 11,
  "influential_citations": 3,
  "tldr": "This work proposes ImplicitRDP, a unified end-to-end visual-force diffusion policy that integrates visual planning and reactive force control within a single network, and introduces Structural Slow-Fast Learning, a mechanism utilizing causal attention to simultaneously process asynchronous visual and force tokens.",
  "doi": "10.1109/LRA.2026.3710031",
  "oa_pdf": "https://arxiv.org/pdf/2512.10946",
  "s2_authors": [
   {
    "name": "Wendi Chen",
    "id": "2326063487",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Han Xue",
    "id": "2351107813",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Yi Wang",
    "id": "2363937133",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Fangyuan Zhou",
    "id": "2325962373",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Jun Lv",
    "id": "2054671126",
    "h_index": 11,
    "papers": 26
   },
   {
    "name": "Yang Jin",
    "id": "2301173136",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Shirun Tang",
    "id": "2397957257",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Chuan Wen",
    "id": "2381956964",
    "h_index": 5,
    "papers": 20
   },
   {
    "name": "Cewu Lu",
    "id": "2301174899",
    "h_index": 7,
    "papers": 29
   }
  ],
  "comment": "Accepted to RA-L 2026. Project page: https://implicit-rdp.github.io",
  "topics": [
   "tactile",
   "imitation-diffusion"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2512.10946v2",
  "pdf_url": "https://arxiv.org/pdf/2512.10946v2",
  "html_url": "https://arxiv.org/html/2512.10946v2",
  "code_url": "https://implicit-rdp.github.io",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.58
 },
 {
  "id": "2512.10675",
  "slug": "evaluating-gemini-robotics-policies-in-a-veo-world-simulator",
  "title": "Evaluating Gemini Robotics Policies in a Veo World Simulator",
  "abstract": "Generative world models hold significant potential for simulating interactions with visuomotor policies in varied environments. Frontier video models can enable generation of realistic observations and environment interactions in a scalable and general manner. However, the use of video models in robotics has been limited primarily to in-distribution evaluations, i.e., scenarios that are similar to ones used to train the policy or fine-tune the base video model. In this report, we demonstrate that video models can be used for the entire spectrum of policy evaluation use cases in robotics: from assessing nominal performance to out-of-distribution (OOD) generalization, and probing physical and semantic safety. We introduce a generative evaluation system built upon a frontier video foundation model (Veo). The system is optimized to support robot action conditioning and multi-view consistency, while integrating generative image-editing and multi-view completion to synthesize realistic variations of real-world scenes along multiple axes of generalization. We demonstrate that the system preserves the base capabilities of the video model to enable accurate simulation of scenes that have been edited to include novel interaction objects, novel visual backgrounds, and novel distractor objects. This fidelity enables accurately predicting the relative performance of different policies in both nominal and OOD conditions, determining the relative impact of different axes of generalization on policy performance, and performing red teaming of policies to expose behaviors that violate physical or semantic safety constraints. We validate these capabilities through 1600+ real-world evaluations of eight Gemini Robotics policy checkpoints and five tasks for a bimanual manipulator.",
  "published": "2025-12-11",
  "updated": "2026-01-06",
  "year": "2025",
  "authors": [
   " Gemini Robotics Team",
   "Krzysztof Choromanski",
   "Coline Devin",
   "Yilun Du",
   "Debidatta Dwibedi",
   "Ruiqi Gao",
   "Abhishek Jindal",
   "Thomas Kipf",
   "Sean Kirmani",
   "Isabel Leal",
   "Fangchen Liu",
   "Anirudha Majumdar",
   "Andrew Marmon",
   "Carolina Parada",
   "Yulia Rubanova",
   "Dhruv Shah",
   "Vikas Sindhwani",
   "Jie Tan",
   "Fei Xia",
   "Ted Xiao",
   "Sherry Yang",
   "Wenhao Yu",
   "Allan Zhou"
  ],
  "author_count": 23,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 47,
  "influential_citations": 2,
  "tldr": "This report introduces a generative evaluation system built upon a frontier video foundation model (Veo), optimized to support robot action conditioning and multi-view consistency, while integrating generative image-editing and multi-view completion to synthesize realistic variations of real-world scenes along multiple axes of generalization.",
  "doi": "10.48550/arXiv.2512.10675",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "G. Team",
    "id": "2316577477",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Google DeepMind",
    "id": "2135683701",
    "h_index": 11,
    "papers": 61
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "sim2real",
   "foundation-pretraining",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2512.10675v2",
  "pdf_url": "https://arxiv.org/pdf/2512.10675v2",
  "html_url": "https://arxiv.org/html/2512.10675v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.68
 },
 {
  "id": "2512.08920",
  "slug": "osmo-open-source-tactile-glove-for-human-to-robot-skill-transfer",
  "title": "OSMO: Open-Source Tactile Glove for Human-to-Robot Skill Transfer",
  "abstract": "Human video demonstrations provide abundant training data for learning robot policies, but video alone cannot capture the rich contact signals critical for mastering manipulation. We introduce OSMO, an open-source wearable tactile glove designed for human-to-robot skill transfer. The glove features 12 three-axis tactile sensors across the fingertips and palm and is designed to be compatible with state-of-the-art hand-tracking methods for in-the-wild data collection. We demonstrate that a robot policy trained exclusively on human demonstrations collected with OSMO, without any real robot data, is capable of executing a challenging contact-rich manipulation task. By equipping both the human and the robot with the same glove, OSMO minimizes the visual and tactile embodiment gap, enabling the transfer of continuous shear and normal force feedback while avoiding the need for image inpainting or other vision-based force inference. On a real-world wiping task requiring sustained contact pressure, our tactile-aware policy achieves a 72% success rate, outperforming vision-only baselines by eliminating contact-related failure modes. We release complete hardware designs, firmware, and assembly instructions to support community adoption.",
  "published": "2025-12-09",
  "updated": "2025-12-09",
  "year": "2025",
  "authors": [
   "Jessica Yin",
   "Haozhi Qi",
   "Youngsun Wi",
   "Sayantan Kundu",
   "Mike Lambeta",
   "William Yang",
   "Changhao Wang",
   "Tingfan Wu",
   "Jitendra Malik",
   "Tess Hellebrekers"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "RA-L",
  "venue_source": "semantic-scholar",
  "citations": 12,
  "influential_citations": 1,
  "tldr": "OSMO, an open-source wearable tactile glove designed for human-to-robot skill transfer, is introduced, demonstrating that a robot policy trained exclusively on human demonstrations collected with OSMO is capable of executing a challenging contact-rich manipulation task.",
  "doi": "10.1109/LRA.2026.3692034",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jessica Yin",
    "id": "1390925803",
    "h_index": 7,
    "papers": 16
   },
   {
    "name": "Haozhi Qi",
    "id": "2247951244",
    "h_index": 11,
    "papers": 18
   },
   {
    "name": "Youngsun Wi",
    "id": "2152020759",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Sayan Kundu",
    "id": "2342400672",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Mike Lambeta",
    "id": "3427691",
    "h_index": 13,
    "papers": 20
   },
   {
    "name": "William Yang",
    "id": "2109423456",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Changhao Wang",
    "id": "2344075962",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Tingfan Wu",
    "id": "2254158966",
    "h_index": 12,
    "papers": 21
   },
   {
    "name": "Jitendra Malik",
    "id": "2242761335",
    "h_index": 15,
    "papers": 32
   },
   {
    "name": "T. Hellebrekers",
    "id": "2576308",
    "h_index": 18,
    "papers": 31
   }
  ],
  "comment": "Project website: https://jessicayin.github.io/osmo_tactile_glove/",
  "topics": [
   "egocentric-data",
   "tactile",
   "data-teleop",
   "hardware-codesign",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2512.08920v1",
  "pdf_url": "https://arxiv.org/pdf/2512.08920v1",
  "html_url": "https://arxiv.org/html/2512.08920v1",
  "code_url": "https://jessicayin.github.io/osmo_tactile_glove/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.61
 },
 {
  "id": "2512.08333",
  "slug": "robust-finetuning-of-vision-language-action-robot-policies-via-paramet",
  "title": "Robust Finetuning of Vision-Language-Action Robot Policies via Parameter Merging",
  "abstract": "Generalist robot policies, trained on large and diverse datasets, have demonstrated the ability to generalize across a wide spectrum of behaviors, enabling a single policy to act in varied real-world environments. However, they still fall short on new tasks not covered in the training data. When finetuned on limited demonstrations of a new task, these policies often overfit to the specific demonstrations--not only losing their prior abilities to solve a wide variety of generalist tasks but also failing to generalize within the new task itself. In this work, we aim to develop a method that preserves the generalization capabilities of the generalist policy during finetuning, allowing a single policy to robustly incorporate a new skill into its repertoire. Our goal is a single policy that both learns to generalize to variations of the new task and retains the broad competencies gained from pretraining. We show that this can be achieved through a simple yet effective strategy: interpolating the weights of a finetuned model with that of the pretrained model. We show, across extensive simulated and real-world experiments, that such model merging produces a single model that inherits the generalist abilities of the base model and learns to solve the new task robustly, outperforming both the pretrained and finetuned model on out-of-distribution variations of the new task. Moreover, we show that model merging performance scales with the amount of pretraining data, and enables continual acquisition of new skills in a lifelong learning setting, without sacrificing previously learned generalist abilities.",
  "published": "2025-12-09",
  "updated": "2026-02-27",
  "year": "2025",
  "authors": [
   "Yajat Yadav",
   "Zhiyuan Zhou",
   "Andrew Wagenmaker",
   "Karl Pertsch",
   "Sergey Levine"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 18,
  "influential_citations": 2,
  "tldr": "This work aims to develop a method that preserves the generalization capabilities of the generalist policy during finetuning, allowing a single policy to robustly incorporate a new skill into its repertoire, and shows that this can be achieved through a simple yet effective strategy: interpolating the weights of a finetuned model with that of the pretrained model.",
  "doi": "10.48550/arXiv.2512.08333",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yajat Yadav",
    "id": "2376531585",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Zhiyuan Zhou",
    "id": "2314073778",
    "h_index": 6,
    "papers": 16
   },
   {
    "name": "Andrew Wagenmaker",
    "id": "9041933",
    "h_index": 16,
    "papers": 33
   },
   {
    "name": "Karl Pertsch",
    "id": "31719101",
    "h_index": 32,
    "papers": 56
   },
   {
    "name": "Sergey Levine",
    "id": "2249615151",
    "h_index": 32,
    "papers": 53
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "foundation-pretraining",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2512.08333v3",
  "pdf_url": "https://arxiv.org/pdf/2512.08333v3",
  "html_url": "https://arxiv.org/html/2512.08333v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.28
 },
 {
  "id": "2512.04040",
  "slug": "relic-interactive-video-world-model-with-long-horizon-memory",
  "title": "RELIC: Interactive Video World Model with Long-Horizon Memory",
  "abstract": "A truly interactive world model requires three key ingredients: real-time long-horizon streaming, consistent spatial memory, and precise user control. However, most existing approaches address only one of these aspects in isolation, as achieving all three simultaneously is highly challenging-for example, long-term memory mechanisms often degrade real-time performance. In this work, we present RELIC, a unified framework that tackles these three challenges altogether. Given a single image and a text description, RELIC enables memory-aware, long-duration exploration of arbitrary scenes in real time. Built upon recent autoregressive video-diffusion distillation techniques, our model represents long-horizon memory using highly compressed historical latent tokens encoded with both relative actions and absolute camera poses within the KV cache. This compact, camera-aware memory structure supports implicit 3D-consistent content retrieval and enforces long-term coherence with minimal computational overhead. In parallel, we fine-tune a bidirectional teacher video model to generate sequences beyond its original 5-second training horizon, and transform it into a causal student generator using a new memory-efficient self-forcing paradigm that enables full-context distillation over long-duration teacher as well as long student self-rollouts. Implemented as a 14B-parameter model and trained on a curated Unreal Engine-rendered dataset, RELIC achieves real-time generation at 16 FPS while demonstrating more accurate action following, more stable long-horizon streaming, and more robust spatial-memory retrieval compared with prior work. These capabilities establish RELIC as a strong foundation for the next generation of interactive world modeling.",
  "published": "2025-12-03",
  "updated": "2025-12-03",
  "year": "2025",
  "authors": [
   "Yicong Hong",
   "Yiqun Mei",
   "Chongjian Ge",
   "Yiran Xu",
   "Yang Zhou",
   "Sai Bi",
   "Yannick Hold-Geoffroy",
   "Mike Roberts",
   "Matthew Fisher",
   "Eli Shechtman",
   "Kalyan Sunkavalli",
   "Feng Liu",
   "Zhengqi Li",
   "Hao Tan"
  ],
  "author_count": 14,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 67,
  "influential_citations": 8,
  "tldr": "Implemented as a 14B-parameter model and trained on a curated Unreal Engine-rendered dataset, RELIC achieves real-time generation at 16 FPS while demonstrating more accurate action following, more stable long-horizon streaming, and more robust spatial-memory retrieval compared with prior work.",
  "doi": "10.48550/arXiv.2512.04040",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yicong Hong",
    "id": "2265727453",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Yiqun Mei",
    "id": "1661057458",
    "h_index": 14,
    "papers": 25
   },
   {
    "name": "Chongjian Ge",
    "id": "2395895392",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Yiran Xu",
    "id": "2395979845",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Yang Zhou",
    "id": "2336017489",
    "h_index": 5,
    "papers": 21
   },
   {
    "name": "Sai Bi",
    "id": "2265648463",
    "h_index": 18,
    "papers": 33
   },
   {
    "name": "Yannick Hold-Geoffroy",
    "id": "1403980495",
    "h_index": 22,
    "papers": 50
   },
   {
    "name": "Mike Roberts",
    "id": "2111886620",
    "h_index": 8,
    "papers": 11
   },
   {
    "name": "Matthew Fisher",
    "id": "2393927817",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Eli Shechtman",
    "id": "2395669368",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Kalyan Sunkavalli",
    "id": "2123318412",
    "h_index": 20,
    "papers": 42
   },
   {
    "name": "Feng Liu",
    "id": "2396713811",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Zhengqi Li",
    "id": "2383845325",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Hao Tan",
    "id": "2325721609",
    "h_index": 3,
    "papers": 8
   }
  ],
  "comment": "22 pages",
  "topics": [
   "world-models",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2512.04040v1",
  "pdf_url": "https://arxiv.org/pdf/2512.04040v1",
  "html_url": "https://arxiv.org/html/2512.04040v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.83
 },
 {
  "id": "2512.03913",
  "slug": "hierarchical-vision-language-action-model-using-success-and-failure-de",
  "title": "Hierarchical Vision Language Action Model Using Success and Failure Demonstrations",
  "abstract": "Prior Vision-Language-Action (VLA) models are typically trained on teleoperated successful demonstrations, while discarding numerous failed attempts that occur naturally during data collection. However, these failures encode where and how policies can be fragile, information that can be exploited to improve robustness. We address this problem by leveraging mixed-quality datasets to learn failure-aware reasoning at planning time. We introduce VINE, a hierarchical vision-language-action model that separates high-level reasoning (System 2) from low-level control (System 1) under a hierarchical reinforcement learning formalism, making failures usable as a structured learning signal rather than noisy supervision. System 2 performs feasibility-guided tree search over a 2D scene-graph abstraction: it proposes subgoal transitions, predicts success probabilities from both successes and failures, and prunes brittle branches before execution, effectively casting plan evaluation as feasibility scoring. The selected subgoal sequence is then passed to System 1, which executes low-level actions without modifying the agent's core skills. Trained entirely from offline teleoperation data, VINE integrates negative experience directly into the decision loop. Across challenging manipulation tasks, this approach consistently improves success rates and robustness, demonstrating that failure data is an essential resource for converting the broad competence of VLAs into robust execution.",
  "published": "2025-12-03",
  "updated": "2025-12-03",
  "year": "2025",
  "authors": [
   "Jeongeun Park",
   "Jihwan Yoon",
   "Byungwoo Jeon",
   "Juhan Park",
   "Jinwoo Shin",
   "Namhoon Cho",
   "Kyungjae Lee",
   "Sangdoo Yun",
   "Sungjoon Choi"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 3,
  "influential_citations": 0,
  "tldr": "VINE is introduced, a hierarchical vision-language-action model that separates high-level reasoning from low-level control (System 1) under a hierarchical reinforcement learning formalism, making failures usable as a structured learning signal rather than noisy supervision.",
  "doi": "10.48550/arXiv.2512.03913",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jeongeun Park",
    "id": "2367343257",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Jihwan Yoon",
    "id": "2386946496",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Byungwoo Jeon",
    "id": "2344744399",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Juhan Park",
    "id": "2328945875",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Jinwoo Shin",
    "id": "2344965378",
    "h_index": 2,
    "papers": 13
   },
   {
    "name": "Namhoon Cho",
    "id": "2395891720",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Kyungjae Lee",
    "id": "2385791632",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Sangdoo Yun",
    "id": "2325167785",
    "h_index": 4,
    "papers": 19
   },
   {
    "name": "Sungjoon Choi",
    "id": "2288142310",
    "h_index": 2,
    "papers": 9
   }
  ],
  "comment": "https://vine-vla.github.io/",
  "topics": [
   "vla",
   "rl-control",
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2512.03913v1",
  "pdf_url": "https://arxiv.org/pdf/2512.03913v1",
  "html_url": "https://arxiv.org/html/2512.03913v1",
  "code_url": "https://vine-vla.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.6
 },
 {
  "id": "2512.03028",
  "slug": "smp-reusable-score-matching-motion-priors-for-physics-based-character",
  "title": "SMP: Reusable Score-Matching Motion Priors for Physics-Based Character Control",
  "abstract": "Data-driven motion priors that can guide agents toward producing naturalistic behaviors play a pivotal role in creating life-like virtual characters. Adversarial imitation learning has been a highly effective method for learning motion priors from reference motion data. However, adversarial priors, with few exceptions, need to be retrained for each new controller, thereby limiting their reusability and necessitating the retention of the reference motion data when applied to downstream tasks. In this work, we present Score-Matching Motion Priors (SMP), which leverages pre-trained motion diffusion models and score distillation sampling (SDS) to create reusable task-agnostic motion priors. SMPs can be pre-trained on a motion dataset, independent of any control policy or task. Once trained, SMPs can be kept frozen and reused as general-purpose reward functions to train new policies to produce naturalistic behaviors for downstream tasks. We show that a general motion prior trained on large-scale datasets can be repurposed into a variety of style-specific priors. Furthermore, SMP can compose different styles to synthesize new styles not present in the original dataset. Our method can create reusable and modular motion priors that produce high-quality motions comparable to state-of-the-art adversarial imitation learning methods. In our experiments, we demonstrate the effectiveness of SMP across a diverse suite of control tasks with physically simulated humanoid characters. Video available at https://youtu.be/jBA2tWk6vzU",
  "published": "2025-12-02",
  "updated": "2026-04-24",
  "year": "2025",
  "authors": [
   "Yuxuan Mu",
   "Ziyu Zhang",
   "Yi Shi",
   "Dun Yang",
   "Minami Matsumoto",
   "Kotaro Imamura",
   "Guy Tevet",
   "Chuan Guo",
   "Michael Taylor",
   "Chang Shu",
   "Pengcheng Xi",
   "Xue Bin Peng"
  ],
  "author_count": 12,
  "categories": [
   "cs.GR",
   "cs.AI",
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.GR",
  "venue": "SIGGRAPH 2026",
  "venue_source": "arxiv-comment",
  "citations": 14,
  "influential_citations": 0,
  "tldr": "This work presents Score-Matching Motion Priors (SMP), which leverages pre-trained motion diffusion models and score distillation sampling (SDS) to create reusable task-agnostic motion priors that produce high-quality motions comparable to state-of-the-art adversarial imitation learning methods.",
  "doi": "10.1145/3811282",
  "oa_pdf": "https://doi.org/10.1145/3811282",
  "s2_authors": [
   {
    "name": "Yuxuan Mu",
    "id": "2359256072",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Ziyu Zhang",
    "id": "2359845052",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Yi Shi",
    "id": "2294546552",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Minami Matsumoto",
    "id": "2395770532",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "K. Imamura",
    "id": "8335221",
    "h_index": 25,
    "papers": 145
   },
   {
    "name": "Guy Tevet",
    "id": "81493694",
    "h_index": 12,
    "papers": 19
   },
   {
    "name": "Chuan Guo",
    "id": "2391164156",
    "h_index": 5,
    "papers": 15
   },
   {
    "name": "Michael Taylor",
    "id": "2359797337",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Chang Shu",
    "id": "2359253613",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Pengcheng Xi",
    "id": "3111197",
    "h_index": 13,
    "papers": 45
   },
   {
    "name": "Xue Bin Peng",
    "id": "2363132278",
    "h_index": 3,
    "papers": 4
   }
  ],
  "comment": "To appear in ACM Transactions on Graphics (SIGGRAPH 2026)",
  "topics": [
   "humanoids",
   "imitation-diffusion",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2512.03028v3",
  "pdf_url": "https://arxiv.org/pdf/2512.03028v3",
  "html_url": "https://arxiv.org/html/2512.03028v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.68
 },
 {
  "id": "2511.21690",
  "slug": "tracegen-world-modeling-in-3d-trace-space-enables-learning-from-cross",
  "title": "TraceGen: World Modeling in 3D Trace Space Enables Learning from Cross-Embodiment Videos",
  "abstract": "Learning new robot tasks on new platforms and in new scenes from only a handful of demonstrations remains challenging. While videos of other embodiments - humans and different robots - are abundant, differences in embodiment, camera, and environment hinder their direct use. We address the small-data problem by introducing a unifying, symbolic representation - a compact 3D \"trace-space\" of scene-level trajectories - that enables learning from cross-embodiment, cross-environment, and cross-task videos. We present TraceGen, a world model that predicts future motion in trace-space rather than pixel space, abstracting away appearance while retaining the geometric structure needed for manipulation. To train TraceGen at scale, we develop TraceForge, a data pipeline that transforms heterogeneous human and robot videos into consistent 3D traces, yielding a corpus of 123K videos and 1.8M observation-trace-language triplets. Pretraining on this corpus produces a transferable 3D motion prior that adapts efficiently: with just five target robot videos, TraceGen attains 80% success across four tasks while offering 50-600x faster inference than state-of-the-art video-based world models. In the more challenging case where only five uncalibrated human demonstration videos captured on a handheld phone are available, it still reaches 67.5% success on a real robot, highlighting TraceGen's ability to adapt across embodiments without relying on object detectors or heavy pixel-space generation.",
  "published": "2025-11-26",
  "updated": "2025-11-26",
  "year": "2025",
  "authors": [
   "Seungjae Lee",
   "Yoonkyo Jung",
   "Inkook Chun",
   "Yao-Chih Lee",
   "Zikui Cai",
   "Hongjia Huang",
   "Aayush Talreja",
   "Tan Dat Dao",
   "Yongyuan Liang",
   "Jia-Bin Huang",
   "Furong Huang"
  ],
  "author_count": 11,
  "categories": [
   "cs.RO",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 12,
  "influential_citations": 2,
  "tldr": "TraceGen is presented, a world model that predicts future motion in trace-space rather than pixel space, abstracting away appearance while retaining the geometric structure needed for manipulation, highlighting TraceGen's ability to adapt across embodiments without relying on object detectors or heavy pixel-space generation.",
  "doi": "10.48550/arXiv.2511.21690",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Seungjae Lee",
    "id": "2394353888",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Y. Jung",
    "id": "73042266",
    "h_index": 3,
    "papers": 16
   },
   {
    "name": "Inkook Chun",
    "id": "2394305764",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Yao-Chih Lee",
    "id": "2371149142",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Zikui Cai",
    "id": "2346643861",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Hongjia Huang",
    "id": "2394649549",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Aayush Talreja",
    "id": "2363397421",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "T. Dao",
    "id": "113752892",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Yongyuan Liang",
    "id": "83158497",
    "h_index": 12,
    "papers": 24
   },
   {
    "name": "Jia-Bin Huang",
    "id": "2213332546",
    "h_index": 41,
    "papers": 108
   },
   {
    "name": "Furong Huang",
    "id": "2347721825",
    "h_index": 3,
    "papers": 7
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "egocentric-data",
   "foundation-pretraining",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2511.21690v1",
  "pdf_url": "https://arxiv.org/pdf/2511.21690v1",
  "html_url": "https://arxiv.org/html/2511.21690v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.11
 },
 {
  "id": "2511.20633",
  "slug": "reinforcing-action-policies-by-prophesying",
  "title": "Reinforcing Action Policies by Prophesying",
  "abstract": "Vision-Language-Action (VLA) policies excel in aligning language, perception, and robot control. However, most VLAs are trained purely by imitation, which overfits to demonstrations, and is brittle under distribution shift. Reinforcement learning (RL) directly optimizes task reward and thus addresses this misalignment, but real-robot interaction is expensive and conventional simulators are hard to engineer and transfer. We address both data efficiency and optimization stability in VLA post-training via a learned world model and an RL procedure tailored to flow-based action heads. Specifically, we first introduce Prophet, a unified action-to-video robot world model pretrained on large-scale, heterogeneous robot data to learn reusable action-outcome dynamics and then few-shot adapted to new robots, objects, and environments, yielding a rollout-ready simulator. Upon Prophet, we reinforce action policies with our proposed FlowScale, which couples Flow-GRPO with intrinsic stepwise reweighting to stabilize gradients. Together, our solution provides a practical, data- and compute-efficient path to VLA post-training. Experiments show 5-17% success gains on public benchmarks and 24-30% on real robots across diverse VLA backbones.",
  "published": "2025-11-25",
  "updated": "2026-08-06",
  "year": "2025",
  "authors": [
   "Jiahui Zhang",
   "Ze Huang",
   "Chun Gu",
   "Zipei Ma",
   "Li Zhang"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 18,
  "influential_citations": 1,
  "tldr": "This work introduces Prophet, a unified action-to-video robot world model pretrained on large-scale, heterogeneous robot data to learn reusable action-outcome dynamics and then few-shot adapted to new robots, objects, and environments, yielding a rollout-ready simulator and a proposed RL procedure tailored to flow-based action heads.",
  "doi": "10.48550/arXiv.2511.20633",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jiahui Zhang",
    "id": "2269777007",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Ze Huang",
    "id": "2269856014",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Chun Gu",
    "id": "2268399619",
    "h_index": 11,
    "papers": 19
   },
   {
    "name": "Zipei Ma",
    "id": "2371082753",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Li Zhang",
    "id": "2371143277",
    "h_index": 4,
    "papers": 8
   }
  ],
  "comment": "https://LogosRoboticsGroup.github.io/ProphRL",
  "topics": [
   "world-models",
   "vla",
   "sim2real",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2511.20633v2",
  "pdf_url": "https://arxiv.org/pdf/2511.20633v2",
  "html_url": "https://arxiv.org/html/2511.20633v2",
  "code_url": "https://LogosRoboticsGroup.github.io/ProphRL",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.28
 },
 {
  "id": "2511.18173",
  "slug": "egocontrol-controllable-egocentric-video-generation-via-3d-full-body-p",
  "title": "EgoControl: Controllable Egocentric Video Generation via 3D Full-Body Poses",
  "abstract": "Egocentric video generation with fine-grained control through body motion is a key requirement towards embodied AI agents that can simulate, predict, and plan actions. In this work, we propose EgoControl, a pose-controllable video diffusion model trained on egocentric data. We train a video prediction model to condition future frame generation on explicit 3D body pose sequences. To achieve precise motion control, we introduce a novel pose representation that captures both global camera dynamics and articulated body movements, and integrate it through a dedicated control mechanism within the diffusion process. Given a short sequence of observed frames and a sequence of target poses, EgoControl generates temporally coherent and visually realistic future frames that align with the provided pose control. Experimental results demonstrate that EgoControl produces high-quality, pose-consistent egocentric videos, paving the way toward controllable embodied video simulation and understanding.",
  "published": "2025-11-22",
  "updated": "2025-11-22",
  "year": "2025",
  "authors": [
   "Enrico Pallotta",
   "Sina Mokhtarzadeh Azar",
   "Lars Doorenbos",
   "Serdar Ozsoy",
   "Umar Iqbal",
   "Juergen Gall"
  ],
  "author_count": 6,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 7,
  "influential_citations": 2,
  "tldr": "This work proposes EgoControl, a pose-controllable video diffusion model trained on egocentric data that introduces a novel pose representation that captures both global camera dynamics and articulated body movements, and integrates it through a dedicated control mechanism within the diffusion process.",
  "doi": "10.48550/arXiv.2511.18173",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Enrico Pallotta",
    "id": "2265490751",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Sina Mokhtarzadeh Azar",
    "id": "51264689",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Lars Doorenbos",
    "id": "2382927409",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Serdar Ozsoy",
    "id": "2185348472",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Umar Iqbal",
    "id": "2353289193",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Juergen Gall",
    "id": "2354258241",
    "h_index": 2,
    "papers": 7
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "egocentric-data",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2511.18173v1",
  "pdf_url": "https://arxiv.org/pdf/2511.18173v1",
  "html_url": "https://arxiv.org/html/2511.18173v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.9
 },
 {
  "id": "2511.16661",
  "slug": "dexterity-from-smart-lenses-multi-fingered-robot-manipulation-with-in",
  "title": "Dexterity from Smart Lenses: Multi-Fingered Robot Manipulation with In-the-Wild Human Demonstrations",
  "abstract": "Learning multi-fingered robot policies from humans performing daily tasks in natural environments has long been a grand goal in the robotics community. Achieving this would mark significant progress toward generalizable robot manipulation in human environments, as it would reduce the reliance on labor-intensive robot data collection. Despite substantial efforts, progress toward this goal has been bottle-necked by the embodiment gap between humans and robots, as well as by difficulties in extracting relevant contextual and motion cues that enable learning of autonomous policies from in-the-wild human videos. We claim that with simple yet sufficiently powerful hardware for obtaining human data and our proposed framework AINA, we are now one significant step closer to achieving this dream. AINA enables learning multi-fingered policies from data collected by anyone, anywhere, and in any environment using Aria Gen 2 glasses. These glasses are lightweight and portable, feature a high-resolution RGB camera, provide accurate on-board 3D head and hand poses, and offer a wide stereo view that can be leveraged for depth estimation of the scene. This setup enables the learning of 3D point-based policies for multi-fingered hands that are robust to background changes and can be deployed directly without requiring any robot data (including online corrections, reinforcement learning, or simulation). We compare our framework against prior human-to-robot policy learning approaches, ablate our design choices, and demonstrate results across nine everyday manipulation tasks. Robot rollouts are best viewed on our website: https://aina-robot.github.io.",
  "published": "2025-11-20",
  "updated": "2025-11-20",
  "year": "2025",
  "authors": [
   "Irmak Guzey",
   "Haozhi Qi",
   "Julen Urain",
   "Changhao Wang",
   "Jessica Yin",
   "Krishna Bodduluri",
   "Mike Lambeta",
   "Lerrel Pinto",
   "Akshara Rai",
   "Jitendra Malik",
   "Tingfan Wu",
   "Akash Sharma",
   "Homanga Bharadhwaj"
  ],
  "author_count": 13,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 16,
  "influential_citations": 0,
  "tldr": "The proposed framework AINA enables the learning of 3D point-based policies for multi-fingered hands that are robust to background changes and can be deployed directly without requiring any robot data (including online corrections, reinforcement learning, or simulation).",
  "doi": "10.48550/arXiv.2511.16661",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Irmak Guzey",
    "id": "2143167646",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Haozhi Qi",
    "id": "2247951244",
    "h_index": 11,
    "papers": 18
   },
   {
    "name": "Julen Urain",
    "id": "51233860",
    "h_index": 13,
    "papers": 22
   },
   {
    "name": "Changhao Wang",
    "id": "2344075962",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Jessica Yin",
    "id": "1390925803",
    "h_index": 7,
    "papers": 15
   },
   {
    "name": "Krishna Bodduluri",
    "id": "2393206157",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Mike Lambeta",
    "id": "3427691",
    "h_index": 13,
    "papers": 20
   },
   {
    "name": "Lerrel Pinto",
    "id": "2320806817",
    "h_index": 10,
    "papers": 12
   },
   {
    "name": "Akshara Rai",
    "id": "2762463",
    "h_index": 26,
    "papers": 45
   },
   {
    "name": "Jitendra Malik",
    "id": "2242761335",
    "h_index": 15,
    "papers": 32
   },
   {
    "name": "Tingfan Wu",
    "id": "2254158966",
    "h_index": 12,
    "papers": 20
   },
   {
    "name": "Akash Sharma",
    "id": "2109364933",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "Homanga Bharadhwaj",
    "id": "51113848",
    "h_index": 23,
    "papers": 59
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "egocentric-data",
   "rl-control",
   "spatial-3d",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2511.16661v1",
  "pdf_url": "https://arxiv.org/pdf/2511.16661v1",
  "html_url": "https://arxiv.org/html/2511.16661v1",
  "code_url": null,
  "presented": [
   {
    "club": "saturday-robotics",
    "club_name": "Saturday Robotics",
    "session_number": 13,
    "session_title": "Robotics & World Models Reading Club 13: HumanEgo: Train Robot Policy from 30 min Egocentric Videos \u2014 SF 0620",
    "date_text": "Saturday, June 20, 2026 | 2:00 PM \u2013 5:00 PM",
    "city": "San Francisco, CA",
    "url": "https://lu.ma/6vkhxnum",
    "listed_as": ""
   }
  ],
  "club_note": "",
  "featured": true,
  "signal": 4.23
 },
 {
  "id": "2511.15704",
  "slug": "in-n-on-scaling-egocentric-manipulation-with-in-the-wild-and-on-task-d",
  "title": "In-N-On: Scaling Egocentric Manipulation with in-the-wild and on-task Data",
  "abstract": "Egocentric videos are a valuable and scalable data source to learn manipulation policies. However, due to significant data heterogeneity, most existing approaches utilize human data for simple pre-training, which does not unlock its full potential. This paper first provides a scalable recipe for collecting and using egocentric data by categorizing human data into two categories: in-the-wild and on-task alongside with systematic analysis on how to use the data. We first curate a dataset, PHSD, which contains over 1,000 hours of diverse in-the-wild egocentric data and over 20 hours of on-task data directly aligned to the target manipulation tasks. This enables learning a large egocentric language-conditioned flow matching policy, Human0. With domain adaptation techniques, Human0 minimizes the gap between humans and humanoids. Empirically, we show Human0 achieves several novel properties from scaling human data, including language following of instructions from only human data, few-shot learning, and improved robustness using on-task data. Project website: https://xiongyicai.github.io/In-N-On/",
  "published": "2025-11-19",
  "updated": "2025-11-19",
  "year": "2025",
  "authors": [
   "Xiongyi Cai",
   "Ri-Zhao Qiu",
   "Geng Chen",
   "Lai Wei",
   "Isabella Liu",
   "Tianshu Huang",
   "Xuxin Cheng",
   "Xiaolong Wang"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 24,
  "influential_citations": 1,
  "tldr": "A scalable recipe for collecting and using egocentric data is provided by categorizing human data into two categories: in-the-wild and on-task alongside with systematic analysis on how to use the data.",
  "doi": "10.48550/arXiv.2511.15704",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xiongyi Cai",
    "id": "2393079892",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Ri-Zhao Qiu",
    "id": "2290904526",
    "h_index": 11,
    "papers": 19
   },
   {
    "name": "Geng Chen",
    "id": "2393143833",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Lai Wei",
    "id": "2392910023",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "I. Liu",
    "id": "2310233934",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Tianshu Huang",
    "id": "2359285165",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Xuxin Cheng",
    "id": "2287822264",
    "h_index": 8,
    "papers": 9
   },
   {
    "name": "Xiaolong Wang",
    "id": "2294782536",
    "h_index": 12,
    "papers": 17
   }
  ],
  "comment": "Project webpage: https://xiongyicai.github.io/In-N-On/",
  "topics": [
   "humanoids",
   "egocentric-data",
   "imitation-diffusion",
   "foundation-pretraining",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2511.15704v1",
  "pdf_url": "https://arxiv.org/pdf/2511.15704v1",
  "html_url": "https://arxiv.org/html/2511.15704v1",
  "code_url": "https://xiongyicai.github.io/In-N-On/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.4
 },
 {
  "id": "2511.14759",
  "slug": "0-6-a-vla-that-learns-from-experience",
  "title": "$\u03c0^{*}_{0.6}$: a VLA That Learns From Experience",
  "abstract": "We study how vision-language-action (VLA) models can improve through real-world deployments via reinforcement learning (RL). We present a general-purpose method, RL with Experience and Corrections via Advantage-conditioned Policies (RECAP), that provides for RL training of VLAs via advantage conditioning. Our method incorporates heterogeneous data into the self-improvement process, including demonstrations, data from on-policy collection, and expert teleoperated interventions provided during autonomous execution. RECAP starts by pre-training a generalist VLA with offline RL, which we call $\u03c0^{*}_{0.6}$, that can then be specialized to attain high performance on downstream tasks through on-robot data collection. We show that the $\u03c0^{*}_{0.6}$ model trained with the full RECAP method can fold laundry in real homes, reliably assemble boxes, and make espresso drinks using a professional espresso machine. On some of the hardest tasks, RECAP more than doubles task throughput and roughly halves the task failure rate.",
  "published": "2025-11-18",
  "updated": "2025-11-19",
  "year": "2025",
  "authors": [
   "Physical Intelligence",
   "Ali Amin",
   "Raichelle Aniceto",
   "Ashwin Balakrishna",
   "Kevin Black",
   "Ken Conley",
   "Grace Connors",
   "James Darpinian",
   "Karan Dhabalia",
   "Jared DiCarlo",
   "Danny Driess",
   "Michael Equi",
   "Adnan Esmail",
   "Yunhao Fang",
   "Chelsea Finn",
   "Catherine Glossop",
   "Thomas Godden",
   "Ivan Goryachev",
   "Lachy Groom",
   "Hunter Hancock",
   "Karol Hausman",
   "Gashon Hussein",
   "Brian Ichter",
   "Szymon Jakubczak",
   "Rowan Jen",
   "Tim Jones",
   "Ben Katz",
   "Liyiming Ke",
   "Chandra Kuchi",
   "Marinda Lamb",
   "Devin LeBlanc",
   "Sergey Levine",
   "Adrian Li-Bell",
   "Yao Lu",
   "Vishnu Mano",
   "Mohith Mothukuri",
   "Suraj Nair",
   "Karl Pertsch",
   "Allen Z. Ren",
   "Charvi Sharma",
   "Lucy Xiaoyang Shi",
   "Laura Smith",
   "Jost Tobias Springenberg",
   "Kyle Stachowicz",
   "Will Stoeckle",
   "Alex Swerdlow",
   "James Tanner",
   "Marcel Torne",
   "Quan Vuong",
   "Anna Walling",
   "Haohuan Wang",
   "Blake Williams",
   "Sukwon Yoo",
   "Lili Yu",
   "Ury Zhilinsky",
   "Zhiyuan Zhou"
  ],
  "author_count": 56,
  "categories": [
   "cs.LG",
   "cs.RO"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 269,
  "influential_citations": 28,
  "tldr": "A general-purpose method that provides for RL training of VLAs via advantage conditioning, which incorporates heterogeneous data into the self-improvement process, including demonstrations, data from on-policy collection, and expert teleoperated interventions provided during autonomous execution.",
  "doi": "10.48550/arXiv.2511.14759",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Physical Intelligence",
    "id": "2356784273",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "A. Amin",
    "id": "2393165614",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Raichelle J. Aniceto",
    "id": "1994633",
    "h_index": 6,
    "papers": 18
   },
   {
    "name": "Ashwin Balakrishna",
    "id": "2348256109",
    "h_index": 8,
    "papers": 11
   },
   {
    "name": "Kevin Black",
    "id": "2258959388",
    "h_index": 13,
    "papers": 15
   },
   {
    "name": "Ken Conley",
    "id": "2392964702",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Grace Connors",
    "id": "2392964751",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "James Darpinian",
    "id": "2356784408",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Karan Dhabalia",
    "id": "2356782366",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Jared DiCarlo",
    "id": "2392963692",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Danny Driess",
    "id": "2283848260",
    "h_index": 27,
    "papers": 35
   },
   {
    "name": "Michael Equi",
    "id": "2298901898",
    "h_index": 10,
    "papers": 15
   },
   {
    "name": "A. Esmail",
    "id": "2332926590",
    "h_index": 7,
    "papers": 12
   },
   {
    "name": "Yunhao Fang",
    "id": "2312873153",
    "h_index": 10,
    "papers": 21
   },
   {
    "name": "Chelsea Finn",
    "id": "2286629816",
    "h_index": 6,
    "papers": 19
   },
   {
    "name": "Catherine Glossop",
    "id": "2257348904",
    "h_index": 9,
    "papers": 13
   },
   {
    "name": "Thomas Godden",
    "id": "2297061888",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "I. Goryachev",
    "id": "2392964584",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Lachy Groom",
    "id": "2332926792",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "H. Hancock",
    "id": "2392965108",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Karol Hausman",
    "id": "1944801",
    "h_index": 47,
    "papers": 122
   },
   {
    "name": "Gashon Hussein",
    "id": "2316434586",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Brian Ichter",
    "id": "2704814",
    "h_index": 37,
    "papers": 60
   },
   {
    "name": "S. Jakubczak",
    "id": "2332926523",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Rowan Jen",
    "id": "2392964552",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Tim Jones",
    "id": "2333409218",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Ben Katz",
    "id": "2392964124",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Liyiming Ke",
    "id": "2332976956",
    "h_index": 7,
    "papers": 8
   },
   {
    "name": "Chandra Kuchi",
    "id": "2392964758",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Marinda Lamb",
    "id": "2392964525",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Devin LeBlanc",
    "id": "2356787296",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Sergey Levine",
    "id": "2257062067",
    "h_index": 14,
    "papers": 18
   },
   {
    "name": "Adrian Li-Bell",
    "id": "2332927424",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Yao Lu",
    "id": "2393033720",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Vishnu Mano",
    "id": "2392965125",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Mohith Mothukuri",
    "id": "2332926533",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Suraj Nair",
    "id": "2286638954",
    "h_index": 12,
    "papers": 19
   },
   {
    "name": "Karl Pertsch",
    "id": "31719101",
    "h_index": 32,
    "papers": 56
   },
   {
    "name": "Allen Z. Ren",
    "id": "2356677628",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Charvi Sharma",
    "id": "2392965859",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "L. Shi",
    "id": "2292341452",
    "h_index": 12,
    "papers": 17
   },
   {
    "name": "Laura Smith",
    "id": "2302181645",
    "h_index": 8,
    "papers": 8
   },
   {
    "name": "Jost Tobias Springenberg",
    "id": "2060551",
    "h_index": 44,
    "papers": 93
   },
   {
    "name": "Kyle Stachowicz",
    "id": "2106415427",
    "h_index": 12,
    "papers": 21
   },
   {
    "name": "Will Stoeckle",
    "id": "2392964442",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Alex Swerdlow",
    "id": "2392964495",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "James Tanner",
    "id": "2332926892",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Marcel Torne",
    "id": "2361252179",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Quan Vuong",
    "id": "2288210223",
    "h_index": 19,
    "papers": 29
   },
   {
    "name": "Anna Walling",
    "id": "2333982746",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Haohuan Wang",
    "id": "2332952253",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Blake Williams",
    "id": "2393372011",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Sukwon Yoo",
    "id": "2393863269",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Lili Yu",
    "id": "2356801221",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Ury Zhilinsky",
    "id": "3187915",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Zhiyuan Zhou",
    "id": "2314073778",
    "h_index": 6,
    "papers": 16
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "rl-control",
   "foundation-pretraining",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2511.14759v2",
  "pdf_url": "https://arxiv.org/pdf/2511.14759v2",
  "html_url": "https://arxiv.org/html/2511.14759v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.43
 },
 {
  "id": "2511.11520",
  "slug": "scalable-policy-evaluation-with-video-world-models",
  "title": "Scalable Policy Evaluation with Video World Models",
  "abstract": "Training generalist policies for robotic manipulation has shown great promise, as they enable language-conditioned, multi-task behaviors across diverse scenarios. However, evaluating these policies remains difficult because real-world testing is expensive, time-consuming, and labor-intensive. It also requires frequent environment resets and carries safety risks when deploying unproven policies on physical robots. Manually creating and populating simulation environments with assets for robotic manipulation has not addressed these issues, primarily due to the significant engineering effort required and the substantial sim-to-real gap, both in terms of physics and rendering. In this paper, we explore the use of action-conditional video generation models as a scalable way to learn world models for policy evaluation. We demonstrate how to incorporate action conditioning into existing pre-trained video generation models. This allows leveraging internet-scale in-the-wild online videos during the pre-training stage and alleviates the need for a large dataset of paired video-action data, which is expensive to collect for robotic manipulation. Our paper examines the effect of dataset diversity, pre-trained weights, and common failure cases for the proposed evaluation pipeline. Our experiments demonstrate that across various metrics, including policy ranking and the correlation between actual policy values and predicted policy values, these models offer a promising approach for evaluating policies without requiring real-world interactions.",
  "published": "2025-11-14",
  "updated": "2025-12-04",
  "year": "2025",
  "authors": [
   "Wei-Cheng Tseng",
   "Jinwei Gu",
   "Qinsheng Zhang",
   "Hanzi Mao",
   "Ming-Yu Liu",
   "Florian Shkurti",
   "Lin Yen-Chen"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 24,
  "influential_citations": 3,
  "tldr": "This paper demonstrates how to incorporate action conditioning into existing pre-trained video generation models, which allows leveraging internet-scale in-the-wild online videos during the pre-training stage and alleviates the need for a large dataset of paired video-action data, which is expensive to collect for robotic manipulation.",
  "doi": "10.48550/arXiv.2511.11520",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Wei-Cheng Tseng",
    "id": "2321873035",
    "h_index": 8,
    "papers": 17
   },
   {
    "name": "Jinwei Gu",
    "id": "2338980231",
    "h_index": 9,
    "papers": 13
   },
   {
    "name": "Qinsheng Zhang",
    "id": "2288856236",
    "h_index": 14,
    "papers": 17
   },
   {
    "name": "Hanzi Mao",
    "id": "2313167739",
    "h_index": 8,
    "papers": 8
   },
   {
    "name": "Ming-Yu Liu",
    "id": "2385747822",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "Florian Shkurti",
    "id": "2355024925",
    "h_index": 11,
    "papers": 20
   },
   {
    "name": "Lin Yen-Chen",
    "id": "1485124622",
    "h_index": 6,
    "papers": 6
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "sim2real",
   "foundation-pretraining",
   "data-teleop",
   "video-generation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2511.11520v3",
  "pdf_url": "https://arxiv.org/pdf/2511.11520v3",
  "html_url": "https://arxiv.org/html/2511.11520v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.4
 },
 {
  "id": "2511.09515",
  "slug": "wmpo-world-model-based-policy-optimization-for-vision-language-action",
  "title": "WMPO: World Model-based Policy Optimization for Vision-Language-Action Models",
  "abstract": "Vision-Language-Action (VLA) models have shown strong potential for general-purpose robotic manipulation, but their reliance on expert demonstrations limits their ability to learn from failures and perform self-corrections. Reinforcement learning (RL) addresses these through self-improving interactions with the physical environment, but suffers from high sample complexity on real robots. We introduce World-Model-based Policy Optimization (WMPO), a principled framework for on-policy VLA RL without interacting with the real environment. In contrast to widely used latent world models, WMPO focuses on pixel-based predictions that align the \"imagined\" trajectories with the VLA features pretrained with web-scale images. Crucially, WMPO enables the policy to perform on-policy GRPO that provides stronger performance than the often-used off-policy methods. Extensive experiments in both simulation and real-robot settings demonstrate that WMPO (i) substantially improves sample efficiency, (ii) achieves stronger overall performance, (iii) exhibits emergent behaviors such as self-correction, and (iv) demonstrates robust generalization and lifelong learning capabilities.",
  "published": "2025-11-12",
  "updated": "2025-11-12",
  "year": "2025",
  "authors": [
   "Fangqi Zhu",
   "Zhengyang Yan",
   "Zicong Hong",
   "Quanxin Shou",
   "Xiao Ma",
   "Song Guo"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 47,
  "influential_citations": 5,
  "tldr": "World-Model-based Policy Optimization (WMPO) is introduced, a principled framework for on-policy VLA RL without interacting with the real environment that enables the policy to perform on-policy GRPO that provides stronger performance than the often-used off-policy methods.",
  "doi": "10.48550/arXiv.2511.09515",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Fangqi Zhu",
    "id": "2307561705",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Zhengyang Yan",
    "id": "2398982194",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Zicong Hong",
    "id": "89600513",
    "h_index": 19,
    "papers": 72
   },
   {
    "name": "Quanxin Shou",
    "id": "2391961599",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Xiao Ma",
    "id": "2125110703",
    "h_index": 13,
    "papers": 45
   },
   {
    "name": "Song Guo",
    "id": "2307558383",
    "h_index": 3,
    "papers": 10
   }
  ],
  "comment": "project website: https://wm-po.github.io",
  "topics": [
   "world-models",
   "vla",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2511.09515v1",
  "pdf_url": "https://arxiv.org/pdf/2511.09515v1",
  "html_url": "https://arxiv.org/html/2511.09515v1",
  "code_url": "https://wm-po.github.io",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.68
 },
 {
  "id": "2511.07820",
  "slug": "sonic-supersizing-motion-tracking-for-natural-humanoid-whole-body-cont",
  "title": "SONIC: Supersizing Motion Tracking for Natural Humanoid Whole-Body Control",
  "abstract": "Despite the rise of billion-parameter foundation models trained across thousands of graphical processing units (GPUs), similar scaling gains have not been shown for humanoid control. Current neural controllers for humanoids remain modest in size, target a limited set of behaviors, and are trained on a handful of GPUs. We show that scaling model capacity, data, and compute yields a generalist humanoid controller capable of natural, robust whole-body movements. We position motion tracking as a scalable task for humanoid control, leveraging dense supervision from diverse motion-capture data to acquire human motion priors without manual reward engineering. We build a foundation model for motion tracking by scaling along three axes: network size (1.2M to 42M parameters), dataset volume (100M+ frames from 700 hours of motion capture), and compute (21k GPU hours). Beyond demonstrating the benefits of scale, we further show downstream utility through a real-time kinematic planner that bridges motion tracking to tasks such as navigation, enabling natural and interactive control, as well as a unified token space that supports virtual reality (VR) teleoperation and vision-language-action (VLA) models with a single policy. Through this interface, we demonstrate autonomous VLA-driven whole-body loco-manipulation requiring coordinated hand and foot placement. Scaling motion tracking exhibits favorable properties: performance improves steadily with compute and data diversity, and learned policies generalize to unseen motions, establishing motion tracking at scale as a practical foundation for humanoid control.",
  "published": "2025-11-11",
  "updated": "2026-08-13",
  "year": "2025",
  "authors": [
   "Zhengyi Luo",
   "Ye Yuan",
   "Tingwu Wang",
   "Chenran Li",
   "Fernando Casta\u00f1eda",
   "Sirui Chen",
   "Zi-Ang Cao",
   "Jiefeng Li",
   "David Minor",
   "Qingwei Ben",
   "Jinhyung Park",
   "David Sami",
   "Zi Wang",
   "Xingye Da",
   "Runyu Ding",
   "Cyrus Hogg",
   "Lina Song",
   "Edy Lim",
   "Eugene Jeong",
   "Tairan He",
   "Haoru Xue",
   "Wenli Xiao",
   "Simon Yuen",
   "Jan Kautz",
   "Yan Chang",
   "Umar Iqbal",
   "Linxi \"Jim\" Fan",
   "Yuke Zhu"
  ],
  "author_count": 28,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.GR",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "Science Robotics",
  "venue_source": "semantic-scholar",
  "citations": 139,
  "influential_citations": 21,
  "tldr": "It is shown that scaling model capacity, data, and compute yields a generalist humanoid controller capable of natural, robust whole-body movements, and a real-time kinematic planner that bridges motion tracking to tasks such as navigation, enabling natural and interactive control.",
  "doi": "10.1126/scirobotics.aed4592",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhengyi Luo",
    "id": "2329051397",
    "h_index": 9,
    "papers": 15
   },
   {
    "name": "Ye Yuan",
    "id": "2391922742",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Tingwu Wang",
    "id": "2392415176",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Chenran Li",
    "id": "2242186839",
    "h_index": 14,
    "papers": 70
   },
   {
    "name": "Sirui Chen",
    "id": "2209905328",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Fernando Casta\u00f1eda",
    "id": "2350859158",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Zi-ang Cao",
    "id": "2359785826",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Jiefeng Li",
    "id": "2335292346",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "David Minor",
    "id": "2073901220",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Qingwei Ben",
    "id": "2293395502",
    "h_index": 11,
    "papers": 16
   },
   {
    "name": "Xingye Da",
    "id": "2350863929",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Runyu Ding",
    "id": "2350936924",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Cyrus Hogg",
    "id": "2391810712",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Lina Song",
    "id": "2241990073",
    "h_index": 4,
    "papers": 14
   },
   {
    "name": "Edy Lim",
    "id": "2391810421",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Eugene Jeong",
    "id": "2391812718",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Tairan He",
    "id": "2055132189",
    "h_index": 20,
    "papers": 29
   },
   {
    "name": "Haoru Xue",
    "id": "2296718361",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "Wenli Xiao",
    "id": "2147212066",
    "h_index": 15,
    "papers": 32
   },
   {
    "name": "Zi Wang",
    "id": "2343669335",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "S. Yuen",
    "id": "31565183",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Jan Kautz",
    "id": "2273651410",
    "h_index": 44,
    "papers": 107
   },
   {
    "name": "Yan Chang",
    "id": "2391927706",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Umar Iqbal",
    "id": "2256734969",
    "h_index": 11,
    "papers": 21
   },
   {
    "name": "LinxiJimFan",
    "id": "2350861618",
    "h_index": 10,
    "papers": 15
   },
   {
    "name": "Yuke Zhu",
    "id": "2338857250",
    "h_index": 7,
    "papers": 14
   }
  ],
  "comment": "Project page: https://nvlabs.github.io/SONIC/",
  "topics": [
   "vla",
   "humanoids",
   "egocentric-data",
   "navigation",
   "foundation-pretraining",
   "data-teleop"
  ],
  "orgs": [
   "NVIDIA"
  ],
  "abs_url": "https://arxiv.org/abs/2511.07820v4",
  "pdf_url": "https://arxiv.org/pdf/2511.07820v4",
  "html_url": "https://arxiv.org/html/2511.07820v4",
  "code_url": "https://nvlabs.github.io/SONIC/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.15
 },
 {
  "id": "2511.07409",
  "slug": "dimo-diverse-3d-motion-generation-for-arbitrary-objects",
  "title": "DIMO: Diverse 3D Motion Generation for Arbitrary Objects",
  "abstract": "We present DIMO, a generative approach capable of generating diverse 3D motions for arbitrary objects from a single image. The core idea of our work is to leverage the rich priors in well-trained video models to extract the common motion patterns and then embed them into a shared low-dimensional latent space. Specifically, we first generate multiple videos of the same object with diverse motions. We then embed each motion into a latent vector and train a shared motion decoder to learn the distribution of motions represented by a structured and compact motion representation, i.e., neural key point trajectories. The canonical 3D Gaussians are then driven by these key points and fused to model the geometry and appearance. During inference time with learned latent space, we can instantly sample diverse 3D motions in a single-forward pass and support several interesting applications including 3D motion interpolation and language-guided motion generation. Our project page is available at https://linzhanm.github.io/dimo.",
  "published": "2025-11-10",
  "updated": "2025-11-10",
  "year": "2025",
  "authors": [
   "Linzhan Mou",
   "Jiahui Lei",
   "Chen Wang",
   "Lingjie Liu",
   "Kostas Daniilidis"
  ],
  "author_count": 5,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "ICCV",
  "venue_source": "semantic-scholar",
  "citations": 7,
  "influential_citations": 0,
  "tldr": "DIMO, a generative approach capable of generating diverse 3D motions for arbitrary objects from a single image, to leverage the rich priors in well-trained video models to extract the common motion patterns and then embed them into a shared low-dimensional latent space.",
  "doi": "10.1109/ICCV51701.2025.01332",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Linzhan Mou",
    "id": "2205658464",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "Jiahui Lei",
    "id": "2052835670",
    "h_index": 13,
    "papers": 41
   },
   {
    "name": "Chen Wang",
    "id": "2319389965",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Lingjie Liu",
    "id": "2268627689",
    "h_index": 9,
    "papers": 18
   },
   {
    "name": "Kostas Daniilidis",
    "id": "2065557091",
    "h_index": 25,
    "papers": 89
   }
  ],
  "comment": "Published in ICCV 2025, project page https://linzhanm.github.io/dimo",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2511.07409v1",
  "pdf_url": "https://arxiv.org/pdf/2511.07409v1",
  "html_url": "https://arxiv.org/html/2511.07409v1",
  "code_url": "https://linzhanm.github.io/dimo",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.4
 },
 {
  "id": "2511.02832",
  "slug": "twist2-scalable-portable-and-holistic-humanoid-data-collection-system",
  "title": "TWIST2: Scalable, Portable, and Holistic Humanoid Data Collection System",
  "abstract": "Large-scale data has driven breakthroughs in robotics, from language models to vision-language-action models in bimanual manipulation. However, humanoid robotics lacks equally effective data collection frameworks. Existing humanoid teleoperation systems either use decoupled control or depend on expensive motion capture setups. We introduce TWIST2, a portable, mocap-free humanoid teleoperation and data collection system that preserves full whole-body control while advancing scalability. Our system leverages PICO4U VR for obtaining real-time whole-body human motions, with a custom 2-DoF robot neck (cost around $250) for egocentric vision, enabling holistic human-to-humanoid control. We demonstrate long-horizon dexterous and mobile humanoid skills and we can collect 100 demonstrations in 15 minutes with an almost 100% success rate. Building on this pipeline, we propose a hierarchical visuomotor policy framework that autonomously controls the full humanoid body based on egocentric vision. Our visuomotor policy successfully demonstrates whole-body dexterous manipulation and dynamic kicking tasks. The entire system is fully reproducible and open-sourced at https://yanjieze.com/TWIST2 . Our collected dataset is also open-sourced at https://twist-data.github.io .",
  "published": "2025-11-04",
  "updated": "2025-11-04",
  "year": "2025",
  "authors": [
   "Yanjie Ze",
   "Siheng Zhao",
   "Weizhuo Wang",
   "Angjoo Kanazawa",
   "Rocky Duan",
   "Pieter Abbeel",
   "Guanya Shi",
   "Jiajun Wu",
   "C. Karen Liu"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 85,
  "influential_citations": 15,
  "tldr": "This work introduces TWIST2, a portable, mocap-free humanoid teleoperation and data collection system that preserves full whole-body control while advancing scalability, and proposes a hierarchical visuomotor policy framework that autonomously controls the full humanoid body based on egocentric vision.",
  "doi": "10.48550/arXiv.2511.02832",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yanjie Ze",
    "id": "2325901084",
    "h_index": 10,
    "papers": 13
   },
   {
    "name": "Siheng Zhao",
    "id": "2243033718",
    "h_index": 12,
    "papers": 19
   },
   {
    "name": "Weizhuo Wang",
    "id": "2290968965",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Angjoo Kanazawa",
    "id": "20615377",
    "h_index": 60,
    "papers": 126
   },
   {
    "name": "Rocky Duan",
    "id": "2381724728",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Pieter Abbeel",
    "id": "2381724317",
    "h_index": 9,
    "papers": 17
   },
   {
    "name": "Guanya Shi",
    "id": "2384824402",
    "h_index": 9,
    "papers": 24
   },
   {
    "name": "Jiajun Wu",
    "id": "2326444500",
    "h_index": 35,
    "papers": 214
   },
   {
    "name": "C. K. Liu",
    "id": "2376138723",
    "h_index": 9,
    "papers": 11
   }
  ],
  "comment": "Website: https://yanjieze.com/TWIST2",
  "topics": [
   "vla",
   "dexterous-manipulation",
   "humanoids",
   "egocentric-data",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2511.02832v1",
  "pdf_url": "https://arxiv.org/pdf/2511.02832v1",
  "html_url": "https://arxiv.org/html/2511.02832v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.93
 },
 {
  "id": "2511.01266",
  "slug": "motionstream-real-time-video-generation-with-interactive-motion-contro",
  "title": "MotionStream: Real-Time Video Generation with Interactive Motion Controls",
  "abstract": "Current motion-conditioned video generation methods suffer from prohibitive latency (minutes per video) and non-causal processing that prevents real-time interaction. We present MotionStream, enabling sub-second latency with up to 29 FPS streaming generation on a single GPU. Our approach begins by augmenting a text-to-video model with motion control, which generates high-quality videos that adhere to the global text prompt and local motion guidance, but does not perform inference on the fly. As such, we distill this bidirectional teacher into a causal student through Self Forcing with Distribution Matching Distillation, enabling real-time streaming inference. Several key challenges arise when generating videos of long, potentially infinite time-horizons -- (1) bridging the domain gap from training on finite length and extrapolating to infinite horizons, (2) sustaining high quality by preventing error accumulation, and (3) maintaining fast inference, without incurring growth in computational cost due to increasing context windows. A key to our approach is introducing carefully designed sliding-window causal attention, combined with attention sinks. By incorporating self-rollout with attention sinks and KV cache rolling during training, we properly simulate inference-time extrapolations with a fixed context window, enabling constant-speed generation of arbitrarily long videos. Our models achieve state-of-the-art results in motion following and video quality while being two orders of magnitude faster, uniquely enabling infinite-length streaming. With MotionStream, users can paint trajectories, control cameras, or transfer motion, and see results unfold in real-time, delivering a truly interactive experience.",
  "published": "2025-11-03",
  "updated": "2026-03-05",
  "year": "2025",
  "authors": [
   "Joonghyuk Shin",
   "Zhengqi Li",
   "Richard Zhang",
   "Jun-Yan Zhu",
   "Jaesik Park",
   "Eli Shechtman",
   "Xun Huang"
  ],
  "author_count": 7,
  "categories": [
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.CV",
  "venue": "ICLR 2026",
  "venue_source": "arxiv-comment",
  "citations": 72,
  "influential_citations": 7,
  "tldr": "This work distills a bidirectional teacher into a causal student through Self Forcing with Distribution Matching Distillation, enabling real-time streaming inference, and achieves state-of-the-art results in motion following and video quality while being two orders of magnitude faster, uniquely enabling infinite-length streaming.",
  "doi": "10.48550/arXiv.2511.01266",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Joonghyuk Shin",
    "id": "2172556590",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Zhengqi Li",
    "id": "2367076348",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Richard Zhang",
    "id": "2401595683",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Jun-Yan Zhu",
    "id": "2284984185",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "Jaesik Park",
    "id": "2423580015",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "E. Schechtman",
    "id": "2408027974",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Xun Huang",
    "id": "2334843806",
    "h_index": 8,
    "papers": 14
   }
  ],
  "comment": "ICLR 2026, Project webpage: https://joonghyuk.com/motionstream-web/",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2511.01266v5",
  "pdf_url": "https://arxiv.org/pdf/2511.01266v5",
  "html_url": "https://arxiv.org/html/2511.01266v5",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.36
 },
 {
  "id": "2511.01177",
  "slug": "scaling-cross-embodiment-world-models-for-dexterous-manipulation",
  "title": "Scaling Cross-Embodiment World Models for Dexterous Manipulation",
  "abstract": "Cross-embodiment learning seeks to build generalist robots that learn from and operate across diverse morphologies, but differences in kinematics and action spaces hinder data sharing and control transfer. We ask: What structure can be shared across embodiments despite these differences? We argue that the physical interactions they induce can be modeled in a shared geometric space, allowing world models to provide a common interface for learning and control. To realize this idea, we represent human and robot hands as sets of 3D particles and define actions as end-effector particle displacement fields. This representation abstracts away embodiment-specific joint spaces while preserving the geometry and motion relevant to physical interaction. We train a graph-based world model on random interaction data from diverse simulated robot hands and real human hands, and integrate it with model-predictive control for deployment on new hardware. Experiments on rigid and deformable manipulation reveal three findings: increasing the diversity of training embodiments improves generalization to unseen hands; appropriately combining simulated and real-world data outperforms either source alone; and the same learned model enables effective control on robotic hands with distinct kinematics and degrees of freedom. These results position particle-based world models as a shared interface for learning from and for heterogeneous embodiments.",
  "published": "2025-11-03",
  "updated": "2026-07-21",
  "year": "2025",
  "authors": [
   "Zihao He",
   "Bo Ai",
   "Tongzhou Mu",
   "Yulin Liu",
   "Weikang Wan",
   "Jiawei Fu",
   "Yilun Du",
   "Henrik I. Christensen",
   "Hao Su"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS 2026",
  "venue_source": "arxiv-comment",
  "citations": 11,
  "influential_citations": 0,
  "tldr": "This work argues that the physical interactions they induce can be modeled in a shared geometric space, allowing world models to provide a common interface for learning and control, and positions particle-based world models as a shared interface for learning from and for heterogeneous embodiments.",
  "doi": "10.48550/arXiv.2511.01177",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zihao He",
    "id": "2392903359",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Bo Ai",
    "id": "2352099808",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Tongzhou Mu",
    "id": "3431352",
    "h_index": 11,
    "papers": 28
   },
   {
    "name": "Yulin Liu",
    "id": "2292207780",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Weikang Wan",
    "id": "2265383014",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Jiawei Fu",
    "id": "2378213165",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Yilun Du",
    "id": "2387117108",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Henrik I. Christensen",
    "id": "2290489704",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Hao Su",
    "id": "2352078206",
    "h_index": 4,
    "papers": 5
   }
  ],
  "comment": "Accepted to IROS 2026, Project Page: https://alan-heoooh.github.io/dexwm.html",
  "topics": [
   "world-models",
   "dexterous-manipulation",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2511.01177v3",
  "pdf_url": "https://arxiv.org/pdf/2511.01177v3",
  "html_url": "https://arxiv.org/html/2511.01177v3",
  "code_url": "https://alan-heoooh.github.io/dexwm.html",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.58
 },
 {
  "id": "2510.26742",
  "slug": "running-vlas-at-real-time-speed",
  "title": "Running VLAs at Real-time Speed",
  "abstract": "In this paper, we show how to run pi0-level multi-view VLA at 30Hz frame rate and at most 480Hz trajectory frequency using a single consumer GPU. This enables dynamic and real-time tasks that were previously believed to be unattainable by large VLA models. To achieve it, we introduce a bag of strategies to eliminate the overheads in model inference. The real-world experiment shows that the pi0 policy with our strategy achieves a 100% success rate in grasping a falling pen task. Based on the results, we further propose a full streaming inference framework for real-time robot control of VLA. Code is available at https://github.com/Dexmal/realtime-vla.",
  "published": "2025-10-30",
  "updated": "2025-10-30",
  "year": "2025",
  "authors": [
   "Yunchao Ma",
   "Yizhuang Zhou",
   "Yunhuan Yang",
   "Tiancai Wang",
   "Haoqiang Fan"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 34,
  "influential_citations": 3,
  "tldr": "This paper shows how to run pi0-level multi-view VLA at 30Hz frame rate and at most 480Hz trajectory frequency using a single consumer GPU, and proposes a full streaming inference framework for real-time robot control of VLA.",
  "doi": "10.48550/arXiv.2510.26742",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yunchao Ma",
    "id": "2388896768",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Yizhuang Zhou",
    "id": "2118764798",
    "h_index": 7,
    "papers": 8
   },
   {
    "name": "Yunhuan Yang",
    "id": "2387115051",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Tiancai Wang",
    "id": "2325923837",
    "h_index": 10,
    "papers": 30
   },
   {
    "name": "Haoqiang Fan",
    "id": "2326357387",
    "h_index": 8,
    "papers": 14
   }
  ],
  "comment": "Code is available at https://github.com/Dexmal/realtime-vla",
  "topics": [
   "vla",
   "dexterous-manipulation"
  ],
  "orgs": [
   "Physical Intelligence"
  ],
  "abs_url": "https://arxiv.org/abs/2510.26742v1",
  "pdf_url": "https://arxiv.org/pdf/2510.26742v1",
  "html_url": "https://arxiv.org/html/2510.26742v1",
  "code_url": "https://github.com/Dexmal/realtime-vla",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.04
 },
 {
  "id": "2510.26433",
  "slug": "co-evolving-latent-action-world-models",
  "title": "Co-Evolving Latent Action World Models",
  "abstract": "Adapting pretrained video generation models into controllable world models via latent actions is a promising step towards creating generalist world models. The dominant paradigm adopts a two-stage approach that trains latent action model (LAM) and the world model separately, resulting in redundant training and limiting their potential for co-adaptation. A conceptually simple and appealing idea is to directly replace the forward dynamic model in LAM with a powerful world model and training them jointly, but it is non-trivial and prone to representational collapse. In this work, we propose CoLA-World, which for the first time successfully realizes this synergistic paradigm, resolving the core challenge in joint learning through a critical warm-up phase that effectively aligns the representations of the from-scratch LAM with the pretrained world model. This unlocks a co-evolution cycle: the world model acts as a knowledgeable tutor, providing gradients to shape a high-quality LAM, while the LAM offers a more precise and adaptable control interface to the world model. Empirically, CoLA-World matches or outperforms prior two-stage methods in both video simulation quality and downstream visual planning, establishing a robust and efficient new paradigm for the field.",
  "published": "2025-10-30",
  "updated": "2026-04-06",
  "year": "2025",
  "authors": [
   "Yucen Wang",
   "Fengming Zhang",
   "De-Chuan Zhan",
   "Li Zhao",
   "Kaixin Wang",
   "Jiang Bian"
  ],
  "author_count": 6,
  "categories": [
   "cs.LG"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 12,
  "influential_citations": 0,
  "tldr": "This work proposes CoLA-World, which for the first time successfully realizes this synergistic paradigm, resolving the core challenge in joint learning through a critical warm-up phase that effectively aligns the representations of the from-scratch LAM with the pretrained world model.",
  "doi": "10.48550/arXiv.2510.26433",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yucen Wang",
    "id": "2220305846",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Feng Zhang",
    "id": "2381987284",
    "h_index": 7,
    "papers": 12
   },
   {
    "name": "De-Chuan Zhan",
    "id": "2291959312",
    "h_index": 6,
    "papers": 18
   },
   {
    "name": "Li Zhao",
    "id": "2218154011",
    "h_index": 7,
    "papers": 25
   },
   {
    "name": "Kaixin Wang",
    "id": "2367933865",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "Jiang Bian",
    "id": "2287806843",
    "h_index": 7,
    "papers": 16
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2510.26433v2",
  "pdf_url": "https://arxiv.org/pdf/2510.26433v2",
  "html_url": "https://arxiv.org/html/2510.26433v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.11
 },
 {
  "id": "2511.00062",
  "slug": "world-simulation-with-video-foundation-models-for-physical-ai",
  "title": "World Simulation with Video Foundation Models for Physical AI",
  "abstract": "We introduce [Cosmos-Predict2.5], the latest generation of the Cosmos World Foundation Models for Physical AI. Built on a flow-based architecture, [Cosmos-Predict2.5] unifies Text2World, Image2World, and Video2World generation in a single model and leverages [Cosmos-Reason1], a Physical AI vision-language model, to provide richer text grounding and finer control of world simulation. Trained on 200M curated video clips and refined with reinforcement learning-based post-training, [Cosmos-Predict2.5] achieves substantial improvements over [Cosmos-Predict1] in video quality and instruction alignment, with models released at 2B and 14B scales. These capabilities enable more reliable synthetic data generation, policy evaluation, and closed-loop simulation for robotics and autonomous systems. We further extend the family with [Cosmos-Transfer2.5], a control-net style framework for Sim2Real and Real2Real world translation. Despite being 3.5$\\times$ smaller than [Cosmos-Transfer1], it delivers higher fidelity and robust long-horizon video generation. Together, these advances establish [Cosmos-Predict2.5] and [Cosmos-Transfer2.5] as versatile tools for scaling embodied intelligence. To accelerate research and deployment in Physical AI, we release source code, pretrained checkpoints, and curated benchmarks under the NVIDIA Open Model License at https://github.com/nvidia-cosmos/cosmos-predict2.5 and https://github.com/nvidia-cosmos/cosmos-transfer2.5. We hope these open resources lower the barrier to adoption and foster innovation in building the next generation of embodied intelligence.",
  "published": "2025-10-28",
  "updated": "2026-02-24",
  "year": "2025",
  "authors": [
   " NVIDIA",
   " :",
   "Arslan Ali",
   "Junjie Bai",
   "Maciej Bala",
   "Yogesh Balaji",
   "Aaron Blakeman",
   "Tiffany Cai",
   "Jiaxin Cao",
   "Tianshi Cao",
   "Elizabeth Cha",
   "Yu-Wei Chao",
   "Prithvijit Chattopadhyay",
   "Mike Chen",
   "Yongxin Chen",
   "Yu Chen",
   "Shuai Cheng",
   "Yin Cui",
   "Jenna Diamond",
   "Yifan Ding",
   "Jiaojiao Fan",
   "Linxi Fan",
   "Liang Feng",
   "Francesco Ferroni",
   "Sanja Fidler",
   "Xiao Fu",
   "Ruiyuan Gao",
   "Yunhao Ge",
   "Jinwei Gu",
   "Aryaman Gupta",
   "Siddharth Gururani",
   "Imad El Hanafi",
   "Ali Hassani",
   "Zekun Hao",
   "Jacob Huffman",
   "Joel Jang",
   "Pooya Jannaty",
   "Jan Kautz",
   "Grace Lam",
   "Xuan Li",
   "Zhaoshuo Li",
   "Maosheng Liao",
   "Chen-Hsuan Lin",
   "Tsung-Yi Lin",
   "Yen-Chen Lin",
   "Huan Ling",
   "Ming-Yu Liu",
   "Xian Liu",
   "Yifan Lu",
   "Alice Luo",
   "Qianli Ma",
   "Hanzi Mao",
   "Kaichun Mo",
   "Seungjun Nah",
   "Yashraj Narang",
   "Abhijeet Panaskar",
   "Lindsey Pavao",
   "Trung Pham",
   "Morteza Ramezanali",
   "Fitsum Reda",
   "Scott Reed",
   "Xuanchi Ren",
   "Haonan Shao",
   "Yue Shen",
   "Stella Shi",
   "Shuran Song",
   "Bartosz Stefaniak",
   "Shangkun Sun",
   "Shitao Tang",
   "Sameena Tasmeen",
   "Lyne Tchapmi",
   "Wei-Cheng Tseng",
   "Jibin Varghese",
   "Andrew Z. Wang",
   "Hao Wang",
   "Haoxiang Wang",
   "Heng Wang",
   "Ting-Chun Wang",
   "Fangyin Wei",
   "Jiashu Xu",
   "Dinghao Yang",
   "Xiaodong Yang",
   "Haotian Ye",
   "Seonghyeon Ye",
   "Xiaohui Zeng",
   "Jing Zhang",
   "Qinsheng Zhang",
   "Kaiwen Zheng",
   "Andrew Zhu",
   "Yuke Zhu"
  ],
  "author_count": 90,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.LG",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 167,
  "influential_citations": 29,
  "tldr": "This work introduces [Cosmos-Predict2.5], the latest generation of the Cosmos World Foundation Models for Physical AI, which unifies Text2World, Image2World, and Video2World generation in a single model and leverages [Cosmos-Reason1], a Physical AI vision-language model, to provide richer text grounding and finer control of world simulation.",
  "doi": "10.48550/arXiv.2511.00062",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Nvidia Arslan Ali",
    "id": "2393035636",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Junjie Bai",
    "id": "2289085383",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "M. Bala",
    "id": "2330191272",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Yogesh Balaji",
    "id": "2310337201",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Aaron Blakeman",
    "id": "2390397153",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Tiffany Cai",
    "id": "2330193931",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Jiaxin Cao",
    "id": "2288916527",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Tianshi Cao",
    "id": "2261282798",
    "h_index": 8,
    "papers": 15
   },
   {
    "name": "Elizabeth Cha",
    "id": "3289717",
    "h_index": 12,
    "papers": 24
   },
   {
    "name": "Yu-Wei Chao",
    "id": "2390395717",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Prithvijit Chattopadhyay",
    "id": "40424000",
    "h_index": 15,
    "papers": 30
   },
   {
    "name": "Mike Chen",
    "id": "2327998508",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Yongxin Chen",
    "id": "2221031822",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Yu Chen",
    "id": "2392209749",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Shuai Cheng",
    "id": "2274105208",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Yin Cui",
    "id": "2299115514",
    "h_index": 10,
    "papers": 17
   },
   {
    "name": "Jenna Diamond",
    "id": "2351057061",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Yifan Ding",
    "id": "2356785654",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Jia-Xin Fan",
    "id": "2367038153",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "L. Fan",
    "id": "2257381161",
    "h_index": 18,
    "papers": 25
   },
   {
    "name": "Liang Feng",
    "id": "2391404071",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Francesco Ferroni",
    "id": "2260336030",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "Sanja Fidler",
    "id": "2261282058",
    "h_index": 20,
    "papers": 41
   },
   {
    "name": "Xiao Fu",
    "id": "2391969274",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Ruiyuan Gao",
    "id": "2366701492",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Yunhao Ge",
    "id": "2299104362",
    "h_index": 11,
    "papers": 20
   },
   {
    "name": "Jinwei Gu",
    "id": "2338980231",
    "h_index": 9,
    "papers": 13
   },
   {
    "name": "Aryaman Gupta",
    "id": "2328251837",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Siddharth Gururani",
    "id": "3454904",
    "h_index": 16,
    "papers": 35
   },
   {
    "name": "Imad El Hanafi",
    "id": "2351050174",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Ali Hassani",
    "id": "2350726194",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Zekun Hao",
    "id": "2313307677",
    "h_index": 8,
    "papers": 11
   },
   {
    "name": "J. Huffman",
    "id": "2298965422",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "J. Jang",
    "id": "2333419032",
    "h_index": 11,
    "papers": 13
   },
   {
    "name": "Pooya Jannaty",
    "id": "2313203519",
    "h_index": 8,
    "papers": 9
   },
   {
    "name": "Jan Kautz",
    "id": "2376331450",
    "h_index": 24,
    "papers": 33
   },
   {
    "name": "Grace Lam",
    "id": "2330190570",
    "h_index": 8,
    "papers": 9
   },
   {
    "name": "Xuan Li",
    "id": "2274429235",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Zhaoshuo Li",
    "id": "2313218585",
    "h_index": 12,
    "papers": 18
   },
   {
    "name": "Maosheng Liao",
    "id": "2390397820",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Chen-Hsuan Lin",
    "id": "2313497801",
    "h_index": 8,
    "papers": 11
   },
   {
    "name": "Tsung-Yi Lin",
    "id": "2300141490",
    "h_index": 15,
    "papers": 22
   },
   {
    "name": "Yen-Chen Lin",
    "id": "2313179579",
    "h_index": 8,
    "papers": 8
   },
   {
    "name": "Huan Ling",
    "id": "18900686",
    "h_index": 28,
    "papers": 44
   },
   {
    "name": "Ming-Yu Liu",
    "id": "2385747822",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "Xian Liu",
    "id": "2323079661",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Yi-Yu Lu",
    "id": "2141567091",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Alice Luo",
    "id": "2330185830",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Qianli Ma",
    "id": "2313310048",
    "h_index": 10,
    "papers": 12
   },
   {
    "name": "Hanzi Mao",
    "id": "2313167739",
    "h_index": 8,
    "papers": 8
   },
   {
    "name": "Kaichun Mo",
    "id": "2261042152",
    "h_index": 8,
    "papers": 14
   },
   {
    "name": "Seungjun Nah",
    "id": "40648435",
    "h_index": 21,
    "papers": 30
   },
   {
    "name": "Yashraj S. Narang",
    "id": "5046361",
    "h_index": 20,
    "papers": 36
   },
   {
    "name": "Abhijeet Panaskar",
    "id": "2390397869",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Lindsey Pavao",
    "id": "2338890844",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "T. Pham",
    "id": "144429629",
    "h_index": 14,
    "papers": 30
   },
   {
    "name": "Morteza Ramezanali",
    "id": "2338890557",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Fitsum Reda",
    "id": "2372721953",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Scott Reed",
    "id": "2344615904",
    "h_index": 8,
    "papers": 9
   },
   {
    "name": "Xuanchi Ren",
    "id": "2271992289",
    "h_index": 12,
    "papers": 23
   },
   {
    "name": "Haonan Shao",
    "id": "2390388135",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Yue Shen",
    "id": "2299728100",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Stella Shi",
    "id": "2330236588",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Shu-Hui Song",
    "id": "2360305532",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Bartosz Stefaniak",
    "id": "2338889798",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Shangkun Sun",
    "id": "2390436164",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Shitao Tang",
    "id": "2338977747",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Sameena Tasmeen",
    "id": "2390394373",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Lyne P. Tchapmi",
    "id": "26917145",
    "h_index": 9,
    "papers": 16
   },
   {
    "name": "Wei-Cheng Tseng",
    "id": "2321873035",
    "h_index": 8,
    "papers": 17
   },
   {
    "name": "J. Varghese",
    "id": "145853825",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Andrew Z. Wang",
    "id": "2351236856",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Hao Wang",
    "id": "2339536289",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Haoxiang Wang",
    "id": "2338962955",
    "h_index": 11,
    "papers": 31
   },
   {
    "name": "Heng-Zhi Wang",
    "id": "2374147184",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Tingjun Wang",
    "id": "2311373240",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Fangyin Wei",
    "id": "2330307301",
    "h_index": 8,
    "papers": 9
   },
   {
    "name": "Jiashu Xu",
    "id": "2301402981",
    "h_index": 11,
    "papers": 25
   },
   {
    "name": "Dinghao Yang",
    "id": "1387901091",
    "h_index": 6,
    "papers": 14
   },
   {
    "name": "Xiaodong Yang",
    "id": "2351209780",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Hao Ye",
    "id": "2335521814",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Seonghyeon Ye",
    "id": "2152111477",
    "h_index": 23,
    "papers": 30
   },
   {
    "name": "Xiaohui Zeng",
    "id": "2299121541",
    "h_index": 10,
    "papers": 14
   },
   {
    "name": "Jing Zhang",
    "id": "2339986899",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Qinsheng Zhang",
    "id": "2288856236",
    "h_index": 14,
    "papers": 17
   },
   {
    "name": "Kaiwen Zheng",
    "id": "1864036526",
    "h_index": 16,
    "papers": 32
   },
   {
    "name": "Andrew Zhu",
    "id": "2390398845",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Yuke Zhu",
    "id": "2258068214",
    "h_index": 13,
    "papers": 21
   }
  ],
  "comment": "",
  "topics": [
   "sim2real",
   "rl-control",
   "foundation-pretraining",
   "video-generation"
  ],
  "orgs": [
   "NVIDIA"
  ],
  "abs_url": "https://arxiv.org/abs/2511.00062v2",
  "pdf_url": "https://arxiv.org/pdf/2511.00062v2",
  "html_url": "https://arxiv.org/html/2511.00062v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.73
 },
 {
  "id": "2510.21571",
  "slug": "scalable-vision-language-action-model-pretraining-for-robotic-manipula",
  "title": "Scalable Vision-Language-Action Model Pretraining for Robotic Manipulation with Real-Life Human Activity Videos",
  "abstract": "This paper presents a novel approach for pretraining robotic manipulation Vision-Language-Action (VLA) models using a large corpus of unscripted real-life video recordings of human hand activities. Treating human hand as dexterous robot end-effector, we show that \"in-the-wild\" egocentric human videos without any annotations can be transformed into data formats fully aligned with existing robotic V-L-A training data in terms of task granularity and labels. This is achieved by the development of a fully-automated holistic human activity analysis approach for arbitrary human hand videos. This approach can generate atomic-level hand activity segments and their language descriptions, each accompanied with framewise 3D hand motion and camera motion. We process a large volume of egocentric videos and create a hand-VLA training dataset containing 1M episodes and 26M frames. This training data covers a wide range of objects and concepts, dexterous manipulation tasks, and environment variations in real life, vastly exceeding the coverage of existing robot data. We design a dexterous hand VLA model architecture and pretrain the model on this dataset. The model exhibits strong zero-shot capabilities on completely unseen real-world observations. Additionally, fine-tuning it on a small amount of real robot action data significantly improves task success rates and generalization to novel objects in real robotic experiments. We also demonstrate the appealing scaling behavior of the model's task performance with respect to pretraining data scale. We believe this work lays a solid foundation for scalable VLA pretraining, advancing robots toward truly generalizable embodied intelligence.",
  "published": "2025-10-24",
  "updated": "2025-10-24",
  "year": "2025",
  "authors": [
   "Qixiu Li",
   "Yu Deng",
   "Yaobo Liang",
   "Lin Luo",
   "Lei Zhou",
   "Chengtang Yao",
   "Lingqi Zeng",
   "Zhiyuan Feng",
   "Huizhi Liang",
   "Sicheng Xu",
   "Yizhong Zhang",
   "Xi Chen",
   "Hao Chen",
   "Lily Sun",
   "Dong Chen",
   "Jiaolong Yang",
   "Baining Guo"
  ],
  "author_count": 17,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 61,
  "influential_citations": 7,
  "tldr": "This paper presents a novel approach for pretraining robotic manipulation Vision-Language-Action (VLA) models using a large corpus of unscripted real-life video recordings of human hand activities, and designs a dexterous hand VLA model architecture and pretrain the model on this dataset.",
  "doi": "10.48550/arXiv.2510.21571",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Qixiu Li",
    "id": "2322728528",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Yu Deng",
    "id": "2387172823",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Yaobo Liang",
    "id": "2333468608",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Lin Luo",
    "id": "2333975449",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Lei Zhou",
    "id": "2333521491",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Chengtang Yao",
    "id": "1739110837",
    "h_index": 7,
    "papers": 15
   },
   {
    "name": "Lingqi Zeng",
    "id": "2269761659",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Zhiyuan Feng",
    "id": "2365226924",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Huizhi Liang",
    "id": "2391890100",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Sicheng Xu",
    "id": "2110510433",
    "h_index": 11,
    "papers": 22
   },
   {
    "name": "Yizhong Zhang",
    "id": "2333252817",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Xi Chen",
    "id": "2333485561",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Hao Chen",
    "id": "2242179023",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Lily Sun",
    "id": "2387876601",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Dong Chen",
    "id": "2333464886",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Jiaolong Yang",
    "id": "2237946707",
    "h_index": 10,
    "papers": 22
   },
   {
    "name": "Baining Guo",
    "id": "2238211403",
    "h_index": 17,
    "papers": 30
   }
  ],
  "comment": "Project page: https://microsoft.github.io/VITRA/",
  "topics": [
   "vla",
   "dexterous-manipulation",
   "egocentric-data",
   "foundation-pretraining",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2510.21571v1",
  "pdf_url": "https://arxiv.org/pdf/2510.21571v1",
  "html_url": "https://arxiv.org/html/2510.21571v1",
  "code_url": "https://microsoft.github.io/VITRA/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.79
 },
 {
  "id": "2510.16240",
  "slug": "cosmos-surg-dvrk-world-foundation-model-based-automated-online-evaluat",
  "title": "Cosmos-Surg-dVRK: World Foundation Model-based Automated Online Evaluation of Surgical Robot Policy Learning",
  "abstract": "The rise of surgical robots and vision-language-action models has accelerated the development of autonomous surgical policies and efficient assessment strategies. However, evaluating these policies directly on physical robotic platforms such as the da Vinci Research Kit (dVRK) remains hindered by high costs, time demands, reproducibility challenges, and variability in execution. World foundation models (WFM) for physical AI offer a transformative approach to simulate complex real-world surgical tasks, such as soft tissue deformation, with high fidelity. This work introduces Cosmos-Surg-dVRK, a surgical finetune of the Cosmos WFM, which, together with a trained video classifier, enables fully automated online evaluation and benchmarking of surgical policies. We evaluate Cosmos-Surg-dVRK using two distinct surgical datasets. On tabletop suture pad tasks, the automated pipeline achieves strong correlation between online rollouts in Cosmos-Surg-dVRK and policy outcomes on the real dVRK Si platform, as well as good agreement between human labelers and the V-JEPA 2-derived video classifier. Additionally, preliminary experiments with ex-vivo porcine cholecystectomy tasks in Cosmos-Surg-dVRK demonstrate promising alignment with real-world evaluations, highlighting the platform's potential for more complex surgical procedures.",
  "published": "2025-10-17",
  "updated": "2025-11-03",
  "year": "2025",
  "authors": [
   "Lukas Zbinden",
   "Nigel Nelson",
   "Juo-Tung Chen",
   "Xinhao Chen",
   "Ji Woong Kim",
   "Mahdi Azizian",
   "Axel Krieger",
   "Sean Huver"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "RA-L",
  "venue_source": "semantic-scholar",
  "citations": 6,
  "influential_citations": 1,
  "tldr": "Cosmos-Surg-dVRK is introduced, a surgical finetune of the Cosmos WFM, which, together with a trained video classifier, enables fully automated online evaluation and benchmarking of surgical policies, highlighting the platform's potential for more complex surgical procedures.",
  "doi": "10.1109/LRA.2026.3675962",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "L. Zbinden",
    "id": "6629817",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Nigel Nelson",
    "id": "2356549198",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Juo-Tung Chen",
    "id": "2361665894",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Xinhao Chen",
    "id": "2307218135",
    "h_index": 3,
    "papers": 14
   },
   {
    "name": "Ji Woong Kim",
    "id": "2277454837",
    "h_index": 8,
    "papers": 22
   },
   {
    "name": "Mahdi Azizian",
    "id": "2300285188",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Axel Krieger",
    "id": "2256991202",
    "h_index": 8,
    "papers": 20
   },
   {
    "name": "Sean Huver",
    "id": "10419594",
    "h_index": 8,
    "papers": 30
   }
  ],
  "comment": "minor metadata and notation fixes; +3 citations",
  "topics": [
   "world-models",
   "vla",
   "foundation-pretraining"
  ],
  "orgs": [
   "NVIDIA"
  ],
  "abs_url": "https://arxiv.org/abs/2510.16240v2",
  "pdf_url": "https://arxiv.org/pdf/2510.16240v2",
  "html_url": "https://arxiv.org/html/2510.16240v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.85
 },
 {
  "id": "2510.15352",
  "slug": "gaussgym-an-open-source-real-to-sim-framework-for-learning-locomotion",
  "title": "GaussGym: An open-source real-to-sim framework for learning locomotion from pixels",
  "abstract": "We present a novel approach for photorealistic robot simulation that integrates 3D Gaussian Splatting as a drop-in renderer within vectorized physics simulators such as IsaacGym. This enables unprecedented speed -- exceeding 100,000 steps per second on consumer GPUs -- while maintaining high visual fidelity, which we showcase across diverse tasks. We additionally demonstrate its applicability in a sim-to-real robotics setting. Beyond depth-based sensing, our results highlight how rich visual semantics improve navigation and decision-making, such as avoiding undesirable regions. We further showcase the ease of incorporating thousands of environments from iPhone scans, large-scale scene datasets (e.g., GrandTour, ARKit), and outputs from generative video models like Veo, enabling rapid creation of realistic training worlds. This work bridges high-throughput simulation and high-fidelity perception, advancing scalable and generalizable robot learning. All code and data will be open-sourced for the community to build upon. Videos, code, and data available at https://escontrela.me/gauss_gym/.",
  "published": "2025-10-17",
  "updated": "2025-10-17",
  "year": "2025",
  "authors": [
   "Alejandro Escontrela",
   "Justin Kerr",
   "Arthur Allshire",
   "Jonas Frey",
   "Rocky Duan",
   "Carmelo Sferrazza",
   "Pieter Abbeel"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.GR"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 29,
  "influential_citations": 1,
  "tldr": "A novel approach for photorealistic robot simulation that integrates 3D Gaussian Splatting as a drop-in renderer within vectorized physics simulators such as IsaacGym is presented, advancing scalable and generalizable robot learning.",
  "doi": "10.48550/arXiv.2510.15352",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Alejandro Escontrela",
    "id": "2008560299",
    "h_index": 11,
    "papers": 17
   },
   {
    "name": "Justin Kerr",
    "id": "2350002804",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Arthur Allshire",
    "id": "2345184559",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Jonas Frey",
    "id": "2249531943",
    "h_index": 10,
    "papers": 29
   },
   {
    "name": "Rocky Duan",
    "id": "2381724728",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Carmelo Sferrazza",
    "id": "47218071",
    "h_index": 21,
    "papers": 44
   },
   {
    "name": "Pieter Abbeel",
    "id": "2381724317",
    "h_index": 9,
    "papers": 17
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "sim2real",
   "spatial-3d",
   "navigation",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2510.15352v1",
  "pdf_url": "https://arxiv.org/pdf/2510.15352v1",
  "html_url": "https://arxiv.org/html/2510.15352v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.48
 },
 {
  "id": "2510.14959",
  "slug": "cbf-rl-safety-filtering-reinforcement-learning-in-training-with-contro",
  "title": "CBF-RL: Safety Filtering Reinforcement Learning in Training with Control Barrier Functions",
  "abstract": "Reinforcement learning (RL), while powerful and expressive, can often prioritize performance at the expense of safety. Yet safety violations can lead to catastrophic outcomes in real-world deployments. Control Barrier Functions (CBFs) offer a principled method to enforce dynamic safety -- traditionally deployed online via safety filters. While the result is safe behavior, the fact that the RL policy does not have knowledge of the CBF can lead to conservative behaviors. This paper proposes CBF-RL, a framework for generating safe behaviors with RL by enforcing CBFs in training. CBF-RL has two key attributes: (1) minimally modifying a nominal RL policy to encode safety constraints via a CBF term, (2) and safety filtering of the policy rollouts in training. Theoretically, we prove that continuous-time safety filters can be deployed via closed-form expressions on discrete-time roll-outs. Practically, we demonstrate that CBF-RL internalizes the safety constraints in the learned policy -- both enforcing safer actions and biasing towards safer rewards -- enabling safe deployment without the need for an online safety filter. We validate our framework through ablation studies on navigation tasks and on the Unitree G1 humanoid robot, where CBF-RL enables safer exploration, faster convergence, and robust performance under uncertainty, enabling the humanoid robot to avoid obstacles and climb stairs safely in real-world settings without a runtime safety filter.",
  "published": "2025-10-16",
  "updated": "2026-06-22",
  "year": "2025",
  "authors": [
   "Lizhi Yang",
   "Blake Werner",
   "Massimiliano de Sa",
   "Aaron D. Ames"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA 2026",
  "venue_source": "arxiv-comment",
  "citations": 18,
  "influential_citations": 1,
  "tldr": "CBF-RL is proposed, a framework for generating safe behaviors with RL by enforcing CBFs in training that internalizes the safety constraints in the learned policy -- both enforcing safer actions and biasing towards safer rewards -- enabling safe deployment without the need for an online safety filter.",
  "doi": "10.48550/arXiv.2510.14959",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Lizhi Yang",
    "id": "2362153740",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Blake Werner",
    "id": "2362086530",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Massimiliano de Sa",
    "id": "2386611670",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Aaron D. Ames",
    "id": "2338277217",
    "h_index": 4,
    "papers": 24
   }
  ],
  "comment": "Accepted to the 2026 IEEE International Conference on Robotics and Automation (ICRA 2026). Copyright transferred to IEEE. Sample code for the navigation example with CBF-RL reward core construction can be found at https://github.com/lzyang2000/cbf-rl-navigation-demo",
  "topics": [
   "humanoids",
   "rl-control",
   "navigation",
   "safety-eval"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2510.14959v6",
  "pdf_url": "https://arxiv.org/pdf/2510.14959v6",
  "html_url": "https://arxiv.org/html/2510.14959v6",
  "code_url": "https://github.com/lzyang2000/cbf-rl-navigation-demo",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.28
 },
 {
  "id": "2510.14955",
  "slug": "realdpo-real-or-not-real-that-is-the-preference",
  "title": "RealDPO: Real or Not Real, that is the Preference",
  "abstract": "Video generative models have recently achieved notable advancements in synthesis quality. However, generating complex motions remains a critical challenge, as existing models often struggle to produce natural, smooth, and contextually consistent movements. This gap between generated and real-world motions limits their practical applicability. To address this issue, we introduce RealDPO, a novel alignment paradigm that leverages real-world data as positive samples for preference learning, enabling more accurate motion synthesis. Unlike traditional supervised fine-tuning (SFT), which offers limited corrective feedback, RealDPO employs Direct Preference Optimization (DPO) with a tailored loss function to enhance motion realism. By contrasting real-world videos with erroneous model outputs, RealDPO enables iterative self-correction, progressively refining motion quality. To support post-training in complex motion synthesis, we propose RealAction-5K, a curated dataset of high-quality videos capturing human daily activities with rich and precise motion details. Extensive experiments demonstrate that RealDPO significantly improves video quality, text alignment, and motion realism compared to state-of-the-art models and existing preference optimization techniques.",
  "published": "2025-10-16",
  "updated": "2025-11-06",
  "year": "2025",
  "authors": [
   "Guo Cheng",
   "Danni Yang",
   "Ziqi Huang",
   "Jianlou Si",
   "Chenyang Si",
   "Ziwei Liu"
  ],
  "author_count": 6,
  "categories": [
   "cs.CV",
   "cs.AI"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 5,
  "influential_citations": 0,
  "tldr": "RealDPO is introduced, a novel alignment paradigm that leverages real-world data as positive samples for preference learning, enabling more accurate motion synthesis and significantly improves video quality, text alignment, and motion realism compared to state-of-the-art models and existing preference optimization techniques.",
  "doi": "10.48550/arXiv.2510.14955",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Guo Cheng",
    "id": "2383915363",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Danni Yang",
    "id": "2376202681",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Ziqi Huang",
    "id": "2243375536",
    "h_index": 13,
    "papers": 28
   },
   {
    "name": "Jianlou Si",
    "id": "2303371732",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Chenyang Si",
    "id": "2243153986",
    "h_index": 14,
    "papers": 23
   },
   {
    "name": "Ziwei Liu",
    "id": "2376170352",
    "h_index": 5,
    "papers": 13
   }
  ],
  "comment": "Code:https://github.com/Vchitect/RealDPO Project Page:https://vchitect.github.io/RealDPO-Project/",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2510.14955v2",
  "pdf_url": "https://arxiv.org/pdf/2510.14955v2",
  "html_url": "https://arxiv.org/html/2510.14955v2",
  "code_url": "https://github.com/Vchitect/RealDPO",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.78
 },
 {
  "id": "2510.14930",
  "slug": "vt-refine-learning-bimanual-assembly-with-visuo-tactile-feedback-via-s",
  "title": "VT-Refine: Learning Bimanual Assembly with Visuo-Tactile Feedback via Simulation Fine-Tuning",
  "abstract": "Humans excel at bimanual assembly tasks by adapting to rich tactile feedback -- a capability that remains difficult to replicate in robots through behavioral cloning alone, due to the suboptimality and limited diversity of human demonstrations. In this work, we present VT-Refine, a visuo-tactile policy learning framework that combines real-world demonstrations, high-fidelity tactile simulation, and reinforcement learning to tackle precise, contact-rich bimanual assembly. We begin by training a diffusion policy on a small set of demonstrations using synchronized visual and tactile inputs. This policy is then transferred to a simulated digital twin equipped with simulated tactile sensors and further refined via large-scale reinforcement learning to enhance robustness and generalization. To enable accurate sim-to-real transfer, we leverage high-resolution piezoresistive tactile sensors that provide normal force signals and can be realistically modeled in parallel using GPU-accelerated simulation. Experimental results show that VT-Refine improves assembly performance in both simulation and the real world by increasing data diversity and enabling more effective policy fine-tuning. Our project page is available at https://binghao-huang.github.io/vt_refine/.",
  "published": "2025-10-16",
  "updated": "2025-10-18",
  "year": "2025",
  "authors": [
   "Binghao Huang",
   "Jie Xu",
   "Iretiayo Akinola",
   "Wei Yang",
   "Balakumar Sundaralingam",
   "Rowland O'Flaherty",
   "Dieter Fox",
   "Xiaolong Wang",
   "Arsalan Mousavian",
   "Yu-Wei Chao",
   "Yunzhu Li"
  ],
  "author_count": 11,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL 2025",
  "venue_source": "arxiv-comment",
  "citations": 25,
  "influential_citations": 2,
  "tldr": "VT-Refine is presented, a visuo-tactile policy learning framework that combines real-world demonstrations, high-fidelity tactile simulation, and reinforcement learning to tackle precise, contact-rich bimanual assembly.",
  "doi": "10.48550/arXiv.2510.14930",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Binghao Huang",
    "id": "2287019710",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Jie Xu",
    "id": "2273556820",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Iretiayo Akinola",
    "id": "2856639",
    "h_index": 19,
    "papers": 38
   },
   {
    "name": "Wei Yang",
    "id": "2266191697",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Balakumar Sundaralingam",
    "id": "32469503",
    "h_index": 26,
    "papers": 47
   },
   {
    "name": "Rowland O\u2019Flaherty",
    "id": "2342857542",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Dieter Fox",
    "id": "2294877174",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Xiaolong Wang",
    "id": "2239141122",
    "h_index": 12,
    "papers": 13
   },
   {
    "name": "A. Mousavian",
    "id": "3040583",
    "h_index": 36,
    "papers": 57
   },
   {
    "name": "Yu-Wei Chao",
    "id": "2306062337",
    "h_index": 7,
    "papers": 12
   },
   {
    "name": "Yunzhu Li",
    "id": "2374457025",
    "h_index": 4,
    "papers": 9
   }
  ],
  "comment": "Accepted by 9th Conference on Robot Learning (CoRL 2025); Website: https://binghao-huang.github.io/vt_refine/",
  "topics": [
   "egocentric-data",
   "tactile",
   "sim2real",
   "imitation-diffusion",
   "rl-control",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2510.14930v2",
  "pdf_url": "https://arxiv.org/pdf/2510.14930v2",
  "html_url": "https://arxiv.org/html/2510.14930v2",
  "code_url": "https://binghao-huang.github.io/vt_refine/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.91
 },
 {
  "id": "2510.13626",
  "slug": "libero-plus-in-depth-robustness-analysis-of-vision-language-action-mod",
  "title": "LIBERO-Plus: In-depth Robustness Analysis of Vision-Language-Action Models",
  "abstract": "Visual-Language-Action (VLA) models report impressive success rates on robotic manipulation benchmarks, yet these results may mask fundamental weaknesses in robustness. We perform a systematic vulnerability analysis by introducing controlled perturbations across seven dimensions: objects layout, camera viewpoints, robot initial states, language instructions, light conditions, background textures and sensor noise. We comprehensively analyzed multiple state-of-the-art models and revealed consistent brittleness beneath apparent competence. Our analysis exposes critical weaknesses: models exhibit extreme sensitivity to perturbation factors, including camera viewpoints and robot initial states, with performance dropping from 95% to below 30% under modest perturbations. Surprisingly, models are largely insensitive to language variations, with further experiments revealing that models tend to ignore language instructions completely. Our findings challenge the assumption that high benchmark scores equate to true competency and highlight the need for evaluation practices that assess reliability under realistic variation.",
  "published": "2025-10-15",
  "updated": "2025-12-26",
  "year": "2025",
  "authors": [
   "Senyu Fei",
   "Siyin Wang",
   "Junhao Shi",
   "Zihao Dai",
   "Jikun Cai",
   "Pengfang Qian",
   "Li Ji",
   "Xinzhe He",
   "Shiduo Zhang",
   "Zhaoye Fei",
   "Jinlan Fu",
   "Jingjing Gong",
   "Xipeng Qiu"
  ],
  "author_count": 13,
  "categories": [
   "cs.RO",
   "cs.CL",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 209,
  "influential_citations": 62,
  "tldr": "This work comprehensively analyzed multiple state-of-the-art VLA models and revealed consistent brittleness beneath apparent competence, challenging the assumption that high benchmark scores equate to true competency and highlighting the need for evaluation practices that assess reliability under realistic variation.",
  "doi": "10.48550/arXiv.2510.13626",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Senyu Fei",
    "id": "2385785349",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Siyin Wang",
    "id": "2182224120",
    "h_index": 9,
    "papers": 24
   },
   {
    "name": "Junhao Shi",
    "id": "2348888995",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Z. G. Dai",
    "id": "2264773695",
    "h_index": 4,
    "papers": 22
   },
   {
    "name": "Jikun Cai",
    "id": "2387222800",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Pengfang Qian",
    "id": "2385785398",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Li Ji",
    "id": "2371306698",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Xinzhe He",
    "id": "2386416920",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Shiduo Zhang",
    "id": "2302704871",
    "h_index": 8,
    "papers": 14
   },
   {
    "name": "Zhaoye Fei",
    "id": "2132200788",
    "h_index": 12,
    "papers": 40
   },
   {
    "name": "Jinlan Fu",
    "id": "2243662006",
    "h_index": 9,
    "papers": 17
   },
   {
    "name": "Jingjing Gong",
    "id": "2371292918",
    "h_index": 6,
    "papers": 20
   },
   {
    "name": "Xipeng Qiu",
    "id": "2350155100",
    "h_index": 9,
    "papers": 20
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2510.13626v3",
  "pdf_url": "https://arxiv.org/pdf/2510.13626v3",
  "html_url": "https://arxiv.org/html/2510.13626v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.32
 },
 {
  "id": "2510.10274",
  "slug": "x-vla-soft-prompted-transformer-as-scalable-cross-embodiment-vision-la",
  "title": "X-VLA: Soft-Prompted Transformer as Scalable Cross-Embodiment Vision-Language-Action Model",
  "abstract": "Successful generalist Vision-Language-Action (VLA) models rely on effective training across diverse robotic platforms with large-scale, cross-embodiment, heterogeneous datasets. To facilitate and leverage the heterogeneity in rich, diverse robotic data sources, we propose a novel Soft Prompt approach with minimally added parameters, by infusing prompt learning concepts into cross-embodiment robot learning and introducing separate sets of learnable embeddings for each distinct data source. These embeddings serve as embodiment-specific prompts, which in unity empower VLA models with effective exploitation of varying cross-embodiment features. Our new X-VLA, a neat flow-matching-based VLA architecture, relies exclusively on soft-prompted standard Transformer encoders, enjoying both scalability and simplicity. Evaluated across 6 simulations as well as 3 real-world robots, our 0.9B instantiation-X-VLA-0.9B simultaneously achieves SOTA performance over a sweep of benchmarks, demonstrating superior results on a wide axes of capabilities, from flexible dexterity to quick adaptation across embodiments, environments, and tasks. Website: https://thu-air-dream.github.io/X-VLA/",
  "published": "2025-10-11",
  "updated": "2025-10-11",
  "year": "2025",
  "authors": [
   "Jinliang Zheng",
   "Jianxiong Li",
   "Zhihao Wang",
   "Dongxiu Liu",
   "Xirui Kang",
   "Yuchun Feng",
   "Yinan Zheng",
   "Jiayin Zou",
   "Yilun Chen",
   "Jia Zeng",
   "Ya-Qin Zhang",
   "Jiangmiao Pang",
   "Jingjing Liu",
   "Tai Wang",
   "Xianyuan Zhan"
  ],
  "author_count": 15,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 246,
  "influential_citations": 31,
  "tldr": "This work proposes a novel Soft Prompt approach with minimally added parameters, by infusing prompt learning concepts into cross-embodiment robot learning and introducing separate sets of learnable embeddings for each distinct data source.",
  "doi": "10.48550/arXiv.2510.10274",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jinliang Zheng",
    "id": "2112524681",
    "h_index": 10,
    "papers": 22
   },
   {
    "name": "Jianxiong Li",
    "id": "2136086524",
    "h_index": 14,
    "papers": 30
   },
   {
    "name": "Zhihao Wang",
    "id": "2313032843",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Dongxiu Liu",
    "id": "2340937401",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Xirui Kang",
    "id": "2385783820",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Yuchun Feng",
    "id": "2385799672",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Yinan Zheng",
    "id": "2224472670",
    "h_index": 12,
    "papers": 22
   },
   {
    "name": "Jiayi Zou",
    "id": "2350341310",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Yilun Chen",
    "id": "2236662733",
    "h_index": 30,
    "papers": 75
   },
   {
    "name": "Jia Zeng",
    "id": "2337356727",
    "h_index": 10,
    "papers": 24
   },
   {
    "name": "Ya-Qin Zhang",
    "id": "2383105350",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Jiangmiao Pang",
    "id": "2377561990",
    "h_index": 10,
    "papers": 28
   },
   {
    "name": "Jingjing Liu",
    "id": "2108435258",
    "h_index": 12,
    "papers": 19
   },
   {
    "name": "Tai Wang",
    "id": "2359108866",
    "h_index": 11,
    "papers": 26
   },
   {
    "name": "Xianyuan Zhan",
    "id": "2242851906",
    "h_index": 21,
    "papers": 55
   }
  ],
  "comment": "preprint, technical report, 33 pages",
  "topics": [
   "vla",
   "dexterous-manipulation",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2510.10274v1",
  "pdf_url": "https://arxiv.org/pdf/2510.10274v1",
  "html_url": "https://arxiv.org/html/2510.10274v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.39
 },
 {
  "id": "2510.10125",
  "slug": "ctrl-world-a-controllable-generative-world-model-for-robot-manipulatio",
  "title": "Ctrl-World: A Controllable Generative World Model for Robot Manipulation",
  "abstract": "Generalist robot policies can now perform a wide range of manipulation skills, but evaluating and improving their ability with unfamiliar objects and instructions remains a significant challenge. Rigorous evaluation requires a large number of real-world rollouts, while systematic improvement demands additional corrective data with expert labels. Both of these processes are slow, costly, and difficult to scale. World models offer a promising, scalable alternative by enabling policies to rollout within imagination space. However, a key challenge is building a controllable world model that can handle multi-step interactions with generalist robot policies. This requires a world model compatible with modern generalist policies by supporting multi-view prediction, fine-grained action control, and consistent long-horizon interactions, which is not achieved by previous works. In this paper, we make a step forward by introducing a controllable multi-view world model that can be used to evaluate and improve the instruction-following ability of generalist robot policies. Our model maintains long-horizon consistency with a pose-conditioned memory retrieval mechanism and achieves precise action control through frame-level action conditioning. Trained on the DROID dataset (95k trajectories, 564 scenes), our model generates spatially and temporally consistent trajectories under novel scenarios and new camera placements for over 20 seconds. We show that our method can accurately rank policy performance without real-world robot rollouts. Moreover, by synthesizing successful trajectories in imagination and using them for supervised fine-tuning, our approach can improve policy success by 44.7\\%.",
  "published": "2025-10-11",
  "updated": "2026-03-01",
  "year": "2025",
  "authors": [
   "Yanjiang Guo",
   "Lucy Xiaoyang Shi",
   "Jianyu Chen",
   "Chelsea Finn"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 135,
  "influential_citations": 31,
  "tldr": "A controllable multi-view world model that can be used to evaluate and improve the instruction-following ability of generalist robot policies is introduced and it is shown that the method can accurately rank policy performance without real-world robot rollouts.",
  "doi": "10.48550/arXiv.2510.10125",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yanjiang Guo",
    "id": "2181339548",
    "h_index": 12,
    "papers": 26
   },
   {
    "name": "L. Shi",
    "id": "2292341452",
    "h_index": 12,
    "papers": 17
   },
   {
    "name": "Jianyu Chen",
    "id": "2280257464",
    "h_index": 10,
    "papers": 20
   },
   {
    "name": "Chelsea Finn",
    "id": "2387218305",
    "h_index": 3,
    "papers": 5
   }
  ],
  "comment": "17 pages",
  "topics": [
   "world-models"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2510.10125v3",
  "pdf_url": "https://arxiv.org/pdf/2510.10125v3",
  "html_url": "https://arxiv.org/html/2510.10125v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.13
 },
 {
  "id": "2510.08568",
  "slug": "novaflow-zero-shot-manipulation-via-actionable-flow-from-generated-vid",
  "title": "NovaFlow: Zero-Shot Manipulation via Actionable Flow from Generated Videos",
  "abstract": "Enabling robots to execute novel manipulation tasks zero-shot is a central goal in robotics. Most existing methods assume in-distribution tasks or rely on fine-tuning with embodiment-matched data, limiting transfer across platforms. We present NovaFlow, an autonomous manipulation framework that converts a task description into an actionable plan for a target robot without any demonstrations. Given a task description, NovaFlow synthesizes a video using a video generation model and distills it into 3D actionable object flow using off-the-shelf perception modules. From the object flow, it computes relative poses for rigid objects and realizes them as robot actions via grasp proposals and trajectory optimization. For deformable objects, this flow serves as a tracking objective for model-based planning with a particle-based dynamics model. By decoupling task understanding from low-level control, NovaFlow naturally transfers across embodiments. We validate on rigid, articulated, and deformable object manipulation tasks using a table-top Franka arm and a Spot quadrupedal mobile robot, and achieve effective zero-shot execution without demonstrations or embodiment-specific training. Project website: https://novaflow.lhy.xyz/.",
  "published": "2025-10-09",
  "updated": "2025-10-09",
  "year": "2025",
  "authors": [
   "Hongyu Li",
   "Lingfeng Sun",
   "Yafei Hu",
   "Duy Ta",
   "Jennifer Barry",
   "George Konidaris",
   "Jiahui Fu"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 49,
  "influential_citations": 3,
  "tldr": "NovaFlow is presented, an autonomous manipulation framework that converts a task description into an actionable plan for a target robot without any demonstrations, and naturally transfers across platforms by decoupling task understanding from low-level control.",
  "doi": "10.48550/arXiv.2510.08568",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hongyu Li",
    "id": "2155109477",
    "h_index": 6,
    "papers": 14
   },
   {
    "name": "Lingfeng Sun",
    "id": "2384856067",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Yafei Hu",
    "id": "2384958063",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "D. Ta",
    "id": "2290009852",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Jennifer L. Barry",
    "id": "144256605",
    "h_index": 8,
    "papers": 14
   },
   {
    "name": "G. Konidaris",
    "id": "1765407",
    "h_index": 49,
    "papers": 269
   },
   {
    "name": "Jiahui Fu",
    "id": "2293908143",
    "h_index": 4,
    "papers": 9
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "humanoids",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2510.08568v1",
  "pdf_url": "https://arxiv.org/pdf/2510.08568v1",
  "html_url": "https://arxiv.org/html/2510.08568v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.7
 },
 {
  "id": "2510.06208",
  "slug": "shapegen4d-towards-high-quality-4d-shape-generation-from-videos",
  "title": "ShapeGen4D: Towards High Quality 4D Shape Generation from Videos",
  "abstract": "Video-conditioned 4D shape generation aims to recover time-varying 3D geometry and view-consistent appearance directly from an input video. In this work, we introduce a native video-to-4D shape generation framework that synthesizes a single dynamic 3D representation end-to-end from the video. Our framework introduces three key components based on large-scale pre-trained 3D models: (i) a temporal attention that conditions generation on all frames while producing a time-indexed dynamic representation; (ii) a time-aware point sampling and 4D latent anchoring that promote temporally consistent geometry and texture; and (iii) noise sharing across frames to enhance temporal stability. Our method accurately captures non-rigid motion, volume changes, and even topological transitions without per-frame optimization. Across diverse in-the-wild videos, our method improves robustness and perceptual fidelity and reduces failure modes compared with the baselines.",
  "published": "2025-10-07",
  "updated": "2025-10-07",
  "year": "2025",
  "authors": [
   "Jiraphon Yenphraphai",
   "Ashkan Mirzaei",
   "Jianqi Chen",
   "Jiaxu Zou",
   "Sergey Tulyakov",
   "Raymond A. Yeh",
   "Peter Wonka",
   "Chaoyang Wang"
  ],
  "author_count": 8,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 19,
  "influential_citations": 4,
  "tldr": "This work introduces a native video-to-4D shape generation framework that synthesizes a single dynamic 3D representation end-to-end from the video that improves robustness and perceptual fidelity and reduces failure modes compared with the baselines.",
  "doi": "10.48550/arXiv.2510.06208",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jiraphon Yenphraphai",
    "id": "2052302061",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Ashkan Mirzaei",
    "id": "2174737496",
    "h_index": 13,
    "papers": 25
   },
   {
    "name": "Jianqi Chen",
    "id": "2349935733",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Jiaxu Zou",
    "id": "2305601075",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Sergey Tulyakov",
    "id": "2292401534",
    "h_index": 17,
    "papers": 63
   },
   {
    "name": "Raymond A. Yeh",
    "id": "2311435672",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Peter Wonka",
    "id": "2262444458",
    "h_index": 14,
    "papers": 82
   },
   {
    "name": "Chaoyang Wang",
    "id": "50097023",
    "h_index": 20,
    "papers": 82
   }
  ],
  "comment": "Project page: https://shapegen4d.github.io/",
  "topics": [
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2510.06208v1",
  "pdf_url": "https://arxiv.org/pdf/2510.06208v1",
  "html_url": "https://arxiv.org/html/2510.06208v1",
  "code_url": "https://shapegen4d.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.3
 },
 {
  "id": "2510.05070",
  "slug": "resmimic-from-general-motion-tracking-to-humanoid-whole-body-loco-mani",
  "title": "ResMimic: From General Motion Tracking to Humanoid Whole-body Loco-Manipulation via Residual Learning",
  "abstract": "Humanoid whole-body loco-manipulation promises transformative capabilities for daily service and warehouse tasks. While recent advances in general motion tracking (GMT) have enabled humanoids to reproduce diverse human motions, these policies lack the precision and object awareness required for loco-manipulation. To this end, we introduce ResMimic, a two-stage residual learning framework for precise and expressive humanoid control from human motion data. First, a GMT policy, trained on large-scale human-only motion, serves as a task-agnostic base for generating human-like whole-body movements. An efficient but precise residual policy is then learned to refine the GMT outputs to improve locomotion and incorporate object interaction. To further facilitate efficient training, we design (i) a point-cloud-based object tracking reward for smoother optimization, (ii) a contact reward that encourages accurate humanoid body-object interactions, and (iii) a curriculum-based virtual object controller to stabilize early training. We evaluate ResMimic in both simulation and on a real Unitree G1 humanoid. Results show substantial gains in task success, training efficiency, and robustness over strong baselines. Videos are available at https://resmimic.github.io/ .",
  "published": "2025-10-06",
  "updated": "2025-10-08",
  "year": "2025",
  "authors": [
   "Siheng Zhao",
   "Yanjie Ze",
   "Yue Wang",
   "C. Karen Liu",
   "Pieter Abbeel",
   "Guanya Shi",
   "Rocky Duan"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 55,
  "influential_citations": 3,
  "tldr": "ResMimic, a two-stage residual learning framework for precise and expressive humanoid control from human motion data, is introduced and results show substantial gains in task success, training efficiency, and robustness over strong baselines.",
  "doi": "10.48550/arXiv.2510.05070",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Siheng Zhao",
    "id": "2243033718",
    "h_index": 12,
    "papers": 19
   },
   {
    "name": "Yanjie Ze",
    "id": "2325901084",
    "h_index": 10,
    "papers": 13
   },
   {
    "name": "Yue Wang",
    "id": "2385729806",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "C. K. Liu",
    "id": "2376138723",
    "h_index": 9,
    "papers": 11
   },
   {
    "name": "Pieter Abbeel",
    "id": "2381724317",
    "h_index": 9,
    "papers": 17
   },
   {
    "name": "Guanya Shi",
    "id": "2384824402",
    "h_index": 9,
    "papers": 24
   },
   {
    "name": "Rocky Duan",
    "id": "2381724728",
    "h_index": 7,
    "papers": 13
   }
  ],
  "comment": "9 pages, 8 figures",
  "topics": [
   "humanoids",
   "safety-eval"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2510.05070v2",
  "pdf_url": "https://arxiv.org/pdf/2510.05070v2",
  "html_url": "https://arxiv.org/html/2510.05070v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.25
 },
 {
  "id": "2510.05057",
  "slug": "stamo-unsupervised-learning-of-generalizable-robot-motion-from-compact",
  "title": "StaMo: Unsupervised Learning of Generalizable Robot Motion from Compact State Representation",
  "abstract": "A fundamental challenge in embodied intelligence is developing expressive and compact state representations for efficient world modeling and decision making. However, existing methods often fail to achieve this balance, yielding representations that are either overly redundant or lacking in task-critical information. We propose an unsupervised approach that learns a highly compressed two-token state representation using a lightweight encoder and a pre-trained Diffusion Transformer (DiT) decoder, capitalizing on its strong generative prior. Our representation is efficient, interpretable, and integrates seamlessly into existing VLA-based models, improving performance by 14.3% on LIBERO and 30% in real-world task success with minimal inference overhead. More importantly, we find that the difference between these tokens, obtained via latent interpolation, naturally serves as a highly effective latent action, which can be further decoded into executable robot actions. This emergent capability reveals that our representation captures structured dynamics without explicit supervision. We name our method StaMo for its ability to learn generalizable robotic Motion from compact State representation, which is encoded from static images, challenging the prevalent dependence to learning latent action on complex architectures and video data. The resulting latent actions also enhance policy co-training, outperforming prior methods by 10.4% with improved interpretability. Moreover, our approach scales effectively across diverse data sources, including real-world robot data, simulation, and human egocentric video.",
  "published": "2025-10-06",
  "updated": "2026-04-12",
  "year": "2025",
  "authors": [
   "Mingyu Liu",
   "Jiuhe Shu",
   "Hui Chen",
   "Zeju Li",
   "Canyu Zhao",
   "Jiange Yang",
   "Shenyuan Gao",
   "Hao Chen",
   "Chunhua Shen"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 12,
  "influential_citations": 1,
  "tldr": "An unsupervised approach that learns a highly compressed two-token state representation using a lightweight encoder and a pre-trained Diffusion Transformer decoder, capitalizing on its strong generative prior, which captures structured dynamics without explicit supervision.",
  "doi": "10.48550/arXiv.2510.05057",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Mingyu Liu",
    "id": "2290717723",
    "h_index": 9,
    "papers": 19
   },
   {
    "name": "Jiuhe Shu",
    "id": "2384125266",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Hui Chen",
    "id": "2384707755",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Zeju Li",
    "id": "2375796515",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Canyu Zhao",
    "id": "2267430102",
    "h_index": 7,
    "papers": 19
   },
   {
    "name": "Jiange Yang",
    "id": "2220590629",
    "h_index": 9,
    "papers": 12
   },
   {
    "name": "Shenyuan Gao",
    "id": "2350220976",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Hao Chen",
    "id": "2363709527",
    "h_index": 6,
    "papers": 21
   },
   {
    "name": "Chunhua Shen",
    "id": "2257324242",
    "h_index": 19,
    "papers": 79
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "egocentric-data"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2510.05057v2",
  "pdf_url": "https://arxiv.org/pdf/2510.05057v2",
  "html_url": "https://arxiv.org/html/2510.05057v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.11
 },
 {
  "id": "2510.03706",
  "slug": "embodiswap-for-zero-shot-robot-imitation-learning",
  "title": "EmbodiSwap for Zero-Shot Robot Imitation Learning",
  "abstract": "We introduce EmbodiSwap - a method for producing photorealistic synthetic robot overlays over human video. We employ EmbodiSwap for zero-shot imitation learning, bridging the embodiment gap between in-the-wild ego-centric human video and a target robot embodiment. We train a closed-loop robot manipulation policy over the data produced by EmbodiSwap. We make novel use of V-JEPA as a visual backbone, repurposing V-JEPA from the domain of video understanding to imitation learning over synthetic robot videos. Adoption of V-JEPA outperforms alternative vision backbones more conventionally used within robotics. In real-world tests, our zero-shot trained V-JEPA model achieves an $82\\%$ success rate, outperforming a few-shot trained $\u03c0_0$ network as well as $\u03c0_0$ trained over data produced by EmbodiSwap. We release (i) code for generating the synthetic robot overlays which takes as input human videos and an arbitrary robot URDF and generates a robot dataset, (ii) the robot dataset we synthesize over EPIC-Kitchens, HOI4D and Ego4D, and (iii) model checkpoints and inference code, to facilitate reproducible research and broader adoption.",
  "published": "2025-10-04",
  "updated": "2025-10-04",
  "year": "2025",
  "authors": [
   "Eadom Dessalene",
   "Pavan Mantripragada",
   "Michael Maynord",
   "Yiannis Aloimonos"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 4,
  "influential_citations": 0,
  "tldr": "This work employs EmbodiSwap for zero-shot imitation learning, bridging the embodiment gap between in-the-wild ego-centric human video and a target robot embodiment, and trains a closed-loop robot manipulation policy over the data produced by EmbodiSwap.",
  "doi": "10.48550/arXiv.2510.03706",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Eadom Dessalene",
    "id": "51230038",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Pavan Mantripragada",
    "id": "2127392338",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Michael Maynord",
    "id": "2739186",
    "h_index": 6,
    "papers": 18
   },
   {
    "name": "Y. Aloimonos",
    "id": "1697493",
    "h_index": 55,
    "papers": 418
   }
  ],
  "comment": "Video link: https://drive.google.com/file/d/1UccngwgPqUwPMhBja7JrXfZoTquCx_Qe/view?usp=sharing",
  "topics": [
   "world-models",
   "egocentric-data",
   "imitation-diffusion"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2510.03706v1",
  "pdf_url": "https://arxiv.org/pdf/2510.03706v1",
  "html_url": "https://arxiv.org/html/2510.03706v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.7
 },
 {
  "id": "2510.02283",
  "slug": "self-forcing-towards-minute-scale-high-quality-video-generation",
  "title": "Self-Forcing++: Towards Minute-Scale High-Quality Video Generation",
  "abstract": "Diffusion models have revolutionized image and video generation, achieving unprecedented visual quality. However, their reliance on transformer architectures incurs prohibitively high computational costs, particularly when extending generation to long videos. Recent work has explored autoregressive formulations for long video generation, typically by distilling from short-horizon bidirectional teachers. Nevertheless, given that teacher models cannot synthesize long videos, the extrapolation of student models beyond their training horizon often leads to pronounced quality degradation, arising from the compounding of errors within the continuous latent space. In this paper, we propose a simple yet effective approach to mitigate quality degradation in long-horizon video generation without requiring supervision from long-video teachers or retraining on long video datasets. Our approach centers on exploiting the rich knowledge of teacher models to provide guidance for the student model through sampled segments drawn from self-generated long videos. Our method maintains temporal consistency while scaling video length by up to 20x beyond teacher's capability, avoiding common issues such as over-exposure and error-accumulation without recomputing overlapping frames like previous methods. When scaling up the computation, our method shows the capability of generating videos up to 4 minutes and 15 seconds, equivalent to 99.9% of the maximum span supported by our base model's position embedding and more than 50x longer than that of our baseline model. Experiments on standard benchmarks and our proposed improved benchmark demonstrate that our approach substantially outperforms baseline methods in both fidelity and consistency. Our long-horizon videos demo can be found at https://self-forcing-plus-plus.github.io/",
  "published": "2025-10-02",
  "updated": "2025-10-02",
  "year": "2025",
  "authors": [
   "Justin Cui",
   "Jie Wu",
   "Ming Li",
   "Tao Yang",
   "Xiaojie Li",
   "Rui Wang",
   "Andrew Bai",
   "Yuanhao Ban",
   "Cho-Jui Hsieh"
  ],
  "author_count": 9,
  "categories": [
   "cs.CV",
   "cs.AI"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 164,
  "influential_citations": 19,
  "tldr": "This paper proposes a simple yet effective approach to mitigate quality degradation in long-horizon video generation without requiring supervision from long-video teachers or retraining on long video datasets, and substantially outperforms baseline methods in both fidelity and consistency.",
  "doi": "10.48550/arXiv.2510.02283",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Justin Cui",
    "id": "2179185249",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Jie Wu",
    "id": "2389783206",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Ming Li",
    "id": "2383497310",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Tao Yang",
    "id": "2334826513",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Xiaojie Li",
    "id": "2355128267",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Rui Wang",
    "id": "2383422629",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Andrew Bai",
    "id": "2279749940",
    "h_index": 5,
    "papers": 15
   },
   {
    "name": "Yuanhao Ban",
    "id": "2304646858",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Cho-Jui Hsieh",
    "id": "2279751532",
    "h_index": 6,
    "papers": 17
   }
  ],
  "comment": "preprint",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2510.02283v1",
  "pdf_url": "https://arxiv.org/pdf/2510.02283v1",
  "html_url": "https://arxiv.org/html/2510.02283v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.22
 },
 {
  "id": "2510.02252",
  "slug": "retargeting-matters-general-motion-retargeting-for-humanoid-motion-tra",
  "title": "Retargeting Matters: General Motion Retargeting for Humanoid Motion Tracking",
  "abstract": "Humanoid motion tracking policies are central to building teleoperation pipelines and hierarchical controllers, yet they face a fundamental challenge: the embodiment gap between humans and humanoid robots. Current approaches address this gap by retargeting human motion data to humanoid embodiments and then training reinforcement learning (RL) policies to imitate these reference trajectories. However, artifacts introduced during retargeting, such as foot sliding, self-penetration, and physically infeasible motion are often left in the reference trajectories for the RL policy to correct. While prior work has demonstrated motion tracking abilities, they often require extensive reward engineering and domain randomization to succeed. In this paper, we systematically evaluate how retargeting quality affects policy performance when excessive reward tuning is suppressed. To address issues that we identify with existing retargeting methods, we propose a new retargeting method, General Motion Retargeting (GMR). We evaluate GMR alongside two open-source retargeters, PHC and ProtoMotions, as well as with a high-quality closed-source dataset from Unitree. Using BeyondMimic for policy training, we isolate retargeting effects without reward tuning. Our experiments on a diverse subset of the LAFAN1 dataset reveal that while most motions can be tracked, artifacts in retargeted data significantly reduce policy robustness, particularly for dynamic or long sequences. GMR consistently outperforms existing open-source methods in both tracking performance and faithfulness to the source motion, achieving perceptual fidelity and policy success rates close to the closed-source baseline. Website: https://jaraujo98.github.io/retargeting_matters. Code: https://github.com/YanjieZe/GMR.",
  "published": "2025-10-02",
  "updated": "2025-10-02",
  "year": "2025",
  "authors": [
   "Joao Pedro Araujo",
   "Yanjie Ze",
   "Pei Xu",
   "Jiajun Wu",
   "C. Karen Liu"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 110,
  "influential_citations": 12,
  "tldr": "This paper systematically evaluates how retargeting quality affects policy performance when excessive reward tuning is suppressed, and proposes a new retargeting method, General Motion Retargeting (GMR), which consistently outperforms existing open-source methods in both tracking performance and faithfulness to the source motion.",
  "doi": "10.48550/arXiv.2510.02252",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Joao Pedro Araujo",
    "id": "2198552611",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Yanjie Ze",
    "id": "2325901084",
    "h_index": 10,
    "papers": 13
   },
   {
    "name": "Pei Xu",
    "id": "2335601372",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Jiajun Wu",
    "id": "2326444500",
    "h_index": 35,
    "papers": 214
   },
   {
    "name": "C. K. Liu",
    "id": "2376138723",
    "h_index": 9,
    "papers": 11
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "sim2real",
   "rl-control",
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2510.02252v1",
  "pdf_url": "https://arxiv.org/pdf/2510.02252v1",
  "html_url": "https://arxiv.org/html/2510.02252v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.55
 },
 {
  "id": "2510.01433",
  "slug": "afford2act-affordance-guided-automatic-keypoint-selection-for-generali",
  "title": "AFFORD2ACT: Affordance-Guided Automatic Keypoint Selection for Generalizable and Lightweight Robotic Manipulation",
  "abstract": "Vision-based robot learning often relies on dense image or point-cloud inputs, which are computationally heavy and entangle irrelevant background features. Existing keypoint-based approaches can focus on manipulation-centric features and be lightweight, but either depend on manual heuristics or task-coupled selection, limiting scalability and semantic understanding. To address this, we propose AFFORD2ACT, an affordance-guided framework that distills a minimal set of semantic 2D keypoints from a text prompt and a single image. AFFORD2ACT follows a three-stage pipeline: affordance filtering, category-level keypoint construction, and transformer-based policy learning with embedded gating to reason about the most relevant keypoints, yielding a compact 38-dimensional state policy that can be trained in 15 minutes, which performs well in real-time without proprioception or dense representations. Across diverse real-world manipulation tasks, AFFORD2ACT consistently improves data efficiency, achieving an 82% success rate on unseen objects, novel categories, backgrounds, and distractors.",
  "published": "2025-10-01",
  "updated": "2026-04-15",
  "year": "2025",
  "authors": [
   "Anukriti Singh",
   "Kasra Torshizi",
   "Khuzema Habib",
   "Kelin Yu",
   "Ruohan Gao",
   "Pratap Tokekar"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 4,
  "influential_citations": 0,
  "tldr": "An affordance-guided framework that distills a minimal set of semantic 2D keypoints from a text prompt and a single image, yielding a compact 38-dimensional state policy that performs well in real-time without proprioception or dense representations.",
  "doi": "10.48550/arXiv.2510.01433",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Anukriti Singh",
    "id": "2272901800",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Kasra Torshizi",
    "id": "2190043473",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Khuzema Habib",
    "id": "2382926469",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Kelin Yu",
    "id": "2376516296",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Ruohan Gao",
    "id": "2376472371",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Pratap Tokekar",
    "id": "2390456",
    "h_index": 31,
    "papers": 192
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2510.01433v2",
  "pdf_url": "https://arxiv.org/pdf/2510.01433v2",
  "html_url": "https://arxiv.org/html/2510.01433v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.7
 },
 {
  "id": "2510.01068",
  "slug": "compose-your-policies-improving-diffusion-based-or-flow-based-robot-po",
  "title": "Compose Your Policies! Improving Diffusion-based or Flow-based Robot Policies via Test-time Distribution-level Composition",
  "abstract": "Diffusion-based models for robotic control, including vision-language-action (VLA) and vision-action (VA) policies, have demonstrated significant capabilities. Yet their advancement is constrained by the high cost of acquiring large-scale interaction datasets. This work introduces an alternative paradigm for enhancing policy performance without additional model training. Perhaps surprisingly, we demonstrate that the composed policies can exceed the performance of either parent policy. Our contribution is threefold. First, we establish a theoretical foundation showing that the convex composition of distributional scores from multiple diffusion models can yield a superior one-step functional objective compared to any individual score. A Gr\u00f6nwall-type bound is then used to show that this single-step improvement propagates through entire generation trajectories, leading to systemic performance gains. Second, motivated by these results, we propose General Policy Composition (GPC), a training-free method that enhances performance by combining the distributional scores of multiple pre-trained policies via a convex combination and test-time search. GPC is versatile, allowing for the plug-and-play composition of heterogeneous policies, including VA and VLA models, as well as those based on diffusion or flow-matching, irrespective of their input visual modalities. Third, we provide extensive empirical validation. Experiments on Robomimic, PushT, and RoboTwin benchmarks, alongside real-world robotic evaluations, confirm that GPC consistently improves performance and adaptability across a diverse set of tasks. Further analysis of alternative composition operators and weighting strategies offers insights into the mechanisms underlying the success of GPC. These results establish GPC as a simple yet effective method for improving control performance by leveraging existing policies.",
  "published": "2025-10-01",
  "updated": "2026-03-10",
  "year": "2025",
  "authors": [
   "Jiahang Cao",
   "Yize Huang",
   "Hanzhong Guo",
   "Rui Zhang",
   "Mu Nan",
   "Weijian Mai",
   "Jiaxu Wang",
   "Hao Cheng",
   "Jingkai Sun",
   "Gang Han",
   "Wen Zhao",
   "Qiang Zhang",
   "Yijie Guo",
   "Qihao Zheng",
   "Chunfeng Song",
   "Xiao Li",
   "Ping Luo",
   "Andrew F. Luo"
  ],
  "author_count": 18,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "ICLR 2026",
  "venue_source": "arxiv-comment",
  "citations": 13,
  "influential_citations": 3,
  "tldr": "This work proposes General Policy Composition (GPC), a training-free method that enhances performance by combining the distributional scores of multiple pre-trained policies via a convex combination and test-time search, and establishes GPC as a simple yet effective method for improving control performance by leveraging existing policies.",
  "doi": "10.48550/arXiv.2510.01068",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jiahang Cao",
    "id": "1406166829",
    "h_index": 13,
    "papers": 47
   },
   {
    "name": "Yize Huang",
    "id": "2383251809",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Hanzhong Guo",
    "id": "2220857922",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Rui Zhang",
    "id": "2363533134",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Mu Nan",
    "id": "2383167589",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Weijian Mai",
    "id": "2331415914",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Jiaxu Wang",
    "id": "2242768739",
    "h_index": 10,
    "papers": 39
   },
   {
    "name": "Haotai Cheng",
    "id": "2217401833",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Jingkai Sun",
    "id": "2243709540",
    "h_index": 10,
    "papers": 34
   },
   {
    "name": "Gang Han",
    "id": "2304988250",
    "h_index": 6,
    "papers": 17
   },
   {
    "name": "Wen Zhao",
    "id": "2304593772",
    "h_index": 7,
    "papers": 16
   },
   {
    "name": "Qiang Zhang",
    "id": "2230253713",
    "h_index": 13,
    "papers": 37
   },
   {
    "name": "Yijie Guo",
    "id": "2304744634",
    "h_index": 6,
    "papers": 17
   },
   {
    "name": "Qihao Zheng",
    "id": "2331607748",
    "h_index": 9,
    "papers": 18
   },
   {
    "name": "Chunfeng Song",
    "id": "2321174558",
    "h_index": 9,
    "papers": 23
   },
   {
    "name": "Xiao Li",
    "id": "2384068086",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Ping Luo",
    "id": "2305385401",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Andrew F. Luo",
    "id": "2375903701",
    "h_index": 2,
    "papers": 8
   }
  ],
  "comment": "Accepted to ICLR 2026. Project Page: https://sagecao1125.github.io/GPC-Site/",
  "topics": [
   "vla",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2510.01068v2",
  "pdf_url": "https://arxiv.org/pdf/2510.01068v2",
  "html_url": "https://arxiv.org/html/2510.01068v2",
  "code_url": "https://sagecao1125.github.io/GPC-Site/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.65
 },
 {
  "id": "2510.00406",
  "slug": "vla-rft-vision-language-action-reinforcement-fine-tuning-with-verified",
  "title": "VLA-RFT: Vision-Language-Action Reinforcement Fine-tuning with Verified Rewards in World Simulators",
  "abstract": "Vision-Language-Action (VLA) models enable embodied decision-making but rely heavily on imitation learning, leading to compounding errors and poor robustness under distribution shift. Reinforcement learning (RL) can mitigate these issues yet typically demands costly real-world interactions or suffers from sim-to-real gaps. We introduce VLA-RFT, a reinforcement fine-tuning framework that leverages a data-driven world model as a controllable simulator. Trained from real interaction data, the simulator predicts future visual observations conditioned on actions, allowing policy rollouts with dense, trajectory-level rewards derived from goal-achieving references. This design delivers an efficient and action-aligned learning signal, drastically lowering sample requirements. With fewer than 400 fine-tuning steps, VLA-RFT surpasses strong supervised baselines and achieves greater efficiency than simulator-based RL. Moreover, it exhibits strong robustness under perturbed conditions, sustaining stable task execution. Our results establish world-model-based RFT as a practical post-training paradigm to enhance the generalization and robustness of VLA models. For more details, please refer to https://vla-rft.github.io/.",
  "published": "2025-10-01",
  "updated": "2025-10-01",
  "year": "2025",
  "authors": [
   "Hengtao Li",
   "Pengxiang Ding",
   "Runze Suo",
   "Yihao Wang",
   "Zirui Ge",
   "Dongyuan Zang",
   "Kexian Yu",
   "Mingyang Sun",
   "Hongyin Zhang",
   "Donglin Wang",
   "Weihua Su"
  ],
  "author_count": 11,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 56,
  "influential_citations": 3,
  "tldr": "This work introduces VLA-RFT, a reinforcement fine-tuning framework that leverages a data-driven world model as a controllable simulator and establishes world-model-based RFT as a practical post-training paradigm to enhance the generalization and robustness of VLA models.",
  "doi": "10.48550/arXiv.2510.00406",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hengtao Li",
    "id": "2218230392",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "Pengxiang Ding",
    "id": "2275186266",
    "h_index": 23,
    "papers": 61
   },
   {
    "name": "Runze Suo",
    "id": "2346981469",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Yihao Wang",
    "id": "2383247029",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Zirui Ge",
    "id": "2359451210",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Dongyuan Zang",
    "id": "2233339711",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Kexian Yu",
    "id": "2384419797",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Mingyang Sun",
    "id": "2334782875",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Hongyin Zhang",
    "id": "2155343887",
    "h_index": 11,
    "papers": 29
   },
   {
    "name": "Donglin Wang",
    "id": "2275282870",
    "h_index": 16,
    "papers": 36
   },
   {
    "name": "Weihua Su",
    "id": "2281117865",
    "h_index": 4,
    "papers": 15
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "vla",
   "sim2real",
   "imitation-diffusion",
   "rl-control",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2510.00406v1",
  "pdf_url": "https://arxiv.org/pdf/2510.00406v1",
  "html_url": "https://arxiv.org/html/2510.00406v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.76
 },
 {
  "id": "2509.26633",
  "slug": "omniretarget-interaction-preserving-data-generation-for-humanoid-whole",
  "title": "OmniRetarget: Interaction-Preserving Data Generation for Humanoid Whole-Body Loco-Manipulation and Scene Interaction",
  "abstract": "A dominant paradigm for teaching humanoid robots complex skills is to retarget human motions as kinematic references to train reinforcement learning (RL) policies. However, existing retargeting pipelines often struggle with the significant embodiment gap between humans and robots, producing physically implausible artifacts like foot-skating and penetration. More importantly, common retargeting methods neglect the rich human-object and human-environment interactions essential for expressive locomotion and loco-manipulation. To address this, we introduce OmniRetarget, an interaction-preserving data generation engine based on an interaction mesh that explicitly models and preserves the crucial spatial and contact relationships between an agent, the terrain, and manipulated objects. By minimizing the Laplacian deformation between the human and robot meshes while enforcing kinematic constraints, OmniRetarget generates kinematically feasible trajectories. Moreover, preserving task-relevant interactions enables efficient data augmentation, from a single demonstration to different robot embodiments, terrains, and object configurations. We comprehensively evaluate OmniRetarget by retargeting motions from OMOMO, LAFAN1, and our in-house MoCap datasets, generating over 8-hour trajectories that achieve better kinematic constraint satisfaction and contact preservation than widely used baselines. Such high-quality data enables proprioceptive RL policies to successfully execute long-horizon (up to 30 seconds) parkour and loco-manipulation skills on a Unitree G1 humanoid, trained with only 5 reward terms and simple domain randomization shared by all tasks, without any learning curriculum.",
  "published": "2025-09-30",
  "updated": "2026-06-15",
  "year": "2025",
  "authors": [
   "Lujie Yang",
   "Xiaoyu Huang",
   "Zhen Wu",
   "Angjoo Kanazawa",
   "Pieter Abbeel",
   "Carmelo Sferrazza",
   "C. Karen Liu",
   "Rocky Duan",
   "Guanya Shi"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 113,
  "influential_citations": 11,
  "tldr": "OmniRetarget is an interaction-preserving data generation engine based on an interaction mesh that explicitly models and preserves the crucial spatial and contact relationships between an agent, the terrain, and manipulated objects, and generates kinematically feasible trajectories.",
  "doi": "10.48550/arXiv.2509.26633",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Lujie Yang",
    "id": "2383205600",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Xiaoyu Huang",
    "id": "2383100448",
    "h_index": 6,
    "papers": 6
   },
   {
    "name": "Zhen Wu",
    "id": "2308574851",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Angjoo Kanazawa",
    "id": "20615377",
    "h_index": 60,
    "papers": 126
   },
   {
    "name": "Pieter Abbeel",
    "id": "2381724317",
    "h_index": 9,
    "papers": 17
   },
   {
    "name": "Carmelo Sferrazza",
    "id": "47218071",
    "h_index": 21,
    "papers": 44
   },
   {
    "name": "C. K. Liu",
    "id": "2376138723",
    "h_index": 9,
    "papers": 11
   },
   {
    "name": "Rocky Duan",
    "id": "2381724728",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Guanya Shi",
    "id": "2384824402",
    "h_index": 9,
    "papers": 24
   }
  ],
  "comment": "Project website: https://omniretarget.github.io",
  "topics": [
   "humanoids",
   "sim2real",
   "rl-control"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2509.26633v3",
  "pdf_url": "https://arxiv.org/pdf/2509.26633v3",
  "html_url": "https://arxiv.org/html/2509.26633v3",
  "code_url": "https://omniretarget.github.io",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.56
 },
 {
  "id": "2509.25747",
  "slug": "best-of-sim-and-real-decoupled-visuomotor-manipulation-via-learning-co",
  "title": "Best of Sim and Real: Decoupled Visuomotor Manipulation via Learning Control in Simulation and Perception in Real",
  "abstract": "Sim-to-real transfer remains a fundamental challenge in robot manipulation due to the entanglement of perception and control in end-to-end learning. We present a decoupled framework that learns each component where it is most reliable: control policies are trained in simulation with privileged state to master spatial layouts and manipulation dynamics, while perception is adapted only at deployment to bridge real observations to the frozen control policy. Our key insight is that control strategies and action patterns are universal across environments and can be learned in simulation through systematic randomization, while perception is inherently domain-specific and must be learned where visual observations are authentic. Unlike existing end-to-end approaches that require extensive real-world data, our method achieves strong performance with only 10-20 real demonstrations by reducing the complex sim-to-real problem to a structured perception alignment task. We validate our approach on tabletop manipulation tasks, demonstrating superior data efficiency and out-of-distribution generalization compared to end-to-end baselines. The learned policies successfully handle object positions and scales beyond the training distribution, confirming that decoupling perception from control fundamentally improves sim-to-real transfer.",
  "published": "2025-09-30",
  "updated": "2025-09-30",
  "year": "2025",
  "authors": [
   "Jialei Huang",
   "Zhaoheng Yin",
   "Yingdong Hu",
   "Shuo Wang",
   "Xingyu Lin",
   "Yang Gao"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 1,
  "tldr": "This work presents a decoupled framework that learns each component where it is most reliable: control policies are trained in simulation with privileged state to master spatial layouts and manipulation dynamics, while perception is adapted only at deployment to bridge real observations to the frozen control policy.",
  "doi": "10.48550/arXiv.2509.25747",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jialei Huang",
    "id": "2326573660",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Zhaoheng Yin",
    "id": "2221218908",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Yingdong Hu",
    "id": "2149297811",
    "h_index": 13,
    "papers": 31
   },
   {
    "name": "Shuo Wang",
    "id": "2155470054",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Xingyu Lin",
    "id": "2115252655",
    "h_index": 15,
    "papers": 31
   },
   {
    "name": "Yang Gao",
    "id": "2382886824",
    "h_index": 2,
    "papers": 8
   }
  ],
  "comment": "10 pages, 6 figures",
  "topics": [
   "sim2real",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2509.25747v1",
  "pdf_url": "https://arxiv.org/pdf/2509.25747v1",
  "html_url": "https://arxiv.org/html/2509.25747v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2509.25161",
  "slug": "rolling-forcing-autoregressive-long-video-diffusion-in-real-time",
  "title": "Rolling Forcing: Autoregressive Long Video Diffusion in Real Time",
  "abstract": "Streaming video generation, as one fundamental component in interactive world models and neural game engines, aims to generate high-quality, low-latency, and temporally coherent long video streams. However, most existing work suffers from severe error accumulation that often significantly degrades the generated stream videos over long horizons. We design Rolling Forcing, a novel video generation technique that enables streaming long videos with minimal error accumulation. Rolling Forcing comes with three novel designs. First, instead of iteratively sampling individual frames, which accelerates error propagation, we design a joint denoising scheme that simultaneously denoises multiple frames with progressively increasing noise levels. This design relaxes the strict causality across adjacent frames, effectively suppressing error growth. Second, we introduce the attention sink mechanism into the long-horizon stream video generation task, which allows the model to keep key value states of initial frames as a global context anchor and thereby enhances long-term global consistency. Third, we design an efficient training algorithm that enables few-step distillation over largely extended denoising windows. This algorithm operates on non-overlapping windows and mitigates exposure bias conditioned on self-generated histories. Extensive experiments show that Rolling Forcing enables real-time streaming generation of multi-minute videos on a single GPU, with substantially reduced error accumulation.",
  "published": "2025-09-29",
  "updated": "2025-09-29",
  "year": "2025",
  "authors": [
   "Kunhao Liu",
   "Wenbo Hu",
   "Jiale Xu",
   "Ying Shan",
   "Shijian Lu"
  ],
  "author_count": 5,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 179,
  "influential_citations": 37,
  "tldr": "Rolling Forcing is designed, a novel video generation technique that enables streaming long videos with minimal error accumulation, and an efficient training algorithm that enables few-step distillation over largely extended denoising windows.",
  "doi": "10.48550/arXiv.2509.25161",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Kunhao Liu",
    "id": "2212199002",
    "h_index": 9,
    "papers": 12
   },
   {
    "name": "Wenbo Hu",
    "id": "2261082315",
    "h_index": 11,
    "papers": 21
   },
   {
    "name": "Jiale Xu",
    "id": "2332484605",
    "h_index": 6,
    "papers": 17
   },
   {
    "name": "Ying Shan",
    "id": "2332408267",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Shijian Lu",
    "id": "2238177864",
    "h_index": 7,
    "papers": 9
   }
  ],
  "comment": "Project page: https://kunhao-liu.github.io/Rolling_Forcing_Webpage/",
  "topics": [
   "world-models",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2509.25161v1",
  "pdf_url": "https://arxiv.org/pdf/2509.25161v1",
  "html_url": "https://arxiv.org/html/2509.25161v1",
  "code_url": "https://kunhao-liu.github.io/Rolling_Forcing_Webpage/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.26
 },
 {
  "id": "2509.24948",
  "slug": "world-env-leveraging-world-model-as-a-virtual-environment-for-vla-post",
  "title": "World-Env: Leveraging World Model as a Virtual Environment for VLA Post-Training",
  "abstract": "Vision-Language-Action (VLA) models trained via imitation learning suffer from significant performance degradation in data-scarce scenarios due to their reliance on large-scale demonstration datasets. Although reinforcement learning (RL)-based post-training has proven effective in addressing data scarcity, its application to VLA models is hindered by the non-resettable nature of real-world environments. This limitation is particularly critical in high-risk domains such as industrial automation, where interactions often induce state changes that are costly or infeasible to revert. Furthermore, existing VLA approaches lack a reliable mechanism for detecting task completion, leading to redundant actions that reduce overall task success rates. To address these challenges, we propose World-Env, an RL-based post-training framework that replaces physical interaction with a low-cost world model-based virtual simulator. World-Env consists of two key components: (1) a physically-consistent world simulator that generates temporally consistent future visual observations, and (2) a vision-language model (VLM)-guided instant reflector that provides continuous reward signals and predicts action termination. This simulated environment enables VLA models to safely explore and generalize beyond their initial imitation learning distribution. Our method achieves notable performance gains with as few as five expert demonstrations per task. Experiments on complex robotic manipulation tasks demonstrate that World-Env effectively overcomes the data inefficiency, safety constraints, and inefficient execution of conventional VLA models that rely on real-world interaction, offering a practical and scalable solution for post-training in resource-constrained settings. Our code is available at https://github.com/amap-cvlab/world-env.",
  "published": "2025-09-29",
  "updated": "2026-04-27",
  "year": "2025",
  "authors": [
   "Junjin Xiao",
   "Yandan Yang",
   "Xinyuan Chang",
   "Ronghan Chen",
   "Feng Xiong",
   "Mu Xu",
   "Wei-Shi Zheng",
   "Qing Zhang"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 49,
  "influential_citations": 6,
  "tldr": "World-Env is proposed, an RL-based post-training framework that replaces physical interaction with a low-cost world model-based virtual simulator that enables VLA models to safely explore and generalize beyond their initial imitation learning distribution.",
  "doi": "10.48550/arXiv.2509.24948",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Junjin Xiao",
    "id": "2291985942",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Yandan Yang",
    "id": "2382929955",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Xinyuan Chang",
    "id": "2320818428",
    "h_index": 8,
    "papers": 25
   },
   {
    "name": "Ronghan Chen",
    "id": "2383246949",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Feng Xiong",
    "id": "2382761717",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Mu Xu",
    "id": "2363407946",
    "h_index": 7,
    "papers": 25
   },
   {
    "name": "Weihua Zheng",
    "id": "2152974380",
    "h_index": 8,
    "papers": 18
   },
   {
    "name": "Qing Zhang",
    "id": "2291975202",
    "h_index": 4,
    "papers": 17
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "vla",
   "sim2real",
   "imitation-diffusion",
   "rl-control",
   "video-generation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2509.24948v6",
  "pdf_url": "https://arxiv.org/pdf/2509.24948v6",
  "html_url": "https://arxiv.org/html/2509.24948v6",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.7
 },
 {
  "id": "2509.24527",
  "slug": "training-agents-inside-of-scalable-world-models",
  "title": "Training Agents Inside of Scalable World Models",
  "abstract": "World models learn general knowledge from videos and simulate experience for training behaviors in imagination, offering a path towards intelligent agents. However, previous world models have been unable to accurately predict object interactions in complex environments. We introduce Dreamer 4, a scalable agent that learns to solve control tasks by reinforcement learning inside of a fast and accurate world model. In the complex video game Minecraft, the world model accurately predicts object interactions and game mechanics, outperforming previous world models by a large margin. The world model achieves real-time interactive inference on a single GPU through a shortcut forcing objective and an efficient transformer architecture. Moreover, the world model learns general action conditioning from only a small amount of data, allowing it to extract the majority of its knowledge from diverse unlabeled videos. We propose the challenge of obtaining diamonds in Minecraft from only offline data, aligning with practical applications such as robotics where learning from environment interaction can be unsafe and slow. This task requires choosing sequences of over 20,000 mouse and keyboard actions from raw pixels. By learning behaviors in imagination, Dreamer 4 is the first agent to obtain diamonds in Minecraft purely from offline data, without environment interaction. Our work provides a scalable recipe for imagination training, marking a step towards intelligent agents.",
  "published": "2025-09-29",
  "updated": "2025-09-29",
  "year": "2025",
  "authors": [
   "Danijar Hafner",
   "Wilson Yan",
   "Timothy Lillicrap"
  ],
  "author_count": 3,
  "categories": [
   "cs.AI",
   "cs.LG",
   "cs.RO",
   "stat.ML"
  ],
  "primary_category": "cs.AI",
  "venue": "",
  "venue_source": "",
  "citations": 117,
  "influential_citations": 6,
  "tldr": "This work proposes the challenge of obtaining diamonds in Minecraft from only offline data, aligning with practical applications such as robotics where learning from environment interaction can be unsafe and slow, and provides a scalable recipe for imagination training.",
  "doi": "10.48550/arXiv.2509.24527",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Danijar Hafner",
    "id": "2285299430",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Wilson Yan",
    "id": "46705049",
    "h_index": 12,
    "papers": 18
   },
   {
    "name": "Timothy P. Lillicrap",
    "id": "2302799561",
    "h_index": 6,
    "papers": 11
   }
  ],
  "comment": "Website: https://danijar.com/dreamer4/",
  "topics": [
   "world-models",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2509.24527v1",
  "pdf_url": "https://arxiv.org/pdf/2509.24527v1",
  "html_url": "https://arxiv.org/html/2509.24527v1",
  "code_url": null,
  "presented": [
   {
    "club": "saturday-robotics",
    "club_name": "Saturday Robotics",
    "session_number": 7,
    "session_title": "Robotics & World Models Reading Club 07: Learning to Dream: World Models, Imagination, Path to Foundation Models for Control \u2014 Los Altos",
    "date_text": "Saturday, May 9, 2026 | 2:00 PM \u2013 5:00 PM",
    "city": "",
    "url": "https://lu.ma/srhe0vuo",
    "listed_as": "Training Agents Inside of Scalable World Models (2025)"
   }
  ],
  "club_note": "Replaces RSSM with diffusion/transformer-based world models",
  "featured": true,
  "signal": 5.07
 },
 {
  "id": "2509.20328",
  "slug": "video-models-are-zero-shot-learners-and-reasoners",
  "title": "Video models are zero-shot learners and reasoners",
  "abstract": "The remarkable zero-shot capabilities of Large Language Models (LLMs) have propelled natural language processing from task-specific models to unified, generalist foundation models. This transformation emerged from simple primitives: large, generative models trained on web-scale data. Curiously, the same primitives apply to today's generative video models. Could video models be on a trajectory towards general-purpose vision understanding, much like LLMs developed general-purpose language understanding? We demonstrate that Veo 3 can solve a broad variety of tasks it wasn't explicitly trained for: segmenting objects, detecting edges, editing images, understanding physical properties, recognizing object affordances, simulating tool use, and more. These abilities to perceive, model, and manipulate the visual world enable early forms of visual reasoning like maze and symmetry solving. Veo's emergent zero-shot capabilities indicate that video models are on a path to becoming unified, generalist vision foundation models.",
  "published": "2025-09-24",
  "updated": "2025-09-29",
  "year": "2025",
  "authors": [
   "Thadd\u00e4us Wiedemer",
   "Yuxuan Li",
   "Paul Vicol",
   "Shixiang Shane Gu",
   "Nick Matarese",
   "Kevin Swersky",
   "Been Kim",
   "Priyank Jaini",
   "Robert Geirhos"
  ],
  "author_count": 9,
  "categories": [
   "cs.LG",
   "cs.AI",
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 200,
  "influential_citations": 24,
  "tldr": "It is demonstrated that Veo 3 can solve a broad variety of tasks it wasn't explicitly trained for: segmenting objects, detecting edges, editing images, understanding physical properties, recognizing object affordances, simulating tool use, and more.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Thaddaus Wiedemer",
    "id": "2347053518",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Yuxuan Li",
    "id": "2380467211",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Paul Vicol",
    "id": "2039154",
    "h_index": 15,
    "papers": 39
   },
   {
    "name": "S. Gu",
    "id": "2253699903",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Nick Matarese",
    "id": "2381836032",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Kevin Swersky",
    "id": "1754860",
    "h_index": 24,
    "papers": 51
   },
   {
    "name": "Been Kim",
    "id": "2372371635",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "P. Jaini",
    "id": "144818264",
    "h_index": 17,
    "papers": 46
   },
   {
    "name": "Robert Geirhos",
    "id": "2132068799",
    "h_index": 7,
    "papers": 14
   }
  ],
  "comment": "Project page: https://video-zero-shot.github.io/",
  "topics": [
   "foundation-pretraining",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2509.20328v2",
  "pdf_url": "https://arxiv.org/pdf/2509.20328v2",
  "html_url": "https://arxiv.org/html/2509.20328v2",
  "code_url": "https://video-zero-shot.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.3
 },
 {
  "id": "2509.20322",
  "slug": "visualmimic-visual-humanoid-loco-manipulation-via-motion-tracking-and",
  "title": "VisualMimic: Visual Humanoid Loco-Manipulation via Motion Tracking and Generation",
  "abstract": "Humanoid loco-manipulation in unstructured environments demands tight integration of egocentric perception and whole-body control. However, existing approaches either depend on external motion capture systems or fail to generalize across diverse tasks. We introduce VisualMimic, a visual sim-to-real framework that unifies egocentric vision with hierarchical whole-body control for humanoid robots. VisualMimic combines a task-agnostic low-level keypoint tracker -- trained from human motion data via a teacher-student scheme -- with a task-specific high-level policy that generates keypoint commands from visual and proprioceptive input. To ensure stable training, we inject noise into the low-level policy and clip high-level actions using human motion statistics. VisualMimic enables zero-shot transfer of visuomotor policies trained in simulation to real humanoid robots, accomplishing a wide range of loco-manipulation tasks such as box lifting, pushing, football dribbling, and kicking. Beyond controlled laboratory settings, our policies also generalize robustly to outdoor environments. Videos are available at: https://visualmimic.github.io .",
  "published": "2025-09-24",
  "updated": "2025-11-13",
  "year": "2025",
  "authors": [
   "Shaofeng Yin",
   "Yanjie Ze",
   "Hong-Xing Yu",
   "C. Karen Liu",
   "Jiajun Wu"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 45,
  "influential_citations": 3,
  "tldr": "VisualMimic is introduced, a visual sim-to-real framework that unifies egocentric vision with hierarchical whole-body control for humanoid robots, and enables zero-shot transfer of visuomotor policies trained in simulation to real humanoid robots, accomplishing a wide range of loco-manipulation tasks.",
  "doi": "10.48550/arXiv.2509.20322",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Shaofeng Yin",
    "id": "2303335516",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Yanjie Ze",
    "id": "2325901084",
    "h_index": 10,
    "papers": 13
   },
   {
    "name": "Hong-Xing Yu",
    "id": "2239448099",
    "h_index": 18,
    "papers": 32
   },
   {
    "name": "C. K. Liu",
    "id": "2376138723",
    "h_index": 9,
    "papers": 11
   },
   {
    "name": "Jiajun Wu",
    "id": "2326444500",
    "h_index": 35,
    "papers": 214
   }
  ],
  "comment": "Website: https://visualmimic.github.io",
  "topics": [
   "humanoids",
   "egocentric-data",
   "sim2real"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2509.20322v2",
  "pdf_url": "https://arxiv.org/pdf/2509.20322v2",
  "html_url": "https://arxiv.org/html/2509.20322v2",
  "code_url": "https://visualmimic.github.io",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.66
 },
 {
  "id": "2509.20021",
  "slug": "embodied-ai-from-llms-to-world-models",
  "title": "Embodied AI: From LLMs to World Models",
  "abstract": "Embodied Artificial Intelligence (AI) is an intelligent system paradigm for achieving Artificial General Intelligence (AGI), serving as the cornerstone for various applications and driving the evolution from cyberspace to physical systems. Recent breakthroughs in Large Language Models (LLMs) and World Models (WMs) have drawn significant attention for embodied AI. On the one hand, LLMs empower embodied AI via semantic reasoning and task decomposition, bringing high-level natural language instructions and low-level natural language actions into embodied cognition. On the other hand, WMs empower embodied AI by building internal representations and future predictions of the external world, facilitating physical law-compliant embodied interactions. As such, this paper comprehensively explores the literature in embodied AI from basics to advances, covering both LLM driven and WM driven works. In particular, we first present the history, key technologies, key components, and hardware systems of embodied AI, as well as discuss its development via looking from unimodal to multimodal angle. We then scrutinize the two burgeoning fields of embodied AI, i.e., embodied AI with LLMs/multimodal LLMs (MLLMs) and embodied AI with WMs, meticulously delineating their indispensable roles in end-to-end embodied cognition and physical laws-driven embodied interactions. Building upon the above advances, we further share our insights on the necessity of the joint MLLM-WM driven embodied AI architecture, shedding light on its profound significance in enabling complex tasks within physical worlds. In addition, we examine representative applications of embodied AI, demonstrating its wide applicability in real-world scenarios. Last but not least, we point out future research directions of embodied AI that deserve further investigation.",
  "published": "2025-09-24",
  "updated": "2025-09-24",
  "year": "2025",
  "authors": [
   "Tongtong Feng",
   "Xin Wang",
   "Yu-Gang Jiang",
   "Wenwu Zhu"
  ],
  "author_count": 4,
  "categories": [
   "cs.AI",
   "cs.CL",
   "cs.RO"
  ],
  "primary_category": "cs.AI",
  "venue": "",
  "venue_source": "",
  "citations": 33,
  "influential_citations": 1,
  "tldr": "This paper comprehensively explores the literature in embodied AI from basics to advances, covering both LLM driven and WM driven works, and meticulously delineates their indispensable roles in end-to-end embodied cognition and physical laws-driven embodied interactions.",
  "doi": "10.1109/MCAS.2025.3603693",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Tongtong Feng",
    "id": "2314118951",
    "h_index": 5,
    "papers": 18
   },
   {
    "name": "Xin Wang",
    "id": "2314795869",
    "h_index": 5,
    "papers": 16
   },
   {
    "name": "Yu-Gang Jiang",
    "id": "2249522707",
    "h_index": 20,
    "papers": 46
   },
   {
    "name": "Wenwu Zhu",
    "id": "2156154955",
    "h_index": 30,
    "papers": 182
   }
  ],
  "comment": "Accepted by IEEE CASM",
  "topics": [
   "world-models"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2509.20021v1",
  "pdf_url": "https://arxiv.org/pdf/2509.20021v1",
  "html_url": "https://arxiv.org/html/2509.20021v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.53
 },
 {
  "id": "2509.19626",
  "slug": "egobridge-domain-adaptation-for-generalizable-imitation-from-egocentri",
  "title": "EgoBridge: Domain Adaptation for Generalizable Imitation from Egocentric Human Data",
  "abstract": "Egocentric human experience data presents a vast resource for scaling up end-to-end imitation learning for robotic manipulation. However, significant domain gaps in visual appearance, sensor modalities, and kinematics between human and robot impede knowledge transfer. This paper presents EgoBridge, a unified co-training framework that explicitly aligns the policy latent spaces between human and robot data using domain adaptation. Through a measure of discrepancy on the joint policy latent features and actions based on Optimal Transport (OT), we learn observation representations that not only align between the human and robot domain but also preserve the action-relevant information critical for policy learning. EgoBridge achieves a significant absolute policy success rate improvement by 44% over human-augmented cross-embodiment baselines in three real-world single-arm and bimanual manipulation tasks. EgoBridge also generalizes to new objects, scenes, and tasks seen only in human data, where baselines fail entirely. Videos and additional information can be found at https://ego-bridge.github.io",
  "published": "2025-09-23",
  "updated": "2025-09-23",
  "year": "2025",
  "authors": [
   "Ryan Punamiya",
   "Dhruv Patel",
   "Patcharapong Aphiwetsa",
   "Pranav Kuppili",
   "Lawrence Y. Zhu",
   "Simar Kareer",
   "Judy Hoffman",
   "Danfei Xu"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 30,
  "influential_citations": 2,
  "tldr": "EgoBridge is presented, a unified co-training framework that explicitly aligns the policy latent spaces between human and robot data using domain adaptation and generalizes to new objects, scenes, and tasks seen only in human data, where baselines fail entirely.",
  "doi": "10.48550/arXiv.2509.19626",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ryan Punamiya",
    "id": "2328411560",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Dhruv Patel",
    "id": "2328566408",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Patcharapong Aphiwetsa",
    "id": "2310435059",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Pranav Kuppili",
    "id": "2378954879",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Lawrence Y. Zhu",
    "id": "2379184973",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Simar Kareer",
    "id": "2188833033",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Judy Hoffman",
    "id": "2328413304",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Danfei Xu",
    "id": "2315451166",
    "h_index": 8,
    "papers": 10
   }
  ],
  "comment": "Accepted at 39th Conference on Neural Information Processing Systems (NeurIPS 2025) and Oral at Conference on Robot Learning (CoRL 2025)",
  "topics": [
   "egocentric-data",
   "imitation-diffusion",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2509.19626v1",
  "pdf_url": "https://arxiv.org/pdf/2509.19626v1",
  "html_url": "https://arxiv.org/html/2509.19626v1",
  "code_url": null,
  "presented": [
   {
    "club": "saturday-robotics",
    "club_name": "Saturday Robotics",
    "session_number": 8,
    "session_title": "Robotics & World Models Reading Club 08: Embodied Human Data as the \u201cInternet of Motion and Behavior\u201d \u2014 San Francisco 0516",
    "date_text": "Saturday, May 16, 2026 | 2:00 PM \u2013 5:00 PM",
    "city": "San Francisco, CA",
    "url": "https://lu.ma/qoxioge7",
    "listed_as": "Scaling Robot Learning with Human Behavior Priors"
   },
   {
    "club": "saturday-robotics",
    "club_name": "Saturday Robotics",
    "session_number": 13,
    "session_title": "Robotics & World Models Reading Club 13: HumanEgo: Train Robot Policy from 30 min Egocentric Videos \u2014 SF 0620",
    "date_text": "Saturday, June 20, 2026 | 2:00 PM \u2013 5:00 PM",
    "city": "San Francisco, CA",
    "url": "https://lu.ma/6vkhxnum",
    "listed_as": ""
   }
  ],
  "club_note": "Pretrains Transformer robot policies on large-scale human behavioral data, then adapts to downstream robotic tasks with low-shot finetuning.",
  "featured": true,
  "signal": 5.99
 },
 {
  "id": "2509.19080",
  "slug": "world4rl-diffusion-world-models-for-policy-refinement-with-reinforceme",
  "title": "World4RL: Diffusion World Models for Policy Refinement with Reinforcement Learning for Robotic Manipulation",
  "abstract": "Robotic manipulation policies are commonly initialized through imitation learning, but their performance is limited by the scarcity and narrow coverage of expert data. Reinforcement learning can refine polices to alleviate this limitation, yet real-robot training is costly and unsafe, while training in simulators suffers from the sim-to-real gap. Recent advances in generative models have demonstrated remarkable capabilities in real-world simulation, with diffusion models in particular excelling at generation. This raises the question of how diffusion model-based world models can be combined to enhance pre-trained policies in robotic manipulation. In this work, we propose World4RL, a framework that employs diffusion-based world models as high-fidelity simulators to refine pre-trained policies entirely in imagined environments for robotic manipulation. Unlike prior works that primarily employ world models for planning, our framework enables direct end-to-end policy optimization. World4RL is designed around two principles: pre-training a diffusion world model that captures diverse dynamics on multi-task datasets and refining policies entirely within a frozen world model to avoid online real-world interactions. We further design a two-hot action encoding scheme tailored for robotic manipulation and adopt diffusion backbones to improve modeling fidelity. Extensive simulation and real-world experiments demonstrate that World4RL provides high-fidelity environment modeling and enables consistent policy refinement, yielding significantly higher success rates compared to imitation learning and other baselines.",
  "published": "2025-09-23",
  "updated": "2026-03-19",
  "year": "2025",
  "authors": [
   "Zhennan Jiang",
   "Kai Liu",
   "Yuxin Qin",
   "Shuai Tian",
   "Yupeng Zheng",
   "Mingcai Zhou",
   "Chao Yu",
   "Haoran Li",
   "Dongbin Zhao"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 31,
  "influential_citations": 2,
  "tldr": "World4RL is a framework that employs diffusion-based world models as high-fidelity simulators to refine pre-trained policies entirely in imagined environments for robotic manipulation and enables consistent policy refinement, yielding significantly higher success rates compared to imitation learning and other baselines.",
  "doi": "10.48550/arXiv.2509.19080",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhen Jiang",
    "id": "2268849443",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Kai Liu",
    "id": "2377763386",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Yuxin Qin",
    "id": "2381783760",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Shuai Tian",
    "id": "2345921369",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Yupeng Zheng",
    "id": "2357068526",
    "h_index": 16,
    "papers": 46
   },
   {
    "name": "Mingcai Zhou",
    "id": "2376443273",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Chao Yu",
    "id": "2322609119",
    "h_index": 6,
    "papers": 18
   },
   {
    "name": "Haoran Li",
    "id": "2347365117",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Dongbin Zhao",
    "id": "2376522592",
    "h_index": 5,
    "papers": 24
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "sim2real",
   "imitation-diffusion",
   "rl-control",
   "foundation-pretraining",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2509.19080v2",
  "pdf_url": "https://arxiv.org/pdf/2509.19080v2",
  "html_url": "https://arxiv.org/html/2509.19080v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.51
 },
 {
  "id": "2509.14353",
  "slug": "dreamcontrol-human-inspired-whole-body-humanoid-control-for-scene-inte",
  "title": "DreamControl: Human-Inspired Whole-Body Humanoid Control for Scene Interaction via Guided Diffusion",
  "abstract": "We introduce DreamControl, a novel methodology for learning autonomous whole-body humanoid skills. DreamControl leverages the strengths of diffusion models and Reinforcement Learning (RL): our core innovation is the use of a diffusion prior trained on human motion data, which subsequently guides an RL policy in simulation to complete specific tasks of interest (e.g., opening a drawer or picking up an object). We demonstrate that this human motion-informed prior allows RL to discover solutions unattainable by direct RL, and that diffusion models inherently promote natural looking motions, aiding in sim-to-real transfer. We validate DreamControl's effectiveness on a Unitree G1 robot across a diverse set of challenging tasks involving simultaneous lower and upper body control and object interaction. Project website at https://genrobo.github.io/DreamControl/",
  "published": "2025-09-17",
  "updated": "2025-09-30",
  "year": "2025",
  "authors": [
   "Dvij Kalaria",
   "Sudarshan S Harithas",
   "Pushkal Katara",
   "Sangkyung Kwak",
   "Sarthak Bhagat",
   "Shankar Sastry",
   "Srinath Sridhar",
   "Sai Vemprala",
   "Ashish Kapoor",
   "Jonathan Chung-Kuan Huang"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 17,
  "influential_citations": 0,
  "tldr": "It is demonstrated that this human motion-informed prior allows RL to discover solutions unattainable by direct RL, and that diffusion models inherently promote natural looking motions, aiding in sim-to-real transfer.",
  "doi": "10.48550/arXiv.2509.14353",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Dvij Kalaria",
    "id": "2126958435",
    "h_index": 8,
    "papers": 19
   },
   {
    "name": "Sudarshan S. Harithas",
    "id": "1703181521",
    "h_index": 4,
    "papers": 15
   },
   {
    "name": "Pushkal Katara",
    "id": "147806740",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Sangkyung Kwak",
    "id": "2214766742",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Sarthak Bhagat",
    "id": "2296703478",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "S. Sastry",
    "id": "2393141608",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Srinath Sridhar",
    "id": "2269469994",
    "h_index": 7,
    "papers": 20
   },
   {
    "name": "Sai H. Vemprala",
    "id": "8355354",
    "h_index": 15,
    "papers": 41
   },
   {
    "name": "Ashish Kapoor",
    "id": "2253595046",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Jonathan Huang",
    "id": "2325100313",
    "h_index": 3,
    "papers": 6
   }
  ],
  "comment": "https://genrobo.github.io/DreamControl/ (under submission)",
  "topics": [
   "humanoids",
   "sim2real",
   "rl-control",
   "video-generation"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2509.14353v3",
  "pdf_url": "https://arxiv.org/pdf/2509.14353v3",
  "html_url": "https://arxiv.org/html/2509.14353v3",
  "code_url": "https://genrobo.github.io/DreamControl/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.76
 },
 {
  "id": "2509.10952",
  "slug": "immimic-cross-domain-imitation-from-human-videos-via-mapping-and-inter",
  "title": "ImMimic: Cross-Domain Imitation from Human Videos via Mapping and Interpolation",
  "abstract": "Learning robot manipulation from abundant human videos offers a scalable alternative to costly robot-specific data collection. However, domain gaps across visual, morphological, and physical aspects hinder direct imitation. To effectively bridge the domain gap, we propose ImMimic, an embodiment-agnostic co-training framework that leverages both human videos and a small amount of teleoperated robot demonstrations. ImMimic uses Dynamic Time Warping (DTW) with either action- or visual-based mapping to map retargeted human hand poses to robot joints, followed by MixUp interpolation between paired human and robot trajectories. Our key insights are (1) retargeted human hand trajectories provide informative action labels, and (2) interpolation over the mapped data creates intermediate domains that facilitate smooth domain adaptation during co-training. Evaluations on four real-world manipulation tasks (Pick and Place, Push, Hammer, Flip) across four robotic embodiments (Robotiq, Fin Ray, Allegro, Ability) show that ImMimic improves task success rates and execution smoothness, highlighting its efficacy to bridge the domain gap for robust robot manipulation. The project website can be found at https://sites.google.com/view/immimic.",
  "published": "2025-09-13",
  "updated": "2025-09-13",
  "year": "2025",
  "authors": [
   "Yangcen Liu",
   "Woo Chul Shin",
   "Yunhai Han",
   "Zhenyang Chen",
   "Harish Ravichandar",
   "Danfei Xu"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 27,
  "influential_citations": 0,
  "tldr": "Evaluations on four real-world manipulation tasks show that ImMimic improves task success rates and execution smoothness, highlighting its efficacy to bridge the domain gap for robust robot manipulation.",
  "doi": "10.48550/arXiv.2509.10952",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yangcen Liu",
    "id": "2297343604",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Woo-Chul Shin",
    "id": "2360757506",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Yunhai Han",
    "id": "2295678712",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Zhenyang Chen",
    "id": "2367302562",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "H. Ravichandar",
    "id": "2138933",
    "h_index": 17,
    "papers": 74
   },
   {
    "name": "Danfei Xu",
    "id": "2315451166",
    "h_index": 8,
    "papers": 10
   }
  ],
  "comment": "Conference of Robot Learning",
  "topics": [
   "egocentric-data",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2509.10952v1",
  "pdf_url": "https://arxiv.org/pdf/2509.10952v1",
  "html_url": "https://arxiv.org/html/2509.10952v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.45
 },
 {
  "id": "2509.10771",
  "slug": "rsl-rl-a-learning-library-for-robotics-research",
  "title": "RSL-RL: A Learning Library for Robotics Research",
  "abstract": "RSL-RL is an open-source Reinforcement Learning library tailored to the specific needs of the robotics community. Unlike broad general-purpose frameworks, its design philosophy prioritizes a compact and easily modifiable codebase, allowing researchers to adapt and extend algorithms with minimal overhead. The library focuses on algorithms most widely adopted in robotics, together with auxiliary techniques that address robotics-specific challenges. Optimized for GPU-only training, RSL-RL achieves high-throughput performance in large-scale simulation environments. Its effectiveness has been validated in both simulation benchmarks and in real-world robotic experiments, demonstrating its utility as a lightweight, extensible, and practical framework to develop learning-based robotic controllers. The library is open-sourced at: https://github.com/leggedrobotics/rsl_rl.",
  "published": "2025-09-13",
  "updated": "2025-09-13",
  "year": "2025",
  "authors": [
   "Clemens Schwarke",
   "Mayank Mittal",
   "Nikita Rudin",
   "David Hoeller",
   "Marco Hutter"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 76,
  "influential_citations": 3,
  "tldr": "RSL-RL is an open-source Reinforcement Learning library tailored to the specific needs of the robotics community, which focuses on algorithms most widely adopted in robotics, together with auxiliary techniques that address robotics-specific challenges.",
  "doi": "10.48550/arXiv.2509.10771",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Clemens Schwarke",
    "id": "2284772421",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Mayank Mittal",
    "id": "2061780867",
    "h_index": 13,
    "papers": 30
   },
   {
    "name": "N. Rudin",
    "id": "2113243810",
    "h_index": 17,
    "papers": 17
   },
   {
    "name": "David Hoeller",
    "id": "71054073",
    "h_index": 18,
    "papers": 19
   },
   {
    "name": "Marco Hutter",
    "id": "2377558922",
    "h_index": 4,
    "papers": 6
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "sim2real",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2509.10771v1",
  "pdf_url": "https://arxiv.org/pdf/2509.10771v1",
  "html_url": "https://arxiv.org/html/2509.10771v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.89
 },
 {
  "id": "2509.09769",
  "slug": "mimicdroid-in-context-learning-for-humanoid-robot-manipulation-from-hu",
  "title": "MimicDroid: In-Context Learning for Humanoid Robot Manipulation from Human Play Videos",
  "abstract": "We aim to enable humanoid robots to efficiently solve new manipulation tasks from a few video examples. In-context learning (ICL) is a promising framework for achieving this goal due to its test-time data efficiency and rapid adaptability. However, current ICL methods rely on labor-intensive teleoperated data for training, which restricts scalability. We propose using human play videos -- continuous, unlabeled videos of people interacting freely with their environment -- as a scalable and diverse training data source. We introduce MimicDroid, which enables humanoids to perform ICL using human play videos as the only training data. MimicDroid extracts trajectory pairs with similar manipulation behaviors and trains the policy to predict the actions of one trajectory conditioned on the other. Through this process, the model acquired ICL capabilities for adapting to novel objects and environments at test time. To bridge the embodiment gap, MimicDroid first retargets human wrist poses estimated from RGB videos to the humanoid, leveraging kinematic similarity. It also applies random patch masking during training to reduce overfitting to human-specific cues and improve robustness to visual differences. To evaluate few-shot learning for humanoids, we introduce an open-source simulation benchmark with increasing levels of generalization difficulty. MimicDroid outperformed state-of-the-art methods and achieved nearly twofold higher success rates in the real world. Additional materials can be found on: ut-austin-rpl.github.io/MimicDroid",
  "published": "2025-09-11",
  "updated": "2025-09-11",
  "year": "2025",
  "authors": [
   "Rutav Shah",
   "Shuijing Liu",
   "Qi Wang",
   "Zhenyu Jiang",
   "Sateesh Kumar",
   "Mingyo Seo",
   "Roberto Mart\u00edn-Mart\u00edn",
   "Yuke Zhu"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 16,
  "influential_citations": 0,
  "tldr": "MimicDroid is introduced, which enables humanoids to perform ICL using human play videos as the only training data and an open-source simulation benchmark with increasing levels of generalization difficulty is introduced to evaluate few-shot learning for humanoids.",
  "doi": "10.48550/arXiv.2509.09769",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Rutav Shah",
    "id": "2117716975",
    "h_index": 12,
    "papers": 17
   },
   {
    "name": "Shuijing Liu",
    "id": "30507647",
    "h_index": 11,
    "papers": 31
   },
   {
    "name": "Qi Wang",
    "id": "2326830174",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Zhenyu Jiang",
    "id": "2246857329",
    "h_index": 8,
    "papers": 13
   },
   {
    "name": "Sateesh Kumar",
    "id": "2374489701",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Mingyo Seo",
    "id": "23190833",
    "h_index": 7,
    "papers": 12
   },
   {
    "name": "Roberto Mart\u00edn-Mart\u00edn",
    "id": "2380440732",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Yuke Zhu",
    "id": "2322736301",
    "h_index": 13,
    "papers": 21
   }
  ],
  "comment": "11 pages, 9 figures, 5 tables",
  "topics": [
   "humanoids",
   "sim2real",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2509.09769v1",
  "pdf_url": "https://arxiv.org/pdf/2509.09769v1",
  "html_url": "https://arxiv.org/html/2509.09769v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.23
 },
 {
  "id": "2509.07962",
  "slug": "ta-vla-elucidating-the-design-space-of-torque-aware-vision-language-ac",
  "title": "TA-VLA: Elucidating the Design Space of Torque-aware Vision-Language-Action Models",
  "abstract": "Many robotic manipulation tasks require sensing and responding to force signals such as torque to assess whether the task has been successfully completed and to enable closed-loop control. However, current Vision-Language-Action (VLA) models lack the ability to integrate such subtle physical feedback. In this work, we explore Torque-aware VLA models, aiming to bridge this gap by systematically studying the design space for incorporating torque signals into existing VLA architectures. We identify and evaluate several strategies, leading to three key findings. First, introducing torque adapters into the decoder consistently outperforms inserting them into the encoder.Third, inspired by joint prediction and planning paradigms in autonomous driving, we propose predicting torque as an auxiliary output, which further improves performance. This strategy encourages the model to build a physically grounded internal representation of interaction dynamics. Extensive quantitative and qualitative experiments across contact-rich manipulation benchmarks validate our findings.",
  "published": "2025-09-09",
  "updated": "2025-09-09",
  "year": "2025",
  "authors": [
   "Zongzheng Zhang",
   "Haobo Xu",
   "Zhuo Yang",
   "Chenghao Yue",
   "Zehao Lin",
   "Huan-ang Gao",
   "Ziwei Wang",
   "Hao Zhao"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL 2025",
  "venue_source": "arxiv-comment",
  "citations": 31,
  "influential_citations": 2,
  "tldr": "This work explores Torque-aware VLA models, aiming to bridge this gap by systematically studying the design space for incorporating torque signals into existing VLA architectures, and identifies and evaluates several strategies, leading to three key findings.",
  "doi": "10.48550/arXiv.2509.07962",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zongzheng Zhang",
    "id": "2294931371",
    "h_index": 7,
    "papers": 23
   },
   {
    "name": "Haobo Xu",
    "id": "2379955487",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Zhuo Yang",
    "id": "2300137968",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Chenghao Yue",
    "id": "2379761890",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Zehao Lin",
    "id": "2146389150",
    "h_index": 11,
    "papers": 29
   },
   {
    "name": "Huan-ang Gao",
    "id": "2221148510",
    "h_index": 14,
    "papers": 33
   },
   {
    "name": "Ziwei Wang",
    "id": "2379837969",
    "h_index": 2,
    "papers": 11
   },
   {
    "name": "Hao Zhao",
    "id": "2379953681",
    "h_index": 4,
    "papers": 6
   }
  ],
  "comment": "Accepted to CoRL 2025, project page: \\url{https://zzongzheng0918.github.io/Torque-Aware-VLA.github.io/}",
  "topics": [
   "vla",
   "tactile",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2509.07962v1",
  "pdf_url": "https://arxiv.org/pdf/2509.07962v1",
  "html_url": "https://arxiv.org/html/2509.07962v1",
  "code_url": "https://zzongzheng0918.github.io/Torque-Aware-VLA.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.01
 },
 {
  "id": "2509.07996",
  "slug": "3d-and-4d-world-modeling-a-survey",
  "title": "3D and 4D World Modeling: A Survey",
  "abstract": "World modeling has become a cornerstone in AI research, enabling agents to understand, represent, and predict the dynamic environments they inhabit. While prior work largely emphasizes generative methods for 2D image and video data, they overlook the rapidly growing body of work that leverages native 3D and 4D representations such as RGB-D imagery, occupancy grids, and LiDAR point clouds for large-scale scene modeling. At the same time, the absence of a standardized definition and taxonomy for \"world models\" has led to fragmented and sometimes inconsistent claims in the literature. This survey addresses these gaps by presenting the first comprehensive review explicitly dedicated to 3D and 4D world modeling and generation. We establish precise definitions, introduce a structured taxonomy spanning video-based (VideoGen), occupancy-based (OccGen), and LiDAR-based (LiDARGen) approaches, and systematically summarize datasets and evaluation metrics tailored to 3D/4D settings. We further discuss practical applications, identify open challenges, and highlight promising research directions, aiming to provide a coherent and foundational reference for advancing the field. A systematic summary of existing literature is available at https://github.com/worldbench/awesome-3d-4d-world-models",
  "published": "2025-09-04",
  "updated": "2026-07-20",
  "year": "2025",
  "authors": [
   "Lingdong Kong",
   "Yu Yang",
   "Jianbiao Mei",
   "Youquan Liu",
   "Ao Liang",
   "Dekai Zhu",
   "Dongyue Lu",
   "Wei Yin",
   "Xiaotao Hu",
   "Mingkai Jia",
   "Junyuan Deng",
   "Kaiwen Zhang",
   "Yang Wu",
   "Tianyi Yan",
   "Shenyuan Gao",
   "Song Wang",
   "Linfeng Li",
   "Liang Pan",
   "Yong Liu",
   "Jianke Zhu",
   "Wei Tsang Ooi",
   "Steven C. H. Hoi",
   "Ziwei Liu"
  ],
  "author_count": 23,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 66,
  "influential_citations": 2,
  "tldr": "This survey addresses gaps in 3D and 4D world modeling and generation by establishing precise definitions, introducing a structured taxonomy spanning video-based (VideoGen), occupancy-based (OccGen), and LiDAR-based (LiDARGen) approaches, and systematically summarize datasets and evaluation metrics tailored to 3D/4D settings.",
  "doi": "10.48550/arXiv.2509.07996",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Lingdong Kong",
    "id": "2335574081",
    "h_index": 9,
    "papers": 34
   },
   {
    "name": "Wesley Yang",
    "id": "2379980485",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Jianbiao Mei",
    "id": "2106504583",
    "h_index": 16,
    "papers": 47
   },
   {
    "name": "Youquan Liu",
    "id": "2373551054",
    "h_index": 7,
    "papers": 15
   },
   {
    "name": "Ao Liang",
    "id": "2372674762",
    "h_index": 7,
    "papers": 21
   },
   {
    "name": "Dekai Zhu",
    "id": "2364574151",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Dongyue Lu",
    "id": "2334806165",
    "h_index": 8,
    "papers": 20
   },
   {
    "name": "Wei Yin",
    "id": "2323515472",
    "h_index": 10,
    "papers": 28
   },
   {
    "name": "Xiaotao Hu",
    "id": "2256224731",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Mingkai Jia",
    "id": "2311359143",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Junyuan Deng",
    "id": "2357641240",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Kaiwen Zhang",
    "id": "2372703092",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Yang Wu",
    "id": "2431340611",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Tianyi Yan",
    "id": "2331322100",
    "h_index": 6,
    "papers": 19
   },
   {
    "name": "Shenyuan Gao",
    "id": "2350220976",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Song Wang",
    "id": "2363511364",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Linfeng Li",
    "id": "2374980628",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Liang Pan",
    "id": "2362530555",
    "h_index": 7,
    "papers": 15
   },
   {
    "name": "Yong Liu",
    "id": "2323469565",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Jianke Zhu",
    "id": "2305333289",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "W. Ooi",
    "id": "2300285563",
    "h_index": 11,
    "papers": 35
   },
   {
    "name": "S. Hoi",
    "id": "2370937932",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Ziwei Liu",
    "id": "2277717448",
    "h_index": 12,
    "papers": 27
   }
  ],
  "comment": "Survey; Project Page at https://worldbench.github.io/survey GitHub Repo at https://github.com/worldbench/awesome-3d-4d-world-models",
  "topics": [
   "world-models",
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2509.07996v4",
  "pdf_url": "https://arxiv.org/pdf/2509.07996v4",
  "html_url": "https://arxiv.org/html/2509.07996v4",
  "code_url": "https://worldbench.github.io/survey",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.83
 },
 {
  "id": "2509.04443",
  "slug": "emma-scaling-mobile-manipulation-via-egocentric-human-data",
  "title": "EMMA: Scaling Mobile Manipulation via Egocentric Human Data",
  "abstract": "Scaling mobile manipulation imitation learning is bottlenecked by expensive mobile robot teleoperation. We present Egocentric Mobile MAnipulation (EMMA), an end-to-end framework training mobile manipulation policies from human mobile manipulation data with static robot data, sidestepping mobile teleoperation. To accomplish this, we co-train human full-body motion data with static robot data. In our experiments across three real-world tasks, EMMA demonstrates comparable performance to baselines trained on teleoperated mobile robot data (Mobile ALOHA), achieving higher or equivalent task performance in full task success. We find that EMMA is able to generalize to new spatial configurations and scenes, and we observe positive performance scaling as we increase the hours of human data, opening new avenues for scalable robotic learning in real-world environments. Details of this project can be found at https://ego-moma.github.io/.",
  "published": "2025-09-04",
  "updated": "2025-12-27",
  "year": "2025",
  "authors": [
   "Lawrence Y. Zhu",
   "Pranav Kuppili",
   "Ryan Punamiya",
   "Patcharapong Aphiwetsa",
   "Dhruv Patel",
   "Simar Kareer",
   "Sehoon Ha",
   "Danfei Xu"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "RA-L",
  "venue_source": "semantic-scholar",
  "citations": 35,
  "influential_citations": 2,
  "tldr": "Egocentric Mobile MAnipulation (EMMA), an end-to-end framework training mobile manipulation policies from human mobile manipulation data with static robot data, sidestepping mobile teleoperation is presented.",
  "doi": "10.1109/LRA.2026.3653320",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Lawrence Y. Zhu",
    "id": "2379184973",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Pranav Kuppili",
    "id": "2378954879",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Ryan Punamiya",
    "id": "2328411560",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Patcharapong Aphiwetsa",
    "id": "2310435059",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Dhruv Patel",
    "id": "2328566408",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Simar Kareer",
    "id": "2188833033",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Sehoon Ha",
    "id": "2378957452",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Danfei Xu",
    "id": "2264393671",
    "h_index": 4,
    "papers": 15
   }
  ],
  "comment": "",
  "topics": [
   "egocentric-data",
   "imitation-diffusion",
   "navigation",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2509.04443v3",
  "pdf_url": "https://arxiv.org/pdf/2509.04443v3",
  "html_url": "https://arxiv.org/html/2509.04443v3",
  "code_url": null,
  "presented": [
   {
    "club": "saturday-robotics",
    "club_name": "Saturday Robotics",
    "session_number": 8,
    "session_title": "Robotics & World Models Reading Club 08: Embodied Human Data as the \u201cInternet of Motion and Behavior\u201d \u2014 San Francisco 0516",
    "date_text": "Saturday, May 16, 2026 | 2:00 PM \u2013 5:00 PM",
    "city": "San Francisco, CA",
    "url": "https://lu.ma/qoxioge7",
    "listed_as": "Cross-Embodiment Policy Learning via Representation Alignment"
   },
   {
    "club": "saturday-robotics",
    "club_name": "Saturday Robotics",
    "session_number": 13,
    "session_title": "Robotics & World Models Reading Club 13: HumanEgo: Train Robot Policy from 30 min Egocentric Videos \u2014 SF 0620",
    "date_text": "Saturday, June 20, 2026 | 2:00 PM \u2013 5:00 PM",
    "city": "San Francisco, CA",
    "url": "https://lu.ma/6vkhxnum",
    "listed_as": ""
   }
  ],
  "club_note": "Bridges humans and robots through shared latent action representations using contrastive alignment and cycle-consistency objectives.",
  "featured": true,
  "signal": 6.06
 },
 {
  "id": "2509.00576",
  "slug": "galaxea-open-world-dataset-and-g0-dual-system-vla-model",
  "title": "Galaxea Open-World Dataset and G0 Dual-System VLA Model",
  "abstract": "We present Galaxea Open-World Dataset, a large-scale, diverse collection of robot behaviors recorded in authentic human living and working environments. All demonstrations are gathered using a consistent robotic embodiment, paired with precise subtask-level language annotations to facilitate both training and evaluation. Building on this dataset, we introduce G0, a dual-system framework that couples a Vision-Language Model (VLM) for multimodal planning with a Vision-Language-Action (VLA) model for fine-grained execution. G0 is trained using a three-stage curriculum: cross-embodiment pre-training, single-embodiment pre-training, and task-specific post-training. A comprehensive benchmark spanning tabletop manipulation, few-shot learning, and long-horizon mobile manipulation, demonstrates the effectiveness of our approach. In particular, we find that the single-embodiment pre-training stage, together with the Galaxea Open-World Dataset, plays a critical role in achieving strong performance.",
  "published": "2025-08-30",
  "updated": "2025-08-30",
  "year": "2025",
  "authors": [
   "Tao Jiang",
   "Tianyuan Yuan",
   "Yicheng Liu",
   "Chenhao Lu",
   "Jianning Cui",
   "Xiao Liu",
   "Shuiqi Cheng",
   "Jiyang Gao",
   "Huazhe Xu",
   "Hang Zhao"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 88,
  "influential_citations": 7,
  "tldr": "G0, a dual-system framework that couples a Vision-Language Model for multimodal planning with a Vision-Language-Action model for fine-grained execution, is introduced, finding that the single-embodiment pre-training stage plays a critical role in achieving strong performance.",
  "doi": "10.48550/arXiv.2509.00576",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Tao Jiang",
    "id": "2384816862",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Tianyuan Yuan",
    "id": "2214583235",
    "h_index": 11,
    "papers": 15
   },
   {
    "name": "Yicheng Liu",
    "id": "2284726857",
    "h_index": 7,
    "papers": 17
   },
   {
    "name": "Chenhao Lu",
    "id": "2265619607",
    "h_index": 8,
    "papers": 20
   },
   {
    "name": "Ji Cui",
    "id": "2087017690",
    "h_index": 15,
    "papers": 34
   },
   {
    "name": "Xiao Liu",
    "id": "2354276733",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Shuiqi Cheng",
    "id": "2312346725",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Jiyang Gao",
    "id": "1474350644",
    "h_index": 7,
    "papers": 8
   },
   {
    "name": "Huazhe Xu",
    "id": "2373743788",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Hang Zhao",
    "id": "2378973882",
    "h_index": 4,
    "papers": 8
   }
  ],
  "comment": "https://opengalaxea.github.io/G0/",
  "topics": [
   "vla",
   "navigation",
   "foundation-pretraining",
   "data-teleop",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2509.00576v1",
  "pdf_url": "https://arxiv.org/pdf/2509.00576v1",
  "html_url": "https://arxiv.org/html/2509.00576v1",
  "code_url": "https://opengalaxea.github.io/G0/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.95
 },
 {
  "id": "2508.21043",
  "slug": "hitter-a-humanoid-table-tennis-robot-via-hierarchical-planning-and-lea",
  "title": "HITTER: A HumanoId Table TEnnis Robot via Hierarchical Planning and Learning",
  "abstract": "Humanoid robots have recently achieved impressive progress in locomotion and whole-body control, yet they remain constrained in tasks that demand rapid interaction with dynamic environments through manipulation. Table tennis exemplifies such a challenge: with ball speeds exceeding 5 m/s, players must perceive, predict, and act within sub-second reaction times, requiring both agility and precision. To address this, we present a hierarchical framework for humanoid table tennis that integrates a model-based planner for ball trajectory prediction and racket target planning with a reinforcement learning-based whole-body controller. The planner determines striking position, velocity and timing, while the controller generates coordinated arm and leg motions that mimic human strikes and maintain stability and agility across consecutive rallies. Moreover, to encourage natural movements, human motion references are incorporated during training. We validate our system on a general-purpose humanoid robot, achieving up to 106 consecutive shots with a human opponent and sustained exchanges against another humanoid. These results demonstrate real-world humanoid table tennis with sub-second reactive control, marking a step toward agile and interactive humanoid behaviors.",
  "published": "2025-08-28",
  "updated": "2025-09-04",
  "year": "2025",
  "authors": [
   "Zhi Su",
   "Bike Zhang",
   "Nima Rahmanian",
   "Yuman Gao",
   "Qiayuan Liao",
   "Caitlin Regan",
   "Koushil Sreenath",
   "S. Shankar Sastry"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 58,
  "influential_citations": 4,
  "tldr": "This work presents a hierarchical framework for humanoid table tennis that integrates a model-based planner for ball trajectory prediction and racket target planning with a reinforcement learning-based whole-body controller, and demonstrates real-world humanoid table tennis with sub-second reactive control.",
  "doi": "10.48550/arXiv.2508.21043",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhi Su",
    "id": "2294040187",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Bike Zhang",
    "id": "4848974",
    "h_index": 16,
    "papers": 24
   },
   {
    "name": "Nima Rahmanian",
    "id": "2142358440",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Yuman Gao",
    "id": "2358305183",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Qiayuan Liao",
    "id": "1713616371",
    "h_index": 13,
    "papers": 20
   },
   {
    "name": "Caitlin Regan",
    "id": "2377794724",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "K. Sreenath",
    "id": "144116765",
    "h_index": 55,
    "papers": 231
   },
   {
    "name": "S. Sastry",
    "id": "2325954829",
    "h_index": 4,
    "papers": 5
   }
  ],
  "comment": "add more references",
  "topics": [
   "humanoids",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2508.21043v2",
  "pdf_url": "https://arxiv.org/pdf/2508.21043v2",
  "html_url": "https://arxiv.org/html/2508.21043v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.77
 },
 {
  "id": "2508.13104",
  "slug": "precise-action-to-video-generation-through-visual-action-prompts",
  "title": "Precise Action-to-Video Generation Through Visual Action Prompts",
  "abstract": "We present visual action prompts, a unified action representation for action-to-video generation of complex high-DoF interactions while maintaining transferable visual dynamics across domains. Action-driven video generation faces a precision-generality trade-off: existing methods using text, primitive actions, or coarse masks offer generality but lack precision, while agent-centric action signals provide precision at the cost of cross-domain transferability. To balance action precision and dynamic transferability, we propose to \"render\" actions into precise visual prompts as domain-agnostic representations that preserve both geometric precision and cross-domain adaptability for complex actions; specifically, we choose visual skeletons for their generality and accessibility. We propose robust pipelines to construct skeletons from two interaction-rich data sources - human-object interactions (HOI) and dexterous robotic manipulation - enabling cross-domain training of action-driven generative models. By integrating visual skeletons into pretrained video generation models via lightweight fine-tuning, we enable precise action control of complex interaction while preserving the learning of cross-domain dynamics. Experiments on EgoVid, RT-1 and DROID demonstrate the effectiveness of our proposed approach. Project page: https://zju3dv.github.io/VAP/.",
  "published": "2025-08-18",
  "updated": "2025-08-18",
  "year": "2025",
  "authors": [
   "Yuang Wang",
   "Chao Wen",
   "Haoyu Guo",
   "Sida Peng",
   "Minghan Qin",
   "Hujun Bao",
   "Xiaowei Zhou",
   "Ruizhen Hu"
  ],
  "author_count": 8,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "ICCV",
  "venue_source": "semantic-scholar",
  "citations": 22,
  "influential_citations": 2,
  "tldr": "This work proposes robust pipelines to construct skeletons from two interaction-rich data sources - human-object interactions (HOI) and dexterous robotic manipulation - enabling cross-domain training of actiondriven generative models, and integrates visual skeletons into pretrained video generation models via lightweight finetuning.",
  "doi": "10.1109/ICCV51701.2025.01181",
  "oa_pdf": "https://arxiv.org/pdf/2508.13104",
  "s2_authors": [
   {
    "name": "Yuang Wang",
    "id": "37075603",
    "h_index": 9,
    "papers": 23
   },
   {
    "name": "Chao Wen",
    "id": "2274104560",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Haoyu Guo",
    "id": "153611272",
    "h_index": 7,
    "papers": 12
   },
   {
    "name": "Sida Peng",
    "id": "2072712025",
    "h_index": 41,
    "papers": 95
   },
   {
    "name": "Minghan Qin",
    "id": "2376193554",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Hujun Bao",
    "id": "2253405497",
    "h_index": 16,
    "papers": 42
   },
   {
    "name": "Xiaowei Zhou",
    "id": "2238207469",
    "h_index": 20,
    "papers": 56
   },
   {
    "name": "Ruizhen Hu",
    "id": "2274775115",
    "h_index": 7,
    "papers": 20
   }
  ],
  "comment": "Accepted to ICCV 2025. Project page: https://zju3dv.github.io/VAP/",
  "topics": [
   "dexterous-manipulation",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2508.13104v1",
  "pdf_url": "https://arxiv.org/pdf/2508.13104v1",
  "html_url": "https://arxiv.org/html/2508.13104v1",
  "code_url": "https://zju3dv.github.io/VAP/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.86
 },
 {
  "id": "2508.13013",
  "slug": "egotwin-dreaming-body-and-view-in-first-person",
  "title": "EgoTwin: Dreaming Body and View in First Person",
  "abstract": "While exocentric video synthesis has achieved great progress, egocentric video generation remains largely underexplored, which requires modeling first-person view content along with camera motion patterns induced by the wearer's body movements. To bridge this gap, we introduce a novel task of joint egocentric video and human motion generation, characterized by two key challenges: 1) Viewpoint Alignment: the camera trajectory in the generated video must accurately align with the head trajectory derived from human motion; 2) Causal Interplay: the synthesized human motion must causally align with the observed visual dynamics across adjacent video frames. To address these challenges, we propose EgoTwin, a joint video-motion generation framework built on the diffusion transformer architecture. Specifically, EgoTwin introduces a head-centric motion representation that anchors the human motion to the head joint and incorporates a cybernetics-inspired interaction mechanism that explicitly captures the causal interplay between video and motion within attention operations. For comprehensive evaluation, we curate a large-scale real-world dataset of synchronized text-video-motion triplets and design novel metrics to assess video-motion consistency. Extensive experiments demonstrate the effectiveness of the EgoTwin framework.",
  "published": "2025-08-18",
  "updated": "2025-08-18",
  "year": "2025",
  "authors": [
   "Jingqiao Xiu",
   "Fangzhou Hong",
   "Yicong Li",
   "Mengze Li",
   "Wentao Wang",
   "Sirui Han",
   "Liang Pan",
   "Ziwei Liu"
  ],
  "author_count": 8,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 13,
  "influential_citations": 1,
  "tldr": "EgoTwin introduces a head-centric motion representation that anchors the human motion to the head joint and incorporates a cybernetics-inspired interaction mechanism that explicitly captures the causal interplay between video and motion within attention operations.",
  "doi": "10.48550/arXiv.2508.13013",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jingqiao Xiu",
    "id": "2327912414",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Fangzhou Hong",
    "id": "1568986485",
    "h_index": 24,
    "papers": 47
   },
   {
    "name": "Yicong Li",
    "id": "2135358934",
    "h_index": 19,
    "papers": 32
   },
   {
    "name": "Mengze Li",
    "id": "2355337672",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Wentao Wang",
    "id": "2385499506",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Sirui Han",
    "id": "2373584517",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Liang Pan",
    "id": "2362530555",
    "h_index": 7,
    "papers": 15
   },
   {
    "name": "Ziwei Liu",
    "id": "2294735512",
    "h_index": 10,
    "papers": 16
   }
  ],
  "comment": "",
  "topics": [
   "egocentric-data",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2508.13013v1",
  "pdf_url": "https://arxiv.org/pdf/2508.13013v1",
  "html_url": "https://arxiv.org/html/2508.13013v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.15
 },
 {
  "id": "2508.13009",
  "slug": "matrix-game-2-0-an-open-source-real-time-and-streaming-interactive-wor",
  "title": "Matrix-game 2.0: An open-source real-time and streaming interactive world model",
  "abstract": "Recent advances in interactive video generations have demonstrated diffusion model's potential as world models by capturing complex physical dynamics and interactive behaviors. However, existing interactive world models depend on bidirectional attention and lengthy inference steps, severely limiting real-time performance. Consequently, they are hard to simulate real-world dynamics, where outcomes must update instantaneously based on historical context and current actions. To address this, we present Matrix-Game 2.0, an interactive world model generates long videos on-the-fly via few-step auto-regressive diffusion. Our framework consists of three key components: (1) A scalable data production pipeline for Unreal Engine and GTA5 environments to effectively produce massive amounts (about 1200 hours) of video data with diverse interaction annotations; (2) An action injection module that enables frame-level mouse and keyboard inputs as interactive conditions; (3) A few-step distillation based on the casual architecture for real-time and streaming video generation. Matrix Game 2.0 can generate high-quality minute-level videos across diverse scenes at an ultra-fast speed of 25 FPS. We open-source our model weights and codebase to advance research in interactive world modeling.",
  "published": "2025-08-18",
  "updated": "2026-04-07",
  "year": "2025",
  "authors": [
   "Xianglong He",
   "Chunli Peng",
   "Zexiang Liu",
   "Boyang Wang",
   "Yifan Zhang",
   "Qi Cui",
   "Fei Kang",
   "Biao Jiang",
   "Mengyin An",
   "Yangyang Ren",
   "Baixin Xu",
   "Hao-Xiang Guo",
   "Kaixiong Gong",
   "Size Wu",
   "Wei Li",
   "Xuchen Song",
   "Yang Liu",
   "Yangguang Li",
   "Yahui Zhou"
  ],
  "author_count": 19,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 140,
  "influential_citations": 41,
  "tldr": "",
  "doi": "10.48550/arXiv.2508.13009",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xianglong He",
    "id": "2257432343",
    "h_index": 7,
    "papers": 17
   },
   {
    "name": "Chunli Peng",
    "id": "2371014883",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Zexiang Liu",
    "id": "2376510137",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Boyang Wang",
    "id": "2371132207",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Yifan Zhang",
    "id": "2325930112",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Qiushi Cui",
    "id": "2302784678",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Fei Kang",
    "id": "2370936390",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Biao Jiang",
    "id": "2371157763",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Mengyin An",
    "id": "2375389951",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Yang Ren",
    "id": "2382776742",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Baixin Xu",
    "id": "2375392473",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Hao Guo",
    "id": "2243372903",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Kaixiong Gong",
    "id": "2325144186",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Cyrus Wu",
    "id": "2376206595",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Wei Li",
    "id": "2347897200",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Xuchen Song",
    "id": "2354290518",
    "h_index": 10,
    "papers": 13
   },
   {
    "name": "Yang Liu",
    "id": "2375532570",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Eric Li",
    "id": "2375390476",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Yahui Zhou",
    "id": "2263493641",
    "h_index": 16,
    "papers": 25
   }
  ],
  "comment": "Project Page: https://matrix-game-v2.github.io",
  "topics": [
   "world-models",
   "data-teleop",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2508.13009v4",
  "pdf_url": "https://arxiv.org/pdf/2508.13009v4",
  "html_url": "https://arxiv.org/html/2508.13009v4",
  "code_url": "https://matrix-game-v2.github.io",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.15
 },
 {
  "id": "2508.11049",
  "slug": "genflowrl-shaping-rewards-with-generative-object-centric-flow-in-visua",
  "title": "GenFlowRL: Shaping Rewards with Generative Object-Centric Flow in Visual Reinforcement Learning",
  "abstract": "Recent advances have shown that video generation models can enhance robot learning by deriving effective robot actions through inverse dynamics. However, these methods heavily depend on the quality of generated data and struggle with fine-grained manipulation due to the lack of environment feedback. While video-based reinforcement learning improves policy robustness, it remains constrained by the uncertainty of video generation and the challenges of collecting large-scale robot datasets for training diffusion models. To address these limitations, we propose GenFlowRL, which derives shaped rewards from generated flow trained from diverse cross-embodiment datasets. This enables learning generalizable and robust policies from diverse demonstrations using low-dimensional, object-centric features. Experiments on 10 manipulation tasks, both in simulation and real-world cross-embodiment evaluations, demonstrate that GenFlowRL effectively leverages manipulation features extracted from generated object-centric flow, consistently achieving superior performance across diverse and challenging scenarios. Our Project Page: https://colinyu1.github.io/genflowrl",
  "published": "2025-08-14",
  "updated": "2025-08-14",
  "year": "2025",
  "authors": [
   "Kelin Yu",
   "Sheng Zhang",
   "Harshit Soora",
   "Furong Huang",
   "Heng Huang",
   "Pratap Tokekar",
   "Ruohan Gao"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "ICCV",
  "venue_source": "semantic-scholar",
  "citations": 10,
  "influential_citations": 1,
  "tldr": "GenFlowRL is proposed, which derives shaped rewards from generated flow trained from diverse cross-embodiment datasets, which enables learning generalizable and robust policies from diverse demonstrations using low-dimensional, object-centric features.",
  "doi": "10.1109/ICCV51701.2025.01225",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Kelin Yu",
    "id": "2376516296",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Sheng Zhang",
    "id": "2384868863",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Harshit Soora",
    "id": "2058394795",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Furong Huang",
    "id": "2288331810",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Heng Huang",
    "id": "2391566480",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Pratap Tokekar",
    "id": "2390456",
    "h_index": 31,
    "papers": 192
   },
   {
    "name": "Ruohan Gao",
    "id": "2376472371",
    "h_index": 2,
    "papers": 4
   }
  ],
  "comment": "Published at ICCV 2025",
  "topics": [
   "rl-control",
   "foundation-pretraining",
   "video-generation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2508.11049v1",
  "pdf_url": "https://arxiv.org/pdf/2508.11049v1",
  "html_url": "https://arxiv.org/html/2508.11049v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.54
 },
 {
  "id": "2508.10898",
  "slug": "puppeteer-rig-and-animate-your-3d-models",
  "title": "Puppeteer: Rig and Animate Your 3D Models",
  "abstract": "Modern interactive applications increasingly demand dynamic 3D content, yet the transformation of static 3D models into animated assets constitutes a significant bottleneck in content creation pipelines. While recent advances in generative AI have revolutionized static 3D model creation, rigging and animation continue to depend heavily on expert intervention. We present Puppeteer, a comprehensive framework that addresses both automatic rigging and animation for diverse 3D objects. Our system first predicts plausible skeletal structures via an auto-regressive transformer that introduces a joint-based tokenization strategy for compact representation and a hierarchical ordering methodology with stochastic perturbation that enhances bidirectional learning capabilities. It then infers skinning weights via an attention-based architecture incorporating topology-aware joint attention that explicitly encodes inter-joint relationships based on skeletal graph distances. Finally, we complement these rigging advances with a differentiable optimization-based animation pipeline that generates stable, high-fidelity animations while being computationally more efficient than existing approaches. Extensive evaluations across multiple benchmarks demonstrate that our method significantly outperforms state-of-the-art techniques in both skeletal prediction accuracy and skinning quality. The system robustly processes diverse 3D content, ranging from professionally designed game assets to AI-generated shapes, producing temporally coherent animations that eliminate the jittering issues common in existing methods.",
  "published": "2025-08-14",
  "updated": "2025-08-14",
  "year": "2025",
  "authors": [
   "Chaoyue Song",
   "Xiu Li",
   "Fan Yang",
   "Zhongcong Xu",
   "Jiacheng Wei",
   "Fayao Liu",
   "Jiashi Feng",
   "Guosheng Lin",
   "Jianfeng Zhang"
  ],
  "author_count": 9,
  "categories": [
   "cs.CV",
   "cs.GR"
  ],
  "primary_category": "cs.CV",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 43,
  "influential_citations": 11,
  "tldr": "Puppeteer is presented, a comprehensive framework that addresses both automatic rigging and animation for diverse 3D objects, and complements these rigging advances with a differentiable optimization-based animation pipeline that generates stable, high-fidelity animations while being computationally more efficient than existing approaches.",
  "doi": "10.48550/arXiv.2508.10898",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Chaoyue Song",
    "id": "1388734273",
    "h_index": 8,
    "papers": 15
   },
   {
    "name": "Xiu Li",
    "id": "2337098657",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Fan Yang",
    "id": "2268853626",
    "h_index": 6,
    "papers": 14
   },
   {
    "name": "Zhongcong Xu",
    "id": "143710117",
    "h_index": 12,
    "papers": 20
   },
   {
    "name": "Jiacheng Wei",
    "id": "145183703",
    "h_index": 11,
    "papers": 25
   },
   {
    "name": "Fayao Liu",
    "id": "2146423021",
    "h_index": 19,
    "papers": 54
   },
   {
    "name": "Jiashi Feng",
    "id": "2255521016",
    "h_index": 12,
    "papers": 20
   },
   {
    "name": "Guosheng Lin",
    "id": "2268646778",
    "h_index": 8,
    "papers": 18
   },
   {
    "name": "Jianfeng Zhang",
    "id": "2107968449",
    "h_index": 21,
    "papers": 34
   }
  ],
  "comment": "Project page: https://chaoyuesong.github.io/Puppeteer/",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2508.10898v1",
  "pdf_url": "https://arxiv.org/pdf/2508.10898v1",
  "html_url": "https://arxiv.org/html/2508.10898v1",
  "code_url": "https://chaoyuesong.github.io/Puppeteer/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.14
 },
 {
  "id": "2508.10104",
  "slug": "dinov3",
  "title": "DINOv3",
  "abstract": "Self-supervised learning holds the promise of eliminating the need for manual data annotation, enabling models to scale effortlessly to massive datasets and larger architectures. By not being tailored to specific tasks or domains, this training paradigm has the potential to learn visual representations from diverse sources, ranging from natural to aerial images -- using a single algorithm. This technical report introduces DINOv3, a major milestone toward realizing this vision by leveraging simple yet effective strategies. First, we leverage the benefit of scaling both dataset and model size by careful data preparation, design, and optimization. Second, we introduce a new method called Gram anchoring, which effectively addresses the known yet unsolved issue of dense feature maps degrading during long training schedules. Finally, we apply post-hoc strategies that further enhance our models' flexibility with respect to resolution, model size, and alignment with text. As a result, we present a versatile vision foundation model that outperforms the specialized state of the art across a broad range of settings, without fine-tuning. DINOv3 produces high-quality dense features that achieve outstanding performance on various vision tasks, significantly surpassing previous self- and weakly-supervised foundation models. We also share the DINOv3 suite of vision models, designed to advance the state of the art on a wide spectrum of tasks and data by providing scalable solutions for diverse resource constraints and deployment scenarios.",
  "published": "2025-08-13",
  "updated": "2025-08-13",
  "year": "2025",
  "authors": [
   "Oriane Sim\u00e9oni",
   "Huy V. Vo",
   "Maximilian Seitzer",
   "Federico Baldassarre",
   "Maxime Oquab",
   "Cijo Jose",
   "Vasil Khalidov",
   "Marc Szafraniec",
   "Seungeun Yi",
   "Micha\u00ebl Ramamonjisoa",
   "Francisco Massa",
   "Daniel Haziza",
   "Luca Wehrstedt",
   "Jianyuan Wang",
   "Timoth\u00e9e Darcet",
   "Th\u00e9o Moutakanni",
   "Leonel Sentana",
   "Claire Roberts",
   "Andrea Vedaldi",
   "Jamie Tolan",
   "John Brandt",
   "Camille Couprie",
   "Julien Mairal",
   "Herv\u00e9 J\u00e9gou",
   "Patrick Labatut",
   "Piotr Bojanowski"
  ],
  "author_count": 26,
  "categories": [
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 1312,
  "influential_citations": 229,
  "tldr": "This technical report introduces DINOv3, a major milestone toward realizing the promise of eliminating the need for manual data annotation, and introduces a new method called Gram anchoring, which effectively addresses the known yet unsolved issue of dense feature maps degrading during long training schedules.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Oriane Sim\u00e9oni",
    "id": "2279694761",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "Huy V. Vo",
    "id": "3323377",
    "h_index": 14,
    "papers": 31
   },
   {
    "name": "Maximilian Seitzer",
    "id": "2375900468",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Federico Baldassarre",
    "id": "2336864698",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Maxime Oquab",
    "id": "2248163611",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "Cijo Jose",
    "id": "2336864882",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Vasil Khalidov",
    "id": "2067158687",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Marc Szafraniec",
    "id": "23994377",
    "h_index": 11,
    "papers": 17
   },
   {
    "name": "Seungeun Yi",
    "id": "2376093138",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Michael Ramamonjisoa",
    "id": "103917092",
    "h_index": 10,
    "papers": 22
   },
   {
    "name": "Francisco Massa",
    "id": "2345005489",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Daniel Haziza",
    "id": "40864100",
    "h_index": 11,
    "papers": 24
   },
   {
    "name": "Luca Wehrstedt",
    "id": "2331511165",
    "h_index": 8,
    "papers": 16
   },
   {
    "name": "Jianyuan Wang",
    "id": "2271268581",
    "h_index": 13,
    "papers": 24
   },
   {
    "name": "Timoth\u00e9e Darcet",
    "id": "2214523349",
    "h_index": 7,
    "papers": 7
   },
   {
    "name": "Th\u00e9o Moutakanni",
    "id": "1752699898",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Leonel Sentana",
    "id": "2375900262",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Claire Roberts",
    "id": "2275844640",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Andrea Vedaldi",
    "id": "2309477007",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Jamie Tolan",
    "id": "2315316051",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "John Brandt",
    "id": "2315312806",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Camille Couprie",
    "id": "2292200003",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "J. Mairal",
    "id": "2599292",
    "h_index": 58,
    "papers": 162
   },
   {
    "name": "Herv\u00e9 J\u00e9gou",
    "id": "2202919559",
    "h_index": 7,
    "papers": 12
   },
   {
    "name": "Patrick Labatut",
    "id": "1744868",
    "h_index": 20,
    "papers": 38
   },
   {
    "name": "Piotr Bojanowski",
    "id": "2329288",
    "h_index": 38,
    "papers": 76
   }
  ],
  "comment": "",
  "topics": [
   "foundation-pretraining",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2508.10104v1",
  "pdf_url": "https://arxiv.org/pdf/2508.10104v1",
  "html_url": "https://arxiv.org/html/2508.10104v1",
  "code_url": null,
  "presented": [
   {
    "club": "saturday-robotics",
    "club_name": "Saturday Robotics",
    "session_number": 9,
    "session_title": "Robotics & World Models Reading Club 09: CVPR Warm-up & Founders Spotlight \u2014 DeltaWorld + VisuoTactile Dexterous Hands | San Francisco 0523",
    "date_text": "Saturday, May 23, 2026 | 2:00 PM \u2013 5:00 PM",
    "city": "San Francisco, CA",
    "url": "https://lu.ma/wooiz0bf",
    "listed_as": "A Frame is Worth One Token: Efficient Generative World Modeling with Delta Tokens"
   }
  ],
  "club_note": "Keynote 2 by by Arjun Subramaniam (Factory Intelligence)",
  "featured": true,
  "signal": 6.0
 },
 {
  "id": "2508.09976",
  "slug": "masquerade-learning-from-in-the-wild-human-videos-using-data-editing",
  "title": "Masquerade: Learning from In-the-wild Human Videos using Data-Editing",
  "abstract": "Robot manipulation research still suffers from significant data scarcity: even the largest robot datasets are orders of magnitude smaller and less diverse than those that fueled recent breakthroughs in language and vision. We introduce Masquerade, a method that edits in-the-wild egocentric human videos to bridge the visual embodiment gap between humans and robots and then learns a robot policy with these edited videos. Our pipeline turns each human video into robotized demonstrations by (i) estimating 3-D hand poses, (ii) inpainting the human arms, and (iii) overlaying a rendered bimanual robot that tracks the recovered end-effector trajectories. Pre-training a visual encoder to predict future 2-D robot keypoints on 675K frames of these edited clips, and continuing that auxiliary loss while fine-tuning a diffusion policy head on only 50 robot demonstrations per task, yields policies that generalize significantly better than prior work. On three long-horizon, bimanual kitchen tasks evaluated in three unseen scenes each, Masquerade outperforms baselines by 5-6x. Ablations show that both the robot overlay and co-training are indispensable, and performance scales logarithmically with the amount of edited human video. These results demonstrate that explicitly closing the visual embodiment gap unlocks a vast, readily available source of data from human videos that can be used to improve robot policies.",
  "published": "2025-08-13",
  "updated": "2025-08-13",
  "year": "2025",
  "authors": [
   "Marion Lepert",
   "Jiaying Fang",
   "Jeannette Bohg"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 53,
  "influential_citations": 4,
  "tldr": "",
  "doi": "10.48550/arXiv.2508.09976",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Marion Lepert",
    "id": "10710717",
    "h_index": 8,
    "papers": 11
   },
   {
    "name": "Jiaying Fang",
    "id": "2344586943",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Jeannette Bohg",
    "id": "2323565347",
    "h_index": 6,
    "papers": 12
   }
  ],
  "comment": "Project website at https://masquerade-robot.github.io/",
  "topics": [
   "egocentric-data",
   "imitation-diffusion",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2508.09976v1",
  "pdf_url": "https://arxiv.org/pdf/2508.09976v1",
  "html_url": "https://arxiv.org/html/2508.09976v1",
  "code_url": "https://masquerade-robot.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.73
 },
 {
  "id": "2508.09071",
  "slug": "geovla-empowering-3d-representations-in-vision-language-action-models",
  "title": "GeoVLA: Empowering 3D Representations in Vision-Language-Action Models",
  "abstract": "Vision-Language-Action (VLA) models have emerged as a promising approach for enabling robots to follow language instructions and predict corresponding actions. However, current VLA models mainly rely on 2D visual inputs, neglecting the rich geometric information in the 3D physical world, which limits their spatial awareness and adaptability. In this paper, we present GeoVLA, a novel VLA framework that effectively integrates 3D information to advance robotic manipulation. It uses a vision-language model (VLM) to process images and language instructions,extracting fused vision-language embeddings. In parallel, it converts depth maps into point clouds and employs a customized point encoder, called Point Embedding Network, to generate 3D geometric embeddings independently. These produced embeddings are then concatenated and processed by our proposed spatial-aware action expert, called 3D-enhanced Action Expert, which combines information from different sensor modalities to produce precise action sequences. Through extensive experiments in both simulation and real-world environments, GeoVLA demonstrates superior performance and robustness. It achieves state-of-the-art results in the LIBERO and ManiSkill2 simulation benchmarks and shows remarkable robustness in real-world tasks requiring height adaptability, scale awareness and viewpoint invariance.",
  "published": "2025-08-12",
  "updated": "2025-08-13",
  "year": "2025",
  "authors": [
   "Lin Sun",
   "Bin Xie",
   "Yingfei Liu",
   "Hao Shi",
   "Tiancai Wang",
   "Jiale Cao"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 58,
  "influential_citations": 4,
  "tldr": "GeoVLA is a novel VLA framework that effectively integrates 3D information to advance robotic manipulation and achieves state-of-the-art results in the LIBERO and ManiSkill2 simulation benchmarks and shows remarkable robustness in real-world tasks requiring height adaptability, scale awareness and viewpoint invariance.",
  "doi": "10.48550/arXiv.2508.09071",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Lin Sun",
    "id": "2376007623",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Bin Xie",
    "id": "2268408921",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Yingfei Liu",
    "id": "2282235642",
    "h_index": 17,
    "papers": 26
   },
   {
    "name": "Hao Shi",
    "id": "2366286284",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "Tiancai Wang",
    "id": "2325923837",
    "h_index": 10,
    "papers": 30
   },
   {
    "name": "Jiale Cao",
    "id": "2348278104",
    "h_index": 2,
    "papers": 3
   }
  ],
  "comment": "The project is visible at https://linsun449.github.io/GeoVLA/",
  "topics": [
   "vla",
   "sim2real",
   "spatial-3d",
   "video-generation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2508.09071v2",
  "pdf_url": "https://arxiv.org/pdf/2508.09071v2",
  "html_url": "https://arxiv.org/html/2508.09071v2",
  "code_url": "https://linsun449.github.io/GeoVLA/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.77
 },
 {
  "id": "2508.08241",
  "slug": "beyondmimic-from-motion-tracking-to-versatile-humanoid-control-via-gui",
  "title": "BeyondMimic: From Motion Tracking to Versatile Humanoid Control via Guided Diffusion",
  "abstract": "The human-like form of humanoid robots positions them uniquely to achieve the agility and versatility in motor skills that humans possess. Learning from human demonstrations offers a scalable approach to acquiring these capabilities. However, prior works either produce unnatural motions or rely on motion-specific tuning to achieve satisfactory naturalness. Furthermore, these methods are often motion- or goal-specific, lacking the versatility to compose diverse skills, especially when solving unseen tasks. We present BeyondMimic, a framework that scales to diverse motions and carries the versatility to compose them seamlessly in tackling unseen downstream tasks. At heart, a compact motion-tracking formulation enables mastering a wide range of radically agile behaviors, including aerial cartwheels, spin-kicks, flip-kicks, and sprinting, with a single setup and shared hyperparameters, all while achieving state-of-the-art human-like performance. Moving beyond the mere imitation of existing motions, we propose a unified latent diffusion model that empowers versatile goal specification, seamless task switching, and dynamic composition of these agile behaviors. Leveraging classifier guidance, a diffusion-specific technique for test-time optimization toward novel objectives, our model extends its capability to solve downstream tasks never encountered during training, including motion inpainting, joystick teleoperation, and obstacle avoidance, and transfers these skills zero-shot to real hardware. This work opens new frontiers for humanoid robots by pushing the limits of scalable human-like motor skill acquisition from human motion and advancing seamless motion synthesis that achieves generalization and versatility beyond training setups.",
  "published": "2025-08-11",
  "updated": "2025-11-13",
  "year": "2025",
  "authors": [
   "Qiayuan Liao",
   "Takara E. Truong",
   "Xiaoyu Huang",
   "Yuman Gao",
   "Guy Tevet",
   "Koushil Sreenath",
   "C. Karen Liu"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 248,
  "influential_citations": 32,
  "tldr": "BeyondMimic is presented, a framework that scales to diverse motions and carries the versatility to compose them seamlessly in tackling unseen downstream tasks, and extends its capability to solve downstream tasks never encountered during training, including motion inpainting, joystick teleoperation, and obstacle avoidance.",
  "doi": "10.48550/arXiv.2508.08241",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Qiayuan Liao",
    "id": "1713616371",
    "h_index": 13,
    "papers": 20
   },
   {
    "name": "Takara Truong",
    "id": "2369918131",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Xiaoyu Huang",
    "id": "2293553057",
    "h_index": 9,
    "papers": 10
   },
   {
    "name": "Guy Tevet",
    "id": "81493694",
    "h_index": 12,
    "papers": 19
   },
   {
    "name": "K. Sreenath",
    "id": "144116765",
    "h_index": 55,
    "papers": 231
   },
   {
    "name": "C. K. Liu",
    "id": "2376138723",
    "h_index": 9,
    "papers": 11
   }
  ],
  "comment": "Project page: https://beyondmimic.github.io/",
  "topics": [
   "humanoids",
   "egocentric-data",
   "navigation",
   "data-teleop",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2508.08241v4",
  "pdf_url": "https://arxiv.org/pdf/2508.08241v4",
  "html_url": "https://arxiv.org/html/2508.08241v4",
  "code_url": "https://beyondmimic.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.4
 },
 {
  "id": "2508.05635",
  "slug": "genie-envisioner-a-unified-world-foundation-platform-for-robotic-manip",
  "title": "Genie Envisioner: A Unified World Foundation Platform for Robotic Manipulation",
  "abstract": "We introduce Genie Envisioner (GE), a unified world foundation platform for robotic manipulation that integrates policy learning, evaluation, and simulation within a single video-generative framework. At its core, GE-Base is a large-scale, instruction-conditioned video diffusion model that captures the spatial, temporal, and semantic dynamics of real-world robotic interactions in a structured latent space. Built upon this foundation, GE-Act maps latent representations to executable action trajectories through a lightweight, flow-matching decoder, enabling precise and generalizable policy inference across diverse embodiments with minimal supervision. To support scalable evaluation and training, GE-Sim serves as an action-conditioned neural simulator, producing high-fidelity rollouts for closed-loop policy development. The platform is further equipped with EWMBench, a standardized benchmark suite measuring visual fidelity, physical consistency, and instruction-action alignment. Together, these components establish Genie Envisioner as a scalable and practical foundation for instruction-driven, general-purpose embodied intelligence. All code, models, and benchmarks will be released publicly.",
  "published": "2025-08-07",
  "updated": "2025-11-04",
  "year": "2025",
  "authors": [
   "Yue Liao",
   "Pengfei Zhou",
   "Siyuan Huang",
   "Donglin Yang",
   "Shengcong Chen",
   "Yuxin Jiang",
   "Yue Hu",
   "Jingbin Cai",
   "Si Liu",
   "Jianlan Luo",
   "Liliang Chen",
   "Shuicheng Yan",
   "Maoqing Yao",
   "Guanghui Ren"
  ],
  "author_count": 14,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 124,
  "influential_citations": 19,
  "tldr": "Genie Envisioner is introduced, a unified world foundation platform for robotic manipulation that integrates policy learning, evaluation, and simulation within a single video-generative framework, and is equipped with EWMBench, a standardized benchmark suite measuring visual fidelity, physical consistency, and instruction-action alignment.",
  "doi": "10.48550/arXiv.2508.05635",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yue Liao",
    "id": "47303356",
    "h_index": 18,
    "papers": 26
   },
   {
    "name": "Yue Liao",
    "id": "2362056474",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Pengfei Zhou",
    "id": "2338818174",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Siyuan Huang",
    "id": "2361492506",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Donglin Yang",
    "id": "2239165347",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Shengcong Chen",
    "id": "2339699627",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Yuxin Jiang",
    "id": "2361654632",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Hu Yue",
    "id": "2361506430",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Jingbin Cai",
    "id": "2375770066",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Si Liu",
    "id": "2291108679",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Jianlan Luo",
    "id": "2349439364",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Liliang Chen",
    "id": "2338767348",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Shuicheng Yan",
    "id": "2287697658",
    "h_index": 6,
    "papers": 14
   },
   {
    "name": "Maoqing Yao",
    "id": "2338695140",
    "h_index": 12,
    "papers": 22
   },
   {
    "name": "Guanghui Ren",
    "id": "2338694069",
    "h_index": 13,
    "papers": 30
   }
  ],
  "comment": "https://genie-envisioner.github.io/",
  "topics": [
   "sim2real",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2508.05635v3",
  "pdf_url": "https://arxiv.org/pdf/2508.05635v3",
  "html_url": "https://arxiv.org/html/2508.05635v3",
  "code_url": "https://genie-envisioner.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.1
 },
 {
  "id": "2508.03218",
  "slug": "actionsink-toward-precise-robot-manipulation-with-dynamic-integration",
  "title": "ActionSink: Toward Precise Robot Manipulation with Dynamic Integration of Action Flow",
  "abstract": "Language-instructed robot manipulation has garnered significant interest due to the potential of learning from collected data. While the challenges in high-level perception and planning are continually addressed along the progress of general large pre-trained models, the low precision of low-level action estimation has emerged as the key limiting factor in manipulation performance. To this end, this paper introduces a novel robot manipulation framework, i.e., ActionSink, to pave the way toward precise action estimations in the field of learning-based robot manipulation. As the name suggests, ActionSink reformulates the actions of robots as action-caused optical flows from videos, called \"action flow\", in a self-supervised manner, which are then used to be retrieved and integrated to enhance the action estimation. Specifically, ActionSink incorporates two primary modules. The first module is a coarse-to-fine action flow matcher, which continuously refines the accuracy of action flow via iterative retrieval and denoising process. The second module is a dynamic action flow integrator, which employs a working memory pool that dynamically and efficiently manages the historical action flows that should be used to integrate to enhance the current action estimation. In this module, a multi-layer fusion module is proposed to integrate direct estimation and action flows from both the current and the working memory, achieving highly accurate action estimation through a series of estimation-integration processes. Our ActionSink framework outperformed prior SOTA on the LIBERO benchmark by a 7.9\\% success rate, and obtained nearly an 8\\% accuracy gain on the challenging long-horizon visual task LIBERO-Long.",
  "published": "2025-08-05",
  "updated": "2025-08-05",
  "year": "2025",
  "authors": [
   "Shanshan Guo",
   "Xiwen Liang",
   "Junfan Lin",
   "Yuzheng Zhuang",
   "Liang Lin",
   "Xiaodan Liang"
  ],
  "author_count": 6,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 1,
  "influential_citations": 0,
  "tldr": "A novel robot manipulation framework that reformulates the actions of robots as action-caused optical flows from videos, called ActionSink, in a self-supervised manner, to pave the way toward precise action estimations in the field of learning-based robot manipulation.",
  "doi": "10.48550/arXiv.2508.03218",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Shanshan Guo",
    "id": "2278586183",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Xiwen Liang",
    "id": "51291599",
    "h_index": 11,
    "papers": 26
   },
   {
    "name": "Junfan Lin",
    "id": "2326803482",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Yuzheng Zhuang",
    "id": "8773733",
    "h_index": 13,
    "papers": 49
   },
   {
    "name": "Liang Lin",
    "id": "2286203284",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Xiaodan Liang",
    "id": "2279329380",
    "h_index": 4,
    "papers": 8
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2508.03218v1",
  "pdf_url": "https://arxiv.org/pdf/2508.03218v1",
  "html_url": "https://arxiv.org/html/2508.03218v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.3
 },
 {
  "id": "2508.03068",
  "slug": "hand-eye-autonomous-delivery-learning-humanoid-navigation-locomotion-a",
  "title": "Hand-Eye Autonomous Delivery: Learning Humanoid Navigation, Locomotion and Reaching",
  "abstract": "We propose Hand-Eye Autonomous Delivery (HEAD), a framework that learns navigation, locomotion, and reaching skills for humanoids, directly from human motion and vision perception data. We take a modular approach where the high-level planner commands the target position and orientation of the hands and eyes of the humanoid, delivered by the low-level policy that controls the whole-body movements. Specifically, the low-level whole-body controller learns to track the three points (eyes, left hand, and right hand) from existing large-scale human motion capture data while high-level policy learns from human data collected by Aria glasses. Our modular approach decouples the ego-centric vision perception from physical actions, promoting efficient learning and scalability to novel scenes. We evaluate our method both in simulation and in the real-world, demonstrating humanoid's capabilities to navigate and reach in complex environments designed for humans.",
  "published": "2025-08-05",
  "updated": "2025-08-07",
  "year": "2025",
  "authors": [
   "Sirui Chen",
   "Yufei Ye",
   "Zi-Ang Cao",
   "Jennifer Lew",
   "Pei Xu",
   "C. Karen Liu"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 14,
  "influential_citations": 0,
  "tldr": "",
  "doi": "10.48550/arXiv.2508.03068",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Sirui Chen",
    "id": "2209905328",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Yufei Ye",
    "id": "9653518",
    "h_index": 10,
    "papers": 18
   },
   {
    "name": "Zi-ang Cao",
    "id": "2359785826",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "J. Lew",
    "id": "46195477",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Pei Xu",
    "id": "2335601372",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "C. K. Liu",
    "id": "2278583770",
    "h_index": 5,
    "papers": 7
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "egocentric-data",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2508.03068v2",
  "pdf_url": "https://arxiv.org/pdf/2508.03068v2",
  "html_url": "https://arxiv.org/html/2508.03068v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.18
 },
 {
  "id": "2507.23785",
  "slug": "gaussian-variation-field-diffusion-for-high-fidelity-video-to-4d-synth",
  "title": "Gaussian Variation Field Diffusion for High-fidelity Video-to-4D Synthesis",
  "abstract": "In this paper, we present a novel framework for video-to-4D generation that creates high-quality dynamic 3D content from single video inputs. Direct 4D diffusion modeling is extremely challenging due to costly data construction and the high-dimensional nature of jointly representing 3D shape, appearance, and motion. We address these challenges by introducing a Direct 4DMesh-to-GS Variation Field VAE that directly encodes canonical Gaussian Splats (GS) and their temporal variations from 3D animation data without per-instance fitting, and compresses high-dimensional animations into a compact latent space. Building upon this efficient representation, we train a Gaussian Variation Field diffusion model with temporal-aware Diffusion Transformer conditioned on input videos and canonical GS. Trained on carefully-curated animatable 3D objects from the Objaverse dataset, our model demonstrates superior generation quality compared to existing methods. It also exhibits remarkable generalization to in-the-wild video inputs despite being trained exclusively on synthetic data, paving the way for generating high-quality animated 3D content. Project page: https://gvfdiffusion.github.io/.",
  "published": "2025-07-31",
  "updated": "2025-07-31",
  "year": "2025",
  "authors": [
   "Bowen Zhang",
   "Sicheng Xu",
   "Chuxin Wang",
   "Jiaolong Yang",
   "Feng Zhao",
   "Dong Chen",
   "Baining Guo"
  ],
  "author_count": 7,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "ICCV",
  "venue_source": "semantic-scholar",
  "citations": 31,
  "influential_citations": 9,
  "tldr": "A Direct 4DMesh-to-GS Variation Field VAE that directly encodes canonical Gaussian Splats (GS) and their temporal variations from 3D animation data without per-instance fitting, and compresses high-dimensional animations into a compact latent space is introduced.",
  "doi": "10.1109/ICCV51701.2025.01162",
  "oa_pdf": "https://arxiv.org/pdf/2507.23785",
  "s2_authors": [
   {
    "name": "Bowen Zhang",
    "id": "2293950022",
    "h_index": 7,
    "papers": 8
   },
   {
    "name": "Sicheng Xu",
    "id": "2110510433",
    "h_index": 11,
    "papers": 22
   },
   {
    "name": "Chuxin Wang",
    "id": "2322452651",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "Jiaolong Yang",
    "id": "2237946707",
    "h_index": 10,
    "papers": 22
   },
   {
    "name": "Feng Zhao",
    "id": "2293908810",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Dong Chen",
    "id": "2333464886",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Baining Guo",
    "id": "2238211403",
    "h_index": 17,
    "papers": 30
   }
  ],
  "comment": "ICCV 2025. Project page: https://gvfdiffusion.github.io/",
  "topics": [
   "spatial-3d",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2507.23785v1",
  "pdf_url": "https://arxiv.org/pdf/2507.23785v1",
  "html_url": "https://arxiv.org/html/2507.23785v1",
  "code_url": "https://gvfdiffusion.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.01
 },
 {
  "id": "2507.23682",
  "slug": "villa-x-enhancing-latent-action-modeling-in-vision-language-action-mod",
  "title": "villa-X: Enhancing Latent Action Modeling in Vision-Language-Action Models",
  "abstract": "Vision-Language-Action (VLA) models have emerged as a popular paradigm for learning robot manipulation policies that can follow language instructions and generalize to novel scenarios. Recent works have begun to explore the incorporation of latent actions, abstract representations of motion between two frames, into VLA pre-training. In this paper, we introduce villa-X, a novel Vision-Language-Latent-Action (ViLLA) framework that advances latent action modeling for learning generalizable robot manipulation policies. Our approach improves both how latent actions are learned and how they are incorporated into VLA pre-training. We demonstrate that villa-X can generate latent action plans in a zero-shot fashion, even for unseen embodiments and open-vocabulary symbolic understanding. This capability enables villa-X to achieve superior performance across diverse simulation tasks in SIMPLER and on two real-world robotic setups involving both gripper and dexterous hand manipulation. These results establish villa-X as a principled and scalable paradigm for learning generalizable robot manipulation policies. We believe it provides a strong foundation for future research.",
  "published": "2025-07-31",
  "updated": "2025-09-25",
  "year": "2025",
  "authors": [
   "Xiaoyu Chen",
   "Hangxing Wei",
   "Pushi Zhang",
   "Chuheng Zhang",
   "Kaixin Wang",
   "Yanjiang Guo",
   "Rushuai Yang",
   "Yucen Wang",
   "Xinquan Xiao",
   "Li Zhao",
   "Jianyu Chen",
   "Jiang Bian"
  ],
  "author_count": 12,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 50,
  "influential_citations": 6,
  "tldr": "It is demonstrated that villa-X can generate latent action plans in a zero-shot fashion, even for unseen embodiments and open-vocabulary symbolic understanding, which enables villa-X to achieve superior performance across diverse simulation tasks in SIMPLER and on two real-world robotic setups involving both gripper and dexterous hand manipulation.",
  "doi": "10.48550/arXiv.2507.23682",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xiaoyu Chen",
    "id": "2329208858",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Hangxing Wei",
    "id": "2374177899",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Pushi Zhang",
    "id": "1570021289",
    "h_index": 9,
    "papers": 19
   },
   {
    "name": "Chuheng Zhang",
    "id": "2293350841",
    "h_index": 3,
    "papers": 15
   },
   {
    "name": "Kaixin Wang",
    "id": "2367933865",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "Yanjiang Guo",
    "id": "2181339548",
    "h_index": 12,
    "papers": 26
   },
   {
    "name": "Rushuai Yang",
    "id": "2373738874",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Yucen Wang",
    "id": "2220305846",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Xinquan Xiao",
    "id": "2374432243",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Li Zhao",
    "id": "2218154011",
    "h_index": 7,
    "papers": 25
   },
   {
    "name": "Jianyu Chen",
    "id": "2280257464",
    "h_index": 10,
    "papers": 20
   },
   {
    "name": "Jiang Bian",
    "id": "2287806843",
    "h_index": 7,
    "papers": 16
   }
  ],
  "comment": "Project page: https://aka.ms/villa-x",
  "topics": [
   "vla",
   "dexterous-manipulation",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2507.23682v3",
  "pdf_url": "https://arxiv.org/pdf/2507.23682v3",
  "html_url": "https://arxiv.org/html/2507.23682v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.71
 },
 {
  "id": "2507.23523",
  "slug": "h-rdt-human-manipulation-enhanced-bimanual-robotic-manipulation",
  "title": "H-RDT: Human Manipulation Enhanced Bimanual Robotic Manipulation",
  "abstract": "Imitation learning for robotic manipulation faces a fundamental challenge: the scarcity of large-scale, high-quality robot demonstration data. Recent robotic foundation models often pre-train on cross-embodiment robot datasets to increase data scale, while they face significant limitations as the diverse morphologies and action spaces across different robot embodiments make unified training challenging. In this paper, we present H-RDT (Human to Robotics Diffusion Transformer), a novel approach that leverages human manipulation data to enhance robot manipulation capabilities. Our key insight is that large-scale egocentric human manipulation videos with paired 3D hand pose annotations provide rich behavioral priors that capture natural manipulation strategies and can benefit robotic policy learning. We introduce a two-stage training paradigm: (1) pre-training on large-scale egocentric human manipulation data, and (2) cross-embodiment fine-tuning on robot-specific data with modular action encoders and decoders. Built on a diffusion transformer architecture with 2B parameters, H-RDT uses flow matching to model complex action distributions. Extensive evaluations encompassing both simulation and real-world experiments, single-task and multitask scenarios, as well as few-shot learning and robustness assessments, demonstrate that H-RDT outperforms training from scratch and existing state-of-the-art methods, including Pi0 and RDT, achieving significant improvements of 13.9% and 40.5% over training from scratch in simulation and real-world experiments, respectively. The results validate our core hypothesis that human manipulation data can serve as a powerful foundation for learning bimanual robotic manipulation policies.",
  "published": "2025-07-31",
  "updated": "2025-08-01",
  "year": "2025",
  "authors": [
   "Hongzhe Bi",
   "Lingxuan Wu",
   "Tianwei Lin",
   "Hengkai Tan",
   "Zhizhong Su",
   "Hang Su",
   "Jun Zhu"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 49,
  "influential_citations": 7,
  "tldr": "H-RDT (Human to Robotics Diffusion Transformer), a novel approach that leverages human manipulation data to enhance robot manipulation capabilities, is presented, demonstrating that H-RDT outperforms training from scratch and existing state-of-the-art methods.",
  "doi": "10.48550/arXiv.2507.23523",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hongzhe Bi",
    "id": "2371070888",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Lingxuan Wu",
    "id": "2294465744",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Tianwei Lin",
    "id": "2268009422",
    "h_index": 8,
    "papers": 15
   },
   {
    "name": "Hengkai Tan",
    "id": "2303970369",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "Zhizhong Su",
    "id": "2374211357",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Hang Su",
    "id": "2374119416",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Jun Zhu",
    "id": "2374341952",
    "h_index": 5,
    "papers": 9
   }
  ],
  "comment": "",
  "topics": [
   "egocentric-data",
   "imitation-diffusion",
   "foundation-pretraining",
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [
   "Physical Intelligence"
  ],
  "abs_url": "https://arxiv.org/abs/2507.23523v2",
  "pdf_url": "https://arxiv.org/pdf/2507.23523v2",
  "html_url": "https://arxiv.org/html/2507.23523v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.2
 },
 {
  "id": "2507.19468",
  "slug": "back-to-the-features-dino-as-a-foundation-for-video-world-models",
  "title": "Back to the Features: DINO as a Foundation for Video World Models",
  "abstract": "We present DINO-world, a powerful generalist video world model trained to predict future frames in the latent space of DINOv2. By leveraging a pre-trained image encoder and training a future predictor on a large-scale uncurated video dataset, DINO-world learns the temporal dynamics of diverse scenes, from driving and indoor scenes to simulated environments. We show that DINO-world outperforms previous models on a variety of video prediction benchmarks, e.g. segmentation and depth forecasting, and demonstrates strong understanding of intuitive physics. Furthermore, we show that it is possible to fine-tune the predictor on observation-action trajectories. The resulting action-conditioned world model can be used for planning by simulating candidate trajectories in latent space.",
  "published": "2025-07-25",
  "updated": "2025-07-25",
  "year": "2025",
  "authors": [
   "Federico Baldassarre",
   "Marc Szafraniec",
   "Basile Terver",
   "Vasil Khalidov",
   "Francisco Massa",
   "Yann LeCun",
   "Patrick Labatut",
   "Maximilian Seitzer",
   "Piotr Bojanowski"
  ],
  "author_count": 9,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 56,
  "influential_citations": 5,
  "tldr": "DINO-world is presented, a powerful generalist video world model trained to predict future frames in the latent space of DINOv2 by leveraging a pre-trained image encoder and training a future predictor on a large-scale uncurated video dataset that outperforms previous models on a variety of video prediction benchmarks.",
  "doi": "10.48550/arXiv.2507.19468",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Federico Baldassarre",
    "id": "2336864698",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Marc Szafraniec",
    "id": "23994377",
    "h_index": 11,
    "papers": 17
   },
   {
    "name": "Basile Terver",
    "id": "2216065931",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Vasil Khalidov",
    "id": "2182694",
    "h_index": 16,
    "papers": 28
   },
   {
    "name": "Francisco Massa",
    "id": "2366427448",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Yann LeCun",
    "id": "2265899558",
    "h_index": 22,
    "papers": 47
   },
   {
    "name": "Patrick Labatut",
    "id": "1744868",
    "h_index": 20,
    "papers": 38
   },
   {
    "name": "Maximilian Seitzer",
    "id": "2348556",
    "h_index": 10,
    "papers": 12
   },
   {
    "name": "Piotr Bojanowski",
    "id": "2329288",
    "h_index": 38,
    "papers": 76
   }
  ],
  "comment": "",
  "topics": [
   "world-models"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2507.19468v1",
  "pdf_url": "https://arxiv.org/pdf/2507.19468v1",
  "html_url": "https://arxiv.org/html/2507.19468v1",
  "code_url": null,
  "presented": [
   {
    "club": "saturday-robotics",
    "club_name": "Saturday Robotics",
    "session_number": 9,
    "session_title": "Robotics & World Models Reading Club 09: CVPR Warm-up & Founders Spotlight \u2014 DeltaWorld + VisuoTactile Dexterous Hands | San Francisco 0523",
    "date_text": "Saturday, May 23, 2026 | 2:00 PM \u2013 5:00 PM",
    "city": "San Francisco, CA",
    "url": "https://lu.ma/wooiz0bf",
    "listed_as": "A Frame is Worth One Token: Efficient Generative World Modeling with Delta Tokens"
   }
  ],
  "club_note": "Keynote 2 by by Arjun Subramaniam (Factory Intelligence)",
  "featured": true,
  "signal": 4.76
 },
 {
  "id": "2507.17294",
  "slug": "vla-touch-enhancing-vision-language-action-models-with-dual-level-tact",
  "title": "VLA-Touch: Enhancing Vision-Language-Action Models with Dual-Level Tactile Feedback",
  "abstract": "Tactile feedback is generally recognized to be crucial for effective interaction with the physical world. However, state-of-the-art Vision-Language-Action (VLA) models lack the ability to interpret and use tactile signals, limiting their effectiveness in contact-rich tasks. Incorporating tactile feedback into these systems is challenging due to the absence of large multi-modal datasets. We present VLA-Touch, an approach that enhances generalist robot policies with tactile sensing \\emph{without fine-tuning} the base VLA. Our method introduces two key innovations: (1) a pipeline that leverages a pretrained tactile-language model that provides semantic tactile feedback for high-level task planning, and (2) a diffusion-based controller that refines VLA-generated actions with tactile signals for contact-rich manipulation. Through real-world experiments, we demonstrate that our dual-level integration of tactile feedback improves task planning efficiency while enhancing execution precision. Code is open-sourced at \\href{https://github.com/jxbi1010/VLA-Touch}{this URL}.",
  "published": "2025-07-23",
  "updated": "2025-07-29",
  "year": "2025",
  "authors": [
   "Jianxin Bi",
   "Kevin Yuchen Ma",
   "Ce Hao",
   "Mike Zheng Shou",
   "Harold Soh"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 60,
  "influential_citations": 1,
  "tldr": "This work introduces two key innovations: a pipeline that leverages a pretrained tactile-language model that provides semantic tactile feedback for high-level task planning, and a diffusion-based controller that refines VLA-generated actions with tactile signals for contact-rich manipulation.",
  "doi": "10.48550/arXiv.2507.17294",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jianxin Bi",
    "id": "2190428643",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "K. Ma",
    "id": "2324345511",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Ce Hao",
    "id": "2306782351",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "M. Shou",
    "id": "2047358650",
    "h_index": 49,
    "papers": 278
   },
   {
    "name": "Harold Soh",
    "id": "2286883070",
    "h_index": 7,
    "papers": 16
   }
  ],
  "comment": "19 pages, 5 figures",
  "topics": [
   "vla",
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2507.17294v2",
  "pdf_url": "https://arxiv.org/pdf/2507.17294v2",
  "html_url": "https://arxiv.org/html/2507.17294v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.79
 },
 {
  "id": "2507.15597",
  "slug": "being-h0-vision-language-action-pretraining-from-large-scale-human-vid",
  "title": "Being-H0: Vision-Language-Action Pretraining from Large-Scale Human Videos",
  "abstract": "We introduce Being-H0, a dexterous Vision-Language-Action model (VLA) trained on large-scale human videos. Existing VLAs struggle with complex manipulation tasks requiring high dexterity and generalize poorly to novel scenarios and tasks, primarily due to their reliance on synthetic data with significant sim-to-real gaps or teleoperated demonstrations lacking scale and diversity. To address this data bottleneck, we propose leveraging human hands as a foundation manipulator, capitalizing on the rich dexterity and scalability present in web data. Our approach centers on physical instruction tuning, a novel training paradigm that combines large-scale VLA pretraining from human videos, physical space alignment for 3D reasoning, and post-training adaptation for robotic tasks. Additionally, we introduce a part-level motion tokenization method which achieves millimeter-level reconstruction accuracy to model precise hand trajectories for action learning. To support our proposed paradigm, we further develop a comprehensive data curation pipeline that integrates heterogeneous sources -- including motion capture, VR, and RGB-only videos -- into a large-scale dataset with millions of motion-based instructional instances. We empirically show the excellence of Being-H0 in hand motion generation and instruction following, and it also scales well with model and data sizes. Importantly, we observe the expected gains of Being-H0 in real-world robotic manipulation as physical instruction tuning is applied. More details are available at https://beingbeyond.github.io/Being-H0.",
  "published": "2025-07-21",
  "updated": "2025-07-21",
  "year": "2025",
  "authors": [
   "Hao Luo",
   "Yicheng Feng",
   "Wanpeng Zhang",
   "Sipeng Zheng",
   "Ye Wang",
   "Haoqi Yuan",
   "Jiazheng Liu",
   "Chaoyi Xu",
   "Qin Jin",
   "Zongqing Lu"
  ],
  "author_count": 10,
  "categories": [
   "cs.CV",
   "cs.LG",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 105,
  "influential_citations": 8,
  "tldr": "This work introduces Being-H0, a dexterous Vision-Language-Action model (VLA) trained on large-scale human videos that combines large-scale VLA pretraining from human videos, physical space alignment for 3D reasoning, and post-training adaptation for robotic tasks, and introduces a part-level motion tokenization method which achieves millimeter-level reconstruction accuracy.",
  "doi": "10.48550/arXiv.2507.15597",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hao Luo",
    "id": "2199828096",
    "h_index": 9,
    "papers": 20
   },
   {
    "name": "Yicheng Feng",
    "id": "2115319860",
    "h_index": 10,
    "papers": 21
   },
   {
    "name": "Wanpeng Zhang",
    "id": "2324490975",
    "h_index": 8,
    "papers": 23
   },
   {
    "name": "Sipeng Zheng",
    "id": "2258682220",
    "h_index": 11,
    "papers": 24
   },
   {
    "name": "Ye Wang",
    "id": "2351216973",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Haoqi Yuan",
    "id": "1429192914",
    "h_index": 12,
    "papers": 37
   },
   {
    "name": "Jiazheng Liu",
    "id": "2258602946",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Chaoyi Xu",
    "id": "2373397887",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Qin Jin",
    "id": "2290783013",
    "h_index": 6,
    "papers": 16
   },
   {
    "name": "Zongqing Lu",
    "id": "2258676670",
    "h_index": 29,
    "papers": 144
   }
  ],
  "comment": "37 pages",
  "topics": [
   "vla",
   "dexterous-manipulation",
   "egocentric-data",
   "sim2real",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2507.15597v1",
  "pdf_url": "https://arxiv.org/pdf/2507.15597v1",
  "html_url": "https://arxiv.org/html/2507.15597v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.03
 },
 {
  "id": "2507.15062",
  "slug": "touch-in-the-wild-learning-fine-grained-manipulation-with-a-portable-v",
  "title": "Touch in the Wild: Learning Fine-Grained Manipulation with a Portable Visuo-Tactile Gripper",
  "abstract": "Handheld grippers are increasingly used to collect human demonstrations due to their ease of deployment and versatility. However, most existing designs lack tactile sensing, despite the critical role of tactile feedback in precise manipulation. We present a portable, lightweight gripper with integrated tactile sensors that enables synchronized collection of visual and tactile data in diverse, real-world, and in-the-wild settings. Building on this hardware, we propose a cross-modal representation learning framework that integrates visual and tactile signals while preserving their distinct characteristics. The learning procedure allows the emergence of interpretable representations that consistently focus on contacting regions relevant for physical interactions. When used for downstream manipulation tasks, these representations enable more efficient and effective policy learning, supporting precise robotic manipulation based on multimodal feedback. We validate our approach on fine-grained tasks such as test tube insertion and pipette-based fluid transfer, demonstrating improved accuracy and robustness under external disturbances. Our project page is available at https://binghao-huang.github.io/touch_in_the_wild/ .",
  "published": "2025-07-20",
  "updated": "2025-11-12",
  "year": "2025",
  "authors": [
   "Xinyue Zhu",
   "Binghao Huang",
   "Yunzhu Li"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 49,
  "influential_citations": 2,
  "tldr": "A portable, lightweight gripper with integrated tactile sensors that enables synchronized collection of visual and tactile data in diverse, real-world, and in-the-wild settings is presented and a cross-modal representation learning framework is proposed that integrates visual and tactile signals while preserving their distinct characteristics.",
  "doi": "10.48550/arXiv.2507.15062",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xinyue Zhu",
    "id": "2283262351",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Binghao Huang",
    "id": "2287019710",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Yunzhu Li",
    "id": "2374457025",
    "h_index": 4,
    "papers": 9
   }
  ],
  "comment": "More videos can be found on our website:https://binghao-huang.github.io/touch_in_the_wild/",
  "topics": [
   "egocentric-data",
   "tactile",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2507.15062v2",
  "pdf_url": "https://arxiv.org/pdf/2507.15062v2",
  "html_url": "https://arxiv.org/html/2507.15062v2",
  "code_url": "https://binghao-huang.github.io/touch_in_the_wild/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.2
 },
 {
  "id": "2507.12462",
  "slug": "spatialtrackerv2-3d-point-tracking-made-easy",
  "title": "SpatialTrackerV2: 3D Point Tracking Made Easy",
  "abstract": "We present SpatialTrackerV2, a feed-forward 3D point tracking method for monocular videos. Going beyond modular pipelines built on off-the-shelf components for 3D tracking, our approach unifies the intrinsic connections between point tracking, monocular depth, and camera pose estimation into a high-performing and feedforward 3D point tracker. It decomposes world-space 3D motion into scene geometry, camera ego-motion, and pixel-wise object motion, with a fully differentiable and end-to-end architecture, allowing scalable training across a wide range of datasets, including synthetic sequences, posed RGB-D videos, and unlabeled in-the-wild footage. By learning geometry and motion jointly from such heterogeneous data, SpatialTrackerV2 outperforms existing 3D tracking methods by 30%, and matches the accuracy of leading dynamic 3D reconstruction approaches while running 50$\\times$ faster.",
  "published": "2025-07-16",
  "updated": "2025-07-19",
  "year": "2025",
  "authors": [
   "Yuxi Xiao",
   "Jianyuan Wang",
   "Nan Xue",
   "Nikita Karaev",
   "Yuri Makarov",
   "Bingyi Kang",
   "Xing Zhu",
   "Hujun Bao",
   "Yujun Shen",
   "Xiaowei Zhou"
  ],
  "author_count": 10,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "ICCV 2025",
  "venue_source": "arxiv-comment",
  "citations": 89,
  "influential_citations": 12,
  "tldr": "SpatialTrackerV2 decomposes world-space 3D motion into scene geometry, camera ego-motion, and pixel-wise object motion, with a fully differentiable and end-to-end architecture, allowing scalable training across a wide range of datasets.",
  "doi": "10.48550/arXiv.2507.12462",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yuxi Xiao",
    "id": "2292535912",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Jianyuan Wang",
    "id": "2271268581",
    "h_index": 13,
    "papers": 24
   },
   {
    "name": "Nan Xue",
    "id": "2292027056",
    "h_index": 11,
    "papers": 31
   },
   {
    "name": "Nikita Karaev",
    "id": "16643610",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "Yuri Makarov",
    "id": "2372596123",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Bingyi Kang",
    "id": "2338689574",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Xing Zhu",
    "id": "2393215354",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Hujun Bao",
    "id": "2253405497",
    "h_index": 16,
    "papers": 42
   },
   {
    "name": "Yujun Shen",
    "id": "2259478633",
    "h_index": 11,
    "papers": 20
   },
   {
    "name": "Xiaowei Zhou",
    "id": "2238207469",
    "h_index": 20,
    "papers": 56
   }
  ],
  "comment": "International Conference on Computer Vision, ICCV 2025. Huggingface Demo: https://huggingface.co/spaces/Yuxihenry/SpatialTrackerV2, Code: https://github.com/henry123-boy/SpaTrackerV2",
  "topics": [
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2507.12462v2",
  "pdf_url": "https://arxiv.org/pdf/2507.12462v2",
  "html_url": "https://arxiv.org/html/2507.12462v2",
  "code_url": "https://github.com/henry123-boy/SpaTrackerV2",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.45
 },
 {
  "id": "2507.12440",
  "slug": "egovla-learning-vision-language-action-models-from-egocentric-human-vi",
  "title": "EgoVLA: Learning Vision-Language-Action Models from Egocentric Human Videos",
  "abstract": "Real robot data collection for imitation learning has led to significant advancements in robotic manipulation. However, the requirement for robot hardware in the process fundamentally constrains the scale of the data. In this paper, we explore training Vision-Language-Action (VLA) models using egocentric human videos. The benefit of using human videos is not only for their scale but more importantly for the richness of scenes and tasks. With a VLA trained on human video that predicts human wrist and hand actions, we can perform Inverse Kinematics and retargeting to convert the human actions to robot actions. We fine-tune the model using a few robot manipulation demonstrations to obtain the robot policy, namely EgoVLA. We propose a simulation benchmark called Ego Humanoid Manipulation Benchmark, where we design diverse bimanual manipulation tasks with demonstrations. We fine-tune and evaluate EgoVLA with Ego Humanoid Manipulation Benchmark and show significant improvements over baselines and ablate the importance of human data. Videos can be found on our website: https://rchalyang.github.io/EgoVLA",
  "published": "2025-07-16",
  "updated": "2025-07-18",
  "year": "2025",
  "authors": [
   "Ruihan Yang",
   "Qinxi Yu",
   "Yecheng Wu",
   "Rui Yan",
   "Borui Li",
   "An-Chieh Cheng",
   "Xueyan Zou",
   "Yunhao Fang",
   "Xuxin Cheng",
   "Ri-Zhao Qiu",
   "Hongxu Yin",
   "Sifei Liu",
   "Song Han",
   "Yao Lu",
   "Xiaolong Wang"
  ],
  "author_count": 15,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 131,
  "influential_citations": 3,
  "tldr": "This paper explores training Vision-Language-Action (VLA) models using egocentric human videos using EgoVLA and fine-tune and evaluate EgoVLA with Ego Humanoid Manipulation Benchmark and show significant improvements over baselines and ablate the importance of human data.",
  "doi": "10.48550/arXiv.2507.12440",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ruihan Yang",
    "id": "143955842",
    "h_index": 18,
    "papers": 22
   },
   {
    "name": "Qinxi Yu",
    "id": "2265469952",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Yecheng Wu",
    "id": "2320183695",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Rui Yan",
    "id": "2395719044",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Borui Li",
    "id": "2381409885",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "An-Chieh Cheng",
    "id": "2366117926",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Xueyan Zou",
    "id": "2319768659",
    "h_index": 7,
    "papers": 15
   },
   {
    "name": "Yunhao Fang",
    "id": "2312873153",
    "h_index": 10,
    "papers": 20
   },
   {
    "name": "Hongxu Yin",
    "id": "1989015",
    "h_index": 40,
    "papers": 72
   },
   {
    "name": "Sifei Liu",
    "id": "2280548977",
    "h_index": 11,
    "papers": 25
   },
   {
    "name": "Song Han",
    "id": "2273855886",
    "h_index": 13,
    "papers": 21
   },
   {
    "name": "Yao Lu",
    "id": "2274926912",
    "h_index": 15,
    "papers": 18
   },
   {
    "name": "Xiaolong Wang",
    "id": "2348408036",
    "h_index": 3,
    "papers": 7
   }
  ],
  "comment": "More videos can be found on our website: https://rchalyang.github.io/EgoVLA",
  "topics": [
   "vla",
   "humanoids",
   "egocentric-data",
   "sim2real",
   "imitation-diffusion",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2507.12440v3",
  "pdf_url": "https://arxiv.org/pdf/2507.12440v3",
  "html_url": "https://arxiv.org/html/2507.12440v3",
  "code_url": "https://rchalyang.github.io/EgoVLA",
  "presented": [
   {
    "club": "saturday-robotics",
    "club_name": "Saturday Robotics",
    "session_number": 13,
    "session_title": "Robotics & World Models Reading Club 13: HumanEgo: Train Robot Policy from 30 min Egocentric Videos \u2014 SF 0620",
    "date_text": "Saturday, June 20, 2026 | 2:00 PM \u2013 5:00 PM",
    "city": "San Francisco, CA",
    "url": "https://lu.ma/6vkhxnum",
    "listed_as": ""
   }
  ],
  "club_note": "",
  "featured": true,
  "signal": 6.12
 },
 {
  "id": "2507.06224",
  "slug": "ec-flow-enabling-versatile-robotic-manipulation-from-action-unlabeled",
  "title": "EC-Flow: Enabling Versatile Robotic Manipulation from Action-Unlabeled Videos via Embodiment-Centric Flow",
  "abstract": "Current language-guided robotic manipulation systems often require low-level action-labeled datasets for imitation learning. While object-centric flow prediction methods mitigate this issue, they remain limited to scenarios involving rigid objects with clear displacement and minimal occlusion. In this work, we present Embodiment-Centric Flow (EC-Flow), a framework that directly learns manipulation from action-unlabeled videos by predicting embodiment-centric flow. Our key insight is that incorporating the embodiment's inherent kinematics significantly enhances generalization to versatile manipulation scenarios, including deformable object handling, occlusions, and non-object-displacement tasks. To connect the EC-Flow with language instructions and object interactions, we further introduce a goal-alignment module by jointly optimizing movement consistency and goal-image prediction. Moreover, translating EC-Flow to executable robot actions only requires a standard robot URDF (Unified Robot Description Format) file to specify kinematic constraints across joints, which makes it easy to use in practice. We validate EC-Flow on both simulation (Meta-World) and real-world tasks, demonstrating its state-of-the-art performance in occluded object handling (62% improvement), deformable object manipulation (45% improvement), and non-object-displacement tasks (80% improvement) than prior state-of-the-art object-centric flow methods. For more information, see our project website at https://ec-flow1.github.io .",
  "published": "2025-07-08",
  "updated": "2025-07-08",
  "year": "2025",
  "authors": [
   "Yixiang Chen",
   "Peiyan Li",
   "Yan Huang",
   "Jiabing Yang",
   "Kehan Chen",
   "Liang Wang"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "ICCV",
  "venue_source": "semantic-scholar",
  "citations": 10,
  "influential_citations": 0,
  "tldr": "Embodiment-Centric Flow (EC-Flow) is presented, a framework that directly learns manipulation from action-unlabeled videos by predicting embodimentcentric flow, and a goal-alignment module is introduced to connect the EC-Flow with language instructions and object interactions.",
  "doi": "10.1109/ICCV51701.2025.01112",
  "oa_pdf": "https://arxiv.org/pdf/2507.06224",
  "s2_authors": [
   {
    "name": "Yixiang Chen",
    "id": "2366155958",
    "h_index": 4,
    "papers": 20
   },
   {
    "name": "Peiyan Li",
    "id": "2305635155",
    "h_index": 7,
    "papers": 26
   },
   {
    "name": "Yan Huang",
    "id": "2258533421",
    "h_index": 14,
    "papers": 39
   },
   {
    "name": "Jiabing Yang",
    "id": "2109723855",
    "h_index": 6,
    "papers": 23
   },
   {
    "name": "Kehan Chen",
    "id": "2335489827",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Liang Wang",
    "id": "2317093163",
    "h_index": 4,
    "papers": 24
   }
  ],
  "comment": "Accepted at ICCV 2025",
  "topics": [
   "imitation-diffusion"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2507.06224v1",
  "pdf_url": "https://arxiv.org/pdf/2507.06224v1",
  "html_url": "https://arxiv.org/html/2507.06224v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.54
 },
 {
  "id": "2507.06261",
  "slug": "gemini-2-5-pushing-the-frontier-with-advanced-reasoning-multimodality",
  "title": "Gemini 2.5: Pushing the Frontier with Advanced Reasoning, Multimodality, Long Context, and Next Generation Agentic Capabilities",
  "abstract": "In this report, we introduce the Gemini 2.X model family: Gemini 2.5 Pro and Gemini 2.5 Flash, as well as our earlier Gemini 2.0 Flash and Flash-Lite models. Gemini 2.5 Pro is our most capable model yet, achieving SoTA performance on frontier coding and reasoning benchmarks. In addition to its incredible coding and reasoning skills, Gemini 2.5 Pro is a thinking model that excels at multimodal understanding and it is now able to process up to 3 hours of video content. Its unique combination of long context, multimodal and reasoning capabilities can be combined to unlock new agentic workflows. Gemini 2.5 Flash provides excellent reasoning abilities at a fraction of the compute and latency requirements and Gemini 2.0 Flash and Flash-Lite provide high performance at low latency and cost. Taken together, the Gemini 2.X model generation spans the full Pareto frontier of model capability vs cost, allowing users to explore the boundaries of what is possible with complex agentic problem solving.",
  "published": "2025-07-07",
  "updated": "2025-12-19",
  "year": "2025",
  "authors": [
   "Gheorghe Comanici",
   "Eric Bieber",
   "Mike Schaekermann",
   "Ice Pasupat",
   "Noveen Sachdeva",
   "Inderjit Dhillon",
   "Marcel Blistein",
   "Ori Ram",
   "Dan Zhang",
   "Evan Rosen",
   "Luke Marris",
   "Sam Petulla",
   "Colin Gaffney",
   "Asaf Aharoni",
   "Nathan Lintz",
   "Tiago Cardal Pais",
   "Henrik Jacobsson",
   "Idan Szpektor",
   "Nan-Jiang Jiang",
   "Krishna Haridasan",
   "Ahmed Omran",
   "Nikunj Saunshi",
   "Dara Bahri",
   "Gaurav Mishra",
   "Eric Chu",
   "Toby Boyd",
   "Brad Hekman",
   "Aaron Parisi",
   "Chaoyi Zhang",
   "Kornraphop Kawintiranon",
   "Tania Bedrax-Weiss",
   "Oliver Wang",
   "Ya Xu",
   "Ollie Purkiss",
   "Uri Mendlovic",
   "Ila\u00ef Deutel",
   "Nam Nguyen",
   "Adam Langley",
   "Flip Korn",
   "Lucia Rossazza",
   "Alexandre Ram\u00e9",
   "Sagar Waghmare",
   "Helen Miller",
   "Nathan Byrd",
   "Ashrith Sheshan",
   "Raia Hadsell",
   "Sangnie Bhardwaj",
   "Pawel Janus",
   "Tero Rissa",
   "Dan Horgan",
   "Alvin Abdagic",
   "Lior Belenki",
   "James Allingham",
   "Anima Singh",
   "Theo Guidroz",
   "Srivatsan Srinivasan",
   "Herman Schmit",
   "Kristen Chiafullo",
   "Andre Elisseeff",
   "Nilpa Jha",
   "Prateek Kolhar",
   "Leonard Berrada",
   "Frank Ding",
   "Xiance Si",
   "Shrestha Basu Mallick",
   "Franz Och",
   "Sofia Erell",
   "Eric Ni",
   "Tejasi Latkar",
   "Sherry Yang",
   "Petar Sirkovic",
   "Ziqiang Feng",
   "Robert Leland",
   "Rachel Hornung",
   "Gang Wu",
   "Charles Blundell",
   "Hamidreza Alvari",
   "Po-Sen Huang",
   "Cathy Yip",
   "Sanja Deur",
   "Li Liu",
   "Gabriela Surita",
   "Pablo Duque",
   "Dima Damen",
   "Johnson Jia",
   "Arthur Guez",
   "Markus Mircea",
   "Animesh Sinha",
   "Alberto Magni",
   "Pawe\u0142 Stradomski",
   "Tal Marian",
   "Vlado Gali\u0107",
   "Wenhu Chen",
   "Hisham Husain",
   "Achintya Singhal",
   "Dominik Grewe",
   "Fran\u00e7ois-Xavier Aubet",
   "Shuang Song",
   "Lorenzo Blanco",
   "Leland Rechis",
   "Lewis Ho",
   "Rich Munoz",
   "Kelvin Zheng",
   "Jessica Hamrick",
   "Kevin Mather",
   "Hagai Taitelbaum",
   "Eliza Rutherford",
   "Yun Lei",
   "Kuangyuan Chen",
   "Anand Shukla",
   "Erica Moreira",
   "Eric Doi",
   "Berivan Isik",
   "Nir Shabat",
   "Dominika Rogozi\u0144ska",
   "Kashyap Kolipaka",
   "Jason Chang",
   "Eugen Vu\u0161ak",
   "Srinivasan Venkatachary",
   "Shadi Noghabi",
   "Tarun Bharti",
   "Younghoon Jun",
   "Aleksandr Zaks",
   "Simon Green",
   "Jeshwanth Challagundla",
   "William Wong",
   "Muqthar Mohammad",
   "Dean Hirsch",
   "Yong Cheng",
   "Iftekhar Naim",
   "Lev Proleev",
   "Damien Vincent",
   "Aayush Singh",
   "Maxim Krikun",
   "Dilip Krishnan",
   "Zoubin Ghahramani",
   "Aviel Atias",
   "Rajeev Aggarwal",
   "Christo Kirov",
   "Dimitrios Vytiniotis",
   "Christy Koh",
   "Alexandra Chronopoulou",
   "Pawan Dogra",
   "Vlad-Doru Ion",
   "Gladys Tyen",
   "Jason Lee",
   "Felix Weissenberger",
   "Trevor Strohman",
   "Ashwin Balakrishna",
   "Jack Rae",
   "Marko Velic",
   "Raoul de Liedekerke",
   "Oded Elyada",
   "Wentao Yuan",
   "Canoee Liu",
   "Lior Shani",
   "Sergey Kishchenko",
   "Bea Alessio",
   "Yandong Li",
   "Richard Song",
   "Sam Kwei",
   "Orion Jankowski",
   "Aneesh Pappu",
   "Youhei Namiki",
   "Yenai Ma",
   "Nilesh Tripuraneni",
   "Colin Cherry",
   "Marissa Ikonomidis",
   "Yu-Cheng Ling",
   "Colin Ji",
   "Beka Westberg",
   "Auriel Wright",
   "Da Yu",
   "David Parkinson",
   "Swaroop Ramaswamy",
   "Jerome Connor",
   "Soheil Hassas Yeganeh",
   "Snchit Grover",
   "George Kenwright",
   "Lubo Litchev",
   "Chris Apps",
   "Alex Tomala",
   "Felix Halim",
   "Alex Castro-Ros",
   "Zefei Li",
   "Anudhyan Boral",
   "Pauline Sho",
   "Michal Yarom",
   "Eric Malmi",
   "David Klinghoffer",
   "Rebecca Lin",
   "Alan Ansell",
   "Pradeep Kumar S",
   "Shubin Zhao",
   "Siqi Zuo",
   "Adam Santoro",
   "Heng-Tze Cheng",
   "Solomon Demmessie",
   "Yuchi Liu",
   "Nicole Brichtova",
   "Allie Culp",
   "Nathaniel Braun",
   "Dan Graur",
   "Will Ng",
   "Nikhil Mehta",
   "Aaron Phillips",
   "Patrik Sundberg",
   "Varun Godbole",
   "Fangyu Liu",
   "Yash Katariya",
   "David Rim",
   "Mojtaba Seyedhosseini",
   "Sean Ammirati",
   "Jonas Valfridsson",
   "Mahan Malihi",
   "Timothy Knight",
   "Andeep Toor",
   "Thomas Lampe",
   "Abe Ittycheriah",
   "Lewis Chiang",
   "Chak Yeung",
   "Alexandre Fr\u00e9chette",
   "Jinmeng Rao",
   "Huisheng Wang",
   "Himanshu Srivastava",
   "Richard Zhang",
   "Rocky Rhodes",
   "Ariel Brand",
   "Dean Weesner",
   "Ilya Figotin",
   "Felix Gimeno",
   "Rachana Fellinger",
   "Pierre Marcenac",
   "Jos\u00e9 Leal",
   "Eyal Marcus",
   "Victor Cotruta",
   "Rodrigo Cabrera",
   "Sheryl Luo",
   "Dan Garrette",
   "Vera Axelrod",
   "Sorin Baltateanu",
   "David Barker",
   "Dongkai Chen",
   "Horia Toma",
   "Ben Ingram",
   "Jason Riesa",
   "Chinmay Kulkarni",
   "Yujing Zhang",
   "Hongbin Liu",
   "Chao Wang",
   "Martin Polacek",
   "Will Wu",
   "Kai Hui",
   "Adrian N Reyes",
   "Yi Su",
   "Megan Barnes",
   "Ishaan Malhi",
   "Anfal Siddiqui",
   "Qixuan Feng",
   "Mihai Damaschin",
   "Daniele Pighin",
   "Andreas Steiner",
   "Samuel Yang",
   "Ramya Sree Boppana",
   "Simeon Ivanov",
   "Arun Kandoor",
   "Aditya Shah",
   "Asier Mujika",
   "Da Huang",
   "Christopher A. Choquette-Choo",
   "Mohak Patel",
   "Tianhe Yu",
   "Toni Creswell",
   " Jerry",
   " Liu",
   "Catarina Barros",
   "Yasaman Razeghi",
   "Aurko Roy",
   "Phil Culliton",
   "Binbin Xiong",
   "Jiaqi Pan",
   "Thomas Strohmann",
   "Tolly Powell",
   "Babi Seal",
   "Doug DeCarlo",
   "Pranav Shyam",
   "Kaan Katircioglu",
   "Xuezhi Wang",
   "Cassidy Hardin",
   "Immanuel Odisho",
   "Josef Broder",
   "Oscar Chang",
   "Arun Nair",
   "Artem Shtefan",
   "Maura O'Brien",
   "Manu Agarwal",
   "Sahitya Potluri",
   "Siddharth Goyal",
   "Amit Jhindal",
   "Saksham Thakur",
   "Yury Stuken",
   "James Lyon",
   "Kristina Toutanova",
   "Fangxiaoyu Feng",
   "Austin Wu",
   "Ben Horn",
   "Alek Wang",
   "Alex Cullum",
   "Gabe Taubman",
   "Disha Shrivastava",
   "Chongyang Shi",
   "Hamish Tomlinson",
   "Roma Patel",
   "Tao Tu",
   "Ada Maksutaj Oflazer",
   "Francesco Pongetti",
   "Mingyao Yang",
   "Adrien Ali Ta\u00efga",
   "Vincent Perot",
   "Nuo Wang Pierse",
   "Feng Han",
   "Yoel Drori",
   "I\u00f1aki Iturrate",
   "Ayan Chakrabarti",
   "Legg Yeung",
   "Dave Dopson",
   "Yi-ting Chen",
   "Apoorv Kulshreshtha",
   "Tongfei Guo",
   "Philip Pham",
   "Tal Schuster",
   "Junquan Chen",
   "Alex Polozov",
   "Jinwei Xing",
   "Huanjie Zhou",
   "Praneeth Kacham",
   "Doron Kukliansky",
   "Antoine Miech",
   "Sergey Yaroshenko",
   "Ed Chi",
   "Sholto Douglas",
   "Hongliang Fei",
   "Mathieu Blondel",
   "Preethi Myla",
   "Lior Madmoni",
   "Xing Wu",
   "Daniel Keysers",
   "Kristian Kjems",
   "Isabela Albuquerque",
   "Lijun Yu",
   "Joel D'sa",
   "Michelle Plantan",
   "Vlad Ionescu",
   "Jaume Sanchez Elias",
   "Abhirut Gupta",
   "Manish Reddy Vuyyuru",
   "Fred Alcober",
   "Tong Zhou",
   "Kaiyang Ji",
   "Florian Hartmann",
   "Subha Puttagunta",
   "Hugo Song",
   "Ehsan Amid",
   "Anca Stefanoiu",
   "Andrew Lee",
   "Paul Pucciarelli",
   "Emma Wang",
   "Amit Raul",
   "Slav Petrov",
   "Isaac Tian",
   "Valentin Anklin",
   "Nana Nti",
   "Victor Gomes",
   "Max Schumacher",
   "Grace Vesom",
   "Alex Panagopoulos",
   "Konstantinos Bousmalis",
   "Daniel Andor",
   "Josh Jacob",
   "Yuan Zhang",
   "Bill Rosgen",
   "Matija Kecman",
   "Matthew Tung",
   "Alexandra Belias",
   "Noah Goodman",
   "Paul Covington",
   "Brian Wieder",
   "Nikita Saxena",
   "Elnaz Davoodi",
   "Muhuan Huang",
   "Sharath Maddineni",
   "Vincent Roulet",
   "Folawiyo Campbell-Ajala",
   "Pier Giuseppe Sessa",
   " Xintian",
   " Wu",
   "Guangda Lai",
   "Paul Collins",
   "Alex Haig",
   "Vytenis Sakenas",
   "Xiaowei Xu",
   "Marissa Giustina",
   "Laurent El Shafey",
   "Pichi Charoenpanit",
   "Shefali Garg",
   "Joshua Ainslie",
   "Boone Severson",
   "Montse Gonzalez Arenas",
   "Shreya Pathak",
   "Sujee Rajayogam",
   "Jie Feng",
   "Michiel Bakker",
   "Sheng Li",
   "Nevan Wichers",
   "Jamie Rogers",
   "Xinyang Geng",
   "Yeqing Li",
   "Rolf Jagerman",
   "Chao Jia",
   "Nadav Olmert",
   "David Sharon",
   "Matthew Mauger",
   "Sandeep Mariserla",
   "Hongxu Ma",
   "Megha Mohabey",
   "Kyuyeun Kim",
   "Alek Andreev",
   "Scott Pollom",
   "Juliette Love",
   "Vihan Jain",
   "Priyanka Agrawal",
   "Yannick Schroecker",
   "Alisa Fortin",
   "Manfred Warmuth",
   "Ji Liu",
   "Andrew Leach",
   "Irina Blok",
   "Ganesh Poomal Girirajan",
   "Roee Aharoni",
   "Benigno Uria",
   "Andrei Sozanschi",
   "Dan Goldberg",
   "Lucian Ionita",
   "Marco Tulio Ribeiro",
   "Martin Zlocha",
   "Vighnesh Birodkar",
   "Sami Lachgar",
   "Liangzhe Yuan",
   "Himadri Choudhury",
   "Matt Ginsberg",
   "Fei Zheng",
   "Gregory Dibb",
   "Emily Graves",
   "Swachhand Lokhande",
   "Gabriel Rasskin",
   "George-Cristian Muraru",
   "Corbin Quick",
   "Sandeep Tata",
   "Pierre Sermanet",
   "Aditya Chawla",
   "Itay Karo",
   "Yan Wang",
   "Susan Zhang",
   "Orgad Keller",
   "Anca Dragan",
   "Guolong Su",
   "Ian Chou",
   "Xi Liu",
   "Yiqing Tao",
   "Shruthi Prabhakara",
   "Marc Wilson",
   "Ruibo Liu",
   "Shibo Wang",
   "Georgie Evans",
   "David Du",
   "Alfonso Casta\u00f1o",
   "Gautam Prasad",
   "Mona El Mahdy",
   "Sebastian Gerlach",
   "Machel Reid",
   "Jarrod Kahn",
   "Amir Zait",
   "Thanumalayan Sankaranarayana Pillai",
   "Thatcher Ulrich",
   "Guanyu Wang",
   "Jan Wassenberg",
   "Efrat Farkash",
   "Kiran Yalasangi",
   "Congchao Wang",
   "Maria Bauza",
   "Simon Bucher",
   "Ting Liu",
   "Jun Yan",
   "Gary Leung",
   "Vikas Sindhwani",
   "Parker Barnes",
   "Avi Singh",
   "Ivan Jurin",
   "Jichuan Chang",
   "Niket Kumar Bhumihar",
   "Sivan Eiger",
   "Gui Citovsky",
   "Ben Withbroe",
   "Zhang Li",
   "Siyang Xue",
   "Niccol\u00f2 Dal Santo",
   "Georgi Stoyanov",
   "Yves Raimond",
   "Steven Zheng",
   "Yilin Gao",
   "V\u00edt List\u00edk",
   "S\u0142awek Kwasiborski",
   "Rachel Saputro",
   "Adnan Ozturel",
   "Ganesh Mallya",
   "Kushal Majmundar",
   "Ross West",
   "Paul Caron",
   "Jinliang Wei",
   "Lluis Castrejon",
   "Sharad Vikram",
   "Deepak Ramachandran",
   "Nikhil Dhawan",
   "Jiho Park",
   "Sara Smoot",
   "George van den Driessche",
   "Yochai Blau",
   "Chase Malik",
   "Wei Liang",
   "Roy Hirsch",
   "Cicero Nogueira dos Santos",
   "Eugene Weinstein",
   "A\u00e4ron van den Oord",
   "Sid Lall",
   "Nicholas FitzGerald",
   "Zixuan Jiang",
   "Xuan Yang",
   "Dale Webster",
   "Ali Elqursh",
   "Aedan Pope",
   "Georges Rotival",
   "David Raposo",
   "Wanzheng Zhu",
   "Jeff Dean",
   "Sami Alabed",
   "Dustin Tran",
   "Arushi Gupta",
   "Zach Gleicher",
   "Jessica Austin",
   "Edouard Rosseel",
   "Megh Umekar",
   "Dipanjan Das",
   "Yinghao Sun",
   "Kai Chen",
   "Karolis Misiunas",
   "Xiang Zhou",
   "Yixian Di",
   "Alyssa Loo",
   "Josh Newlan",
   "Bo Li",
   "Vinay Ramasesh",
   "Ying Xu",
   "Alex Chen",
   "Sudeep Gandhe",
   "Radu Soricut",
   "Nikita Gupta",
   "Shuguang Hu",
   "Seliem El-Sayed",
   "Xavier Garcia",
   "Idan Brusilovsky",
   "Pu-Chin Chen",
   "Andrew Bolt",
   "Lu Huang",
   "Alex Gurney",
   "Zhiying Zhang",
   "Alexander Pritzel",
   "Jarek Wilkiewicz",
   "Bryan Seybold",
   "Bhargav Kanagal Shamanna",
   "Felix Fischer",
   "Josef Dean",
   "Karan Gill",
   "Ross Mcilroy",
   "Abhishek Bhowmick",
   "Jeremy Selier",
   "Antoine Yang",
   "Derek Cheng",
   "Vladimir Magay",
   "Jie Tan",
   "Dhriti Varma",
   "Christian Walder",
   "Tomas Kocisky",
   "Ryo Nakashima",
   "Paul Natsev",
   "Mike Kwong",
   "Ionel Gog",
   "Chiyuan Zhang",
   "Sander Dieleman",
   "Thomas Jimma",
   "Andrey Ryabtsev",
   "Siddhartha Brahma",
   "David Steiner",
   "Dayou Du",
   "Ante \u017du\u017eul",
   "Mislav \u017dani\u0107",
   "Mukund Raghavachari",
   "Willi Gierke",
   "Zeyu Zheng",
   "Dessie Petrova",
   "Yann Dauphin",
   "Yuchuan Liu",
   "Ido Kessler",
   "Steven Hand",
   "Chris Duvarney",
   "Seokhwan Kim",
   "Hyo Lee",
   "L\u00e9onard Hussenot",
   "Jeffrey Hui",
   "Josh Smith",
   "Deepali Jain",
   "Jiawei Xia",
   "Gaurav Singh Tomar",
   "Keyvan Amiri",
   "Du Phan",
   "Fabian Fuchs",
   "Tobias Weyand",
   "Nenad Tomasev",
   "Alexandra Cordell",
   "Xin Liu",
   "Jonathan Mallinson",
   "Pankaj Joshi",
   "Andy Crawford",
   "Arun Suggala",
   "Steve Chien",
   "Nick Fernando",
   "Mariella Sanchez-Vargas",
   "Duncan Williams",
   "Phil Crone",
   "Xiyang Luo",
   "Igor Karpov",
   "Jyn Shan",
   "Terry Thurk",
   "Robin Strudel",
   "Paul Voigtlaender",
   "Piyush Patil",
   "Tim Dozat",
   "Ali Khodaei",
   "Sahil Singla",
   "Piotr Ambroszczyk",
   "Qiyin Wu",
   "Yifan Chang",
   "Brian Roark",
   "Chaitra Hegde",
   "Tianli Ding",
   "Angelos Filos",
   "Zhongru Wu",
   "Andr\u00e9 Susano Pinto",
   "Shuang Liu",
   "Saarthak Khanna",
   "Aditya Pandey",
   "Siobhan Mcloughlin",
   "Qiujia Li",
   "Sam Haves",
   "Allan Zhou",
   "Elena Buchatskaya",
   "Isabel Leal",
   "Peter de Boursac",
   "Nami Akazawa",
   "Nina Anderson",
   "Terry Chen",
   "Krishna Somandepalli",
   "Chen Liang",
   "Sheela Goenka",
   "Stephanie Winkler",
   "Alexander Grushetsky",
   "Yifan Ding",
   "Jamie Smith",
   "Fan Ye",
   "Jordi Pont-Tuset",
   "Eric Li",
   "Ruichao Li",
   "Tomer Golany",
   "Dawid Wegner",
   "Tao Jiang",
   "Omer Barak",
   "Yuan Shangguan",
   "Eszter V\u00e9rtes",
   "Renee Wong",
   "J\u00f6rg Bornschein",
   "Alex Tudor",
   "Michele Bevilacqua",
   "Tom Schaul",
   "Ankit Singh Rawat",
   "Yang Zhao",
   "Kyriakos Axiotis",
   "Lei Meng",
   "Cory McLean",
   "Jonathan Lai",
   "Jennifer Beattie",
   "Nate Kushman",
   "Yaxin Liu",
   "Blair Kutzman",
   "Fiona Lang",
   "Jingchen Ye",
   "Praneeth Netrapalli",
   "Pushkar Mishra",
   "Myriam Khan",
   "Megha Goel",
   "Rob Willoughby",
   "David Tian",
   "Honglei Zhuang",
   "JD Chen",
   "Zak Tsai",
   "Tasos Kementsietsidis",
   "Arjun Khare",
   "James Keeling",
   "Keyang Xu",
   "Nathan Waters",
   "Florent Altch\u00e9",
   "Ashok Popat",
   "Bhavishya Mittal",
   "David Saxton",
   "Dalia El Badawy",
   "Michael Mathieu",
   "Zheng Zheng",
   "Hao Zhou",
   "Nishant Ranka",
   "Richard Shin",
   "Qingnan Duan",
   "Tim Salimans",
   "Ioana Mihailescu",
   "Uri Shaham",
   "Ming-Wei Chang",
   "Yannis Assael",
   "Nishanth Dikkala",
   "Martin Izzard",
   "Vincent Cohen-Addad",
   "Cat Graves",
   "Vlad Feinberg",
   "Grace Chung",
   "DJ Strouse",
   "Danny Karmon",
   "Sahand Sharifzadeh",
   "Zoe Ashwood",
   "Khiem Pham",
   "Jon Blanton",
   "Alex Vasiloff",
   "Jarred Barber",
   "Mark Geller",
   "Aurick Zhou",
   "Fedir Zubach",
   "Tzu-Kuo Huang",
   "Lei Zhang",
   "Himanshu Gupta",
   "Matt Young",
   "Julia Proskurnia",
   "Ronny Votel",
   "Valentin Gabeur",
   "Gabriel Barcik",
   "Aditya Tripathi",
   "Hongkun Yu",
   "Geng Yan",
   "Beer Changpinyo",
   "Filip Paveti\u0107",
   "Amy Coyle",
   "Yasuhisa Fujii",
   "Jorge Gonzalez Mendez",
   "Tianhao Zhou",
   "Harish Rajamani",
   "Blake Hechtman",
   "Eddie Cao",
   "Da-Cheng Juan",
   "Yi-Xuan Tan",
   "Valentin Dalibard",
   "Yilun Du",
   "Natalie Clay",
   "Kaisheng Yao",
   "Wenhao Jia",
   "Dimple Vijaykumar",
   "Yuxiang Zhou",
   "Xinyi Bai",
   "Wei-Chih Hung",
   "Steven Pecht",
   "Georgi Todorov",
   "Nikhil Khadke",
   "Pramod Gupta",
   "Preethi Lahoti",
   "Arnaud Autef",
   "Karthik Duddu",
   "James Lee-Thorp",
   "Alexander Bykovsky",
   "Tautvydas Misiunas",
   "Sebastian Flennerhag",
   "Santhosh Thangaraj",
   "Jed McGiffin",
   "Zack Nado",
   "Markus Kunesch",
   "Andreas Noever",
   "Amir Hertz",
   "Marco Liang",
   "Victor Stone",
   "Evan Palmer",
   "Samira Daruki",
   "Arijit Pramanik",
   "Siim P\u00f5der",
   "Austin Kyker",
   "Mina Khan",
   "Evgeny Sluzhaev",
   "Marvin Ritter",
   "Avraham Ruderman",
   "Wenlei Zhou",
   "Chirag Nagpal",
   "Kiran Vodrahalli",
   "George Necula",
   "Paul Barham",
   "Ellie Pavlick",
   "Jay Hartford",
   "Izhak Shafran",
   "Long Zhao",
   "Maciej Miku\u0142a",
   "Tom Eccles",
   "Hidetoshi Shimokawa",
   "Kanav Garg",
   "Luke Vilnis",
   "Hanwen Chen",
   "Ilia Shumailov",
   "Kuang-Huei Lee",
   "Abdelrahman Abdelhamed",
   "Meiyan Xie",
   "Vered Cohen",
   "Ester Hlavnova",
   "Dan Malkin",
   "Chawin Sitawarin",
   "James Lottes",
   "Pauline Coquinot",
   "Tianli Yu",
   "Sandeep Kumar",
   "Jingwei Zhang",
   "Aroma Mahendru",
   "Zafarali Ahmed",
   "James Martens",
   "Tao Chen",
   "Aviel Boag",
   "Daiyi Peng",
   "Coline Devin",
   "Arseniy Klimovskiy",
   "Mary Phuong",
   "Danny Vainstein",
   "Jin Xie",
   "Bhuvana Ramabhadran",
   "Nathan Howard",
   "Xinxin Yu",
   "Gitartha Goswami",
   "Jingyu Cui",
   "Sam Shleifer",
   "Mario Pinto",
   "Chih-Kuan Yeh",
   "Ming-Hsuan Yang",
   "Sara Javanmardi",
   "Dan Ethier",
   "Chace Lee",
   "Jordi Orbay",
   "Suyog Kotecha",
   "Carla Bromberg",
   "Pete Shaw",
   "James Thornton",
   "Adi Gerzi Rosenthal",
   "Shane Gu",
   "Matt Thomas",
   "Ian Gemp",
   "Aditya Ayyar",
   "Asahi Ushio",
   "Aarush Selvan",
   "Joel Wee",
   "Chenxi Liu",
   "Maryam Majzoubi",
   "Weiren Yu",
   "Jake Abernethy",
   "Tyler Liechty",
   "Renke Pan",
   "Hoang Nguyen",
   " Qiong",
   " Hu",
   "Sarah Perrin",
   "Abhinav Arora",
   "Emily Pitler",
   "Weiyi Wang",
   "Kaushik Shivakumar",
   "Flavien Prost",
   "Ben Limonchik",
   "Jing Wang",
   "Yi Gao",
   "Timothee Cour",
   "Shyamal Buch",
   "Huan Gui",
   "Maria Ivanova",
   "Philipp Neubeck",
   "Kelvin Chan",
   "Lucy Kim",
   "Huizhong Chen",
   "Naman Goyal",
   "Da-Woon Chung",
   "Lu Liu",
   "Yao Su",
   "Anastasia Petrushkina",
   "Jiajun Shen",
   "Armand Joulin",
   "Yuanzhong Xu",
   "Stein Xudong Lin",
   "Yana Kulizhskaya",
   "Ciprian Chelba",
   "Shobha Vasudevan",
   "Eli Collins",
   "Vasilisa Bashlovkina",
   "Tony Lu",
   "Doug Fritz",
   "Jongbin Park",
   "Yanqi Zhou",
   "Chen Su",
   "Richard Tanburn",
   "Mikhail Sushkov",
   "Mitchelle Rasquinha",
   "Jinning Li",
   "Jennifer Prendki",
   "Yiming Li",
   "Pallavi LV",
   "Shriya Sharma",
   "Hen Fitoussi",
   "Hui Huang",
   "Andrew Dai",
   "Phuong Dao",
   "Mike Burrows",
   "Henry Prior",
   "Danfeng Qin",
   "Golan Pundak",
   "Lars Lowe Sjoesund",
   "Art Khurshudov",
   "Zhenkai Zhu",
   "Albert Webson",
   "Elizabeth Kemp",
   "Tat Tan",
   "Saurabh Agrawal",
   "Susie Sargsyan",
   "Liqun Cheng",
   "Jim Stephan",
   "Tom Kwiatkowski",
   "David Reid",
   "Arunkumar Byravan",
   "Assaf Hurwitz Michaely",
   "Nicolas Heess",
   "Luowei Zhou",
   "Sonam Goenka",
   "Viral Carpenter",
   "Anselm Levskaya",
   "Bo Wang",
   "Reed Roberts",
   "R\u00e9mi Leblond",
   "Sharat Chikkerur",
   "Stav Ginzburg",
   "Max Chang",
   "Robert Riachi",
   " Chuqiao",
   " Xu",
   "Zal\u00e1n Borsos",
   "Michael Pliskin",
   "Julia Pawar",
   "Morgane Lustman",
   "Hannah Kirkwood",
   "Ankit Anand",
   "Aditi Chaudhary",
   "Norbert Kalb",
   "Kieran Milan",
   "Sean Augenstein",
   "Anna Goldie",
   "Laurel Prince",
   "Karthik Raman",
   "Yanhua Sun",
   "Vivian Xia",
   "Aaron Cohen",
   "Zhouyuan Huo",
   "Josh Camp",
   "Seher Ellis",
   "Lukas Zilka",
   "David Vilar Torres",
   "Lisa Patel",
   "Sho Arora",
   "Betty Chan",
   "Jonas Adler",
   "Kareem Ayoub",
   "Jacky Liang",
   "Fayaz Jamil",
   "Jiepu Jiang",
   "Simon Baumgartner",
   "Haitian Sun",
   "Yael Karov",
   "Yaroslav Akulov",
   "Hui Zheng",
   "Irene Cai",
   "Claudio Fantacci",
   "James Rubin",
   "Alex Rav Acha",
   "Mengchao Wang",
   "Nina D'Souza",
   "Rohit Sathyanarayana",
   "Shengyang Dai",
   "Simon Rowe",
   "Andrey Simanovsky",
   "Omer Goldman",
   "Yuheng Kuang",
   "Xiaoyue Pan",
   "Andrew Rosenberg",
   "Tania Rojas-Esponda",
   "Praneet Dutta",
   "Amy Zeng",
   "Irina Jurenka",
   "Greg Farquhar",
   "Yamini Bansal",
   "Shariq Iqbal",
   "Becca Roelofs",
   "Ga-Young Joung",
   "Parker Beak",
   "Changwan Ryu",
   "Ryan Poplin",
   "Yan Wu",
   "Jean-Baptiste Alayrac",
   "Senaka Buthpitiya",
   "Olaf Ronneberger",
   "Caleb Habtegebriel",
   "Wei Li",
   "Paul Cavallaro",
   "Aurora Wei",
   "Guy Bensky",
   "Timo Denk",
   "Harish Ganapathy",
   "Jeff Stanway",
   "Pratik Joshi",
   "Francesco Bertolini",
   "Jessica Lo",
   "Olivia Ma",
   "Zachary Charles",
   "Geta Sampemane",
   "Himanshu Sahni",
   "Xu Chen",
   "Harry Askham",
   "David Gaddy",
   "Peter Young",
   "Jiewen Tan",
   "Matan Eyal",
   "Arthur Bra\u017einskas",
   "Li Zhong",
   "Zhichun Wu",
   "Mark Epstein",
   "Kai Bailey",
   "Andrew Hard",
   "Kamyu Lee",
   "Sasha Goldshtein",
   "Alex Ruiz",
   "Mohammed Badawi",
   "Matthias Lochbrunner",
   "JK Kearns",
   "Ashley Brown",
   "Fabio Pardo",
   "Theophane Weber",
   "Haichuan Yang",
   "Pan-Pan Jiang",
   "Berkin Akin",
   "Zhao Fu",
   "Marcus Wainwright",
   "Chi Zou",
   "Meenu Gaba",
   "Pierre-Antoine Manzagol",
   "Wendy Kan",
   "Yang Song",
   "Karina Zainullina",
   "Rui Lin",
   "Jeongwoo Ko",
   "Salil Deshmukh",
   "Apoorv Jindal",
   "James Svensson",
   "Divya Tyam",
   "Heri Zhao",
   "Christine Kaeser-Chen",
   "Scott Baird",
   "Pooya Moradi",
   "Jamie Hall",
   "Qiuchen Guo",
   "Vincent Tsang",
   "Bowen Liang",
   "Fernando Pereira",
   "Suhas Ganesh",
   "Ivan Korotkov",
   "Jakub Adamek",
   "Sridhar Thiagarajan",
   "Vinh Tran",
   "Charles Chen",
   "Chris Tar",
   "Sanil Jain",
   "Ishita Dasgupta",
   "Taylan Bilal",
   "David Reitter",
   "Kai Zhao",
   "Giulia Vezzani",
   "Yasmin Gehman",
   "Pulkit Mehta",
   "Lauren Beltrone",
   "Xerxes Dotiwalla",
   "Sergio Guadarrama",
   "Zaheer Abbas",
   "Stefani Karp",
   "Petko Georgiev",
   "Chun-Sung Ferng",
   "Marc Brockschmidt",
   "Liqian Peng",
   "Christoph Hirnschall",
   "Vikas Verma",
   "Yingying Bi",
   "Ying Xiao",
   "Avigail Dabush",
   "Kelvin Xu",
   "Phil Wallis",
   "Randall Parker",
   "Qifei Wang",
   "Yang Xu",
   "Ilkin Safarli",
   "Dinesh Tewari",
   "Yin Zhang",
   "Seungyeon Kim",
   "Andrea Gesmundo",
   "Mackenzie Thomas",
   "Sergey Levi",
   "Ahmed Chowdhury",
   "Kanishka Rao",
   "Peter Garst",
   "Sam Conway-Rahman",
   "Helen Ran",
   "Kay McKinney",
   "Zhisheng Xiao",
   "Wenhao Yu",
   "Rohan Agrawal",
   "Axel Stjerngren",
   "Catalin Ionescu",
   "Jingjing Chen",
   "Vivek Sharma",
   "Justin Chiu",
   "Fei Liu",
   "Ken Franko",
   "Clayton Sanford",
   "Xingyu Cai",
   "Paul Michel",
   "Sanjay Ganapathy",
   "Jane Labanowski",
   "Zachary Garrett",
   "Ben Vargas",
   "Sean Sun",
   "Bryan Gale",
   "Thomas Buschmann",
   "Guillaume Desjardins",
   "Nimesh Ghelani",
   "Palak Jain",
   "Mudit Verma",
   "Chulayuth Asawaroengchai",
   "Julian Eisenschlos",
   "Jitendra Harlalka",
   "Hideto Kazawa",
   "Don Metzler",
   "Joshua Howland",
   "Ying Jian",
   "Jake Ades",
   "Viral Shah",
   "Tynan Gangwani",
   "Seungji Lee",
   "Roman Ring",
   "Steven M. Hernandez",
   "Dean Reich",
   "Amer Sinha",
   "Ashutosh Sathe",
   "Joe Kovac",
   "Ashleah Gill",
   "Ajay Kannan",
   "Andrea D'olimpio",
   "Martin Sevenich",
   "Jay Whang",
   "Been Kim",
   "Khe Chai Sim",
   "Jilin Chen",
   "Jiageng Zhang",
   "Shuba Lall",
   "Yossi Matias",
   "Bill Jia",
   "Abe Friesen",
   "Sara Nasso",
   "Ashish Thapliyal",
   "Bryan Perozzi",
   "Ting Yu",
   "Anna Shekhawat",
   "Safeen Huda",
   "Peter Grabowski",
   "Eric Wang",
   "Ashwin Sreevatsa",
   "Hilal Dib",
   "Mehadi Hassen",
   "Parker Schuh",
   "Vedrana Milutinovic",
   "Chris Welty",
   "Michael Quinn",
   "Ali Shah",
   "Bangju Wang",
   "Gabe Barth-Maron",
   "Justin Frye",
   "Natalie Axelsson",
   "Tao Zhu",
   "Yukun Ma",
   "Irene Giannoumis",
   "Hanie Sedghi",
   "Chang Ye",
   "Yi Luan",
   "Kevin Aydin",
   "Bilva Chandra",
   "Vivek Sampathkumar",
   "Ronny Huang",
   "Victor Lavrenko",
   "Ahmed Eleryan",
   "Zhi Hong",
   "Steven Hansen",
   "Sara Mc Carthy",
   "Bidisha Samanta",
   "Domagoj \u0106evid",
   "Xin Wang",
   "Fangtao Li",
   "Michael Voznesensky",
   "Matt Hoffman",
   "Andreas Terzis",
   "Vikash Sehwag",
   "Gil Fidel",
   "Luheng He",
   "Mu Cai",
   "Yanzhang He",
   "Alex Feng",
   "Martin Nikoltchev",
   "Samrat Phatale",
   "Jason Chase",
   "Rory Lawton",
   "Ming Zhang",
   "Tom Ouyang",
   "Manuel Tragut",
   "Mehdi Hafezi Manshadi",
   "Arjun Narayanan",
   "Jiaming Shen",
   "Xu Gao",
   "Tolga Bolukbasi",
   "Nick Roy",
   "Xin Li",
   "Daniel Golovin",
   "Liviu Panait",
   "Zhen Qin",
   "Guangxing Han",
   "Thomas Anthony",
   "Sneha Kudugunta",
   "Viorica Patraucean",
   "Aniket Ray",
   "Xinyun Chen",
   "Xiaochen Yang",
   "Tanuj Bhatia",
   "Pranav Talluri",
   "Alex Morris",
   "Andrija Ra\u017enatovi\u0107",
   "Bethanie Brownfield",
   "James An",
   "Sheng Peng",
   "Patrick Kane",
   "Ce Zheng",
   "Nico Duduta",
   "Joshua Kessinger",
   "James Noraky",
   "Siqi Liu",
   "Keran Rong",
   "Petar Veli\u010dkovi\u0107",
   "Keith Rush",
   "Alex Goldin",
   "Fanny Wei",
   "Shiva Mohan Reddy Garlapati",
   "Caroline Pantofaru",
   "Okwan Kwon",
   "Jianmo Ni",
   "Eric Noland",
   "Julia Di Trapani",
   "Fran\u00e7oise Beaufays",
   "Abhijit Guha Roy",
   "Yinlam Chow",
   "Aybuke Turker",
   "Geoffrey Cideron",
   "Lantao Mei",
   "Jon Clark",
   "Qingyun Dou",
   "Matko Bo\u0161njak",
   "Ralph Leith",
   "Yuqing Du",
   "Amir Yazdanbakhsh",
   "Milad Nasr",
   "Chester Kwak",
   "Suraj Satishkumar Sheth",
   "Alex Kaskasoli",
   "Ankesh Anand",
   "Balaji Lakshminarayanan",
   "Sammy Jerome",
   "David Bieber",
   "Chun-Te Chu",
   "Alexandre Senges",
   "Tianxiao Shen",
   "Mukund Sridhar",
   "Ndaba Ndebele",
   "Benjamin Beyret",
   "Shakir Mohamed",
   "Mia Chen",
   "Markus Freitag",
   "Jiaxian Guo",
   "Luyang Liu",
   "Paul Roit",
   "Heng Chen",
   "Shen Yan",
   "Tom Stone",
   "JD Co-Reyes",
   "Jeremy Cole",
   "Salvatore Scellato",
   "Shekoofeh Azizi",
   "Hadi Hashemi",
   "Alicia Jin",
   "Anand Iyer",
   "Marcella Valentine",
   "Andr\u00e1s Gy\u00f6rgy",
   "Arun Ahuja",
   "Daniel Hernandez Diaz",
   "Chen-Yu Lee",
   "Nathan Clement",
   "Weize Kong",
   "Drew Garmon",
   "Ishaan Watts",
   "Kush Bhatia",
   "Khyatti Gupta",
   "Matt Miecnikowski",
   "Hugo Vallet",
   "Ankur Taly",
   "Edward Loper",
   "Saket Joshi",
   "James Atwood",
   "Jo Chick",
   "Mark Collier",
   "Fotis Iliopoulos",
   "Ryan Trostle",
   "Beliz Gunel",
   "Ramiro Leal-Cavazos",
   "Arnar Mar Hrafnkelsson",
   "Michael Guzman",
   "Xiaoen Ju",
   "Andy Forbes",
   "Jesse Emond",
   "Kushal Chauhan",
   "Ben Caine",
   "Li Xiao",
   "Wenjun Zeng",
   "Alexandre Moufarek",
   "Daniel Murphy",
   "Maya Meng",
   "Nitish Gupta",
   "Felix Riedel",
   "Anil Das",
   "Elijah Lawal",
   "Shashi Narayan",
   "Tiberiu Sosea",
   "James Swirhun",
   "Linda Friso",
   "Behnam Neyshabur",
   "Jing Lu",
   "Sertan Girgin",
   "Michael Wunder",
   "Edouard Yvinec",
   "Aroonalok Pyne",
   "Victor Carbune",
   "Shruti Rijhwani",
   "Yang Guo",
   "Tulsee Doshi",
   "Anton Briukhov",
   "Max Bain",
   "Ayal Hitron",
   "Xuanhui Wang",
   "Ashish Gupta",
   "Ke Chen",
   "Cosmo Du",
   "Weiyang Zhang",
   "Dhruv Shah",
   "Arjun Akula",
   "Max Dylla",
   "Ashyana Kachra",
   "Weicheng Kuo",
   "Tingting Zou",
   "Lily Wang",
   "Luyao Xu",
   "Jifan Zhu",
   "Justin Snyder",
   "Sachit Menon",
   "Orhan Firat",
   "Igor Mordatch",
   "Yuan Yuan",
   "Natalia Ponomareva",
   "Rory Blevins",
   "Lawrence Moore",
   "Weijun Wang",
   "Phil Chen",
   "Martin Scholz",
   "Artur Dwornik",
   "Jason Lin",
   "Sicheng Li",
   "Diego Antognini",
   "Te I",
   "Xiaodan Song",
   "Matt Miller",
   "Uday Kalra",
   "Adam Raveret",
   "Oscar Akerlund",
   "Felix Wu",
   "Andrew Nystrom",
   "Namrata Godbole",
   "Tianqi Liu",
   "Hannah DeBalsi",
   "Jewel Zhao",
   "Buhuang Liu",
   "Avi Caciularu",
   "Lauren Lax",
   "Urvashi Khandelwal",
   "Victoria Langston",
   "Eric Bailey",
   "Silvio Lattanzi",
   "Yufei Wang",
   "Neel Kovelamudi",
   "Sneha Mondal",
   "Guru Guruganesh",
   "Nan Hua",
   "Ofir Roval",
   "Pawe\u0142 Weso\u0142owski",
   "Rishikesh Ingale",
   "Jonathan Halcrow",
   "Tim Sohn",
   "Christof Angermueller",
   "Bahram Raad",
   "Eli Stickgold",
   "Eva Lu",
   "Alec Kosik",
   "Jing Xie",
   "Timothy Lillicrap",
   "Austin Huang",
   "Lydia Lihui Zhang",
   "Dominik Paulus",
   "Clement Farabet",
   "Alex Wertheim",
   "Bing Wang",
   "Rishabh Joshi",
   "Chu-ling Ko",
   "Yonghui Wu",
   "Shubham Agrawal",
   "Lily Lin",
   "XiangHai Sheng",
   "Peter Sung",
   "Tyler Breland-King",
   "Christina Butterfield",
   "Swapnil Gawde",
   "Sumeet Singh",
   "Qiao Zhang",
   "Raj Apte",
   "Shilpa Shetty",
   "Adrian Hutter",
   "Tao Li",
   "Elizabeth Salesky",
   "Federico Lebron",
   "Jonni Kanerva",
   "Michela Paganini",
   "Arthur Nguyen",
   "Rohith Vallu",
   "Jan-Thorsten Peter",
   "Sarmishta Velury",
   "David Kao",
   "Jay Hoover",
   "Anna Bortsova",
   "Colton Bishop",
   "Shoshana Jakobovits",
   "Alessandro Agostini",
   "Alekh Agarwal",
   "Chang Liu",
   "Charles Kwong",
   "Sasan Tavakkol",
   "Ioana Bica",
   "Alex Greve",
   "Anirudh GP",
   "Jake Marcus",
   "Le Hou",
   "Tom Duerig",
   "Rivka Moroshko",
   "Dave Lacey",
   "Andy Davis",
   "Julien Amelot",
   "Guohui Wang",
   "Frank Kim",
   "Theofilos Strinopoulos",
   "Hui Wan",
   "Charline Le Lan",
   "Shankar Krishnan",
   "Haotian Tang",
   "Peter Humphreys",
   "Junwen Bai",
   "Idan Heimlich Shtacher",
   "Diego Machado",
   "Chenxi Pang",
   "Ken Burke",
   "Dangyi Liu",
   "Renga Aravamudhan",
   "Yue Song",
   "Ed Hirst",
   "Abhimanyu Singh",
   "Brendan Jou",
   "Liang Bai",
   "Francesco Piccinno",
   "Chuyuan Kelly Fu",
   "Robin Alazard",
   "Barak Meiri",
   "Daniel Winter",
   "Charlie Chen",
   "Mingda Zhang",
   "Jens Heitkaemper",
   "John Lambert",
   "Jinhyuk Lee",
   "Alexander Fr\u00f6mmgen",
   "Sergey Rogulenko",
   "Pranav Nair",
   "Paul Niemczyk",
   "Anton Bulyenov",
   "Bibo Xu",
   "Hadar Shemtov",
   "Morteza Zadimoghaddam",
   "Serge Toropov",
   "Mateo Wirth",
   "Hanjun Dai",
   "Sreenivas Gollapudi",
   "Daniel Zheng",
   "Alex Kurakin",
   "Chansoo Lee",
   "Kalesha Bullard",
   "Nicolas Serrano",
   "Ivana Balazevic",
   "Yang Li",
   "Johan Schalkwyk",
   "Mark Murphy",
   "Mingyang Zhang",
   "Kevin Sequeira",
   "Romina Datta",
   "Nishant Agrawal",
   "Charles Sutton",
   "Nithya Attaluri",
   "Mencher Chiang",
   "Wael Farhan",
   "Gregory Thornton",
   "Kate Lin",
   "Travis Choma",
   "Hung Nguyen",
   "Kingshuk Dasgupta",
   "Dirk Robinson",
   "Iulia Com\u015fa",
   "Michael Riley",
   "Arjun Pillai",
   "Basil Mustafa",
   "Ben Golan",
   "Amir Zandieh",
   "Jean-Baptiste Lespiau",
   "Billy Porter",
   "David Ross",
   "Sujeevan Rajayogam",
   "Mohit Agarwal",
   "Subhashini Venugopalan",
   "Bobak Shahriari",
   "Qiqi Yan",
   "Hao Xu",
   "Taylor Tobin",
   "Pavel Dubov",
   "Hongzhi Shi",
   "Adri\u00e0 Recasens",
   "Anton Kovsharov",
   "Sebastian Borgeaud",
   "Lucio Dery",
   "Shanthal Vasanth",
   "Elena Gribovskaya",
   "Linhai Qiu",
   "Mahdis Mahdieh",
   "Wojtek Skut",
   "Elizabeth Nielsen",
   "CJ Zheng",
   "Adams Yu",
   "Carrie Grimes Bostock",
   "Shaleen Gupta",
   "Aaron Archer",
   "Chris Rawles",
   "Elinor Davies",
   "Alexey Svyatkovskiy",
   "Tomy Tsai",
   "Yoni Halpern",
   "Christian Reisswig",
   "Bartek Wydrowski",
   "Bo Chang",
   "Joan Puigcerver",
   "Mor Hazan Taege",
   "Jian Li",
   "Eva Schnider",
   "Xinjian Li",
   "Dragos Dena",
   "Yunhan Xu",
   "Umesh Telang",
   "Tianze Shi",
   "Heiga Zen",
   "Kyle Kastner",
   "Yeongil Ko",
   "Neesha Subramaniam",
   "Aviral Kumar",
   "Pete Blois",
   "Zhuyun Dai",
   "John Wieting",
   "Yifeng Lu",
   "Yoel Zeldes",
   "Tian Xie",
   "Anja Hauth",
   "Alexandru \u0162ifrea",
   "Yuqi Li",
   "Sam El-Husseini",
   "Dan Abolafia",
   "Howard Zhou",
   "Wen Ding",
   "Sahra Ghalebikesabi",
   "Carlos Gu\u00eda",
   "Andrii Maksai",
   "\u00c1goston Weisz",
   "Sercan Arik",
   "Nick Sukhanov",
   "Aga \u015awietlik",
   "Xuhui Jia",
   "Luo Yu",
   "Weiyue Wang",
   "Mark Brand",
   "Dawn Bloxwich",
   "Sean Kirmani",
   "Zhe Chen",
   "Alec Go",
   "Pablo Sprechmann",
   "Nithish Kannen",
   "Alen Carin",
   "Paramjit Sandhu",
   "Isabel Edkins",
   "Leslie Nooteboom",
   "Jai Gupta",
   "Loren Maggiore",
   "Javad Azizi",
   "Yael Pritch",
   "Pengcheng Yin",
   "Mansi Gupta",
   "Danny Tarlow",
   "Duncan Smith",
   "Desi Ivanov",
   "Mohammad Babaeizadeh",
   "Ankita Goel",
   "Satish Kambala",
   "Grace Chu",
   "Matej Kastelic",
   "Michelle Liu",
   "Hagen Soltau",
   "Austin Stone",
   "Shivani Agrawal",
   "Min Kim",
   "Kedar Soparkar",
   "Srinivas Tadepalli",
   "Oskar Bunyan",
   "Rachel Soh",
   "Arvind Kannan",
   "DY Kim",
   "Blake JianHang Chen",
   "Afief Halumi",
   "Sudeshna Roy",
   "Yulong Wang",
   "Olcan Sercinoglu",
   "Gena Gibson",
   "Sijal Bhatnagar",
   "Motoki Sano",
   "Daniel von Dincklage",
   "Qingchun Ren",
   "Blagoj Mitrevski",
   "Mirek Ol\u0161\u00e1k",
   "Jennifer She",
   "Carl Doersch",
   " Jilei",
   " Wang",
   "Bingyuan Liu",
   "Qijun Tan",
   "Tamar Yakar",
   "Tris Warkentin",
   "Alex Ramirez",
   "Carl Lebsack",
   "Josh Dillon",
   "Rajiv Mathews",
   "Tom Cobley",
   "Zelin Wu",
   "Zhuoyuan Chen",
   "Jon Simon",
   "Swaroop Nath",
   "Tara Sainath",
   "Alexei Bendebury",
   "Ryan Julian",
   "Bharath Mankalale",
   "Daria \u0106urko",
   "Paulo Zacchello",
   "Adam R. Brown",
   "Kiranbir Sodhia",
   "Heidi Howard",
   "Sergi Caelles",
   "Abhinav Gupta",
   "Gareth Evans",
   "Anna Bulanova",
   "Lesley Katzen",
   "Roman Goldenberg",
   "Anton Tsitsulin",
   "Joe Stanton",
   "Benoit Schillings",
   "Vitaly Kovalev",
   "Corey Fry",
   "Rushin Shah",
   "Kuo Lin",
   "Shyam Upadhyay",
   "Cheng Li",
   "Soroush Radpour",
   "Marcello Maggioni",
   "Jing Xiong",
   "Lukas Haas",
   "Jenny Brennan",
   "Aishwarya Kamath",
   "Nikolay Savinov",
   "Arsha Nagrani",
   "Trevor Yacovone",
   "Ryan Kappedal",
   "Kostas Andriopoulos",
   "Li Lao",
   "YaGuang Li",
   "Grigory Rozhdestvenskiy",
   "Kazuma Hashimoto",
   "Andrew Audibert",
   "Sophia Austin",
   "Daniel Rodriguez",
   "Anian Ruoss",
   "Garrett Honke",
   "Deep Karkhanis",
   "Xi Xiong",
   "Qing Wei",
   "James Huang",
   "Zhaoqi Leng",
   "Vittal Premachandran",
   "Stan Bileschi",
   "Georgios Evangelopoulos",
   "Thomas Mensink",
   "Jay Pavagadhi",
   "Denis Teplyashin",
   "Paul Chang",
   "Linting Xue",
   "Garrett Tanzer",
   "Sally Goldman",
   "Kaushal Patel",
   "Shixin Li",
   "Jeremy Wiesner",
   "Ivy Zheng",
   "Ian Stewart-Binks",
   "Jie Han",
   "Zhi Li",
   "Liangchen Luo",
   "Karel Lenc",
   "Mario Lu\u010di\u0107",
   "Fuzhao Xue",
   "Ryan Mullins",
   "Alexey Guseynov",
   "Chung-Ching Chang",
   "Isaac Galatzer-Levy",
   "Adam Zhang",
   "Garrett Bingham",
   "Grace Hu",
   "Ale Hartman",
   "Yue Ma",
   "Jordan Griffith",
   "Alex Irpan",
   "Carey Radebaugh",
   "Summer Yue",
   "Lijie Fan",
   "Victor Ungureanu",
   "Christina Sorokin",
   "Hannah Teufel",
   "Peiran Li",
   "Rohan Anil",
   "Dimitris Paparas",
   "Todd Wang",
   "Chu-Cheng Lin",
   "Hui Peng",
   "Megan Shum",
   "Goran Petrovic",
   "Demetra Brady",
   "Richard Nguyen",
   "Klaus Macherey",
   "Zhihao Li",
   "Harman Singh",
   "Madhavi Yenugula",
   "Mariko Iinuma",
   "Xinyi Chen",
   "Kavya Kopparapu",
   "Alexey Stern",
   "Shachi Dave",
   "Chandu Thekkath",
   "Florence Perot",
   "Anurag Kumar",
   "Fangda Li",
   "Yang Xiao",
   "Matthew Bilotti",
   "Mohammad Hossein Bateni",
   "Isaac Noble",
   "Lisa Lee",
   "Amelio V\u00e1zquez-Reina",
   "Julian Salazar",
   "Xiaomeng Yang",
   "Boyu Wang",
   "Ela Gruzewska",
   "Anand Rao",
   "Sindhu Raghuram",
   "Zheng Xu",
   "Eyal Ben-David",
   "Jieru Mei",
   "Sid Dalmia",
   "Zhaoyi Zhang",
   "Yuchen Liu",
   "Gagan Bansal",
   "Helena Pankov",
   "Steven Schwarcz",
   "Andrea Burns",
   "Christine Chan",
   "Sumit Sanghai",
   "Ricky Liang",
   "Ethan Liang",
   "Antoine He",
   "Amy Stuart",
   "Arun Narayanan",
   "Yukun Zhu",
   "Christian Frank",
   "Bahar Fatemi",
   "Amit Sabne",
   "Oran Lang",
   "Indro Bhattacharya",
   "Shane Settle",
   "Maria Wang",
   "Brendan McMahan",
   "Andrea Tacchetti",
   "Livio Baldini Soares",
   "Majid Hadian",
   "Serkan Cabi",
   "Timothy Chung",
   "Nikita Putikhin",
   "Gang Li",
   "Jeremy Chen",
   "Austin Tarango",
   "Henryk Michalewski",
   "Mehran Kazemi",
   "Hussain Masoom",
   "Hila Sheftel",
   "Rakesh Shivanna",
   "Archita Vadali",
   "Ramona Comanescu",
   "Doug Reid",
   "Joss Moore",
   "Arvind Neelakantan",
   "Micha\u00ebl Sander",
   "Jonathan Herzig",
   "Aviv Rosenberg",
   "Mostafa Dehghani",
   "JD Choi",
   "Michael Fink",
   "Reid Hayes",
   "Eric Ge",
   "Shitao Weng",
   "Chia-Hua Ho",
   "John Karro",
   "Kalpesh Krishna",
   "Lam Nguyen Thiet",
   "Amy Skerry-Ryan",
   "Daniel Eppens",
   "Marco Andreetto",
   "Navin Sarma",
   "Silvano Bonacina",
   "Burcu Karagol Ayan",
   "Megha Nawhal",
   "Zhihao Shan",
   "Mike Dusenberry",
   "Shantanu Thakoor",
   "Sagar Gubbi",
   "Duc Dung Nguyen",
   "Reut Tsarfaty",
   "Samuel Albanie",
   "Jovana Mitrovi\u0107",
   "Meet Gandhi",
   "Bo-Juen Chen",
   "Alessandro Epasto",
   "Georgi Stephanov",
   "Ye Jin",
   "Samuel Gehman",
   "Aida Amini",
   "Jack Weber",
   "Feryal Behbahani",
   "Shawn Xu",
   "Miltos Allamanis",
   "Xi Chen",
   "Myle Ott",
   "Claire Sha",
   "Michal Jastrzebski",
   "Hang Qi",
   "David Greene",
   "Xinyi Wu",
   "Abodunrinwa Toki",
   "Daniel Vlasic",
   "Jane Shapiro",
   "Ragha Kotikalapudi",
   "Zhe Shen",
   "Takaaki Saeki",
   "Sirui Xie",
   "Albin Cassirer",
   "Shikhar Bharadwaj",
   "Tatsuya Kiyono",
   "Srinadh Bhojanapalli",
   "Elan Rosenfeld",
   "Sam Ritter",
   "Jieming Mao",
   "Jo\u00e3o Gabriel Oliveira",
   "Zoltan Egyed",
   "Bernd Bandemer",
   "Emilio Parisotto",
   "Keisuke Kinoshita",
   "Juliette Pluto",
   "Petros Maniatis",
   "Steve Li",
   "Yaohui Guo",
   "Golnaz Ghiasi",
   "Jean Tarbouriech",
   "Srimon Chatterjee",
   "Julie Jin",
   " Katrina",
   " Xu",
   "Jennimaria Palomaki",
   "S\u00e9b Arnold",
   "Madhavi Sewak",
   "Federico Piccinini",
   "Mohit Sharma",
   "Ben Albrecht",
   "Sean Purser-haskell",
   "Ashwin Vaswani",
   "Chongyan Chen",
   "Matheus Wisniewski",
   "Qin Cao",
   "John Aslanides",
   "Nguyet Minh Phu",
   "Maximilian Sieb",
   "Lauren Agubuzu",
   "Anne Zheng",
   "Daniel Sohn",
   "Marco Selvi",
   "Anders Andreassen",
   "Krishan Subudhi",
   "Prem Eruvbetine",
   "Oliver Woodman",
   "Tomas Mery",
   "Sebastian Krause",
   "Xiaoqi Ren",
   "Xiao Ma",
   "Jincheng Luo",
   "Dawn Chen",
   "Wei Fan",
   "Henry Griffiths",
   "Christian Schuler",
   "Alice Li",
   "Shujian Zhang",
   "Jean-Michel Sarr",
   "Shixin Luo",
   "Riccardo Patana",
   "Matthew Watson",
   "Dani Naboulsi",
   "Michael Collins",
   "Sailesh Sidhwani",
   "Emiel Hoogeboom",
   "Sharon Silver",
   "Emily Caveness",
   "Xiaokai Zhao",
   "Mikel Rodriguez",
   "Maxine Deines",
   "Libin Bai",
   "Patrick Griffin",
   "Marco Tagliasacchi",
   "Emily Xue",
   "Spandana Raj Babbula",
   "Bo Pang",
   "Nan Ding",
   "Gloria Shen",
   "Elijah Peake",
   "Remi Crocker",
   "Shubha Srinivas Raghvendra",
   "Danny Swisher",
   "Woohyun Han",
   "Richa Singh",
   "Ling Wu",
   "Vladimir Pchelin",
   "Tsendsuren Munkhdalai",
   "Dana Alon",
   "Geoff Bacon",
   "Efren Robles",
   "Jannis Bulian",
   "Melvin Johnson",
   "George Powell",
   "Felipe Tiengo Ferreira",
   "Yaoyiran Li",
   "Frederik Benzing",
   "Mihajlo Velimirovi\u0107",
   "Hubert Soyer",
   "William Kong",
   " Tony",
   " Nguy\u00ean",
   "Zhen Yang",
   "Jeremiah Liu",
   "Joost van Amersfoort",
   "Daniel Gillick",
   "Baochen Sun",
   "Nathalie Rauschmayr",
   "Katie Zhang",
   "Serena Zhan",
   "Tao Zhou",
   "Alexey Frolov",
   "Chengrun Yang",
   "Denis Vnukov",
   "Louis Rouillard",
   "Hongji Li",
   "Amol Mandhane",
   "Nova Fallen",
   "Rajesh Venkataraman",
   "Clara Huiyi Hu",
   "Jennifer Brennan",
   "Jenny Lee",
   "Jerry Chang",
   "Martin Sundermeyer",
   "Zhufeng Pan",
   "Rosemary Ke",
   "Simon Tong",
   "Alex Fabrikant",
   "William Bono",
   "Jindong Gu",
   "Ryan Foley",
   "Yiran Mao",
   "Manolis Delakis",
   "Dhruva Bhaswar",
   "Roy Frostig",
   "Nick Li",
   "Avital Zipori",
   "Cath Hope",
   "Olga Kozlova",
   "Swaroop Mishra",
   "Josip Djolonga",
   "Craig Schiff",
   "Majd Al Merey",
   "Eleftheria Briakou",
   "Peter Morgan",
   "Andy Wan",
   "Avinatan Hassidim",
   "RJ Skerry-Ryan",
   "Kuntal Sengupta",
   "Mary Jasarevic",
   "Praveen Kallakuri",
   "Paige Kunkle",
   "Hannah Brennan",
   "Tom Lieber",
   "Hassan Mansoor",
   "Julian Walker",
   "Bing Zhang",
   "Annie Xie",
   "Goran \u017du\u017ei\u0107",
   "Adaeze Chukwuka",
   "Alex Druinsky",
   "Donghyun Cho",
   "Rui Yao",
   "Ferjad Naeem",
   "Shiraz Butt",
   "Eunyoung Kim",
   "Zhipeng Jia",
   "Mandy Jordan",
   "Adam Lelkes",
   "Mark Kurzeja",
   "Sophie Wang",
   "James Zhao",
   "Andrew Over",
   "Abhishek Chakladar",
   "Marcel Prasetya",
   "Neha Jha",
   "Sriram Ganapathy",
   "Yale Cong",
   "Prakash Shroff",
   "Carl Saroufim",
   "Sobhan Miryoosefi",
   "Mohamed Hammad",
   "Tajwar Nasir",
   "Weijuan Xi",
   "Yang Gao",
   "Young Maeng",
   "Ben Hora",
   "Chin-Yi Cheng",
   "Parisa Haghani",
   "Yoad Lewenberg",
   "Caden Lu",
   "Martin Matysiak",
   "Naina Raisinghani",
   "Huiyu Wang",
   "Lexi Baugher",
   "Rahul Sukthankar",
   "Minh Giang",
   "John Schultz",
   "Noah Fiedel",
   "Minmin Chen",
   "Cheng-Chun Lee",
   "Tapomay Dey",
   "Hao Zheng",
   "Shachi Paul",
   "Celine Smith",
   "Andy Ly",
   "Yicheng Wang",
   "Rishabh Bansal",
   "Bartek Perz",
   "Susanna Ricco",
   "Stasha Blank",
   "Vaishakh Keshava",
   "Deepak Sharma",
   "Marvin Chow",
   "Kunal Lad",
   "Komal Jalan",
   "Simon Osindero",
   "Craig Swanson",
   "Jacob Scott",
   "Anastasija Ili\u0107",
   "Xiaowei Li",
   "Siddhartha Reddy Jonnalagadda",
   "Afzal Shama Soudagar",
   "Yan Xiong",
   "Bat-Orgil Batsaikhan",
   "Daniel Jarrett",
   "Naveen Kumar",
   "Maulik Shah",
   "Matt Lawlor",
   "Austin Waters",
   "Mark Graham",
   "Rhys May",
   "Sabela Ramos",
   "Sandra Lefdal",
   "Zeynep Cankara",
   "Nacho Cano",
   "Brendan O'Donoghue",
   "Jed Borovik",
   "Frederick Liu",
   "Jordan Grimstad",
   "Mahmoud Alnahlawi",
   "Katerina Tsihlas",
   "Tom Hudson",
   "Nikolai Grigorev",
   "Yiling Jia",
   "Terry Huang",
   "Tobenna Peter Igwe",
   "Sergei Lebedev",
   "Xiaodan Tang",
   "Igor Krivokon",
   "Frankie Garcia",
   "Melissa Tan",
   "Eric Jia",
   "Peter Stys",
   "Shikhar Vashishth",
   "Yu Liang",
   "Balaji Venkatraman",
   "Chenjie Gu",
   "Anastasios Kementsietsidis",
   "Chen Zhu",
   "Junehyuk Jung",
   "Yunfei Bai",
   "Mohammad Javad Hosseini",
   "Faruk Ahmed",
   "Aditya Gupta",
   "Xin Yuan",
   "Shereen Ashraf",
   "Shitij Nigam",
   "Gautam Vasudevan",
   "Pranjal Awasthi",
   "Adi Mayrav Gilady",
   "Zelda Mariet",
   "Ramy Eskander",
   "Haiguang Li",
   "Hexiang Hu",
   "Guillermo Garrido",
   "Philippe Schlattner",
   "George Zhang",
   "Rohun Saxena",
   "Petar Devi\u0107",
   "Kritika Muralidharan",
   "Ashwin Murthy",
   "Yiqian Zhou",
   "Min Choi",
   "Arissa Wongpanich",
   "Zhengdong Wang",
   "Premal Shah",
   "Yuntao Xu",
   "Yiling Huang",
   "Stephen Spencer",
   "Alice Chen",
   "James Cohan",
   "Junjie Wang",
   "Jonathan Tompson",
   "Junru Wu",
   "Ruba Haroun",
   "Haiqiong Li",
   "Blanca Huergo",
   "Fan Yang",
   "Tongxin Yin",
   "James Wendt",
   "Michael Bendersky",
   "Rahma Chaabouni",
   "Javier Snaider",
   "Johan Ferret",
   "Abhishek Jindal",
   "Tara Thompson",
   "Andrew Xue",
   "Will Bishop",
   "Shubham Milind Phal",
   "Archit Sharma",
   "Yunhsuan Sung",
   "Prabakar Radhakrishnan",
   "Mo Shomrat",
   "Reeve Ingle",
   "Roopali Vij",
   "Justin Gilmer",
   "Mihai Dorin Istin",
   "Sam Sobell",
   "Yang Lu",
   "Emily Nottage",
   "Dorsa Sadigh",
   "Jeremiah Willcock",
   "Tingnan Zhang",
   "Steve Xu",
   "Sasha Brown",
   "Katherine Lee",
   "Gary Wang",
   "Yun Zhu",
   "Yi Tay",
   "Cheolmin Kim",
   "Audrey Gutierrez",
   "Abhanshu Sharma",
   "Yongqin Xian",
   "Sungyong Seo",
   "Claire Cui",
   "Elena Pochernina",
   "Cip Baetu",
   "Krzysztof Jastrz\u0119bski",
   "Mimi Ly",
   "Mohamed Elhawaty",
   "Dan Suh",
   "Eren Sezener",
   "Pidong Wang",
   "Nancy Yuen",
   "George Tucker",
   "Jiahao Cai",
   "Zuguang Yang",
   "Cindy Wang",
   "Alex Muzio",
   "Hai Qian",
   "Jae Yoo",
   "Derek Lockhart",
   "Kevin R. McKee",
   "Mandy Guo",
   "Malika Mehrotra",
   "Artur Mendon\u00e7a",
   "Sanket Vaibhav Mehta",
   "Sherry Ben",
   "Chetan Tekur",
   "Jiaqi Mu",
   "Muye Zhu",
   "Victoria Krakovna",
   "Hongrae Lee",
   "AJ Maschinot",
   "S\u00e9bastien Cevey",
   "HyunJeong Choe",
   "Aijun Bai",
   "Hansa Srinivasan",
   "Derek Gasaway",
   "Nick Young",
   "Patrick Siegler",
   "Dan Holtmann-Rice",
   "Vihari Piratla",
   "Kate Baumli",
   "Roey Yogev",
   "Alex Hofer",
   "Hado van Hasselt",
   "Svetlana Grant",
   "Yuri Chervonyi",
   "David Silver",
   "Andrew Hogue",
   "Ayushi Agarwal",
   "Kathie Wang",
   "Preeti Singh",
   "Four Flynn",
   "Josh Lipschultz",
   "Robert David",
   "Lizzetth Bellot",
   "Yao-Yuan Yang",
   "Long Le",
   "Filippo Graziano",
   "Kate Olszewska",
   "Kevin Hui",
   "Akanksha Maurya",
   "Nikos Parotsidis",
   "Weijie Chen",
   "Tayo Oguntebi",
   "Joe Kelley",
   "Anirudh Baddepudi",
   "Johannes Mauerer",
   "Gregory Shaw",
   "Alex Siegman",
   "Lin Yang",
   "Shravya Shetty",
   "Subhrajit Roy",
   "Yunting Song",
   "Wojciech Stokowiec",
   "Ryan Burnell",
   "Omkar Savant",
   "Robert Busa-Fekete",
   "Jin Miao",
   "Samrat Ghosh",
   "Liam MacDermed",
   "Phillip Lippe",
   "Mikhail Dektiarev",
   "Zach Behrman",
   "Fabian Mentzer",
   "Kelvin Nguyen",
   "Meng Wei",
   "Siddharth Verma",
   "Chris Knutsen",
   "Sudeep Dasari",
   "Zhipeng Yan",
   "Petr Mitrichev",
   "Xingyu Wang",
   "Virat Shejwalkar",
   "Jacob Austin",
   "Srinivas Sunkara",
   "Navneet Potti",
   "Yan Virin",
   "Christian Wright",
   "Ga\u00ebl Liu",
   "Oriana Riva",
   "Etienne Pot",
   "Greg Kochanski",
   "Quoc Le",
   "Gargi Balasubramaniam",
   "Arka Dhar",
   "Yuguo Liao",
   "Adam Bloniarz",
   "Divyansh Shukla",
   "Elizabeth Cole",
   "Jong Lee",
   "Sheng Zhang",
   "Sushant Kafle",
   "Siddharth Vashishtha",
   "Parsa Mahmoudieh",
   "Grace Chen",
   "Raphael Hoffmann",
   "Pranesh Srinivasan",
   "Agustin Dal Lago",
   "Yoav Ben Shalom",
   "Zi Wang",
   "Michael Elabd",
   "Anuj Sharma",
   "Junhyuk Oh",
   "Suraj Kothawade",
   "Maigo Le",
   "Marianne Monteiro",
   "Shentao Yang",
   "Kaiz Alarakyia",
   "Robert Geirhos",
   "Diana Mincu",
   "H\u00e5vard Garnes",
   "Hayato Kobayashi",
   "Soroosh Mariooryad",
   "Kacper Krasowiak",
   " Zhixin",
   " Lai",
   "Shibl Mourad",
   "Mingqiu Wang",
   "Fan Bu",
   "Ophir Aharoni",
   "Guanjie Chen",
   "Abhimanyu Goyal",
   "Vadim Zubov",
   "Ankur Bapna",
   "Elahe Dabir",
   "Nisarg Kothari",
   "Kay Lamerigts",
   "Nicola De Cao",
   "Jeremy Shar",
   "Christopher Yew",
   "Nitish Kulkarni",
   "Dre Mahaarachchi",
   "Mandar Joshi",
   "Zhenhai Zhu",
   "Jared Lichtarge",
   "Yichao Zhou",
   "Hannah Muckenhirn",
   "Vittorio Selo",
   "Oriol Vinyals",
   "Peter Chen",
   "Anthony Brohan",
   "Vaibhav Mehta",
   "Sarah Cogan",
   "Ruth Wang",
   "Ty Geri",
   "Wei-Jen Ko",
   "Wei Chen",
   "Fabio Viola",
   "Keshav Shivam",
   "Lisa Wang",
   "Madeleine Clare Elish",
   "Raluca Ada Popa",
   "S\u00e9bastien Pereira",
   "Jianqiao Liu",
   "Raphael Koster",
   "Donnie Kim",
   "Gufeng Zhang",
   "Sayna Ebrahimi",
   "Partha Talukdar",
   "Yanyan Zheng",
   "Petra Poklukar",
   "Ales Mikhalap",
   "Dale Johnson",
   "Anitha Vijayakumar",
   "Mark Omernick",
   "Matt Dibb",
   "Ayush Dubey",
   "Qiong Hu",
   "Apurv Suman",
   "Vaibhav Aggarwal",
   "Ilya Kornakov",
   "Fei Xia",
   "Wing Lowe",
   "Alexey Kolganov",
   "Ted Xiao",
   "Vitaly Nikolaev",
   "Steven Hemingray",
   "Bonnie Li",
   "Joana Iljazi",
   "Miko\u0142aj Rybi\u0144ski",
   "Ballie Sandhu",
   "Peggy Lu",
   "Thang Luong",
   "Rodolphe Jenatton",
   "Vineetha Govindaraj",
   " Hui",
   " Li",
   "Gabriel Dulac-Arnold",
   "Wonpyo Park",
   "Henry Wang",
   "Abhinit Modi",
   "Jean Pouget-Abadie",
   "Kristina Greller",
   "Rahul Gupta",
   "Robert Berry",
   "Prajit Ramachandran",
   "Jinyu Xie",
   "Liam McCafferty",
   "Jianling Wang",
   "Kilol Gupta",
   "Hyeontaek Lim",
   "Bla\u017e Bratani\u010d",
   "Andy Brock",
   "Ilia Akolzin",
   "Jim Sproch",
   "Dan Karliner",
   "Duhyeon Kim",
   "Adrian Goedeckemeyer",
   "Noam Shazeer",
   "Cordelia Schmid",
   "Daniele Calandriello",
   "Parul Bhatia",
   "Krzysztof Choromanski",
   "Ceslee Montgomery",
   "Dheeru Dua",
   "Ana Ramalho",
   "Helen King",
   "Yue Gao",
   "Lynn Nguyen",
   "David Lindner",
   "Divya Pitta",
   "Oleaser Johnson",
   "Khalid Salama",
   "Diego Ardila",
   "Michael Han",
   "Erin Farnese",
   "Seth Odoom",
   "Ziyue Wang",
   "Xiangzhuo Ding",
   "Norman Rink",
   "Ray Smith",
   "Harshal Tushar Lehri",
   "Eden Cohen",
   "Neera Vats",
   "Tong He",
   "Parthasarathy Gopavarapu",
   "Adam Paszke",
   "Miteyan Patel",
   "Wouter Van Gansbeke",
   "Lucia Loher",
   "Luis Castro",
   "Maria Voitovich",
   "Tamara von Glehn",
   "Nelson George",
   "Simon Niklaus",
   "Zach Eaton-Rosen",
   "Nemanja Raki\u0107evi\u0107",
   "Erik Jue",
   "Sagi Perel",
   "Carrie Zhang",
   "Yuval Bahat",
   "Ang\u00e9line Pouget",
   "Zhi Xing",
   "Fantine Huot",
   "Ashish Shenoy",
   "Taylor Bos",
   "Vincent Coriou",
   "Bryan Richter",
   "Natasha Noy",
   "Yaqing Wang",
   "Santiago Ontanon",
   "Siyang Qin",
   "Gleb Makarchuk",
   "Demis Hassabis",
   "Zhuowan Li",
   "Mandar Sharma",
   "Kumaran Venkatesan",
   "Iurii Kemaev",
   "Roxanne Daniel",
   "Shiyu Huang",
   "Saloni Shah",
   "Octavio Ponce",
   " Warren",
   " Chen",
   "Manaal Faruqui",
   "Jialin Wu",
   "Slavica Anda\u010di\u0107",
   "Szabolcs Payrits",
   "Daniel McDuff",
   "Tom Hume",
   "Yuan Cao",
   "MH Tessler",
   "Qingze Wang",
   "Yinan Wang",
   "Ivor Rendulic",
   "Eirikur Agustsson",
   "Matthew Johnson",
   "Tanya Lando",
   "Andrew Howard",
   "Sri Gayatri Sundara Padmanabhan",
   "Mayank Daswani",
   "Andrea Banino",
   "Michael Kilgore",
   "Jonathan Heek",
   "Ziwei Ji",
   "Alvaro Caceres",
   "Conglong Li",
   "Nora Kassner",
   "Alexey Vlaskin",
   "Zeyu Liu",
   "Alex Grills",
   "Yanhan Hou",
   "Roykrong Sukkerd",
   "Gowoon Cheon",
   "Nishita Shetty",
   "Larisa Markeeva",
   "Piotr Stanczyk",
   "Tejas Iyer",
   "Yuan Gong",
   "Shawn Gao",
   "Keerthana Gopalakrishnan",
   "Tim Blyth",
   "Malcolm Reynolds",
   "Avishkar Bhoopchand",
   "Misha Bilenko",
   "Dero Gharibian",
   "Vicky Zayats",
   "Aleksandra Faust",
   "Abhinav Singh",
   "Min Ma",
   "Hongyang Jiao",
   "Sudheendra Vijayanarasimhan",
   "Lora Aroyo",
   "Vikas Yadav",
   "Sarah Chakera",
   "Ashwin Kakarla",
   "Vilobh Meshram",
   "Karol Gregor",
   "Gabriela Botea",
   "Evan Senter",
   "Dawei Jia",
   "Geza Kovacs",
   "Neha Sharma",
   "Sebastien Baur",
   "Kai Kang",
   "Yifan He",
   "Lin Zhuo",
   "Marija Kostelac",
   "Itay Laish",
   "Songyou Peng",
   "Louis O'Bryan",
   "Daniel Kasenberg",
   "Girish Ramchandra Rao",
   "Edouard Leurent",
   "Biao Zhang",
   "Sage Stevens",
   "Ana Salazar",
   "Ye Zhang",
   "Ivan Lobov",
   "Jake Walker",
   "Allen Porter",
   "Morgan Redshaw",
   "Han Ke",
   "Abhishek Rao",
   "Alex Lee",
   "Hoi Lam",
   "Michael Moffitt",
   "Jaeyoun Kim",
   "Siyuan Qiao",
   "Terry Koo",
   "Robert Dadashi",
   "Xinying Song",
   "Mukund Sundararajan",
   "Peng Xu",
   "Chizu Kawamoto",
   "Yan Zhong",
   "Clara Barbu",
   "Apoorv Reddy",
   "Mauro Verzetti",
   "Leon Li",
   "George Papamakarios",
   "Hanna Klimczak-Pluci\u0144ska",
   "Mary Cassin",
   "Koray Kavukcuoglu",
   "Rigel Swavely",
   "Alain Vaucher",
   "Jeffrey Zhao",
   "Ross Hemsley",
   "Michael Tschannen",
   "Heming Ge",
   "Gaurav Menghani",
   "Yang Yu",
   "Natalie Ha",
   "Wei He",
   "Xiao Wu",
   "Maggie Song",
   "Rachel Sterneck",
   "Stefan Zinke",
   "Dan A. Calian",
   "Annie Marsden",
   "Alejandro Cruzado Ruiz",
   "Matteo Hessel",
   "Almog Gueta",
   "Benjamin Lee",
   "Brian Farris",
   "Manish Gupta",
   "Yunjie Li",
   "Mohammad Saleh",
   "Vedant Misra",
   "Kefan Xiao",
   "Piermaria Mendolicchio",
   "Gavin Buttimore",
   "Varvara Krayvanova",
   "Nigamaa Nayakanti",
   "Matthew Wiethoff",
   "Yash Pande",
   "Azalia Mirhoseini",
   "Ni Lao",
   "Jasmine Liu",
   "Yiqing Hua",
   "Angie Chen",
   "Yury Malkov",
   "Dmitry Kalashnikov",
   "Shubham Gupta",
   "Kartik Audhkhasi",
   "Yuexiang Zhai",
   "Sudhindra Kopalle",
   "Prateek Jain",
   "Eran Ofek",
   "Clemens Meyer",
   "Khuslen Baatarsukh",
   "Hana Strej\u010dek",
   "Jun Qian",
   "James Freedman",
   "Ricardo Figueira",
   "Michal Sokolik",
   "Olivier Bachem",
   "Raymond Lin",
   "Dia Kharrat",
   "Chris Hidey",
   "Pingmei Xu",
   "Dennis Duan",
   "Yin Li",
   "Muge Ersoy",
   "Richard Everett",
   "Kevin Cen",
   "Rebeca Santamaria-Fernandez",
   "Amir Taubenfeld",
   "Ian Mackinnon",
   "Linda Deng",
   "Polina Zablotskaia",
   "Shashank Viswanadha",
   "Shivanker Goel",
   "Damion Yates",
   "Yunxiao Deng",
   "Peter Choy",
   "Mingqing Chen",
   "Abhishek Sinha",
   "Alex Mossin",
   "Yiming Wang",
   "Arthur Szlam",
   "Susan Hao",
   "Paul Kishan Rubenstein",
   "Metin Toksoz-Exley",
   "Miranda Aperghis",
   "Yin Zhong",
   "Junwhan Ahn",
   "Michael Isard",
   "Olivier Lacombe",
   "Florian Luisier",
   "Chrysovalantis Anastasiou",
   "Yogesh Kalley",
   "Utsav Prabhu",
   "Emma Dunleavy",
   "Shaan Bijwadia",
   "Justin Mao-Jones",
   "Kelly Chen",
   "Rama Pasumarthi",
   "Emily Wood",
   "Adil Dostmohamed",
   "Nate Hurley",
   "Jiri Simsa",
   "Alicia Parrish",
   "Mantas Pajarskas",
   "Matt Harvey",
   "Ondrej Skopek",
   "Yony Kochinski",
   "Javier Rey",
   "Verena Rieser",
   "Denny Zhou",
   "Sun Jae Lee",
   "Trilok Acharya",
   "Guowang Li",
   "Joe Jiang",
   "Xiaofan Zhang",
   "Bryant Gipson",
   "Ethan Mahintorabi",
   "Marco Gelmi",
   "Nima Khajehnouri",
   "Angel Yeh",
   "Kayi Lee",
   "Loic Matthey",
   "Leslie Baker",
   "Trang Pham",
   "Han Fu",
   "Alex Pak",
   "Prakhar Gupta",
   "Cristina Vasconcelos",
   "Adam Sadovsky",
   "Brian Walker",
   "Sissie Hsiao",
   "Patrik Zochbauer",
   "Andreea Marzoca",
   "Noam Velan",
   "Junhao Zeng",
   "Gilles Baechler",
   "Danny Driess",
   "Divya Jain",
   "Yanping Huang",
   "Lizzie Tao",
   "John Maggs",
   "Nir Levine",
   "Jon Schneider",
   "Erika Gemzer",
   "Samuel Petit",
   "Shan Han",
   "Zach Fisher",
   "Dustin Zelle",
   "Courtney Biles",
   "Eugene Ie",
   "Asya Fadeeva",
   "Casper Liu",
   "Juliana Vicente Franco",
   "Adrian Collister",
   "Hao Zhang",
   "Renshen Wang",
   "Ruizhe Zhao",
   "Leandro Kieliger",
   "Kurt Shuster",
   "Rui Zhu",
   "Boqing Gong",
   "Lawrence Chan",
   "Ruoxi Sun",
   "Sujoy Basu",
   "Roland Zimmermann",
   "Jamie Hayes",
   "Abhishek Bapna",
   "Jasper Snoek",
   "Weel Yang",
   "Puranjay Datta",
   "Jad Al Abdallah",
   "Kevin Kilgour",
   "Lu Li",
   "SQ Mah",
   "Yennie Jun",
   "Morgane Rivi\u00e8re",
   "Abhijit Karmarkar",
   "Tammo Spalink",
   "Tao Huang",
   "Lucas Gonzalez",
   "Duc-Hieu Tran",
   "Averi Nowak",
   "John Palowitch",
   "Martin Chadwick",
   "Ellie Talius",
   "Harsh Mehta",
   "Thibault Sellam",
   "Philipp Fr\u00e4nken",
   "Massimo Nicosia",
   "Kyle He",
   "Aditya Kini",
   "David Amos",
   "Sugato Basu",
   "Harrison Jobe",
   "Eleni Shaw",
   "Qiantong Xu",
   "Colin Evans",
   "Daisuke Ikeda",
   "Chaochao Yan",
   "Larry Jin",
   "Lun Wang",
   "Sachin Yadav",
   "Ilia Labzovsky",
   "Ramesh Sampath",
   "Ada Ma",
   "Candice Schumann",
   "Aditya Siddhant",
   "Rohin Shah",
   "John Youssef",
   "Rishabh Agarwal",
   "Natalie Dabney",
   "Alessio Tonioni",
   "Moran Ambar",
   "Jing Li",
   "Isabelle Guyon",
   "Benny Li",
   "David Soergel",
   "Boya Fang",
   "Georgi Karadzhov",
   "Cristian Udrescu",
   "Trieu Trinh",
   "Vikas Raunak",
   "Seb Noury",
   "Dee Guo",
   "Sonal Gupta",
   "Mara Finkelstein",
   "Denis Petek",
   "Lihao Liang",
   "Greg Billock",
   "Pei Sun",
   "David Wood",
   "Yiwen Song",
   "Xiaobin Yu",
   "Tatiana Matejovicova",
   "Regev Cohen",
   "Kalyan Andra",
   "David D'Ambrosio",
   "Zhiwei Deng",
   "Vincent Nallatamby",
   "Ebrahim Songhori",
   "Rumen Dangovski",
   "Andrew Lampinen",
   "Pankil Botadra",
   "Adam Hillier",
   "Jiawei Cao",
   "Nagabhushan Baddi",
   "Adhi Kuncoro",
   "Toshihiro Yoshino",
   "Ankit Bhagatwala",
   "Marc\u00e1urelio Ranzato",
   "Rylan Schaeffer",
   "Tianlin Liu",
   "Shuai Ye",
   "Obaid Sarvana",
   "John Nham",
   "Chenkai Kuang",
   "Isabel Gao",
   "Jinoo Baek",
   "Shubham Mittal",
   "Ayzaan Wahid",
   "Anita Gergely",
   "Bin Ni",
   "Josh Feldman",
   "Carrie Muir",
   "Pascal Lamblin",
   "Wolfgang Macherey",
   "Ethan Dyer",
   "Logan Kilpatrick",
   "V\u00edctor Campos",
   "Mukul Bhutani",
   "Stanislav Fort",
   "Yanif Ahmad",
   "Aliaksei Severyn",
   "Kleopatra Chatziprimou",
   "Oleksandr Ferludin",
   "Mason Dimarco",
   "Aditya Kusupati",
   "Joe Heyward",
   "Dan Bahir",
   "Kevin Villela",
   "Katie Millican",
   "Dror Marcus",
   "Sanaz Bahargam",
   "Caglar Unlu",
   "Nicholas Roth",
   "Zichuan Wei",
   "Siddharth Gopal",
   "Deepanway Ghoshal",
   "Edward Lee",
   "Sharon Lin",
   "Jennie Lees",
   "Dayeong Lee",
   "Anahita Hosseini",
   "Connie Fan",
   "Seth Neel",
   "Marcus Wu",
   "Yasemin Altun",
   "Honglong Cai",
   "Enrique Piqueras",
   "Josh Woodward",
   "Alessandro Bissacco",
   "Salem Haykal",
   "Mahyar Bordbar",
   "Prasha Sundaram",
   "Sarah Hodkinson",
   "Daniel Toyama",
   "George Polovets",
   "Austin Myers",
   "Anu Sinha",
   "Tomer Levinboim",
   "Kashyap Krishnakumar",
   "Rachita Chhaparia",
   "Tatiana Sholokhova",
   "Nitesh Bharadwaj Gundavarapu",
   "Ganesh Jawahar",
   "Haroon Qureshi",
   "Jieru Hu",
   "Nikola Momchev",
   "Matthew Rahtz",
   "Renjie Wu",
   "Aishwarya P S",
   "Kedar Dhamdhere",
   "Meiqi Guo",
   "Umang Gupta",
   "Ali Eslami",
   "Mariano Schain",
   "Michiel Blokzijl",
   "David Welling",
   "Dave Orr",
   "Levent Bolelli",
   "Nicolas Perez-Nieves",
   "Mikhail Sirotenko",
   "Aman Prasad",
   "Arjun Kar",
   "Borja De Balle Pigem",
   "Tayfun Terzi",
   "Gell\u00e9rt Weisz",
   "Dipankar Ghosh",
   "Aditi Mavalankar",
   "Dhruv Madeka",
   "Kaspar Daugaard",
   "Hartwig Adam",
   "Viraj Shah",
   "Dana Berman",
   "Maggie Tran",
   "Steven Baker",
   "Ewa Andrejczuk",
   "Grishma Chole",
   "Ganna Raboshchuk",
   "Mahdi Mirzazadeh",
   "Thais Kagohara",
   "Shimu Wu",
   "Christian Schallhart",
   "Bernett Orlando",
   "Chen Wang",
   "Alban Rrustemi",
   "Hao Xiong",
   "Hao Liu",
   "Arpi Vezer",
   "Nolan Ramsden",
   "Shuo-yiin Chang",
   "Sidharth Mudgal",
   "Yan Li",
   "Nino Vieillard",
   "Yedid Hoshen",
   "Farooq Ahmad",
   "Ambrose Slone",
   "Amy Hua",
   "Natan Potikha",
   "Mirko Rossini",
   "Jon Stritar",
   "Sushant Prakash",
   "Zifeng Wang",
   "Xuanyi Dong",
   "Alireza Nazari",
   "Efrat Nehoran",
   "Kaan Tekelioglu",
   "Yinxiao Li",
   "Kartikeya Badola",
   "Tom Funkhouser",
   "Yuanzhen Li",
   "Varun Yerram",
   "Ramya Ganeshan",
   "Daniel Formoso",
   "Karol Langner",
   "Tian Shi",
   "Huijian Li",
   "Yumeya Yamamori",
   "Amayika Panda",
   "Alaa Saade",
   "Angelo Scorza Scarpati",
   "Chris Breaux",
   "CJ Carey",
   "Zongwei Zhou",
   "Cho-Jui Hsieh",
   "Sophie Bridgers",
   "Alena Butryna",
   "Nishesh Gupta",
   "Vaibhav Tulsyan",
   "Sanghyun Woo",
   "Evgenii Eltyshev",
   "Will Grathwohl",
   "Chanel Parks",
   "Seth Benjamin",
   "Rina Panigrahy",
   "Shenil Dodhia",
   "Daniel De Freitas",
   "Chris Sauer",
   "Will Song",
   "Ferran Alet",
   "Jackson Tolins",
   "Cosmin Paduraru",
   "Xingyi Zhou",
   "Brian Albert",
   "Zizhao Zhang",
   "Lei Shu",
   "Mudit Bansal",
   "Sarah Nguyen",
   "Amir Globerson",
   "Owen Xiao",
   "James Manyika",
   "Tom Hennigan",
   "Rong Rong",
   "Josip Matak",
   "Anton Bakalov",
   "Ankur Sharma",
   "Danila Sinopalnikov",
   "Andrew Pierson",
   "Stephen Roller",
   "Geoff Brown",
   "Mingcen Gao",
   "Toshiyuki Fukuzawa",
   "Amin Ghafouri",
   "Kenny Vassigh",
   "Iain Barr",
   "Zhicheng Wang",
   "Anna Korsun",
   "Rajesh Jayaram",
   "Lijie Ren",
   "Tim Zaman",
   "Samira Khan",
   "Yana Lunts",
   "Dan Deutsch",
   "Dave Uthus",
   "Nitzan Katz",
   "Masha Samsikova",
   "Amr Khalifa",
   "Nikhil Sethi",
   "Jiao Sun",
   "Luming Tang",
   "Uri Alon",
   "Xianghong Luo",
   "Dian Yu",
   "Abhishek Nayyar",
   "Bryce Petrini",
   "Will Truong",
   "Vincent Hellendoorn",
   "Nikolai Chinaev",
   "Chris Alberti",
   "Wei Wang",
   "Jingcao Hu",
   "Vahab Mirrokni",
   "Ananth Balashankar",
   "Avia Aharon",
   "Aahil Mehta",
   "Ahmet Iscen",
   "Joseph Kready",
   "Lucas Manning",
   "Anhad Mohananey",
   "Yuankai Chen",
   "Anshuman Tripathi",
   "Allen Wu",
   "Igor Petrovski",
   "Dawsen Hwang",
   "Martin Baeuml",
   "Shreyas Chandrakaladharan",
   "Yuan Liu",
   "Rey Coaguila",
   "Maxwell Chen",
   "Sally Ma",
   "Pouya Tafti",
   "Susheel Tatineni",
   "Terry Spitz",
   "Jiayu Ye",
   "Paul Vicol",
   "Mihaela Rosca",
   "Adri\u00e0 Puigdom\u00e8nech",
   "Zohar Yahav",
   "Sanjay Ghemawat",
   "Hanzhao Lin",
   "Phoebe Kirk",
   "Zaid Nabulsi",
   "Sergey Brin",
   "Bernd Bohnet",
   "Ken Caluwaerts",
   "Aditya Srikanth Veerubhotla",
   "Dan Zheng",
   "Zihang Dai",
   "Petre Petrov",
   "Yichong Xu",
   "Ramin Mehran",
   "Zhuo Xu",
   "Luisa Zintgraf",
   "Jiho Choi",
   "Spurthi Amba Hombaiah",
   "Romal Thoppilan",
   "Sashank Reddi",
   "Lukasz Lew",
   "Li Li",
   "Kellie Webster",
   "KP Sawhney",
   "Lampros Lamprou",
   "Siamak Shakeri",
   "Mayank Lunayach",
   "Jianmin Chen",
   "Sumit Bagri",
   "Alex Salcianu",
   "Ying Chen",
   "Yani Donchev",
   "Charlotte Magister",
   "Signe N\u00f8rly",
   "Vitor Rodrigues",
   "Tomas Izo",
   "Hila Noga",
   "Joe Zou",
   "Thomas K\u00f6ppe",
   "Wenxuan Zhou",
   "Kenton Lee",
   "Xiangzhu Long",
   "Danielle Eisenbud",
   "Anthony Chen",
   "Connor Schenck",
   "Chi Ming To",
   "Peilin Zhong",
   "Emanuel Taropa",
   "Minh Truong",
   "Omer Levy",
   "Danilo Martins",
   "Zhiyuan Zhang",
   "Christopher Semturs",
   "Kelvin Zhang",
   "Alex Yakubovich",
   "Pol Moreno",
   "Lara McConnaughey",
   "Di Lu",
   "Sam Redmond",
   "Lotte Weerts",
   "Yonatan Bitton",
   "Tiziana Refice",
   "Nicolas Lacasse",
   "Arthur Conmy",
   "Corentin Tallec",
   "Julian Odell",
   "Hannah Forbes-Pollard",
   "Arkadiusz Socala",
   "Jonathan Hoech",
   "Pushmeet Kohli",
   "Alanna Walton",
   "Rui Wang",
   "Mikita Sazanovich",
   "Kexin Zhu",
   "Andrei Kapishnikov",
   "Rich Galt",
   "Matthew Denton",
   "Ben Murdoch",
   "Caitlin Sikora",
   "Kareem Mohamed",
   "Wei Wei",
   "Uri First",
   "Tim McConnell",
   "Luis C. Cobo",
   "James Qin",
   "Thi Avrahami",
   "Daniel Balle",
   "Yu Watanabe",
   "Annie Louis",
   "Adam Kraft",
   "Setareh Ariafar",
   "Yiming Gu",
   "Eug\u00e9nie Rives",
   "Charles Yoon",
   "Andrei Rusu",
   "James Cobon-Kerr",
   "Chris Hahn",
   "Jiaming Luo",
   " Yuvein",
   " Zhu",
   "Niharika Ahuja",
   "Rodrigo Benenson",
   "Rapha\u00ebl Lopez Kaufman",
   "Honglin Yu",
   "Lloyd Hightower",
   "Junlin Zhang",
   "Darren Ni",
   "Lisa Anne Hendricks",
   "Gabby Wang",
   "Gal Yona",
   "Lalit Jain",
   "Pablo Barrio",
   "Surya Bhupatiraju",
   "Siva Velusamy",
   "Allan Dafoe",
   "Sebastian Riedel",
   "Tara Thomas",
   "Zhe Yuan",
   "Mathias Bellaiche",
   "Sheena Panthaplackel",
   "Klemen Kloboves",
   "Sarthak Jauhari",
   "Canfer Akbulut",
   "Todor Davchev",
   "Evgeny Gladchenko",
   "David Madras",
   "Aleksandr Chuklin",
   "Tyrone Hill",
   "Quan Yuan",
   "Mukundan Madhavan",
   "Luke Leonhard",
   "Dylan Scandinaro",
   "Qihang Chen",
   "Ning Niu",
   "Arthur Douillard",
   "Bogdan Damoc",
   "Yasumasa Onoe",
   "Fabian Pedregosa",
   "Fred Bertsch",
   "Chas Leichner",
   "Joseph Pagadora",
   "Jonathan Malmaud",
   "Sameera Ponda",
   "Andy Twigg",
   "Oleksii Duzhyi",
   "Jingwei Shen",
   "Miaosen Wang",
   "Roopal Garg",
   "Jing Chen",
   "Utku Evci",
   "Jonathan Lee",
   "Leon Liu",
   "Koji Kojima",
   "Masa Yamaguchi",
   "Arunkumar Rajendran",
   "AJ Piergiovanni",
   "Vinodh Kumar Rajendran",
   "Marco Fornoni",
   "Gabriel Ibagon",
   "Harry Ragan",
   "Sadh MNM Khan",
   "John Blitzer",
   "Andrew Bunner",
   "Guan Sun",
   "Takahiro Kosakai",
   "Scott Lundberg",
   "Ndidi Elue",
   "Kelvin Guu",
   "SK Park",
   "Jane Park",
   "Arunachalam Narayanaswamy",
   "Chengda Wu",
   "Jayaram Mudigonda",
   "Trevor Cohn",
   "Hairong Mu",
   "Ravi Kumar",
   "Laura Graesser",
   "Yichi Zhang",
   "Richard Killam",
   "Vincent Zhuang",
   "Mai Gim\u00e9nez",
   "Wael Al Jishi",
   "Ruy Ley-Wild",
   "Alex Zhai",
   "Kazuki Osawa",
   "Diego Cedillo",
   "Jialu Liu",
   "Mayank Upadhyay",
   "Marcin Sieniek",
   "Roshan Sharma",
   "Tom Paine",
   "Anelia Angelova",
   "Sravanti Addepalli",
   "Carolina Parada",
   "Kingshuk Majumder",
   "Avery Lamp",
   "Sanjiv Kumar",
   "Xiang Deng",
   "Artiom Myaskovsky",
   "Tea Saboli\u0107",
   "Jeffrey Dudek",
   "Sarah York",
   "F\u00e9lix de Chaumont Quitry",
   "Jiazhong Nie",
   "Dee Cattle",
   "Alok Gunjan",
   "Bilal Piot",
   "Waleed Khawaja",
   "Seojin Bang",
   "Simon Wang",
   "Siavash Khodadadeh",
   "Raghavender R",
   "Praynaa Rawlani",
   "Richard Powell",
   "Kevin Lee",
   "Johannes Griesser",
   "GS Oh",
   "Cesar Magalhaes",
   "Yujia Li",
   "Simon Tokumine",
   "Hadas Natalie Vogel",
   "Dennis Hsu",
   "Arturo BC",
   "Disha Jindal",
   "Matan Cohen",
   "Zi Yang",
   "Junwei Yuan",
   "Dario de Cesare",
   "Tony Bruguier",
   "Jun Xu",
   "Monica Roy",
   "Alon Jacovi",
   "Dan Belov",
   "Rahul Arya",
   "Phoenix Meadowlark",
   "Shlomi Cohen-Ganor",
   "Wenting Ye",
   "Patrick Morris-Suzuki",
   "Praseem Banzal",
   "Gan Song",
   "Pranavaraj Ponnuramu",
   "Fred Zhang",
   "George Scrivener",
   "Salah Zaiem",
   "Alif Raditya Rochman",
   "Kehang Han",
   "Badih Ghazi",
   "Kate Lee",
   "Shahar Drath",
   "Daniel Suo",
   "Antonious Girgis",
   "Pradeep Shenoy",
   "Duy Nguyen",
   "Douglas Eck",
   "Somit Gupta",
   "Le Yan",
   "Joao Carreira",
   "Anmol Gulati",
   "Ruoxin Sang",
   "Daniil Mirylenka",
   "Emma Cooney",
   "Edward Chou",
   "Mingyang Ling",
   "Cindy Fan",
   "Ben Coleman",
   "Guilherme Tubone",
   "Ravin Kumar",
   "Jason Baldridge",
   "Felix Hernandez-Campos",
   "Angeliki Lazaridou",
   "James Besley",
   "Itay Yona",
   "Neslihan Bulut",
   "Quentin Wellens",
   "AJ Pierigiovanni",
   "Jasmine George",
   "Richard Green",
   "Pu Han",
   "Connie Tao",
   "Geoff Clark",
   "Chong You",
   "Abbas Abdolmaleki",
   "Justin Fu",
   "Tongzhou Chen",
   "Ashwin Chaugule",
   "Angad Chandorkar",
   "Altaf Rahman",
   "Will Thompson",
   "Penporn Koanantakool",
   "Mike Bernico",
   "Jie Ren",
   "Andrey Vlasov",
   "Sergei Vassilvitskii",
   "Maciej Kula",
   "Yizhong Liang",
   "Dahun Kim",
   "Yangsibo Huang",
   "Chengxi Ye",
   "Dmitry Lepikhin",
   "Wesley Helmholz"
  ],
  "author_count": 3435,
  "categories": [
   "cs.CL",
   "cs.AI"
  ],
  "primary_category": "cs.CL",
  "venue": "",
  "venue_source": "",
  "citations": 3917,
  "influential_citations": 1029,
  "tldr": "The Gemini 2.X model generation spans the full Pareto frontier of model capability vs cost, allowing users to explore the boundaries of what is possible with complex agentic problem solving.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Gheorghe Comanici",
    "id": "1819858",
    "h_index": 12,
    "papers": 22
   },
   {
    "name": "E. Bieber",
    "id": "1910984",
    "h_index": 18,
    "papers": 81
   },
   {
    "name": "Mike Schaekermann",
    "id": "10685155",
    "h_index": 18,
    "papers": 48
   },
   {
    "name": "Ice Pasupat",
    "id": "2316577900",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Noveen Sachdeva",
    "id": "40705044",
    "h_index": 15,
    "papers": 32
   },
   {
    "name": "Inderjit S. Dhillon",
    "id": "2329738870",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Marcel Blistein",
    "id": "2358036487",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Ori Ram",
    "id": "73775461",
    "h_index": 15,
    "papers": 20
   },
   {
    "name": "Dan Zhang",
    "id": "2188417102",
    "h_index": 6,
    "papers": 25
   },
   {
    "name": "Evan Rosen",
    "id": "2275187560",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Luke Marris",
    "id": "2327046455",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Sam Petulla",
    "id": "2373037038",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Colin Gaffney",
    "id": "2160887964",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "A. Aharoni",
    "id": "2270131335",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Nathan Lintz",
    "id": "2104609067",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "T. C. Pais",
    "id": "34853087",
    "h_index": 8,
    "papers": 15
   },
   {
    "name": "Henrik Jacobsson",
    "id": "2275187124",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Idan Szpektor",
    "id": "1711977",
    "h_index": 37,
    "papers": 118
   },
   {
    "name": "Nan-Jiang Jiang",
    "id": "2316624506",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Krishna Haridasan",
    "id": "2256873459",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Ahmed Omran",
    "id": "2358036121",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Nikunj Saunshi",
    "id": "10769461",
    "h_index": 18,
    "papers": 29
   },
   {
    "name": "Dara Bahri",
    "id": "2119725651",
    "h_index": 15,
    "papers": 21
   },
   {
    "name": "Gaurav Mishra",
    "id": "2299042877",
    "h_index": 5,
    "papers": 17
   },
   {
    "name": "Eric Chu",
    "id": "2253520497",
    "h_index": 8,
    "papers": 15
   },
   {
    "name": "Toby Boyd",
    "id": "2283932206",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Brad Hekman",
    "id": "2373036337",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Aaron Parisi",
    "id": "2266464462",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Chaoyi Zhang",
    "id": "2256775724",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Kornraphop Kawintiranon",
    "id": "9257981",
    "h_index": 9,
    "papers": 14
   },
   {
    "name": "Tania Bedrax-Weiss",
    "id": "2113816700",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "O. Wang",
    "id": "2316578444",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Ya Xu",
    "id": "2177018597",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Ollie Purkiss",
    "id": "2373037663",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Uri Mendlovic",
    "id": "2329488",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Ilai Deutel",
    "id": "2373037841",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Nam Nguyen",
    "id": "2307557596",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "A. Langley",
    "id": "144805083",
    "h_index": 10,
    "papers": 24
   },
   {
    "name": "Flip Korn",
    "id": "2325726493",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Lucia Rossazza",
    "id": "28348625",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Alexandre Ram'e",
    "id": "2280134846",
    "h_index": 9,
    "papers": 12
   },
   {
    "name": "Sagar M. Waghmare",
    "id": "2243335607",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Helen Miller",
    "id": "2275121046",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Vaishakh Keshava",
    "id": "17320214",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Ying Jian",
    "id": "2372545670",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Xiaofan Zhang",
    "id": "2391001123",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "R. A. Popa",
    "id": "2350518852",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Kedar Dhamdhere",
    "id": "1696833",
    "h_index": 17,
    "papers": 31
   },
   {
    "name": "Blavz Bratanivc",
    "id": "1988215994",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Kyuyeun Kim",
    "id": "2313697996",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Terry Koo",
    "id": "2295987918",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Ferran Alet",
    "id": "2266458515",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Yi-ting Chen",
    "id": "2372910358",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Arsha Nagrani",
    "id": "2140304926",
    "h_index": 12,
    "papers": 17
   },
   {
    "name": "Hannah Muckenhirn",
    "id": "4563878",
    "h_index": 11,
    "papers": 14
   },
   {
    "name": "Zhiyuan Zhang",
    "id": "2374256817",
    "h_index": 5,
    "papers": 20
   },
   {
    "name": "Corbin Quick",
    "id": "51260442",
    "h_index": 9,
    "papers": 11
   },
   {
    "name": "Filip Paveti'c",
    "id": "2170163036",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "D. D. Nguyen",
    "id": "2112293680",
    "h_index": 6,
    "papers": 29
   },
   {
    "name": "Jo\u00e3o Carreira",
    "id": "2257349317",
    "h_index": 9,
    "papers": 20
   },
   {
    "name": "Michael Elabd",
    "id": "2161688185",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Ha-roon Qureshi",
    "id": "2275177545",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Fabian Mentzer",
    "id": "3468078",
    "h_index": 21,
    "papers": 27
   },
   {
    "name": "Yao Yang",
    "id": "2325208195",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Danielle Eisenbud",
    "id": "2351908333",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Anmol Gulati",
    "id": "4478284",
    "h_index": 14,
    "papers": 26
   },
   {
    "name": "Ellie Talius",
    "id": "2161825243",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Eric Ni",
    "id": "2291065245",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "Sahra Ghalebikesabi",
    "id": "2052021549",
    "h_index": 10,
    "papers": 21
   },
   {
    "name": "Edouard Yvinec",
    "id": "1632928879",
    "h_index": 11,
    "papers": 34
   },
   {
    "name": "Alaa Saade",
    "id": "16927419",
    "h_index": 13,
    "papers": 31
   },
   {
    "name": "Thatcher Ulrich",
    "id": "2372984025",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Lorenzo Blanco",
    "id": "2275186192",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Dan A. Calian",
    "id": "2792016",
    "h_index": 12,
    "papers": 18
   },
   {
    "name": "Muhua Huang",
    "id": "2336922270",
    "h_index": 5,
    "papers": 16
   },
   {
    "name": "A\u00e4ron van den Oord",
    "id": "3422336",
    "h_index": 42,
    "papers": 58
   },
   {
    "name": "Naman Goyal",
    "id": "39589154",
    "h_index": 2,
    "papers": 11
   },
   {
    "name": "Terry Chen",
    "id": "2296952760",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Praynaa Rawlani",
    "id": "1960281",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "C. Schallhart",
    "id": "1766300",
    "h_index": 22,
    "papers": 78
   },
   {
    "name": "S. Lokhande",
    "id": "2265975898",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Xianghong Luo",
    "id": "2115828242",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Jyn Shan",
    "id": "2372322231",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Ceslee Montgomery",
    "id": "2142412422",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Victoria Krakovna",
    "id": "2578985",
    "h_index": 16,
    "papers": 31
   },
   {
    "name": "Federico Piccinini",
    "id": "2373043187",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Omer Barak",
    "id": "22718593",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Jingyu Cui",
    "id": "2114354459",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Yiling Jia",
    "id": "2257230381",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Mikhail Dektiarev",
    "id": "2210805140",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "A. Kolganov",
    "id": "31216267",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Shiyu Huang",
    "id": "2305795673",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "Zhe Chen",
    "id": "2275189004",
    "h_index": 6,
    "papers": 20
   },
   {
    "name": "Xingyu Wang",
    "id": "2275584090",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Jessica Austin",
    "id": "2290488307",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Peter de Boursac",
    "id": "2373036647",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Evgeny Sluzhaev",
    "id": "2373037228",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "F. Ding",
    "id": "2064424741",
    "h_index": 8,
    "papers": 36
   },
   {
    "name": "Huijian Li",
    "id": "2108561551",
    "h_index": 9,
    "papers": 30
   },
   {
    "name": "Surya Bhupatiraju",
    "id": "9692128",
    "h_index": 11,
    "papers": 37
   },
   {
    "name": "M. Agarwal",
    "id": "2328331148",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Slawek Kwasiborski",
    "id": "2373037142",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Paramjit Sandhu",
    "id": "2307452920",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Patrick Siegler",
    "id": "39558817",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Ahmet Iscen",
    "id": "2910193",
    "h_index": 20,
    "papers": 47
   },
   {
    "name": "Eyal Ben-David",
    "id": "2304952646",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Shiraz Butt",
    "id": "7443818",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Miltos Allamanis",
    "id": "2283771805",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Seth Benjamin",
    "id": "2324801063",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "R. Busa-Fekete",
    "id": "1398396696",
    "h_index": 31,
    "papers": 101
   },
   {
    "name": "F\u00e9lix Hern\u00e1ndez-Campos",
    "id": "1403807088",
    "h_index": 19,
    "papers": 30
   },
   {
    "name": "S. Goldshtein",
    "id": "35540270",
    "h_index": 9,
    "papers": 25
   },
   {
    "name": "Matt Dibb",
    "id": "2373042188",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Weiyan Zhang",
    "id": "2401736227",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Annie Marsden",
    "id": "2329105772",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Carey Radebaugh",
    "id": "146419516",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Stephen Roller",
    "id": "3849208",
    "h_index": 21,
    "papers": 49
   },
   {
    "name": "Abhishek Nayyar",
    "id": "52187649",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Jacob Austin",
    "id": "2288056644",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Tayfun Terzi",
    "id": "114261380",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Bhargav Kanagal Shamanna",
    "id": "10327989",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Peter Shaw",
    "id": "2291050711",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Aayush Singh",
    "id": "2403534827",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Florian Luisier",
    "id": "2243272671",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Artur Mendoncca",
    "id": "2373034370",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "V. Aggarwal",
    "id": "2296788650",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "L. Markeeva",
    "id": "72361999",
    "h_index": 11,
    "papers": 26
   },
   {
    "name": "Claudio Fantacci",
    "id": "2312744677",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Sergey Brin",
    "id": "1786259",
    "h_index": 17,
    "papers": 28
   },
   {
    "name": "HyunJeong Choe",
    "id": "2316578116",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Guanyu Wang",
    "id": "2152582278",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Hartwig Adam",
    "id": "2595180",
    "h_index": 44,
    "papers": 70
   },
   {
    "name": "Avigail Dabush",
    "id": "2157242156",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "T. Kiyono",
    "id": "104759432",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "E. Marcus",
    "id": "152844537",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Jeremy R. Cole",
    "id": "2294173946",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "T. Weber",
    "id": "143947744",
    "h_index": 33,
    "papers": 67
   },
   {
    "name": "Hongrae Lee",
    "id": "2294439237",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Ronny Huang",
    "id": "2162739754",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Alex Muzio",
    "id": "2294360330",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Leandro Kieliger",
    "id": "71397721",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Maigo Le",
    "id": "2275163719",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Courtney Biles",
    "id": "118927199",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Long Le",
    "id": "2407283916",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Archit Sharma",
    "id": "2373105745",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Chengrun Yang",
    "id": "2349435196",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Avery Lamp",
    "id": "2373038080",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Dave Dopson",
    "id": "52355309",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "N. Hurley",
    "id": "151149353",
    "h_index": 9,
    "papers": 23
   },
   {
    "name": "Katrina Xu",
    "id": "2372464102",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Zhihao Shan",
    "id": "96887312",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Shuang Song",
    "id": "2372598739",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Jiewen Tan",
    "id": "2374093844",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Alexandre Senges",
    "id": "2253693024",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "G. Zhang",
    "id": "2308825234",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Chong You",
    "id": "2355321069",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Yennie Jun",
    "id": "2266398790",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "David Raposo",
    "id": "2285203932",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Susanna Ricco",
    "id": "2266519913",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Xuan Yang",
    "id": "2350843695",
    "h_index": 9,
    "papers": 24
   },
   {
    "name": "Weijie Chen",
    "id": "2292207373",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Prakhar Gupta",
    "id": "1491232062",
    "h_index": 14,
    "papers": 28
   },
   {
    "name": "Arthur Szlam",
    "id": "3149531",
    "h_index": 51,
    "papers": 145
   },
   {
    "name": "Kevin Villela",
    "id": "2275154010",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Chun-Sung Ferng",
    "id": "3340602",
    "h_index": 14,
    "papers": 27
   },
   {
    "name": "Daniel Kasenberg",
    "id": "2311700011",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Chen Liang",
    "id": "2316790498",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Rui Zhu",
    "id": "2070271342",
    "h_index": 7,
    "papers": 12
   },
   {
    "name": "Arunachalam Narayanaswamy",
    "id": "50484974",
    "h_index": 17,
    "papers": 29
   },
   {
    "name": "Florence Perot",
    "id": "2373034187",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Paul Pucciarelli",
    "id": "2373042315",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "A. Shekhawat",
    "id": "2401818282",
    "h_index": 8,
    "papers": 27
   },
   {
    "name": "A. Stern",
    "id": "2250191674",
    "h_index": 4,
    "papers": 17
   },
   {
    "name": "Rishikesh Ingale",
    "id": "2366937440",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Stefani Karp",
    "id": "121202667",
    "h_index": 7,
    "papers": 12
   },
   {
    "name": "Sanaz Bahargam",
    "id": "3297718",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Adrian Goedeckemeyer",
    "id": "2275188881",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Jie Han",
    "id": "2373240009",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Sicheng Li",
    "id": "2361908461",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "A. Tacchetti",
    "id": "2333917647",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Dian Yu",
    "id": "2256337021",
    "h_index": 7,
    "papers": 14
   },
   {
    "name": "A. Chakladar",
    "id": "2257341041",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Zhiying Zhang",
    "id": "2117992811",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Mona Mahdy",
    "id": "2137130739",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Xu Gao",
    "id": "2370147130",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Dale S. Johnson",
    "id": "2150445792",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Samrat Phatale",
    "id": "2067196169",
    "h_index": 7,
    "papers": 15
   },
   {
    "name": "A. Piergiovanni",
    "id": "8797855",
    "h_index": 27,
    "papers": 63
   },
   {
    "name": "Hyeontaek Lim",
    "id": "2275798209",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "C. Farabet",
    "id": "2256269",
    "h_index": 27,
    "papers": 50
   },
   {
    "name": "C. Lebsack",
    "id": "47248074",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Theo Guidroz",
    "id": "2300092706",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "John Blitzer",
    "id": "2116927",
    "h_index": 23,
    "papers": 48
   },
   {
    "name": "Nico Duduta",
    "id": "2373042209",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "David Madras",
    "id": "40373515",
    "h_index": 13,
    "papers": 33
   },
   {
    "name": "Steve Li",
    "id": "2316582169",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "D. V. Dincklage",
    "id": "1982980",
    "h_index": 9,
    "papers": 14
   },
   {
    "name": "X. Li",
    "id": "2316642629",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Mahdis Mahdieh",
    "id": "101247384",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "George Tucker",
    "id": "2275183383",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Ganesh Jawahar",
    "id": "2065043351",
    "h_index": 7,
    "papers": 15
   },
   {
    "name": "Yunxuan Owen Xiao",
    "id": "66162105",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Daniel Tarlow",
    "id": "2288056916",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Robert Geirhos",
    "id": "2132068799",
    "h_index": 7,
    "papers": 14
   },
   {
    "name": "Noam Velan",
    "id": "2373037200",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Daniel Vlasic",
    "id": "2258550828",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Kalesha Bullard",
    "id": "2302656",
    "h_index": 9,
    "papers": 27
   },
   {
    "name": "S. Park",
    "id": "150304350",
    "h_index": 6,
    "papers": 24
   },
   {
    "name": "Nishesh Gupta",
    "id": "2290474186",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Kellie Webster",
    "id": "2289184641",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Ayal Hitron",
    "id": "1800569",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Jieming Mao",
    "id": "2365393520",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "J. Eisenschlos",
    "id": "2221122676",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Laurel Prince",
    "id": "2258722620",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Nina D'souza",
    "id": "2373042884",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "K. Zheng",
    "id": "2269249729",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Sara Nasso",
    "id": "2373042206",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Gabriela Botea",
    "id": "2324052920",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Carl Doersch",
    "id": "2786693",
    "h_index": 33,
    "papers": 57
   },
   {
    "name": "Caglar Unlu",
    "id": "2373036606",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Chris Alberti",
    "id": "114577307",
    "h_index": 21,
    "papers": 34
   },
   {
    "name": "A. Svyatkovskiy",
    "id": "2256640504",
    "h_index": 74,
    "papers": 289
   },
   {
    "name": "Ankit Goel",
    "id": "49645115",
    "h_index": 6,
    "papers": 27
   },
   {
    "name": "Krzysztof Choromanski",
    "id": "2243007794",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Pan-Pan Jiang",
    "id": "2009106510",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "R. Nguyen",
    "id": "2298010005",
    "h_index": 6,
    "papers": 23
   },
   {
    "name": "Four Flynn",
    "id": "2350518566",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Daria \u0106urko",
    "id": "100646464",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Peter Chen",
    "id": "2359907160",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Nicholas Roth",
    "id": "2352934829",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Kieran Milan",
    "id": "8181864",
    "h_index": 6,
    "papers": 17
   },
   {
    "name": "Caleb Habtegebriel",
    "id": "2373036674",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Shashi Narayan",
    "id": "2276244046",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "M. Moffitt",
    "id": "2390395727",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Jake Marcus",
    "id": "2059685709",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "T. Anthony",
    "id": "2250318251",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Brendan McMahan",
    "id": "2286331246",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Gowoon Cheon",
    "id": "2268315391",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Ruibo Liu",
    "id": "2307497329",
    "h_index": 4,
    "papers": 20
   },
   {
    "name": "Megan Barnes",
    "id": "2275180117",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Lukasz Lew",
    "id": "2294360121",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Re-beca Santamaria-Fernandez",
    "id": "2275188901",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Mayank Upadhyay",
    "id": "2314621386",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Arjun Akula",
    "id": "2329736986",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "A. M. Hrafnkelsson",
    "id": "2259962018",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "A. Caceres",
    "id": "2372576380",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Andrew Bunner",
    "id": "2209988826",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Michal Sokolik",
    "id": "2357083684",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Subha Puttagunta",
    "id": "2373036671",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "L. Moore",
    "id": "144567536",
    "h_index": 6,
    "papers": 18
   },
   {
    "name": "Berivan Isik",
    "id": "1707440322",
    "h_index": 12,
    "papers": 32
   },
   {
    "name": "Jay Hartford",
    "id": "2098504",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Lawrence Chan",
    "id": "2307002182",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "P. Shenoy",
    "id": "2345185681",
    "h_index": 4,
    "papers": 16
   },
   {
    "name": "D. Holtmann-Rice",
    "id": "1404655176",
    "h_index": 13,
    "papers": 29
   },
   {
    "name": "Jane Park",
    "id": "2275788749",
    "h_index": 5,
    "papers": 17
   },
   {
    "name": "Fabio Viola",
    "id": "2290484786",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Alexandru Salcianu",
    "id": "3251354",
    "h_index": 12,
    "papers": 18
   },
   {
    "name": "Sujeevan Rajayogam",
    "id": "2324800990",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Ian Stewart-Binks",
    "id": "2373036534",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Zelin Wu",
    "id": "2303979221",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Richard Everett",
    "id": "2268661989",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Xi Xiong",
    "id": "2258552923",
    "h_index": 7,
    "papers": 21
   },
   {
    "name": "Pierre-Antoine Manzagol",
    "id": "1798462",
    "h_index": 13,
    "papers": 16
   },
   {
    "name": "Gary Leung",
    "id": "2059835325",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Carl Saroufim",
    "id": "2275186597",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Bo Pang",
    "id": "2324110631",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Dawid Wegner",
    "id": "2373038139",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "G. Papamakarios",
    "id": "3065681",
    "h_index": 20,
    "papers": 35
   },
   {
    "name": "J. Palomaki",
    "id": "52578817",
    "h_index": 8,
    "papers": 23
   },
   {
    "name": "Helena Pankov",
    "id": "2373034055",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Guangda Lai",
    "id": "2064491739",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "G. Tubone",
    "id": "2309580105",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Shubin Zhao",
    "id": "2370038802",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "T. Strinopoulos",
    "id": "102732276",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Seth Neel",
    "id": "2273685865",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Mingqiu Wang",
    "id": "2249764807",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Joe Kelley",
    "id": "2307455862",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Li Li",
    "id": "2374428675",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Ping-mei Xu",
    "id": "2352600773",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Anitha Vijayakumar",
    "id": "2275186554",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Andrea D'olimpio",
    "id": "2373037329",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Omer Levy",
    "id": "2326991363",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Massimo Nicosia",
    "id": "2327549684",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Grigory Rozhdestvenskiy",
    "id": "2291067913",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Ni Lao",
    "id": "2373037523",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Sirui Xie",
    "id": "2283263582",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Yash Katariya",
    "id": "71749200",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Jon Simon",
    "id": "2324821384",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Sanjiv Kumar",
    "id": "2283420123",
    "h_index": 9,
    "papers": 20
   },
   {
    "name": "Florian Hartmann",
    "id": "2275151970",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "M. Kilgore",
    "id": "2373042855",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Jinhyuk Lee",
    "id": "2275576296",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Aroma Mahendru",
    "id": "3158336",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Roman Ring",
    "id": "81387328",
    "h_index": 9,
    "papers": 20
   },
   {
    "name": "T. Hennigan",
    "id": "2146532222",
    "h_index": 9,
    "papers": 33
   },
   {
    "name": "Fiona Lang",
    "id": "39194601",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Colin Cherry",
    "id": "2257002653",
    "h_index": 9,
    "papers": 21
   },
   {
    "name": "David Steiner",
    "id": "2275188258",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Dawsen Hwang",
    "id": "2372331762",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Ray Smith",
    "id": "2032936454",
    "h_index": 4,
    "papers": 17
   },
   {
    "name": "Pidong Wang",
    "id": "2164862499",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Jeremy Chen",
    "id": "2275275439",
    "h_index": 5,
    "papers": 15
   },
   {
    "name": "Ming Yang",
    "id": "2228360903",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "S. Kwei",
    "id": "2124722379",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Philippe Schlattner",
    "id": "1471879817",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Donnie Kim",
    "id": "2373583624",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Ganesh Poomal Girirajan",
    "id": "2373034476",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Nikola Momchev",
    "id": "1470531643",
    "h_index": 12,
    "papers": 26
   },
   {
    "name": "Ayushi Agarwal",
    "id": "2240030337",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Xingyi Zhou",
    "id": "2289777650",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Ilkin Safarli",
    "id": "2529219",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Zachary Garrett",
    "id": "40449749",
    "h_index": 12,
    "papers": 21
   },
   {
    "name": "AJ Pierigiovanni",
    "id": "2373036516",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Sarthak Jauhari",
    "id": "72483847",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Alif Raditya Rochman",
    "id": "2373038199",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Shikhar Vashishth",
    "id": "3404827",
    "h_index": 18,
    "papers": 37
   },
   {
    "name": "Quan Yuan",
    "id": "2275194184",
    "h_index": 6,
    "papers": 23
   },
   {
    "name": "Christof Angermueller",
    "id": "2269460640",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Jon Blanton",
    "id": "2373036784",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Xiny-ing Song",
    "id": "2316781375",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "N. B. Gundavarapu",
    "id": "1387987945",
    "h_index": 8,
    "papers": 13
   },
   {
    "name": "Thi Avrahami",
    "id": "2261737895",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Maxine Deines",
    "id": "2373037549",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Subhrajit Roy",
    "id": "2331267066",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Manish Gupta",
    "id": "2374370621",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Christopher Semturs",
    "id": "52326632",
    "h_index": 17,
    "papers": 25
   },
   {
    "name": "Shobha Vasudevan",
    "id": "2351724140",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Aditya Srikanth Veerubhotla",
    "id": "2123019049",
    "h_index": 8,
    "papers": 11
   },
   {
    "name": "Shriya Sharma",
    "id": "2257432740",
    "h_index": 4,
    "papers": 32
   },
   {
    "name": "Joshy Jacob",
    "id": "37032402",
    "h_index": 12,
    "papers": 44
   },
   {
    "name": "Zheng Yang",
    "id": "2454981467",
    "h_index": 6,
    "papers": 19
   },
   {
    "name": "Andreas Terzis",
    "id": "2282843407",
    "h_index": 9,
    "papers": 17
   },
   {
    "name": "Dan Karliner",
    "id": "2284349593",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Auriel Wright",
    "id": "2217344072",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Tania Rojas-Esponda",
    "id": "1405815602",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Ashley Brown",
    "id": "2372865260",
    "h_index": 4,
    "papers": 17
   },
   {
    "name": "A. Roy",
    "id": "23360006",
    "h_index": 27,
    "papers": 50
   },
   {
    "name": "Pawan Dogra",
    "id": "2373037807",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "A. Kapishnikov",
    "id": "2362444163",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Peter Young",
    "id": "2372449022",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "W. Kan",
    "id": "2372998708",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Vinodh K. Rajendran",
    "id": "2327433",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "M. Ivanova",
    "id": "2298281047",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "S. Deshmukh",
    "id": "2397225559",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Chia-Hua Ho",
    "id": "2286697952",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Michael Kwong",
    "id": "2275182209",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "S. Ginzburg",
    "id": "2372745899",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Annie Louis",
    "id": "2275175393",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "KP Sawhney",
    "id": "2373037630",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Slav Petrov",
    "id": "2257293575",
    "h_index": 6,
    "papers": 17
   },
   {
    "name": "Jing Xie",
    "id": "2375047506",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Yunfei Bai",
    "id": "2287670042",
    "h_index": 11,
    "papers": 51
   },
   {
    "name": "G. Stoyanov",
    "id": "145936086",
    "h_index": 8,
    "papers": 19
   },
   {
    "name": "Alex Fabrikant",
    "id": "1983135",
    "h_index": 20,
    "papers": 46
   },
   {
    "name": "Rajesh Jayaram",
    "id": "2295519962",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Yuqi Li",
    "id": "2342368239",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Joseph Heyward",
    "id": "69425681",
    "h_index": 10,
    "papers": 18
   },
   {
    "name": "Justin Gilmer",
    "id": "2243002880",
    "h_index": 7,
    "papers": 16
   },
   {
    "name": "Yaqing Wang",
    "id": "2363668756",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Radu Soricut",
    "id": "1737285",
    "h_index": 40,
    "papers": 104
   },
   {
    "name": "Lu Liu",
    "id": "2275576625",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Qing-ping Duan",
    "id": "2026510850",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Jamie Hayes",
    "id": "2283307020",
    "h_index": 13,
    "papers": 28
   },
   {
    "name": "Maura O'Brien",
    "id": "1576527177",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Gau-rav Singh Tomar",
    "id": "2348274256",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Sivan Eiger",
    "id": "2348098272",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Bahare Fatemi",
    "id": "3422551",
    "h_index": 16,
    "papers": 36
   },
   {
    "name": "Jeffrey Hui",
    "id": "2134586809",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Catarina Barros",
    "id": "2292141175",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "A. Chukwuka",
    "id": "2047687817",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Alena Butryna",
    "id": "1724714282",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Saksham Thakur",
    "id": "2358038670",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Austin Huang",
    "id": "2349227176",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Zhufeng Pan",
    "id": "2291169360",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Haotian Tang",
    "id": "2322083245",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Serkan Cabi",
    "id": "12159303",
    "h_index": 19,
    "papers": 29
   },
   {
    "name": "Tulsee Doshi",
    "id": "2314111664",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Michiel A. Bakker",
    "id": "2238422815",
    "h_index": 10,
    "papers": 15
   },
   {
    "name": "S. Bagri",
    "id": "2275191535",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Ruy Ley-Wild",
    "id": "1403893997",
    "h_index": 10,
    "papers": 23
   },
   {
    "name": "\u00c1. Lelkes",
    "id": "143828990",
    "h_index": 9,
    "papers": 22
   },
   {
    "name": "J. Lees",
    "id": "2372995356",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "P. Kane",
    "id": "2275181138",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "David Greene",
    "id": "2211734549",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Shimu Wu",
    "id": "2275313266",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "J. Bornschein",
    "id": "2298192258",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "G. Surita",
    "id": "1956049835",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Sarah Hodkinson",
    "id": "2265053608",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Fangtao Li",
    "id": "2287741374",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Chris Hidey",
    "id": "2324798859",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "S\u00e9bastien Pereira",
    "id": "2325139101",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Sean Ammirati",
    "id": "2063300576",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Phillip Lippe",
    "id": "2350751902",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Adam Kraft",
    "id": "2314693663",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Pu Han",
    "id": "2372331845",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Sebastian Gerlach",
    "id": "12588585",
    "h_index": 7,
    "papers": 16
   },
   {
    "name": "Zifeng Wang",
    "id": "2135785111",
    "h_index": 26,
    "papers": 44
   },
   {
    "name": "Liviu Panait",
    "id": "1703826",
    "h_index": 29,
    "papers": 50
   },
   {
    "name": "Feng Han",
    "id": "2349560824",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "B. Farris",
    "id": "120845002",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Y. Bi",
    "id": "2324793814",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Hannah DeBalsi",
    "id": "2373034973",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Miaosen Wang",
    "id": "5116578",
    "h_index": 10,
    "papers": 17
   },
   {
    "name": "Gladys Tyen",
    "id": "2266751853",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "James Cohan",
    "id": "2311700270",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Susan Zhang",
    "id": "2238121623",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Jarred Barber",
    "id": "152630175",
    "h_index": 9,
    "papers": 18
   },
   {
    "name": "D. Chung",
    "id": "2275180366",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Jaeyoung Kim",
    "id": "2292140501",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "M. Kunesch",
    "id": "12224272",
    "h_index": 15,
    "papers": 24
   },
   {
    "name": "S. Pecht",
    "id": "2097371855",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Nami Akazawa",
    "id": "2253598113",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Abe Friesen",
    "id": "2312322254",
    "h_index": 6,
    "papers": 19
   },
   {
    "name": "James Lyon",
    "id": "2086684749",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Ali Eslami",
    "id": "2344753162",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Junru Wu",
    "id": "2261361394",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Jiewen Tan",
    "id": "2374093844",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Yue Song",
    "id": "2306137723",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Ravin Kumar",
    "id": "2290629265",
    "h_index": 9,
    "papers": 27
   },
   {
    "name": "Christoper A. Welty",
    "id": "2062396538",
    "h_index": 7,
    "papers": 27
   },
   {
    "name": "Ilia Akolzin",
    "id": "2373042232",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Gena Gibson",
    "id": "2362503754",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Sean Augenstein",
    "id": "2304454639",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Arjun Pillai",
    "id": "2064837578",
    "h_index": 8,
    "papers": 18
   },
   {
    "name": "N. Yuen",
    "id": "46863475",
    "h_index": 7,
    "papers": 18
   },
   {
    "name": "Du Phan",
    "id": "2273784044",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Xin Wang",
    "id": "2374174561",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Iain Barr",
    "id": "2159207795",
    "h_index": 8,
    "papers": 21
   },
   {
    "name": "H. Zen",
    "id": "1691713",
    "h_index": 52,
    "papers": 153
   },
   {
    "name": "Nan Hua",
    "id": "2316578703",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Casper Liu",
    "id": "2373590704",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Jilei Wang",
    "id": "2273067401",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "T. Bhatia",
    "id": "2280878104",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Hao Xu",
    "id": "2282141807",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Oded Elyada",
    "id": "3365544",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Pushmeet Kohli",
    "id": "143967473",
    "h_index": 119,
    "papers": 372
   },
   {
    "name": "Mirek Olvs'ak",
    "id": "2211433817",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Kelly Chen",
    "id": "2374234216",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Azalia Mirhoseini",
    "id": "1861312",
    "h_index": 36,
    "papers": 127
   },
   {
    "name": "Noam Shazeer",
    "id": "1846258",
    "h_index": 40,
    "papers": 146
   },
   {
    "name": "Shoshana Jakobovits",
    "id": "72153548",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Maggie Tran",
    "id": "2372466401",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Nolan Ramsden",
    "id": "2373036506",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "T. Bharti",
    "id": "89672436",
    "h_index": 4,
    "papers": 15
   },
   {
    "name": "Fred Alcober",
    "id": "2275177971",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Yunjie Li",
    "id": "2276036296",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "S. Shetty",
    "id": "145594228",
    "h_index": 13,
    "papers": 54
   },
   {
    "name": "Jing Chen",
    "id": "2373516242",
    "h_index": 6,
    "papers": 22
   },
   {
    "name": "Dmitry Kalashnikov",
    "id": "48313860",
    "h_index": 21,
    "papers": 34
   },
   {
    "name": "Megha Nawhal",
    "id": "8080821",
    "h_index": 9,
    "papers": 14
   },
   {
    "name": "Sercan \u00d6. Arik",
    "id": "2676352",
    "h_index": 47,
    "papers": 137
   },
   {
    "name": "Hanwen Chen",
    "id": "2363319132",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "M. Blokzijl",
    "id": "14564467",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Shubham Gupta",
    "id": "2300566992",
    "h_index": 4,
    "papers": 16
   },
   {
    "name": "J. Rubin",
    "id": "2265401139",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Rigel Swavely",
    "id": "2037297118",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Sophie Bridgers",
    "id": "2273670422",
    "h_index": 8,
    "papers": 23
   },
   {
    "name": "I. Gemp",
    "id": "8616871",
    "h_index": 14,
    "papers": 47
   },
   {
    "name": "Chenlin Su",
    "id": "2378480006",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "A. Suggala",
    "id": "2304967488",
    "h_index": 5,
    "papers": 16
   },
   {
    "name": "Juliette Pluto",
    "id": "2362496726",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Mary Cassin",
    "id": "147433059",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Alain C. Vaucher",
    "id": "3363862",
    "h_index": 19,
    "papers": 53
   },
   {
    "name": "Kaiyang Ji",
    "id": "2374473800",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Jiahao Cai",
    "id": "2374430619",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Andrew Audibert",
    "id": "31728717",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Animesh Sinha",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "David Tian",
    "id": "2352238566",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "E. Farkash",
    "id": "2265182544",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Amy Hua",
    "id": "2076366241",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Jilin Chen",
    "id": "2249566095",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Duc Tran",
    "id": "2354146605",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "E. Loper",
    "id": "3213150",
    "h_index": 8,
    "papers": 34
   },
   {
    "name": "Nicole Brichtova",
    "id": "2373037875",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Lara McConnaughey",
    "id": "2065132876",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Ballie Sandhu",
    "id": "2373038091",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Robert Leland",
    "id": "2320955679",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Douglas DeCarlo",
    "id": "2245226118",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "A. Over",
    "id": "2373034976",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "J. Huang",
    "id": "2372612723",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Xing Wu",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "C. Fan",
    "id": "2351579038",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Eric Li",
    "id": "2326301768",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Yun-Peng Lei",
    "id": "2262425515",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Deepak Sharma",
    "id": "2374196709",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Cosmin Paduraru",
    "id": "3316271",
    "h_index": 21,
    "papers": 43
   },
   {
    "name": "Luo Yu",
    "id": "2372353361",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Matko Bovsnjak",
    "id": "1471340201",
    "h_index": 7,
    "papers": 7
   },
   {
    "name": "Phuong Dao",
    "id": "2275181671",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Min Choi",
    "id": "2349735740",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Sneha Kudugunta",
    "id": "2131614068",
    "h_index": 11,
    "papers": 16
   },
   {
    "name": "Jakub Adamek",
    "id": "50290651",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "C. Guia",
    "id": "49692846",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Ali Khodaei",
    "id": "2316558711",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Jie Feng",
    "id": "2350685577",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Wenjun Zeng",
    "id": "2313916188",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "David Welling",
    "id": "2373036660",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Sandeep Tata",
    "id": "2273559338",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Christina Butterfield",
    "id": "2275166845",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "A. Vlasov",
    "id": "2255064302",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Seliem El-Sayed",
    "id": "2292259110",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Swaroop Mishra",
    "id": "2287936369",
    "h_index": 11,
    "papers": 26
   },
   {
    "name": "Tara N. Sainath",
    "id": "2279918122",
    "h_index": 7,
    "papers": 17
   },
   {
    "name": "Shentao Yang",
    "id": "2316258854",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "R. Skerry-Ryan",
    "id": "1380248814",
    "h_index": 18,
    "papers": 25
   },
   {
    "name": "Jeremy Shar",
    "id": "2363345850",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Robert Berry",
    "id": "2302801613",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "A. Rajendran",
    "id": "2242617242",
    "h_index": 5,
    "papers": 34
   },
   {
    "name": "A. Kandoor",
    "id": "2349411",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Andrea Burns",
    "id": "2300099978",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Deepali Jain",
    "id": "2058985645",
    "h_index": 17,
    "papers": 36
   },
   {
    "name": "Tom Stone",
    "id": "2241763292",
    "h_index": 1,
    "papers": 24
   },
   {
    "name": "Wonpyo Park",
    "id": "2307279019",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Shibo Wang",
    "id": "2108553866",
    "h_index": 10,
    "papers": 27
   },
   {
    "name": "Albin Cassirer",
    "id": "51042571",
    "h_index": 9,
    "papers": 18
   },
   {
    "name": "Guohui Wang",
    "id": "2110630493",
    "h_index": 7,
    "papers": 14
   },
   {
    "name": "H. Kobayashi",
    "id": "2109756179",
    "h_index": 4,
    "papers": 29
   },
   {
    "name": "Sergey Rogulenko",
    "id": "2373042758",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Vineetha Govindaraj",
    "id": "2113709953",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Mikolaj Rybi'nski",
    "id": "2275177997",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Nadav Olmert",
    "id": "2373036464",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Colin Evans",
    "id": "2275183014",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Po-Sen Huang",
    "id": "2268826600",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Kelvin Xu",
    "id": "2266735761",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Premal Shah",
    "id": "2275185092",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Terry Thurk",
    "id": "2373038097",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Caitlin Sikora",
    "id": "2303406428",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Mu Cai",
    "id": "2387640027",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Jinyu Xie",
    "id": "2375047508",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Elahe Dabir",
    "id": "2290484418",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Saloni Shah",
    "id": "2340552517",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Norbert Kalb",
    "id": "2093692922",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Carrie Zhang",
    "id": "2373943628",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Shruthi Prabhakara",
    "id": "2192960",
    "h_index": 13,
    "papers": 30
   },
   {
    "name": "Amit Sabne",
    "id": "2339784756",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Artiom Myaskovsky",
    "id": "2346506169",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Vikas Raunak",
    "id": "24025563",
    "h_index": 17,
    "papers": 26
   },
   {
    "name": "Blanca Huergo",
    "id": "2373037821",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Behnam Neyshabur",
    "id": "3007442",
    "h_index": 46,
    "papers": 79
   },
   {
    "name": "Jon Clark",
    "id": "2336243148",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Ye Zhang",
    "id": "2275538925",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Shankar Krishnan",
    "id": "2229092194",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Eden Cohen",
    "id": "2372767666",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Dinesh Tewari",
    "id": "2330077472",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "James Lottes",
    "id": "2266398107",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Yumeya Yamamori",
    "id": "2316842382",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Hui Li",
    "id": "2275921865",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Mohamed Elhawaty",
    "id": "2275176049",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Ada Maksutaj Oflazer",
    "id": "2373036457",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Adri\u00e0 Recasens",
    "id": "39257069",
    "h_index": 20,
    "papers": 33
   },
   {
    "name": "S. Luo",
    "id": "2390889121",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Duy Nguyen",
    "id": "2314778432",
    "h_index": 6,
    "papers": 18
   },
   {
    "name": "Taylor Bos",
    "id": "2150572221",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "K. Andra",
    "id": "2320494114",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Ana S. Salazar",
    "id": "2143109714",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "E. Chi",
    "id": "2253469026",
    "h_index": 9,
    "papers": 30
   },
   {
    "name": "J. Ko",
    "id": "37614272",
    "h_index": 13,
    "papers": 24
   },
   {
    "name": "Matthew L. Ginsberg",
    "id": "2285567191",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Anders Andreassen",
    "id": "2314121408",
    "h_index": 6,
    "papers": 17
   },
   {
    "name": "Anian Ruoss",
    "id": "12114187",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Todor Davchev",
    "id": "121676884",
    "h_index": 15,
    "papers": 30
   },
   {
    "name": "Elnaz Davoodi",
    "id": "3232032",
    "h_index": 17,
    "papers": 60
   },
   {
    "name": "Chenxi Liu",
    "id": "2145153958",
    "h_index": 12,
    "papers": 33
   },
   {
    "name": "M. Kim",
    "id": "2145928629",
    "h_index": 6,
    "papers": 40
   },
   {
    "name": "Santiago Ontanon",
    "id": "2256997247",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "C. To",
    "id": "2373030472",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Dawei Jia",
    "id": "2275186531",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Rosemary Ke",
    "id": "2292140085",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Jing Wang",
    "id": "2372633183",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "A. Korsun",
    "id": "2069138315",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Moran Ambar",
    "id": "2311700543",
    "h_index": 6,
    "papers": 6
   },
   {
    "name": "Ilya Kornakov",
    "id": "2373038058",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Irene Giannoumis",
    "id": "1986487993",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Toni Creswell",
    "id": "2373038111",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Denny Zhou",
    "id": "2263221363",
    "h_index": 7,
    "papers": 23
   },
   {
    "name": "Yi Su",
    "id": "2321906289",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Ishaan Watts",
    "id": "2297767230",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Aleksandr Zaks",
    "id": "153263165",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Evgenii Eltyshev",
    "id": "2275187189",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Ziqiang Feng",
    "id": "2276189588",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Sidharth Mudgal",
    "id": "40171292",
    "h_index": 8,
    "papers": 15
   },
   {
    "name": "Alex Kaskasoli",
    "id": "2275186627",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "J Christopher Love",
    "id": "2253158807",
    "h_index": 7,
    "papers": 15
   },
   {
    "name": "Kingshuk Dasgupta",
    "id": "2290487762",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Sam Shleifer",
    "id": "88728159",
    "h_index": 16,
    "papers": 50
   },
   {
    "name": "R. Green",
    "id": "2348648285",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Sungyong Seo",
    "id": "2306121535",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Chansoo Lee",
    "id": "2109425498",
    "h_index": 10,
    "papers": 16
   },
   {
    "name": "Dale R. Webster",
    "id": "2324791213",
    "h_index": 7,
    "papers": 19
   },
   {
    "name": "Prakash Shroff",
    "id": "2275187236",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Ganna Raboshchuk",
    "id": "2069131",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Isabel Leal",
    "id": "2057988112",
    "h_index": 12,
    "papers": 15
   },
   {
    "name": "J. Manyika",
    "id": "2254252286",
    "h_index": 15,
    "papers": 37
   },
   {
    "name": "Sofia Erell",
    "id": "2101317386",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "D. Murphy",
    "id": "144062443",
    "h_index": 5,
    "papers": 21
   },
   {
    "name": "Zhisheng Xiao",
    "id": "117362006",
    "h_index": 15,
    "papers": 20
   },
   {
    "name": "Anton Bulyenov",
    "id": "2373037920",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Julian Walker",
    "id": "2110651361",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Mark Collier",
    "id": "153247100",
    "h_index": 13,
    "papers": 30
   },
   {
    "name": "Matej Kastelic",
    "id": "2282960175",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "N. George",
    "id": "67087104",
    "h_index": 10,
    "papers": 25
   },
   {
    "name": "Sushant Prakash",
    "id": "2268674039",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Sailesh Sidhwani",
    "id": "114254769",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Alexey I. Frolov",
    "id": "2060969884",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Steven Hansen",
    "id": "2275188563",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Petko Georgiev",
    "id": "1737522",
    "h_index": 22,
    "papers": 46
   },
   {
    "name": "Tiberiu Sosea",
    "id": "2008183567",
    "h_index": 10,
    "papers": 20
   },
   {
    "name": "Chris Apps",
    "id": "2293394525",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Aishwarya B Kamath",
    "id": "2269391198",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "David Reid",
    "id": "2275179889",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Emma Cooney",
    "id": "87458877",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Charlotte Magister",
    "id": "2373034125",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "O. Riva",
    "id": "2275053596",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Alec Go",
    "id": "2304454410",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Pu-Chin Chen",
    "id": "2115950770",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Se-bastian Krause",
    "id": "2275186493",
    "h_index": 4,
    "papers": 14
   },
   {
    "name": "Nir Levine",
    "id": "153898744",
    "h_index": 15,
    "papers": 21
   },
   {
    "name": "M. Fornoni",
    "id": "2274633",
    "h_index": 9,
    "papers": 14
   },
   {
    "name": "Ilya Figotin",
    "id": "2790301",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Nick Roy",
    "id": "2352018552",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Parsa Mahmoudieh",
    "id": "2972575",
    "h_index": 8,
    "papers": 13
   },
   {
    "name": "Vladimir Magay",
    "id": "2373038135",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Mukundan Madhavan",
    "id": "38470310",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Jin Miao",
    "id": "2275181068",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Jianmo Ni",
    "id": "2348507846",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Yasuhisa Fujii",
    "id": "2114175058",
    "h_index": 11,
    "papers": 20
   },
   {
    "name": "I-Fan Chou",
    "id": "2192207285",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "G. Scrivener",
    "id": "2373036432",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Zak Tsai",
    "id": "2372711703",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "S. Mcloughlin",
    "id": "50440804",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Jeremy Selier",
    "id": "2373036477",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Sandra Lefdal",
    "id": "2311700306",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Jeffrey Zhao",
    "id": "2316582764",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Abhijit Karmarkar",
    "id": "2078909017",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Kushal Chauhan",
    "id": "70083483",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Shivank Goel",
    "id": "2067569194",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Zhaoyi Zhang",
    "id": "2261733216",
    "h_index": 4,
    "papers": 16
   },
   {
    "name": "Vihan Jain",
    "id": "2314113713",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Parisa Haghani",
    "id": "2059866453",
    "h_index": 17,
    "papers": 27
   },
   {
    "name": "Mostafa Dehghani",
    "id": "2256989598",
    "h_index": 5,
    "papers": 22
   },
   {
    "name": "Jacob Scott",
    "id": "1665051132",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Erin Farnese",
    "id": "2373036439",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Anastasija Ili'c",
    "id": "2279830514",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Steven Baker",
    "id": "2372297822",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Julia Pawar",
    "id": "2156928197",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Li Zhong",
    "id": "2109823962",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Josh Camp",
    "id": "2345218391",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Yoel Zeldes",
    "id": "1912547",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "S. Shetty",
    "id": "2894170",
    "h_index": 32,
    "papers": 68
   },
   {
    "name": "Anand Iyer",
    "id": "2349535533",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "V'it List'ik",
    "id": "2373034650",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Jiaxian Guo",
    "id": "2366104775",
    "h_index": 7,
    "papers": 29
   },
   {
    "name": "Luming Tang",
    "id": "34689393",
    "h_index": 11,
    "papers": 14
   },
   {
    "name": "Mark Geller",
    "id": "2275182380",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Simon Bucher",
    "id": "2168128459",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Yifan Ding",
    "id": "2290634593",
    "h_index": 4,
    "papers": 14
   },
   {
    "name": "Hongzhi Shi",
    "id": "2325144265",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Carrie Muir",
    "id": "2275184985",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Dominik Grewe",
    "id": "2401609",
    "h_index": 14,
    "papers": 24
   },
   {
    "name": "R. Eskander",
    "id": "37558091",
    "h_index": 19,
    "papers": 48
   },
   {
    "name": "Octavio Ponce",
    "id": "2373038177",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Boqing Gong",
    "id": "2257000670",
    "h_index": 9,
    "papers": 16
   },
   {
    "name": "Derek Gasaway",
    "id": "2373036582",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Samir Khan",
    "id": "2341800704",
    "h_index": 1,
    "papers": 11
   },
   {
    "name": "Umang Gupta",
    "id": "2125189545",
    "h_index": 6,
    "papers": 17
   },
   {
    "name": "Angelos Filos",
    "id": "146066583",
    "h_index": 14,
    "papers": 30
   },
   {
    "name": "Weicheng Kuo",
    "id": "7987770",
    "h_index": 21,
    "papers": 30
   },
   {
    "name": "K. Kloboves",
    "id": "69476549",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Jennifer Beattie",
    "id": "2275186613",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Christiana Wright",
    "id": "2397431822",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Le Li",
    "id": "2348264004",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "A. Jin",
    "id": "2150572756",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Sandeep Mariserla",
    "id": "116991105",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Miteyan Patel",
    "id": "2275771582",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Jens Heitkaemper",
    "id": "2292597574",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Dilip Krishnan",
    "id": "2254258596",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "V. Sharma",
    "id": "2290595918",
    "h_index": 4,
    "papers": 16
   },
   {
    "name": "David Bieber",
    "id": "2355624801",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Christian Frank",
    "id": "2290488254",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "John Lambert",
    "id": "2372347332",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Paul Caron",
    "id": "2351909416",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "M. Polacek",
    "id": "35930544",
    "h_index": 11,
    "papers": 33
   },
   {
    "name": "Mai Gim'enez",
    "id": "2275181171",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Himadri Choudhury",
    "id": "2324798781",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Xing Yu",
    "id": "2373736130",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Sasan Tavakkol",
    "id": "2353997275",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Arun Ahuja",
    "id": "2275185727",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "F. Och",
    "id": "2002316",
    "h_index": 47,
    "papers": 79
   },
   {
    "name": "Rodolphe Jenatton",
    "id": "2068720",
    "h_index": 34,
    "papers": 59
   },
   {
    "name": "Wojtek Skut",
    "id": "2373034634",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Bryan Richter",
    "id": "2217682955",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "D. Gaddy",
    "id": "2259285276",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Andy Ly",
    "id": "2064878079",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Misha Bilenko",
    "id": "2297766346",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Megh Umekar",
    "id": "2372489847",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Ethan Liang",
    "id": "2304467077",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Martin Sevenich",
    "id": "2373036709",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Mandar Joshi",
    "id": "2365705213",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "H. Mansoor",
    "id": "2265752172",
    "h_index": 9,
    "papers": 24
   },
   {
    "name": "R. Lin",
    "id": "48756096",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Sumit K. Sanghai",
    "id": "144074891",
    "h_index": 17,
    "papers": 53
   },
   {
    "name": "Abhimanyu Singh",
    "id": "2385601890",
    "h_index": 3,
    "papers": 13
   },
   {
    "name": "Xiaowei Li",
    "id": "2275306281",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Sudheendra Vijayanarasimhan",
    "id": "2259154",
    "h_index": 24,
    "papers": 36
   },
   {
    "name": "Zaheer Abbas",
    "id": "47738035",
    "h_index": 11,
    "papers": 61
   },
   {
    "name": "Yonatan Bitton",
    "id": "1938499056",
    "h_index": 21,
    "papers": 47
   },
   {
    "name": "Hansa Srinivasan",
    "id": "2261494252",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Manish Reddy Vuyyuru",
    "id": "2127729353",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Alexander Frommgen",
    "id": "2373034614",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Yanhua Sun",
    "id": "2265240845",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Ralph Leith",
    "id": "2373038647",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Alfonso Casta\u00f1o",
    "id": "2275182237",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "D. Strouse",
    "id": "69925460",
    "h_index": 14,
    "papers": 21
   },
   {
    "name": "Le Yan",
    "id": "2275103811",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Austin Kyker",
    "id": "2373036728",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "S. Kambala",
    "id": "2066427",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Mary Jasarevic",
    "id": "2373038513",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Thibault Sellam",
    "id": "145450400",
    "h_index": 18,
    "papers": 69
   },
   {
    "name": "Chao Jia",
    "id": "2275129088",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "A. Pritzel",
    "id": "1863250",
    "h_index": 29,
    "papers": 39
   },
   {
    "name": "R. Raghavender",
    "id": "2373038142",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Huizhong Chen",
    "id": "2266066105",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Natalie Clay",
    "id": "2201776471",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Sudeep Gandhe",
    "id": "2285302874",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Sean Kirmani",
    "id": "51881277",
    "h_index": 21,
    "papers": 29
   },
   {
    "name": "Sayna Ebrahimi",
    "id": "27556211",
    "h_index": 21,
    "papers": 43
   },
   {
    "name": "Hannah Kirkwood",
    "id": "47801016",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Jonathan Mallinson",
    "id": "2280144293",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Chao Wang",
    "id": "2400098382",
    "h_index": 7,
    "papers": 16
   },
   {
    "name": "Adnan Ozturel",
    "id": "2275159497",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Kuo-Chin Lin",
    "id": "39810995",
    "h_index": 5,
    "papers": 20
   },
   {
    "name": "Shyam Upadhyay",
    "id": "2254265068",
    "h_index": 6,
    "papers": 14
   },
   {
    "name": "Vincent Cohen-Addad",
    "id": "2248219819",
    "h_index": 11,
    "papers": 32
   },
   {
    "name": "Sean Purser-Haskell",
    "id": "2112206162",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Yichong Xu",
    "id": "2110197273",
    "h_index": 20,
    "papers": 39
   },
   {
    "name": "Ebrahim M. Songhori",
    "id": "2714003",
    "h_index": 15,
    "papers": 40
   },
   {
    "name": "Babi Seal",
    "id": "2373038315",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Al-berto Magni",
    "id": "2275182383",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Almog Gueta",
    "id": "2204956989",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Tingting Zou",
    "id": "145656311",
    "h_index": 6,
    "papers": 20
   },
   {
    "name": "Guru Guruganesh",
    "id": "1947314",
    "h_index": 16,
    "papers": 42
   },
   {
    "name": "Thais Kagohara",
    "id": "2275186582",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Hung Nguyen",
    "id": "2324895068",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Khalid Salama",
    "id": "2324798792",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Alejandro Ruiz",
    "id": "2079785300",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Justin Frye",
    "id": "2275193725",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Zhenkai Zhu",
    "id": "2275539055",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "M. Lochbrunner",
    "id": "40829060",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Simon Osindero",
    "id": "2217144",
    "h_index": 36,
    "papers": 85
   },
   {
    "name": "Wentao Yuan",
    "id": "2352610181",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "Lisa Lee",
    "id": "2275291886",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Aman Prasad",
    "id": "2372462269",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Lam Nguyen Thiet",
    "id": "2275177831",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Daniele Calandriello",
    "id": "2439765",
    "h_index": 26,
    "papers": 56
   },
   {
    "name": "V. Stone",
    "id": "2373038623",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Qixuang Feng",
    "id": "2191691224",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Han Ke",
    "id": "2065995558",
    "h_index": 4,
    "papers": 18
   },
   {
    "name": "M. Voitovich",
    "id": "2061512997",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Geta Sampemane",
    "id": "2373038076",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "L. Chiang",
    "id": "2372305348",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Ling-da Wu",
    "id": "2116666487",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Alexander Bykovsky",
    "id": "2373036930",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Matt Young",
    "id": "2349752516",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Luke Vilnis",
    "id": "2289035179",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Ishita Dasgupta",
    "id": "2263548023",
    "h_index": 7,
    "papers": 15
   },
   {
    "name": "Aditya Chawla",
    "id": "1864148966",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Qin Cao",
    "id": "2285985741",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Bowen Liang",
    "id": "2306102444",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Daniel Toyama",
    "id": "1393948967",
    "h_index": 11,
    "papers": 19
   },
   {
    "name": "S. Payrits",
    "id": "2530102",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Anca Stefanoiu",
    "id": "3417870",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Dimitrios Vytiniotis",
    "id": "1757457",
    "h_index": 29,
    "papers": 92
   },
   {
    "name": "Ankesh Anand",
    "id": "12679121",
    "h_index": 14,
    "papers": 28
   },
   {
    "name": "Tianxiao Shen",
    "id": "2372386646",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "B. Mitrevski",
    "id": "48207181",
    "h_index": 22,
    "papers": 64
   },
   {
    "name": "Michael Tschannen",
    "id": "143902495",
    "h_index": 37,
    "papers": 72
   },
   {
    "name": "Sreenivas Gollapudi",
    "id": "144979147",
    "h_index": 26,
    "papers": 125
   },
   {
    "name": "S. AishwaryaP",
    "id": "2283933275",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "J. Leal",
    "id": "1919812",
    "h_index": 19,
    "papers": 152
   },
   {
    "name": "Zhe Shen",
    "id": "2314348713",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Han Fu",
    "id": "2344205869",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Wei Wang",
    "id": "2402897909",
    "h_index": 8,
    "papers": 45
   },
   {
    "name": "Arvind Kannan",
    "id": "2300997775",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Doron Kukliansky",
    "id": "2771709",
    "h_index": 7,
    "papers": 8
   },
   {
    "name": "Sergey Yaroshenko",
    "id": "2102918715",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Svetlana Grant",
    "id": "2311701299",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Umesh Telang",
    "id": "2047820428",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "D. Wood",
    "id": "2114999609",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "A. Chronopoulou",
    "id": "2324583448",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Alexandru cTifrea",
    "id": "2122910191",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "T. Zhou",
    "id": "144137446",
    "h_index": 5,
    "papers": 16
   },
   {
    "name": "Tony Nguyen",
    "id": "2349547163",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Muge Ersoy",
    "id": "2373042833",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Anima Singh",
    "id": "2111008861",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Meiyan Xie",
    "id": "6638834",
    "h_index": 7,
    "papers": 17
   },
   {
    "name": "Emanuel Taropa",
    "id": "2779842",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Woohyun Han",
    "id": "2215449616",
    "h_index": 7,
    "papers": 15
   },
   {
    "name": "E. Agustsson",
    "id": "2794259",
    "h_index": 32,
    "papers": 49
   },
   {
    "name": "Andrei Sozanschi",
    "id": "2275182611",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Hui Peng",
    "id": "2395868237",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Alex Chen",
    "id": "2374015015",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Y. Drori",
    "id": "2144598002",
    "h_index": 15,
    "papers": 21
   },
   {
    "name": "Efren Robles",
    "id": "2281508638",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Yang Gao",
    "id": "2336312153",
    "h_index": 5,
    "papers": 16
   },
   {
    "name": "Xerxes Dotiwalla",
    "id": "1404332584",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Ying Chen",
    "id": "2377245226",
    "h_index": 6,
    "papers": 14
   },
   {
    "name": "Anudhyan Boral",
    "id": "11167300",
    "h_index": 10,
    "papers": 22
   },
   {
    "name": "Alexei Bendebury",
    "id": "2351911525",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "John Nham",
    "id": "4111313",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "C. Tar",
    "id": "7887562",
    "h_index": 10,
    "papers": 15
   },
   {
    "name": "Luis Castro",
    "id": "2055134235",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Jiepu Jiang",
    "id": "2260169185",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Canoee Liu",
    "id": "2218423887",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Felix Halim",
    "id": "2073773301",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Jinoo Baek",
    "id": "2352935625",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "A. Wan",
    "id": "32652863",
    "h_index": 5,
    "papers": 18
   },
   {
    "name": "Jeremiah Liu",
    "id": "2273905457",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "Yuan Cao",
    "id": "2295912737",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Shengyang Dai",
    "id": "2314112926",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "T. Acharya",
    "id": "8647819",
    "h_index": 18,
    "papers": 55
   },
   {
    "name": "Ruoxi Sun",
    "id": "2298925536",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Fuzhao Xue",
    "id": "2313476433",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Saket Joshi",
    "id": "145593737",
    "h_index": 9,
    "papers": 27
   },
   {
    "name": "Morgane Lustman",
    "id": "134426275",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Yongqin Xian",
    "id": "2256223080",
    "h_index": 10,
    "papers": 20
   },
   {
    "name": "Rishabh Joshi",
    "id": "2258551072",
    "h_index": 10,
    "papers": 17
   },
   {
    "name": "Deep Karkhanis",
    "id": "1740653882",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Nora Kassner",
    "id": "2306326044",
    "h_index": 1,
    "papers": 10
   },
   {
    "name": "Jamie Hall",
    "id": "2344103953",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Xiangzhuo Ding",
    "id": "2117436426",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Gan Song",
    "id": "2323463296",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Gang Li",
    "id": "2324109428",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Chen Zhu",
    "id": "2325192664",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Yana Kulizhskaya",
    "id": "2275177999",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Bin Ni",
    "id": "2307454954",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "A. Vlaskin",
    "id": "2333889933",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Solomon Demmessie",
    "id": "119710664",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Lucio M. Dery",
    "id": "2349820533",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Salah Zaiem",
    "id": "2402718424",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Yanping Huang",
    "id": "2349831834",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Cindy Fan",
    "id": "2375323483",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Felix Gimeno",
    "id": "49423009",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "Ananth Balashankar",
    "id": "2593082",
    "h_index": 11,
    "papers": 47
   },
   {
    "name": "K. Kojima",
    "id": "2400914413",
    "h_index": 17,
    "papers": 104
   },
   {
    "name": "Hagai Taitelbaum",
    "id": "51258885",
    "h_index": 6,
    "papers": 16
   },
   {
    "name": "M. Meng",
    "id": "2054282873",
    "h_index": 3,
    "papers": 16
   },
   {
    "name": "Dero Gharibian",
    "id": "69997688",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Sahil Singla",
    "id": "2306781072",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Wei Chen",
    "id": "2383310127",
    "h_index": 6,
    "papers": 17
   },
   {
    "name": "Ambrose Slone",
    "id": "133666998",
    "h_index": 9,
    "papers": 29
   },
   {
    "name": "Guanjie Chen",
    "id": "2307435117",
    "h_index": 6,
    "papers": 17
   },
   {
    "name": "Sujeevan Rajayogam",
    "id": "2324800990",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Max Schumacher",
    "id": "2326100273",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "S. Kotecha",
    "id": "2329864197",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "R. Blevins",
    "id": "46901218",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Qifei Wang",
    "id": "2401629878",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "M. Taege",
    "id": "47975898",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "A. Morris",
    "id": "2275162692",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Xin Liu",
    "id": "2373691831",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Fayaz Jamil",
    "id": "2261083305",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Richard Zhang",
    "id": "2268175436",
    "h_index": 3,
    "papers": 13
   },
   {
    "name": "Pratik Joshi",
    "id": "2336086099",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "B. Ingram",
    "id": "2280241913",
    "h_index": 5,
    "papers": 29
   },
   {
    "name": "T. Liechty",
    "id": "2275188012",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Ahmed Eleryan",
    "id": "2338055207",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Scott P. Baird",
    "id": "2361300324",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Alex Grills",
    "id": "2373034554",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Gagan Bansal",
    "id": "2253477657",
    "h_index": 9,
    "papers": 15
   },
   {
    "name": "Shan Han",
    "id": "2372769193",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Kiran Yalasangi",
    "id": "2158997234",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Shawn Xu",
    "id": "2300139620",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Majd Al Merey",
    "id": "2089890458",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Isabel Gao",
    "id": "2290513267",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Felix Weissenberger",
    "id": "10805574",
    "h_index": 7,
    "papers": 15
   },
   {
    "name": "Igor Karpov",
    "id": "50355738",
    "h_index": 11,
    "papers": 23
   },
   {
    "name": "Robert Riachi",
    "id": "2373037960",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Ankit Anand",
    "id": "2311701214",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Gautam Prasad",
    "id": "2775959",
    "h_index": 7,
    "papers": 7
   },
   {
    "name": "Kay Lamerigts",
    "id": "2324052906",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Reid Hayes",
    "id": "39336466",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "J. Daniel Rogers",
    "id": "2252992217",
    "h_index": 8,
    "papers": 9
   },
   {
    "name": "Mandy Guo",
    "id": "2275169672",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Ashish Shenoy",
    "id": "2275187275",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Qiong Hu",
    "id": "2113362191",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Kyle He",
    "id": "2211736022",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Yuchen Liu",
    "id": "2373587444",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Polina Zablotskaia",
    "id": "7164154",
    "h_index": 8,
    "papers": 16
   },
   {
    "name": "S. Gubbi",
    "id": "2306041520",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Yifan Chang",
    "id": "2302821979",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Jay Pavagadhi",
    "id": "2275177854",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Kristian Kjems",
    "id": "2095741288",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Archita Vadali",
    "id": "2373036839",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Diego Machado",
    "id": "2372609648",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Yeqing Li",
    "id": "2357984168",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Ren-shen Wang",
    "id": "2290529512",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Dipankar Ghosh",
    "id": "2375477259",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "A. Mehta",
    "id": "2390726404",
    "h_index": 3,
    "papers": 14
   },
   {
    "name": "Dana Alon",
    "id": "2261282871",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "George Polovets",
    "id": "1402376936",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "A. Tonioni",
    "id": "20406113",
    "h_index": 23,
    "papers": 60
   },
   {
    "name": "Nate Kushman",
    "id": "1684887",
    "h_index": 24,
    "papers": 45
   },
   {
    "name": "J. D'sa",
    "id": "1404679583",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Lin Zhuo",
    "id": "2375701819",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Allen Wu",
    "id": "2406429212",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Rohin Shah",
    "id": "2359788174",
    "h_index": 8,
    "papers": 13
   },
   {
    "name": "J. Youssef",
    "id": "119729048",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Jiayu Ye",
    "id": "2266807828",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Justin Snyder",
    "id": "2052437158",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Karel Lenc",
    "id": "3257286",
    "h_index": 14,
    "papers": 26
   },
   {
    "name": "S. Buthpitiya",
    "id": "2620680",
    "h_index": 13,
    "papers": 30
   },
   {
    "name": "Matthew Tung",
    "id": "2275176212",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Jichuan Chang",
    "id": "1698747",
    "h_index": 21,
    "papers": 49
   },
   {
    "name": "Tao Chen",
    "id": "2152186563",
    "h_index": 10,
    "papers": 25
   },
   {
    "name": "David Saxton",
    "id": "2320805428",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Jenny Lee",
    "id": "2108442117",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Lydia Zhang",
    "id": "2410298092",
    "h_index": 3,
    "papers": 14
   },
   {
    "name": "James Qin",
    "id": "2316813695",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "P. Radhakrishnan",
    "id": "2242989573",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Maxwell Chen",
    "id": "2372304819",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Piotr Ambroszczyk",
    "id": "2373039607",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Metin Toksoz-Exley",
    "id": "2101721125",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Yan Zhong",
    "id": "2326297398",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Nitzan Katz",
    "id": "2373039934",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Brendan O'Donoghue",
    "id": "2064271917",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Tamara von Glehn",
    "id": "51029932",
    "h_index": 11,
    "papers": 23
   },
   {
    "name": "A. Rosenthal",
    "id": "2137312054",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Agnieszka Swietlik",
    "id": "2280143694",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Xiaokai Zhao",
    "id": "2325483290",
    "h_index": 8,
    "papers": 27
   },
   {
    "name": "Nicholas Fernando",
    "id": "2186403526",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Jinliang Wei",
    "id": "2275213068",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Jieru Mei",
    "id": "10407760",
    "h_index": 20,
    "papers": 37
   },
   {
    "name": "Sergei Vassilvitskii",
    "id": "2282962033",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Diego Cedillo",
    "id": "2373038416",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Pranjal Awasthi",
    "id": "144030228",
    "h_index": 33,
    "papers": 128
   },
   {
    "name": "Hui-Wen Zheng",
    "id": "2392412725",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "K. Kavukcuoglu",
    "id": "2645384",
    "h_index": 76,
    "papers": 124
   },
   {
    "name": "I. Laish",
    "id": "2265190644",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Joseph Pagadora",
    "id": "2288558325",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Marc Brockschmidt",
    "id": "2107692",
    "h_index": 35,
    "papers": 70
   },
   {
    "name": "Christopher A. Choquette-Choo",
    "id": "2314115870",
    "h_index": 14,
    "papers": 38
   },
   {
    "name": "Arun Byravan",
    "id": "2283762951",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Yifeng Lu",
    "id": "2316615958",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Xu Chen",
    "id": "2363371728",
    "h_index": 5,
    "papers": 15
   },
   {
    "name": "Mianna Chen",
    "id": "2275287759",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Kenton Lee",
    "id": "2324512904",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "R. Pasumarthi",
    "id": "2121709804",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Sijal Bhatnagar",
    "id": "2351909196",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Aditya Shah",
    "id": "2223213194",
    "h_index": 4,
    "papers": 16
   },
   {
    "name": "Qiyin Wu",
    "id": "31354662",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Zhuoyuan Chen",
    "id": "2382500477",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Zachary Nado",
    "id": "81408931",
    "h_index": 21,
    "papers": 36
   },
   {
    "name": "Bartek Perz",
    "id": "2275180701",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Zixuan Jiang",
    "id": "2274903344",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "D. Kao",
    "id": "145230530",
    "h_index": 16,
    "papers": 59
   },
   {
    "name": "G. Mallya",
    "id": "2354179184",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Nino Vieillard",
    "id": "2308037523",
    "h_index": 10,
    "papers": 31
   },
   {
    "name": "Lantao Mei",
    "id": "48176111",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Sertan Girgin",
    "id": "35022714",
    "h_index": 23,
    "papers": 78
   },
   {
    "name": "M. Jordan",
    "id": "2283786324",
    "h_index": 2,
    "papers": 24
   },
   {
    "name": "Yeongil Ko",
    "id": "2275191840",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Alekh Agarwal",
    "id": "2274120058",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "Yaxin Liu",
    "id": "2290524803",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Yasemin Altun",
    "id": "2306632499",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Raoul de Liedekerke",
    "id": "2275184736",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Anastasios Kementsietsidis",
    "id": "2182066815",
    "h_index": 3,
    "papers": 31
   },
   {
    "name": "Daiyi Peng",
    "id": "2324800542",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Dangyi Liu",
    "id": "2290539499",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Utku Evci",
    "id": "3399348",
    "h_index": 19,
    "papers": 33
   },
   {
    "name": "Peter C. Humphreys",
    "id": "2275186692",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Austin Tarango",
    "id": "2079629578",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Xiang Deng",
    "id": "2330085651",
    "h_index": 6,
    "papers": 19
   },
   {
    "name": "Yoad Lewenberg",
    "id": "2291654",
    "h_index": 12,
    "papers": 25
   },
   {
    "name": "Kevin Aydin",
    "id": "2032042",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Chengda Wu",
    "id": "2372367965",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Bhavishya Mittal",
    "id": "2275177915",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Tsendsuren Munkhdalai",
    "id": "2227827",
    "h_index": 26,
    "papers": 64
   },
   {
    "name": "K. Chatziprimou",
    "id": "2702929",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Rodrigo Benenson",
    "id": "2264186625",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Uri First",
    "id": "2373038126",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Xiao Ma",
    "id": "2275564095",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Jinning Li",
    "id": "2373558300",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Armand Joulin",
    "id": "2319608",
    "h_index": 72,
    "papers": 151
   },
   {
    "name": "Hamish Tomlinson",
    "id": "2293394534",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Tingnan Zhang",
    "id": "2240715659",
    "h_index": 14,
    "papers": 25
   },
   {
    "name": "Milad Nasr",
    "id": "3490923",
    "h_index": 33,
    "papers": 85
   },
   {
    "name": "Zhi Hong",
    "id": "2384423628",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Michael E. Sander",
    "id": "2372860785",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "L. Hendricks",
    "id": "2258347245",
    "h_index": 12,
    "papers": 30
   },
   {
    "name": "Anuj Sharma",
    "id": "2297251026",
    "h_index": 6,
    "papers": 56
   },
   {
    "name": "Andrew Bolt",
    "id": "2290488166",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Eszter V'ertes",
    "id": "2136446499",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Ji\u0159\u00ed \u0160im\u0161a",
    "id": "38300863",
    "h_index": 17,
    "papers": 37
   },
   {
    "name": "Tomer Levinboim",
    "id": "2900341",
    "h_index": 11,
    "papers": 21
   },
   {
    "name": "O. Sercinoglu",
    "id": "2099860914",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Divyanshu Shukla",
    "id": "2283646257",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Austin Wu",
    "id": "2147235308",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Craig Swanson",
    "id": "2275181554",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Danny Vainstein",
    "id": "2265084287",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "F. Bu",
    "id": "2326301838",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Boyu Wang",
    "id": "2373739775",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Ryan C. Julian",
    "id": "144885996",
    "h_index": 19,
    "papers": 34
   },
   {
    "name": "Charles Yoon",
    "id": "2141654755",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "S. Lebedev",
    "id": "2387718810",
    "h_index": 1,
    "papers": 9
   },
   {
    "name": "Antonious M. Girgis",
    "id": "38485018",
    "h_index": 11,
    "papers": 28
   },
   {
    "name": "B. Bandemer",
    "id": "2687860",
    "h_index": 16,
    "papers": 35
   },
   {
    "name": "David Du",
    "id": "2268904939",
    "h_index": 1,
    "papers": 9
   },
   {
    "name": "Todd Wang",
    "id": "2162794697",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Xi Chen",
    "id": "2275535939",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Ying Xiao",
    "id": "2363983417",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Peggy Lu",
    "id": "2374129689",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Natalie Ha",
    "id": "2410068702",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Vlad Ionescu",
    "id": "2313685593",
    "h_index": 7,
    "papers": 34
   },
   {
    "name": "Simon Rowe",
    "id": "2282688147",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "J. Matak",
    "id": "87071549",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "F. Lebron",
    "id": "2275184616",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Andreas Steiner",
    "id": "2350755056",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Lalit Jain",
    "id": "2372815741",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Manaal Faruqui",
    "id": "1779225",
    "h_index": 37,
    "papers": 81
   },
   {
    "name": "Nicolas Lacasse",
    "id": "2373034652",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "G. Evans",
    "id": "52319489",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Neesha Subramaniam",
    "id": "25579814",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "D. Reich",
    "id": "2373036330",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Giulia Vezzani",
    "id": "2315317835",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "A. Pandey",
    "id": "2357852189",
    "h_index": 19,
    "papers": 482
   },
   {
    "name": "Joe Stanton",
    "id": "2275190309",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "T. Zhou",
    "id": "2374199642",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Liam McCafferty",
    "id": "48291331",
    "h_index": 9,
    "papers": 15
   },
   {
    "name": "Henry Griffiths",
    "id": "1962358666",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Verena Rieser",
    "id": "2327053658",
    "h_index": 5,
    "papers": 15
   },
   {
    "name": "S. Yeganeh",
    "id": "1735318",
    "h_index": 16,
    "papers": 33
   },
   {
    "name": "Eleftheria Briakou",
    "id": "2343507548",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Lu Huang",
    "id": "2373441435",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Zichuan Wei",
    "id": "2314328859",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "Liangchen Luo",
    "id": "2256991052",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Erik Jue",
    "id": "2373042274",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Gabby Wang",
    "id": "122337061",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Victor Cotruta",
    "id": "2275189218",
    "h_index": 6,
    "papers": 20
   },
   {
    "name": "M. Khan",
    "id": "66145271",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Jongbin Park",
    "id": "2109284811",
    "h_index": 1,
    "papers": 12
   },
   {
    "name": "Qi Guo",
    "id": "2355368662",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Peiran Li",
    "id": "2325254893",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Rong Rong",
    "id": "2065927841",
    "h_index": 3,
    "papers": 20
   },
   {
    "name": "Diego Antognini",
    "id": "26399699",
    "h_index": 12,
    "papers": 30
   },
   {
    "name": "A. Petrushkina",
    "id": "2275187155",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Chetan Tekur",
    "id": "118505443",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Eli Collins",
    "id": "2275181648",
    "h_index": 8,
    "papers": 23
   },
   {
    "name": "Parul Bhatia",
    "id": "2339692051",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Chester Kwak",
    "id": "2324801379",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Wenhu Chen",
    "id": "2362858999",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Arvind Neelakantan",
    "id": "2072676",
    "h_index": 24,
    "papers": 75
   },
   {
    "name": "Immanuel Odisho",
    "id": "2373036281",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Sheng Peng",
    "id": "2280210471",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Vincent Nallatamby",
    "id": "2373038250",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Vaibhav Tulsyan",
    "id": "84171112",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "F. Pedregosa",
    "id": "2310335993",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Peng Xu",
    "id": "2329711463",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Raymond Lin",
    "id": "2068171220",
    "h_index": 4,
    "papers": 14
   },
   {
    "name": "Yulong Wang",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Emma Wang",
    "id": "2314334001",
    "h_index": 6,
    "papers": 16
   },
   {
    "name": "Sholto Douglas",
    "id": "2269733876",
    "h_index": 8,
    "papers": 19
   },
   {
    "name": "Reut Tsarfaty",
    "id": "2799181",
    "h_index": 30,
    "papers": 127
   },
   {
    "name": "E. Gribovskaya",
    "id": "1980809",
    "h_index": 18,
    "papers": 34
   },
   {
    "name": "Renga Aravamudhan",
    "id": "5130509",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Manu Agarwal",
    "id": "2372584064",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Mara Finkelstein",
    "id": "2257001597",
    "h_index": 9,
    "papers": 21
   },
   {
    "name": "Qiao Zhang",
    "id": "2197671266",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Elizabeth Cole",
    "id": "2275178399",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Phil Crone",
    "id": "2275183277",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Sarmishta Velury",
    "id": "47101851",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Anil Das",
    "id": "2352906247",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "C. Sauer",
    "id": "2514884",
    "h_index": 24,
    "papers": 92
   },
   {
    "name": "Luyao Xu",
    "id": "2324836588",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Danfeng Qin",
    "id": "2296785667",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Chenjie Gu",
    "id": "2275149073",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Dror Marcus",
    "id": "2307471502",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "CJ Zheng",
    "id": "2244626915",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Wouter Van Gansbeke",
    "id": "66314383",
    "h_index": 10,
    "papers": 13
   },
   {
    "name": "Sobhan Miryoosefi",
    "id": "2373035138",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Haitian Sun",
    "id": "3456820",
    "h_index": 14,
    "papers": 32
   },
   {
    "name": "Yaguang Li",
    "id": "2261797906",
    "h_index": 10,
    "papers": 28
   },
   {
    "name": "Charlie Chen",
    "id": "2182971260",
    "h_index": 10,
    "papers": 22
   },
   {
    "name": "Jaewook Yoo",
    "id": "2115662588",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "P. Dubov",
    "id": "2373036748",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Alex Tomala",
    "id": "2275176047",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Adams Yu",
    "id": "2312325791",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Pawe\u0142 Weso\u0142owski",
    "id": "2295556765",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Alok Gunjan",
    "id": "2373036891",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Eddie Cao",
    "id": "2372474857",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Jiaming Luo",
    "id": "19236313",
    "h_index": 9,
    "papers": 17
   },
   {
    "name": "Nikhil Sethi",
    "id": "2275187415",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Arkadiusz Socala",
    "id": "3048839",
    "h_index": 9,
    "papers": 18
   },
   {
    "name": "Laura Graesser",
    "id": "30131402",
    "h_index": 12,
    "papers": 19
   },
   {
    "name": "Tom\u00e1s Kocisk\u00fd",
    "id": "2367821",
    "h_index": 17,
    "papers": 41
   },
   {
    "name": "BC Arturo",
    "id": "2373036812",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Minmin Chen",
    "id": "1743082",
    "h_index": 30,
    "papers": 100
   },
   {
    "name": "Edward Lee",
    "id": "2337235089",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Sophie Wang",
    "id": "2396521977",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Weize Kong",
    "id": "2275163940",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Qian Xu",
    "id": "2400632478",
    "h_index": 19,
    "papers": 287
   },
   {
    "name": "Nilesh Tripuraneni",
    "id": "1925801",
    "h_index": 20,
    "papers": 33
   },
   {
    "name": "Yiming Li",
    "id": "2322365558",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Xinxin Yu",
    "id": "2335544759",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "A. Porter",
    "id": "2055473371",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "P. Voigtlaender",
    "id": "2767859",
    "h_index": 24,
    "papers": 40
   },
   {
    "name": "Biao Zhang",
    "id": "2275942342",
    "h_index": 6,
    "papers": 17
   },
   {
    "name": "Arpi Vezer",
    "id": "2275188533",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Sarah York",
    "id": "143981350",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Qinglan Wei",
    "id": "2271279019",
    "h_index": 3,
    "papers": 22
   },
   {
    "name": "Geoffrey Cideron",
    "id": "2282966842",
    "h_index": 8,
    "papers": 16
   },
   {
    "name": "Mark Kurzeja",
    "id": "2373035013",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Seungyeon Kim",
    "id": "2328160131",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Benny Li",
    "id": "2373540467",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Ang\u00e9line Pouget",
    "id": "2093477010",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Hyo Lee",
    "id": "2275320829",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Kaspar Daugaard",
    "id": "2373038780",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Yang Li",
    "id": "2367447542",
    "h_index": 10,
    "papers": 33
   },
   {
    "name": "David Uthus",
    "id": "2267341621",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Aditya Siddhant",
    "id": "2373042288",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Paul Cavallaro",
    "id": "2373036851",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Sriram Ganapathy",
    "id": "1726355",
    "h_index": 31,
    "papers": 203
   },
   {
    "name": "Maulik Shah",
    "id": "2325749645",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "R. Jagerman",
    "id": "1886219",
    "h_index": 15,
    "papers": 33
   },
   {
    "name": "J. Stanway",
    "id": "35149729",
    "h_index": 10,
    "papers": 32
   },
   {
    "name": "Piermaria Mendolicchio",
    "id": "2156929516",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Li Xiao",
    "id": "2302929199",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Kayi Lee",
    "id": "2374238525",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Tara Thompson",
    "id": "2054787395",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Shubham Milind Phal",
    "id": "1491516919",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Jason A. Chase",
    "id": "2372500857",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Sun Jae Lee",
    "id": "2336914594",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Adrian N. Reyes",
    "id": "2371075311",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Disha Shrivastava",
    "id": "2275113487",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Zhen Qin",
    "id": "2319877894",
    "h_index": 2,
    "papers": 17
   },
   {
    "name": "Roykrong Sukkerd",
    "id": "3100287",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "S. Odoom",
    "id": "2275182230",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Lior Madmoni",
    "id": "10425319",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "John Aslanides",
    "id": "9958912",
    "h_index": 14,
    "papers": 19
   },
   {
    "name": "Jonathan Herzig",
    "id": "2253566854",
    "h_index": 11,
    "papers": 20
   },
   {
    "name": "Elena Pochernina",
    "id": "2157966428",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Sheng Zhang",
    "id": "2396557516",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Parker Barnes",
    "id": "80940648",
    "h_index": 8,
    "papers": 21
   },
   {
    "name": "Daisuke Ikeda",
    "id": "2311130335",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Qiujia Li",
    "id": "2373501095",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Shuo-Yiin Chang",
    "id": "2275193337",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Shakir Mohamed",
    "id": "2312128006",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Jim Sproch",
    "id": "1769078",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Richard Powell",
    "id": "2067745837",
    "h_index": 7,
    "papers": 14
   },
   {
    "name": "Bidisha Samanta",
    "id": "2277741562",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Domagoj Cevid",
    "id": "1456876802",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Anton Kovsharov",
    "id": "2373037002",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Shrestha Basu Mallick",
    "id": "48204180",
    "h_index": 6,
    "papers": 6
   },
   {
    "name": "Srinivas Tadepalli",
    "id": "2313410903",
    "h_index": 3,
    "papers": 39
   },
   {
    "name": "A. Zheng",
    "id": "2407459786",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Kareem W. Ayoub",
    "id": "34122449",
    "h_index": 8,
    "papers": 14
   },
   {
    "name": "Andreas Noever",
    "id": "3376211",
    "h_index": 8,
    "papers": 17
   },
   {
    "name": "C. Reisswig",
    "id": "9992847",
    "h_index": 33,
    "papers": 67
   },
   {
    "name": "Zhuo Xu",
    "id": "2265456732",
    "h_index": 14,
    "papers": 18
   },
   {
    "name": "Junhyuk Oh",
    "id": "2275114643",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Martin Matysiak",
    "id": "2073180748",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Tim Blyth",
    "id": "2221119859",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Shereen Ashraf",
    "id": "2275181309",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "J. Amelot",
    "id": "2506388",
    "h_index": 10,
    "papers": 23
   },
   {
    "name": "Boone Severson",
    "id": "2373038822",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Michele Bevilacqua",
    "id": "2253475117",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Motoki Sano",
    "id": "3151314",
    "h_index": 8,
    "papers": 25
   },
   {
    "name": "Ethan Dyer",
    "id": "2275180676",
    "h_index": 4,
    "papers": 14
   },
   {
    "name": "Ofir Roval",
    "id": "2275182219",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Anu Sinha",
    "id": "2079873547",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Y. Zhong",
    "id": "2258684869",
    "h_index": 18,
    "papers": 246
   },
   {
    "name": "Sagi Perel",
    "id": "3274881",
    "h_index": 12,
    "papers": 22
   },
   {
    "name": "Tea Saboli'c",
    "id": "2373038419",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Johannes Mauerer",
    "id": "2373037187",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "W. Gierke",
    "id": "145556052",
    "h_index": 8,
    "papers": 15
   },
   {
    "name": "Mauro Verzetti",
    "id": "2246352916",
    "h_index": 6,
    "papers": 6
   },
   {
    "name": "Rodrigo Cabrera",
    "id": "2056624878",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Alvin Abdagic",
    "id": "3026185",
    "h_index": 6,
    "papers": 16
   },
   {
    "name": "S. Hemingray",
    "id": "102876101",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Austin Stone",
    "id": "2316127614",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Jong Lee",
    "id": "2275753935",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Farooq Ahmad",
    "id": "2053051132",
    "h_index": 5,
    "papers": 15
   },
   {
    "name": "K. Raman",
    "id": "2062947723",
    "h_index": 12,
    "papers": 27
   },
   {
    "name": "Lior Shani",
    "id": "38274824",
    "h_index": 12,
    "papers": 22
   },
   {
    "name": "J. Lai",
    "id": "2276422942",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Orhan Firat",
    "id": "2273534960",
    "h_index": 10,
    "papers": 33
   },
   {
    "name": "Nathan Waters",
    "id": "114958554",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Eric Ge",
    "id": "2373042464",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Mo Shomrat",
    "id": "2373038609",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Himanshu Gupta",
    "id": "2324799492",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "R. Aggarwal",
    "id": "2220733426",
    "h_index": 3,
    "papers": 14
   },
   {
    "name": "Tom Hudson",
    "id": "2275187110",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Bill Jia",
    "id": "2372590728",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Simon Baumgartner",
    "id": "2282531735",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Palak Jain",
    "id": "2261758456",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "J. Kovac",
    "id": "14784693",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Junehyuk Jung",
    "id": "2344187360",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Ante vZuvzul",
    "id": "2373038413",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "W. Truong",
    "id": "12816249",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Morteza Zadimoghaddam",
    "id": "1724391",
    "h_index": 33,
    "papers": 95
   },
   {
    "name": "Songyou Peng",
    "id": "2350462602",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "M. Liang",
    "id": "144472614",
    "h_index": 9,
    "papers": 100
   },
   {
    "name": "Rachel Sterneck",
    "id": "1932270322",
    "h_index": 7,
    "papers": 17
   },
   {
    "name": "Balaji Lakshminarayanan",
    "id": "40627523",
    "h_index": 46,
    "papers": 92
   },
   {
    "name": "Machel Reid",
    "id": "1557386977",
    "h_index": 19,
    "papers": 35
   },
   {
    "name": "Oliver Woodman",
    "id": "2275186839",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Tong Zhou",
    "id": "2374199644",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Jianling Wang",
    "id": "2264885653",
    "h_index": 6,
    "papers": 24
   },
   {
    "name": "Vincent Coriou",
    "id": "2352939443",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Arjun Narayanan",
    "id": "2372997777",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "J. Hoover",
    "id": "2275186591",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Yenai Ma",
    "id": "2327754",
    "h_index": 8,
    "papers": 17
   },
   {
    "name": "Apoorv Jindal",
    "id": "2199404295",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Clayton Sanford",
    "id": "2303654564",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Doug Reid",
    "id": "2351913146",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Swaroop Indra Ramaswamy",
    "id": "9219142",
    "h_index": 11,
    "papers": 20
   },
   {
    "name": "Alexey Kurakin",
    "id": "2284694544",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Roland S. Zimmermann",
    "id": "2359000043",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Yana Lunts",
    "id": "2373037085",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "D. Dena",
    "id": "9211802",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Zal\u00e1n Borsos",
    "id": "144494941",
    "h_index": 15,
    "papers": 27
   },
   {
    "name": "Vered Cohen",
    "id": "2275184693",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Shujian Zhang",
    "id": "2315457348",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Will Grathwohl",
    "id": "2279228337",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Robert Dadashi",
    "id": "51914693",
    "h_index": 20,
    "papers": 40
   },
   {
    "name": "Morgan Redshaw",
    "id": "2073190396",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Joshua Kessinger",
    "id": "2324801905",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "J. Odell",
    "id": "2054927716",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Silvano Bonacina",
    "id": "2373034817",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Zihang Dai",
    "id": "2285185684",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Grace Chen",
    "id": "2373505603",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Ayush Dubey",
    "id": "2292313225",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "P. Sprechmann",
    "id": "2905900",
    "h_index": 28,
    "papers": 54
   },
   {
    "name": "Mantas Pajarskas",
    "id": "2146532125",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Wenxuan Zhou",
    "id": "2246658976",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "N. Ahuja",
    "id": "2324798245",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "T. Thomas",
    "id": "49555331",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Martin Nikoltchev",
    "id": "1850250539",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Matija Kecman",
    "id": "2279546451",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Bharath Mankalale",
    "id": "2373038593",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Andrey Ryabtsev",
    "id": "2068646799",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Jennifer She",
    "id": "2311701560",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Christian J. Walder",
    "id": "2362630425",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Jiaming Shen",
    "id": "2266463492",
    "h_index": 8,
    "papers": 13
   },
   {
    "name": "Lu Li",
    "id": "2275716550",
    "h_index": 5,
    "papers": 17
   },
   {
    "name": "Carolina Parada",
    "id": "2238125998",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Sheena Panthaplackel",
    "id": "1468751197",
    "h_index": 9,
    "papers": 13
   },
   {
    "name": "Okwan Kwon",
    "id": "2746997",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Matthew Lawlor",
    "id": "120457427",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Utsav Prabhu",
    "id": "2310437864",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Yannick Schroecker",
    "id": "2274104424",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Marc'Aurelio Ranzato",
    "id": "1706809",
    "h_index": 63,
    "papers": 123
   },
   {
    "name": "Pete Blois",
    "id": "2373036856",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Iurii Kemaev",
    "id": "51883910",
    "h_index": 7,
    "papers": 9
   },
   {
    "name": "Ting Yu",
    "id": "2314504391",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "Dmitry Lepikhin",
    "id": "150077954",
    "h_index": 13,
    "papers": 25
   },
   {
    "name": "Hao Xiong",
    "id": "2325957537",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Sahand Sharifzadeh",
    "id": "7782886",
    "h_index": 14,
    "papers": 28
   },
   {
    "name": "Oleaser Johnson",
    "id": "2372564511",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Jeremiah Willcock",
    "id": "144233173",
    "h_index": 19,
    "papers": 51
   },
   {
    "name": "Rui Yao",
    "id": "2352107831",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Gregory Farquhar",
    "id": "2358997148",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Sujoy Basu",
    "id": "2266467648",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "H. Shimokawa",
    "id": "30581571",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "N. Anderson",
    "id": "2068945909",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Haiguang Li",
    "id": "2190035791",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Khiem Pham",
    "id": "2055776505",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Yizhong Liang",
    "id": "2372424449",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Sebastian Borgeaud",
    "id": "148016269",
    "h_index": 20,
    "papers": 52
   },
   {
    "name": "Alexandre Moufarek",
    "id": "2275184966",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "H. Kazawa",
    "id": "1754386",
    "h_index": 14,
    "papers": 32
   },
   {
    "name": "Blair Kutzman",
    "id": "2373036903",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Marcin Sieniek",
    "id": "1717409",
    "h_index": 10,
    "papers": 35
   },
   {
    "name": "Sara Smoot",
    "id": "2351911771",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Ruth Wang",
    "id": "2394564212",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Natalie Axelsson",
    "id": "2373036874",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Nova Fallen",
    "id": "2296785713",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "P. Sundaram",
    "id": "2098302716",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Yuexiang Zhai",
    "id": "119692515",
    "h_index": 19,
    "papers": 30
   },
   {
    "name": "Varun Godbole",
    "id": "40156666",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "P. Maniatis",
    "id": "2329474887",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "A. Wang",
    "id": "2290580195",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Ilia Shumailov",
    "id": "47473421",
    "h_index": 28,
    "papers": 111
   },
   {
    "name": "Santhosh Thangaraj",
    "id": "2342530265",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Remi Crocker",
    "id": "2275186244",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Nikita Gupta",
    "id": "2373651741",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Gang Wu",
    "id": "2349556246",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Phil Chen",
    "id": "2307558539",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "G. Weisz",
    "id": "39752522",
    "h_index": 10,
    "papers": 21
   },
   {
    "name": "C\u00e9line Smith",
    "id": "2373570986",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Mojtaba Seyedhosseini",
    "id": "2237806111",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Bo Fang",
    "id": "2327132910",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Xiyang Luo",
    "id": "2338563754",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Roey Yogev",
    "id": "2324801367",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Zeynep Cankara",
    "id": "2275184671",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Andrew Straiton Hard",
    "id": "2063986787",
    "h_index": 8,
    "papers": 15
   },
   {
    "name": "Helen Ran",
    "id": "2372812004",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Rahul Sukthankar",
    "id": "2242609051",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "George Necula",
    "id": "2278591322",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Gael Liu",
    "id": "2352144736",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Honglong Cai",
    "id": "2275868595",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Praseem Banzal",
    "id": "2275187506",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Daniel Keysers",
    "id": "2064752644",
    "h_index": 8,
    "papers": 15
   },
   {
    "name": "S. Ghemawat",
    "id": "1780892",
    "h_index": 19,
    "papers": 43
   },
   {
    "name": "Connie Tao",
    "id": "2218112465",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Emma Dunleavy",
    "id": "2292141184",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Aditi Chaudhary",
    "id": "2285811461",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Wei Li",
    "id": "2378198189",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Maciej Miku\u0142a",
    "id": "2291068287",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Chen-Yu Lee",
    "id": "2278969944",
    "h_index": 18,
    "papers": 28
   },
   {
    "name": "Tiziana Refice",
    "id": "1951539",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Krishna Somandepalli",
    "id": "2301091006",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Alexandre Fr\u00e9chette",
    "id": "2156930381",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "D. Bahir",
    "id": "2129052169",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "John E. Karro",
    "id": "2056044",
    "h_index": 17,
    "papers": 44
   },
   {
    "name": "K. Rush",
    "id": "2304457219",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Sarah Perrin",
    "id": "2312326042",
    "h_index": 8,
    "papers": 22
   },
   {
    "name": "B. Rosgen",
    "id": "2080520726",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Xiaomeng Yang",
    "id": "2344183862",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "C. Hu",
    "id": "2275282496",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Mah-moud Alnahlawi",
    "id": "2275186342",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "J. Mao-Jones",
    "id": "1423275766",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Roopal Garg",
    "id": "2271225517",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Hoang-Phi Nguyen",
    "id": "2249871123",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Bat-Orgil Batsaikhan",
    "id": "2322326223",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "I. Iturrate",
    "id": "3268647",
    "h_index": 27,
    "papers": 64
   },
   {
    "name": "Anselm Levskaya",
    "id": "6639036",
    "h_index": 15,
    "papers": 34
   },
   {
    "name": "Avi Singh",
    "id": "2258779676",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Ashyana Kachra",
    "id": "2373042595",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Tony Lu",
    "id": "14041374",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Denis Petek",
    "id": "32328579",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Zhen Xu",
    "id": "2367769998",
    "h_index": 6,
    "papers": 18
   },
   {
    "name": "Mark Graham",
    "id": "144643908",
    "h_index": 50,
    "papers": 158
   },
   {
    "name": "Luk\u00e1s Zilka",
    "id": "1780245",
    "h_index": 11,
    "papers": 27
   },
   {
    "name": "Yael Karov",
    "id": "2103866144",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Marija Kostelac",
    "id": "2373037050",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Fangyu Liu",
    "id": "2307000286",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Yaohui Guo",
    "id": "2378650072",
    "h_index": 3,
    "papers": 13
   },
   {
    "name": "Weiyue Wang",
    "id": "2369269577",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Bernd Bohnet",
    "id": "2266464503",
    "h_index": 8,
    "papers": 24
   },
   {
    "name": "Emily Pitler",
    "id": "2585932",
    "h_index": 23,
    "papers": 58
   },
   {
    "name": "Tony Bruguier",
    "id": "2100307572",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Keisuke Kinoshita",
    "id": "2336957321",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Chrysovalantis Anastasiou",
    "id": "2029521",
    "h_index": 7,
    "papers": 21
   },
   {
    "name": "Nilpa Jha",
    "id": "147676500",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Ting Liu",
    "id": "2265693798",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Jerome T. Connor",
    "id": "2007915382",
    "h_index": 8,
    "papers": 14
   },
   {
    "name": "Phillip Wallis",
    "id": "2365931454",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Philip Pham",
    "id": "1899637431",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "E. Bailey",
    "id": "145270817",
    "h_index": 3,
    "papers": 14
   },
   {
    "name": "Shixin Li",
    "id": "2117889450",
    "h_index": 9,
    "papers": 51
   },
   {
    "name": "Heng-tze Cheng",
    "id": "2257129838",
    "h_index": 10,
    "papers": 17
   },
   {
    "name": "Sally Ma",
    "id": "2349767490",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Haiqiong Li",
    "id": "17950489",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Akanksha Maurya",
    "id": "2362584655",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Kate Olszewska",
    "id": "2275180557",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "M. Warmuth",
    "id": "2286778409",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Christy Koh",
    "id": "2275179007",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Dominik Paulus",
    "id": "2324801134",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Siddhartha R. Jonnalagadda",
    "id": "32421806",
    "h_index": 27,
    "papers": 82
   },
   {
    "name": "Enrique Piqueras",
    "id": "2275183119",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Ali Elqursh",
    "id": "2544590",
    "h_index": 7,
    "papers": 14
   },
   {
    "name": "Geoff Brown",
    "id": "2259937157",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Hadar Shemtov",
    "id": "3039533",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Lorenzo Maggiore",
    "id": "108173905",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Fei Xia",
    "id": "2290487337",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Ryan Foley",
    "id": "2275187696",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Beka Westberg",
    "id": "2373042542",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "George van den Driessche",
    "id": "47568983",
    "h_index": 14,
    "papers": 29
   },
   {
    "name": "Livio Baldini Soares",
    "id": "2258550407",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "A. Kar",
    "id": "2372652908",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Michael Quinn",
    "id": "2290485761",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Siqi Zuo",
    "id": "2307456167",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Jialin Wu",
    "id": "2315630152",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Kyle Kastner",
    "id": "2289037014",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Anna Bortsova",
    "id": "2275181572",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Aijun Bai",
    "id": "2324782053",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Ales Mikhalap",
    "id": "2373036816",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Lu-owei Zhou",
    "id": "2275297082",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Jenny Brennan",
    "id": "2275186701",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "V. Ramasesh",
    "id": "96641652",
    "h_index": 19,
    "papers": 55
   },
   {
    "name": "Honglei Zhuang",
    "id": "39371343",
    "h_index": 26,
    "papers": 77
   },
   {
    "name": "J. Maggs",
    "id": "2257812070",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "J. Schalkwyk",
    "id": "2271748248",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Yuntao Xu",
    "id": "2281339106",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Hui Huang",
    "id": "2403067431",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Andrew Howard",
    "id": "2239236365",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Sasha Brown",
    "id": "2292270307",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "L. Xue",
    "id": "2326922813",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Gloria Shen",
    "id": "2171017357",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "B. Albert",
    "id": "2373036502",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "N.K. Jha",
    "id": "2284257855",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Daniel Zheng",
    "id": "2337362505",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Varvara Krayvanova",
    "id": "3337905",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Spurthi Amba Hombaiah",
    "id": "2078501964",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Olivier Lacombe",
    "id": "2373039186",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Gautam Vasudevan",
    "id": "1986538793",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Dan Graur",
    "id": "2285439952",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Tian Xie",
    "id": "2262217036",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Meet Gandhi",
    "id": "2237567221",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Bangju Wang",
    "id": "47780402",
    "h_index": 9,
    "papers": 15
   },
   {
    "name": "Dustin Zelle",
    "id": "1389613483",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "Harman Singh",
    "id": "2298362472",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Dahun Kim",
    "id": "2237957528",
    "h_index": 9,
    "papers": 19
   },
   {
    "name": "S'ebastien Cevey",
    "id": "2275180682",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Victor Ungureanu",
    "id": "2282960170",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Natasha Noy",
    "id": "2351907620",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Fei Liu",
    "id": "2324851408",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Annie Xie",
    "id": "2345006140",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Fangxi-aoyu Feng",
    "id": "2275173841",
    "h_index": 5,
    "papers": 18
   },
   {
    "name": "Katerina Tsihlas",
    "id": "2275185589",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Daniel Formoso",
    "id": "2373036849",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Neera Vats",
    "id": "2290485442",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Quentin Wellens",
    "id": "2103788835",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Yinan Wang",
    "id": "2386081027",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Niket Kumar Bhumihar",
    "id": "2373042549",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Samrat Ghosh",
    "id": "2143032776",
    "h_index": 5,
    "papers": 31
   },
   {
    "name": "Matt Hoffman",
    "id": "2312323782",
    "h_index": 6,
    "papers": 14
   },
   {
    "name": "Tom Lieber",
    "id": "2373039470",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Oran Lang",
    "id": "2275176236",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "K. Bhatia",
    "id": "2285776242",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "T. Paine",
    "id": "40470211",
    "h_index": 25,
    "papers": 43
   },
   {
    "name": "Aroonalok Pyne",
    "id": "3165987",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Ronny Votel",
    "id": "69423660",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Madeleine Elish",
    "id": "2327199842",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Beno\u00eet Schillings",
    "id": "66272155",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "A. Panagopoulos",
    "id": "121984201",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Haichuan Yang",
    "id": "2118697626",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "Adam Raveret",
    "id": "2373037956",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Zohar Yahav",
    "id": "2373036898",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Shuang Liu",
    "id": "2400686608",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "D. Badawy",
    "id": "2489672",
    "h_index": 8,
    "papers": 15
   },
   {
    "name": "Nishant Agrawal",
    "id": "2201464144",
    "h_index": 11,
    "papers": 49
   },
   {
    "name": "M. Badawi",
    "id": "2300321956",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Mahdi Mirzazadeh",
    "id": "2062997707",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "C. Bromberg",
    "id": "2263779641",
    "h_index": 18,
    "papers": 90
   },
   {
    "name": "Fan Ye",
    "id": "2053897305",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Chang Liu",
    "id": "2340379413",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Tatiana Sholokhova",
    "id": "2373036970",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "G. Muraru",
    "id": "2091489438",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Gargi Balasubramaniam",
    "id": "1486413740",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "J. Malmaud",
    "id": "3274291",
    "h_index": 13,
    "papers": 19
   },
   {
    "name": "Alen Carin",
    "id": "2373042629",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Danilo Martins",
    "id": "2290487616",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Irina Jurenka",
    "id": "2311700390",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Pankil Botadra",
    "id": "2373038554",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Dave Lacey",
    "id": "2290485332",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Richa Singh",
    "id": "2265062251",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Mariano Schain",
    "id": "39190484",
    "h_index": 13,
    "papers": 29
   },
   {
    "name": "Daniel Zheng",
    "id": "2337362505",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Isabelle Guyon",
    "id": "2273585010",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "V. Lavrenko",
    "id": "2270531936",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Seungjin Lee",
    "id": "2297330525",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Xiang Zhou",
    "id": "2326647420",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "D. Hassabis",
    "id": "48987704",
    "h_index": 92,
    "papers": 160
   },
   {
    "name": "Jeshwanth Challagundla",
    "id": "101092080",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "D. Cheng",
    "id": "2352214485",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "N. Mehta",
    "id": "2356475030",
    "h_index": 3,
    "papers": 22
   },
   {
    "name": "M. Mauger",
    "id": "2251517316",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "M. Paganini",
    "id": "2264591527",
    "h_index": 10,
    "papers": 28
   },
   {
    "name": "Pushkar Mishra",
    "id": "2349237731",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Katherine Lee",
    "id": "2374238527",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Zhang Li",
    "id": "2374262906",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Lexi Baugher",
    "id": "2217251550",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Ondrej Skopek",
    "id": "52018133",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Max Chang",
    "id": "2211743212",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Amir Zait",
    "id": "2322447364",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Gaurav Menghani",
    "id": "2171591",
    "h_index": 7,
    "papers": 16
   },
   {
    "name": "Lizzetth Bellot",
    "id": "2069098611",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Guangxing Han",
    "id": "2327047819",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "J. Sarr",
    "id": "1947638484",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "S. Chikkerur",
    "id": "7489841",
    "h_index": 20,
    "papers": 45
   },
   {
    "name": "H. Sahni",
    "id": "34594615",
    "h_index": 10,
    "papers": 23
   },
   {
    "name": "Rohan Anil",
    "id": "1508890387",
    "h_index": 21,
    "papers": 57
   },
   {
    "name": "Arun Narayanan",
    "id": "2359509380",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Chandu Thekkath",
    "id": "89250086",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Daniele Pighin",
    "id": "2285298435",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Hana Strejvcek",
    "id": "2373036994",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "M. Velic",
    "id": "1753131",
    "h_index": 8,
    "papers": 15
   },
   {
    "name": "Fred Bertsch",
    "id": "21267179",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Manuel Tragut",
    "id": "2149496620",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Keran Rong",
    "id": "1996199677",
    "h_index": 10,
    "papers": 20
   },
   {
    "name": "Alicia Parrish",
    "id": "2292197465",
    "h_index": 5,
    "papers": 17
   },
   {
    "name": "Kai Bailey",
    "id": "2331856541",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Jiho Park",
    "id": "2339972538",
    "h_index": 3,
    "papers": 15
   },
   {
    "name": "Isabela Albuquerque",
    "id": "2298274437",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Abhishek Bapna",
    "id": "2277740200",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Rajesh Venkataraman",
    "id": "2277038451",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Alec Kosik",
    "id": "90781259",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Johannes Griesser",
    "id": "2373038452",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Zhiwei Deng",
    "id": "2258679614",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Alek Andreev",
    "id": "2290741315",
    "h_index": 8,
    "papers": 20
   },
   {
    "name": "Q. Dou",
    "id": "2372523401",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Kevin Hui",
    "id": "2290487874",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Fanny Wei",
    "id": "2374192377",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Xiaobing Yu",
    "id": "2321487156",
    "h_index": 4,
    "papers": 22
   },
   {
    "name": "Lei Shu",
    "id": "2257004117",
    "h_index": 10,
    "papers": 16
   },
   {
    "name": "Avia Aharon",
    "id": "2270662216",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "David Barker",
    "id": "2290481818",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Badih Ghazi",
    "id": "2529354",
    "h_index": 23,
    "papers": 108
   },
   {
    "name": "Sebastian Flennerhag",
    "id": "46212062",
    "h_index": 13,
    "papers": 21
   },
   {
    "name": "Chris Breaux",
    "id": "118587653",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Yuchuan Liu",
    "id": "2269509219",
    "h_index": 6,
    "papers": 20
   },
   {
    "name": "Matthew Bilotti",
    "id": "2371114965",
    "h_index": 1,
    "papers": 9
   },
   {
    "name": "J. Woodward",
    "id": "2332083213",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Uri Alon",
    "id": "2268672727",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Stephanie Winkler",
    "id": "2218062983",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Tzu-Kuo Huang",
    "id": "2215571407",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Kostas Andriopoulos",
    "id": "2373038307",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Jo\u00e3o Gabriel Oliveira",
    "id": "145539000",
    "h_index": 9,
    "papers": 13
   },
   {
    "name": "Penporn Koanantakool",
    "id": "3344182",
    "h_index": 8,
    "papers": 13
   },
   {
    "name": "Berkin Akin",
    "id": "17853782",
    "h_index": 18,
    "papers": 30
   },
   {
    "name": "M. Wunder",
    "id": "1708399",
    "h_index": 28,
    "papers": 104
   },
   {
    "name": "Cicero Nogueira dos Santos",
    "id": "2267546965",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Mohammad Hossein Bateni",
    "id": "153924932",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Ling Yang",
    "id": "2363283279",
    "h_index": 9,
    "papers": 18
   },
   {
    "name": "Dan Horgan",
    "id": "48257711",
    "h_index": 11,
    "papers": 18
   },
   {
    "name": "Beer Changpinyo",
    "id": "2158369306",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Keyvan Amiri",
    "id": "2275190104",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Min Ma",
    "id": "2352024723",
    "h_index": 7,
    "papers": 16
   },
   {
    "name": "Dayeong Lee",
    "id": "2309772989",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Lihao Liang",
    "id": "2351001357",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Anirudh Baddepudi",
    "id": "2275186584",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Tejasi Latkar",
    "id": "2275186806",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "R. Hadsell",
    "id": "2315504",
    "h_index": 51,
    "papers": 111
   },
   {
    "name": "Jun Xu",
    "id": "2275619274",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Hairong Mu",
    "id": "2372557318",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Shiyi Han",
    "id": "2176923703",
    "h_index": 7,
    "papers": 12
   },
   {
    "name": "Aedan Pope",
    "id": "20702300",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "Snchit Grover",
    "id": "2372879210",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Frank Kim",
    "id": "2372411840",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Ankit V. Bhagatwala",
    "id": "1976725",
    "h_index": 15,
    "papers": 37
   },
   {
    "name": "Guan Sun",
    "id": "2374206175",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Yamini Bansal",
    "id": "2275111016",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "A. Globerson",
    "id": "2264110817",
    "h_index": 7,
    "papers": 17
   },
   {
    "name": "Ali Nazari",
    "id": "2341715213",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Samira Daruki",
    "id": "2255862",
    "h_index": 6,
    "papers": 17
   },
   {
    "name": "Hagen Soltau",
    "id": "2324368866",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Jane Labanowski",
    "id": "2275184618",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Laurent El Shafey",
    "id": "2121764",
    "h_index": 17,
    "papers": 39
   },
   {
    "name": "M. Harvey",
    "id": "2290487101",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Yanif Ahmad",
    "id": "1841612",
    "h_index": 19,
    "papers": 113
   },
   {
    "name": "Elan Rosenfeld",
    "id": "49686853",
    "h_index": 13,
    "papers": 22
   },
   {
    "name": "W. Kong",
    "id": "50340269",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Etienne Pot",
    "id": "38627717",
    "h_index": 9,
    "papers": 18
   },
   {
    "name": "Yi-Xuan Tan",
    "id": "2325707139",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "A. Wei",
    "id": "2242967654",
    "h_index": 2,
    "papers": 27
   },
   {
    "name": "Victoria Langston",
    "id": "2066201331",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "M. Prasetya",
    "id": "2372917550",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Petar Velivckovi'c",
    "id": "1742197495",
    "h_index": 23,
    "papers": 43
   },
   {
    "name": "R. Killam",
    "id": "108238204",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Robin Strudel",
    "id": "86898863",
    "h_index": 11,
    "papers": 19
   },
   {
    "name": "Darren Ni",
    "id": "2372463822",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Zhen-Xing Zhu",
    "id": "2357490076",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Aaron Archer",
    "id": "2275054971",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Kavya Kopparapu",
    "id": "1751654639",
    "h_index": 8,
    "papers": 14
   },
   {
    "name": "Lynn Nguyen",
    "id": "2374129910",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Emilio Parisotto",
    "id": "3166516",
    "h_index": 26,
    "papers": 44
   },
   {
    "name": "Hussain Masoom",
    "id": "2274104517",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Sravanti Addepalli",
    "id": "1398341975",
    "h_index": 12,
    "papers": 26
   },
   {
    "name": "Jordan Grimstad",
    "id": "2275175432",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Hexiang Hu",
    "id": "2324220123",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Joss Moore",
    "id": "2284725355",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Avinatan Hassidim",
    "id": "1809983",
    "h_index": 46,
    "papers": 211
   },
   {
    "name": "Le Hou",
    "id": "2274787555",
    "h_index": 3,
    "papers": 19
   },
   {
    "name": "M. Raghavachari",
    "id": "3080747",
    "h_index": 15,
    "papers": 35
   },
   {
    "name": "Jared Lichtarge",
    "id": "51888730",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Adam R. Brown",
    "id": "2254150367",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Hilal Dib",
    "id": "2373037091",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "N. Ponomareva",
    "id": "2282969107",
    "h_index": 8,
    "papers": 21
   },
   {
    "name": "Justin Fu",
    "id": "2550764",
    "h_index": 22,
    "papers": 31
   },
   {
    "name": "Yujing Zhang",
    "id": "2275534739",
    "h_index": 6,
    "papers": 18
   },
   {
    "name": "Altafur Rahman",
    "id": "2115323576",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Joana Iljazi",
    "id": "2543878",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Edouard Leurent",
    "id": "2275189310",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Gabriel Dulac-Arnold",
    "id": "1387885286",
    "h_index": 26,
    "papers": 43
   },
   {
    "name": "Cosmo Du",
    "id": "2042634588",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "Chulayuth Asawaroengchai",
    "id": "50844587",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Larry Jin",
    "id": "50496734",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Ela Gruzewska",
    "id": "2373036985",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Z. Ji",
    "id": "2253445320",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Benigno Uria",
    "id": "2825051",
    "h_index": 14,
    "papers": 21
   },
   {
    "name": "Daniel De Freitas",
    "id": "1490889580",
    "h_index": 5,
    "papers": 17
   },
   {
    "name": "P. Barham",
    "id": "152399055",
    "h_index": 12,
    "papers": 32
   },
   {
    "name": "Lauren Beltrone",
    "id": "2373042906",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Victor F. Campos",
    "id": "2275193990",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Jun Yan",
    "id": "49781448",
    "h_index": 21,
    "papers": 32
   },
   {
    "name": "Neel Kovelamudi",
    "id": "2303849635",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Arthur Nguyen",
    "id": "48721230",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Elinor Davies",
    "id": "118137348",
    "h_index": 3,
    "papers": 26
   },
   {
    "name": "Zhi Wu",
    "id": "2411135698",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Z. Egyed",
    "id": "41183279",
    "h_index": 22,
    "papers": 103
   },
   {
    "name": "Kristina Toutanova",
    "id": "2288931206",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Nithya Attaluri",
    "id": "80930649",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Hongliang Fei",
    "id": "2280063566",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Peter Stys",
    "id": "2250968390",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Siddhartha Brahma",
    "id": "1791585",
    "h_index": 15,
    "papers": 67
   },
   {
    "name": "M. Izzard",
    "id": "40315108",
    "h_index": 13,
    "papers": 31
   },
   {
    "name": "S. Velusamy",
    "id": "2387708955",
    "h_index": 3,
    "papers": 18
   },
   {
    "name": "Scott M. Lundberg",
    "id": "2371074977",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Vincent Zhuang",
    "id": "13165193",
    "h_index": 9,
    "papers": 19
   },
   {
    "name": "Kevin Sequeira",
    "id": "2373037368",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Adam Santoro",
    "id": "2253463637",
    "h_index": 7,
    "papers": 15
   },
   {
    "name": "Ehsan Amid",
    "id": "2142862",
    "h_index": 16,
    "papers": 47
   },
   {
    "name": "Ophir Aharoni",
    "id": "2373038591",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Shuai Ye",
    "id": "2298199025",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Mukund Sundararajan",
    "id": "2301077834",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Lijun Yu",
    "id": "2373562397",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Yu-Cheng Ling",
    "id": "2372548100",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Stephen Spencer",
    "id": "2135383313",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Hugo Song",
    "id": "2372451184",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "J. Djolonga",
    "id": "2941141",
    "h_index": 24,
    "papers": 41
   },
   {
    "name": "Christo Kirov",
    "id": "2022649",
    "h_index": 20,
    "papers": 44
   },
   {
    "name": "Sonal Gupta",
    "id": "2285627848",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Alessandro Bissacco",
    "id": "2285540285",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Clemens Meyer",
    "id": "1406288863",
    "h_index": 12,
    "papers": 18
   },
   {
    "name": "Mukul Bhutani",
    "id": "26320815",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Andrew M. Dai",
    "id": "2273563615",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Weiyi Wang",
    "id": "2372309435",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Siqi Liu",
    "id": "2288039328",
    "h_index": 4,
    "papers": 14
   },
   {
    "name": "Ashwin Sreevatsa",
    "id": "2106628426",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Qijun Tan",
    "id": "2261496951",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Maria Wang",
    "id": "2374466860",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Lucy Kim",
    "id": "2290490486",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Yicheng Wang",
    "id": "2290627844",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "A. Irpan",
    "id": "17818078",
    "h_index": 22,
    "papers": 32
   },
   {
    "name": "Yang Xiao",
    "id": "2307314606",
    "h_index": 9,
    "papers": 19
   },
   {
    "name": "Stanislav Fort",
    "id": "2299002813",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Yifan He",
    "id": "2324897112",
    "h_index": 6,
    "papers": 22
   },
   {
    "name": "A. Gurney",
    "id": "66004200",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Bryan Gale",
    "id": "2290265302",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Yue Ma",
    "id": "2347173134",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Monica Roy",
    "id": "46318978",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Viorica Patraucean",
    "id": "1756112",
    "h_index": 15,
    "papers": 44
   },
   {
    "name": "Taylan Bilal",
    "id": "153289063",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Golnaz Ghiasi",
    "id": "1898210",
    "h_index": 26,
    "papers": 37
   },
   {
    "name": "Anahita Hosseini",
    "id": "2314696131",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Melvin Johnson",
    "id": "2275525680",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Zhuowan Li",
    "id": "2312752299",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Y. Tay",
    "id": "2286237772",
    "h_index": 3,
    "papers": 18
   },
   {
    "name": "Benjamin Beyret",
    "id": "102928633",
    "h_index": 7,
    "papers": 9
   },
   {
    "name": "Katie Millican",
    "id": "2143434227",
    "h_index": 11,
    "papers": 29
   },
   {
    "name": "Josef Broder",
    "id": "2290485721",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Mayank Lunayach",
    "id": "1382193868",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Danny Swisher",
    "id": "2373042878",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Eugen Vuvsak",
    "id": "2373036895",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "David Parkinson",
    "id": "2059592938",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Mh Tessler",
    "id": "2264208229",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Adi Mayrav Gilady",
    "id": "2348098277",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "R. Song",
    "id": "2067622212",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Allan Dafoe",
    "id": "2265490911",
    "h_index": 20,
    "papers": 38
   },
   {
    "name": "Yves Raimond",
    "id": "2288705691",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Masataka Yamaguchi",
    "id": "70204030",
    "h_index": 4,
    "papers": 24
   },
   {
    "name": "Itay Karo",
    "id": "2324801958",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Elizabeth Nielsen",
    "id": "2345929345",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Kevin Kilgour",
    "id": "3336784",
    "h_index": 17,
    "papers": 49
   },
   {
    "name": "Mike Dusenberry",
    "id": "2253347588",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Rajiv Mathews",
    "id": "144068963",
    "h_index": 17,
    "papers": 41
   },
   {
    "name": "Jiho Choi",
    "id": "2336394540",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Siyuan Qiao",
    "id": "2275178766",
    "h_index": 8,
    "papers": 19
   },
   {
    "name": "Harsh Mehta",
    "id": "2337578397",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Sahitya Potluri",
    "id": "2261493257",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Chris Knutsen",
    "id": "2353383530",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Jialu Liu",
    "id": "2239559694",
    "h_index": 7,
    "papers": 14
   },
   {
    "name": "T. Tan",
    "id": "2242104010",
    "h_index": 6,
    "papers": 34
   },
   {
    "name": "K. Sengupta",
    "id": "144037321",
    "h_index": 18,
    "papers": 69
   },
   {
    "name": "K. Gopalakrishnan",
    "id": "2161342233",
    "h_index": 17,
    "papers": 25
   },
   {
    "name": "Abodunrinwa Toki",
    "id": "146206843",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Mencher Chiang",
    "id": "2372312682",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Mike Burrows",
    "id": "2062790552",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Grace Vesom",
    "id": "2419315",
    "h_index": 8,
    "papers": 13
   },
   {
    "name": "Zafarali Ahmed",
    "id": "2275130349",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Ilia Labzovsky",
    "id": "102066123",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Siddharth Vashishtha",
    "id": "68972934",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "Preeti Singh",
    "id": "2286994643",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Ankur Sharma",
    "id": "2109669524",
    "h_index": 6,
    "papers": 17
   },
   {
    "name": "Ada Ma",
    "id": "2275786213",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Jinyu Xie",
    "id": "2375047508",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Pranav Talluri",
    "id": "2373034955",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Hannah Forbes-Pollard",
    "id": "2373042691",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Aarush Selvan",
    "id": "2316995650",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Joel Wee",
    "id": "2280145442",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "L. Matthey",
    "id": "2367480",
    "h_index": 23,
    "papers": 37
   },
   {
    "name": "Tom Funkhouser",
    "id": "2348779239",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Parthasarathy Gopavarapu",
    "id": "3412286",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Lev Proleev",
    "id": "2161966573",
    "h_index": 8,
    "papers": 16
   },
   {
    "name": "Cheng Li",
    "id": "2275767067",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Matt Thomas",
    "id": "2325813886",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "K. Kolipaka",
    "id": "2839851",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Zhipeng Jia",
    "id": "2366583171",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Ashwin Kakarla",
    "id": "1399592827",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Srinivas Sunkara",
    "id": "31801337",
    "h_index": 10,
    "papers": 31
   },
   {
    "name": "J. Puigcerver",
    "id": "1794202",
    "h_index": 26,
    "papers": 46
   },
   {
    "name": "S. Sheth",
    "id": "2237560999",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "E. Graves",
    "id": "49217404",
    "h_index": 5,
    "papers": 16
   },
   {
    "name": "Chen Wang",
    "id": "2364017135",
    "h_index": 6,
    "papers": 25
   },
   {
    "name": "Sadhia Khan",
    "id": "2182009616",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Kai Kang",
    "id": "2275567101",
    "h_index": 7,
    "papers": 14
   },
   {
    "name": "S. Buch",
    "id": "8983218",
    "h_index": 19,
    "papers": 27
   },
   {
    "name": "Fred Zhang",
    "id": "2109244660",
    "h_index": 13,
    "papers": 20
   },
   {
    "name": "Omkar Savant",
    "id": "2373038454",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "David Soergel",
    "id": "46773550",
    "h_index": 6,
    "papers": 17
   },
   {
    "name": "Kevin Lee",
    "id": "2397384216",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Linda Friso",
    "id": "2324801106",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Xuanyi Dong",
    "id": "2316301140",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Rahul Arya",
    "id": "2274235016",
    "h_index": 3,
    "papers": 16
   },
   {
    "name": "Shreyas Chandrakaladharan",
    "id": "150293130",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Connor Schenck",
    "id": "2310607473",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Greg Billock",
    "id": "1869575",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Tejas Iyer",
    "id": "2268310853",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "A. Bakalov",
    "id": "3058597",
    "h_index": 7,
    "papers": 15
   },
   {
    "name": "Leslie W. Baker",
    "id": "2150777298",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Alex D. Ruiz",
    "id": "2373641419",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Angad Chandorkar",
    "id": "2279923995",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Trieu H. Trinh",
    "id": "40895509",
    "h_index": 12,
    "papers": 20
   },
   {
    "name": "Matt Miecnikowski",
    "id": "2290487829",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Yanqi Zhou",
    "id": "2389316",
    "h_index": 30,
    "papers": 73
   },
   {
    "name": "Yangsibo Huang",
    "id": "2283138638",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Jiazhong Nie",
    "id": "38113961",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Ali Shah",
    "id": "2374302233",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "A. Thapliyal",
    "id": "2265495656",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Sam Haves",
    "id": "2292141400",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Lun Wang",
    "id": "2407661284",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Uri Shaham",
    "id": "50482645",
    "h_index": 12,
    "papers": 16
   },
   {
    "name": "Patrick Morris-Suzuki",
    "id": "2373035225",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Soroush Radpour",
    "id": "46801205",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Leonard Berrada",
    "id": "48092709",
    "h_index": 12,
    "papers": 19
   },
   {
    "name": "Thomas Strohmann",
    "id": "2931575",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Chaochao Yan",
    "id": "2273547767",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Jingwei Shen",
    "id": "2374955996",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Sonam Goenka",
    "id": "2096063076",
    "h_index": 7,
    "papers": 12
   },
   {
    "name": "Tris Warkentin",
    "id": "1986491804",
    "h_index": 13,
    "papers": 29
   },
   {
    "name": "Petar Devi'c",
    "id": "2373038261",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Daniel Belov",
    "id": "2360503108",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Albert Webson",
    "id": "2291172852",
    "h_index": 10,
    "papers": 22
   },
   {
    "name": "Madhavi Yenugula",
    "id": "2915507",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Puranjay Datta",
    "id": "2323842478",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Jerry Chang",
    "id": "143934592",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Nimesh Ghelani",
    "id": "3404697",
    "h_index": 8,
    "papers": 18
   },
   {
    "name": "Aviral Kumar",
    "id": "2275526115",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Vincent Perot",
    "id": "2307469987",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Jessica Lo",
    "id": "2351910249",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Yang Song",
    "id": "2376525175",
    "h_index": 8,
    "papers": 34
   },
   {
    "name": "Herman Schmit",
    "id": "2241622658",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Jianmin Chen",
    "id": "2400017849",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Vasilisa Bashlovkina",
    "id": "3289612",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Xiaoyue Pan",
    "id": "2281793927",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Diana Mincu",
    "id": "2007712128",
    "h_index": 13,
    "papers": 26
   },
   {
    "name": "Paul Roit",
    "id": "1400349617",
    "h_index": 8,
    "papers": 11
   },
   {
    "name": "Isabel Edkins",
    "id": "2373038651",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Andy Davis",
    "id": "2275128167",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Yujia Li",
    "id": "2275290025",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Ben Horn",
    "id": "2324801500",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Xinjian Li",
    "id": "2392129099",
    "h_index": 4,
    "papers": 15
   },
   {
    "name": "S. Pradeepkumar",
    "id": "2266948617",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Eric Doi",
    "id": "2408240425",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Wanzheng Zhu",
    "id": "2405547",
    "h_index": 8,
    "papers": 10
   },
   {
    "name": "S. Padmanabhan",
    "id": "2251814340",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Siddharth Verma",
    "id": "2376089629",
    "h_index": 1,
    "papers": 9
   },
   {
    "name": "Jasmine Liu",
    "id": "2275539011",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Heng Chen",
    "id": "2407138532",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Mihajlo Velimirovi'c",
    "id": "2220407384",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Malcolm Reynolds",
    "id": "2295675672",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Priyanka Agrawal",
    "id": "2266842599",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "N. Sukhanov",
    "id": "91662218",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Abhinit Modi",
    "id": "2372917827",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Siddharth Goyal",
    "id": "2286036332",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "John Palowitch",
    "id": "1798404",
    "h_index": 11,
    "papers": 26
   },
   {
    "name": "N. Khajehnouri",
    "id": "3084406",
    "h_index": 9,
    "papers": 21
   },
   {
    "name": "Wing W. Lowe",
    "id": "47335637",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "David Klinghoffer",
    "id": "13184027",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "S. Silver",
    "id": "39868660",
    "h_index": 4,
    "papers": 26
   },
   {
    "name": "Vinh Q. Tran",
    "id": "2317984275",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Candice Schumann",
    "id": "2266521235",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "F. Piccinno",
    "id": "2261956317",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Xi Liu",
    "id": "2249435718",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Mario Luvci'c",
    "id": "2170162986",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Xiaochen Yang",
    "id": "2283447442",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Sandeep Kumar",
    "id": "2290882352",
    "h_index": 3,
    "papers": 16
   },
   {
    "name": "A. Kannan",
    "id": "2275186247",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Raghavendra Kotikalapudi",
    "id": "2076881",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Mudit Bansal",
    "id": "2324798410",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Fabian B Fuchs",
    "id": "2300352685",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Mohammad Javad Hosseini",
    "id": "2286294605",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "A. Abdelhamed",
    "id": "2313559209",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Dawn Bloxwich",
    "id": "2275185808",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Tianhe Yu",
    "id": "2290486855",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Ruoxin Sang",
    "id": "2275189194",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Gregory Thornton",
    "id": "2005813",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Karan Gill",
    "id": "2349484203",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Yuchi Liu",
    "id": "2282764952",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Virat Shejwalkar",
    "id": "148318826",
    "h_index": 14,
    "papers": 32
   },
   {
    "name": "Jason Lin",
    "id": "2283168763",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Zhipeng Yan",
    "id": "2363839442",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Kehang Han",
    "id": "2273880591",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Thomas Buschmann",
    "id": "2352107861",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "M. Pliskin",
    "id": "46354713",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Zhiqiang Xing",
    "id": "2233582288",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Susheel Tatineni",
    "id": "2373042631",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Junlin Zhang",
    "id": "50562008",
    "h_index": 19,
    "papers": 57
   },
   {
    "name": "Sissie Hsiao",
    "id": "2372558052",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Gavin Buttimore",
    "id": "1390054453",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Marcus Wu",
    "id": "2275284716",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Zefei Li",
    "id": "2308646578",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Geza Kovacs",
    "id": "2329372194",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Legg Yeung",
    "id": "2164123008",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Tao Huang",
    "id": "2330359656",
    "h_index": 4,
    "papers": 14
   },
   {
    "name": "Aaron Cohen",
    "id": "2374363382",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Bethanie Brownfield",
    "id": "2156928199",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Averi Nowak",
    "id": "2306632822",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Mikel Rodriguez",
    "id": "2275529269",
    "h_index": 8,
    "papers": 10
   },
   {
    "name": "Tianze Shi",
    "id": "2372754776",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "H. V. Hasselt",
    "id": "7634925",
    "h_index": 40,
    "papers": 67
   },
   {
    "name": "K. Cen",
    "id": "2373018239",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Deepanway Ghosal",
    "id": "32528506",
    "h_index": 25,
    "papers": 57
   },
   {
    "name": "Kushal Majmundar",
    "id": "1724897679",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Weiren Yu",
    "id": "2300029894",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "W. Chen",
    "id": "2275535281",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Danila Sinopalnikov",
    "id": "2359017585",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Hao Zhang",
    "id": "2281494006",
    "h_index": 9,
    "papers": 27
   },
   {
    "name": "Vlatko Gali\u0107",
    "id": "2337699760",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Di Lu",
    "id": "2387129493",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Zeyu Zheng",
    "id": "1500655637",
    "h_index": 20,
    "papers": 188
   },
   {
    "name": "M. Song",
    "id": "2336586866",
    "h_index": 7,
    "papers": 19
   },
   {
    "name": "Gary Wang",
    "id": "2182522748",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Gui Citovsky",
    "id": "2649516",
    "h_index": 8,
    "papers": 21
   },
   {
    "name": "Swapnil Gawde",
    "id": "2373035204",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Isaac R. Galatzer-Levy",
    "id": "2287936651",
    "h_index": 7,
    "papers": 23
   },
   {
    "name": "David Silver",
    "id": "2275185813",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Ivana Balazevic",
    "id": "3451828",
    "h_index": 9,
    "papers": 15
   },
   {
    "name": "D. Das",
    "id": "2275184585",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Kingshuk Majumder",
    "id": "2301421594",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Yale Cong",
    "id": "2372800299",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Praneet Dutta",
    "id": "9076891",
    "h_index": 7,
    "papers": 14
   },
   {
    "name": "Dustin Tran",
    "id": "2273790995",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Hui Wan",
    "id": "2376189576",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Junwei Yuan",
    "id": "2338787772",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "D. Eppens",
    "id": "102776053",
    "h_index": 14,
    "papers": 20
   },
   {
    "name": "Alanna Walton",
    "id": "2275185047",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Been Kim",
    "id": "2372371635",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Harry Ragan",
    "id": "2373038567",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "James Cobon-Kerr",
    "id": "2275185511",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Lu Liu",
    "id": "2275576625",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Weijun Wang",
    "id": "2235256501",
    "h_index": 8,
    "papers": 17
   },
   {
    "name": "Bryce Petrini",
    "id": "2181215497",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Jack W. Rae",
    "id": "2275178294",
    "h_index": 7,
    "papers": 26
   },
   {
    "name": "Rakesh Shivanna",
    "id": "2934334",
    "h_index": 9,
    "papers": 20
   },
   {
    "name": "Yan Xiong",
    "id": "2410665153",
    "h_index": 4,
    "papers": 14
   },
   {
    "name": "Chace Lee",
    "id": "2159593330",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Pauline Coquinot",
    "id": "2373037073",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Yiming Gu",
    "id": "2275725490",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "L. Patel",
    "id": "66625829",
    "h_index": 5,
    "papers": 19
   },
   {
    "name": "Blake A. Hechtman",
    "id": "3135881",
    "h_index": 16,
    "papers": 29
   },
   {
    "name": "A. Boag",
    "id": "2252554409",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Orion Jankowski",
    "id": "2373042575",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Alex Wertheim",
    "id": "2373035396",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Alex X. Lee",
    "id": "2315463075",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Paul Covington",
    "id": "2328307676",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Hila Noga",
    "id": "34094590",
    "h_index": 7,
    "papers": 12
   },
   {
    "name": "Sam Sobell",
    "id": "2373037197",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "S. Vasanth",
    "id": "2318393584",
    "h_index": 4,
    "papers": 22
   },
   {
    "name": "William Bono",
    "id": "2373038427",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Chirag Nagpal",
    "id": "2963503",
    "h_index": 16,
    "papers": 47
   },
   {
    "name": "W. Fan",
    "id": "2275288688",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Xavier Garc\u00eda",
    "id": "2275181515",
    "h_index": 6,
    "papers": 24
   },
   {
    "name": "K. Soparkar",
    "id": "2275185640",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Aybuke Turker",
    "id": "16660786",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Nathan Howard",
    "id": "2319368470",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Sachit Menon",
    "id": "46245898",
    "h_index": 10,
    "papers": 17
   },
   {
    "name": "Yuankai Chen",
    "id": "2363818874",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Vikas Verma",
    "id": "2344274213",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "V. Pchelin",
    "id": "2169522430",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Harish Rajamani",
    "id": "34789806",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Valentin Dalibard",
    "id": "2315322153",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Ana Ramalho",
    "id": "2372545171",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Yang Guo",
    "id": "2155599159",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Kartikeya Badola",
    "id": "2051018967",
    "h_index": 9,
    "papers": 19
   },
   {
    "name": "Seojin Bang",
    "id": "2382623450",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "N. Rauschmayr",
    "id": "120529492",
    "h_index": 66,
    "papers": 478
   },
   {
    "name": "Julia Proskurnia",
    "id": "3461305",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "S. Dasari",
    "id": "36076404",
    "h_index": 23,
    "papers": 39
   },
   {
    "name": "Xinyun Chen",
    "id": "2343591975",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Mikhail Sushkov",
    "id": "41032394",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "A. Hauth",
    "id": "119556335",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "P. Sho",
    "id": "2373042531",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Abhinavkumar Singh",
    "id": "9985822",
    "h_index": 5,
    "papers": 22
   },
   {
    "name": "Bilva Chandra",
    "id": "2372466083",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Allie Culp",
    "id": "2373035255",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "M. Dylla",
    "id": "30736359",
    "h_index": 15,
    "papers": 23
   },
   {
    "name": "Olivier Bachem",
    "id": "1936951",
    "h_index": 41,
    "papers": 85
   },
   {
    "name": "James Besley",
    "id": "2275186515",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "He Zhao",
    "id": "2263783762",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Timothy P. Lillicrap",
    "id": "2302799561",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Wei Wei",
    "id": "2314672859",
    "h_index": 6,
    "papers": 17
   },
   {
    "name": "Wael Al Jishi",
    "id": "2373039022",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Ning Niu",
    "id": "2275178929",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Alban Rrustemi",
    "id": "2275186093",
    "h_index": 8,
    "papers": 16
   },
   {
    "name": "Raphael Lopez Kaufman",
    "id": "31713635",
    "h_index": 8,
    "papers": 13
   },
   {
    "name": "R. Poplin",
    "id": "48663822",
    "h_index": 22,
    "papers": 36
   },
   {
    "name": "Jewel Zhao",
    "id": "2373586063",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Minh Truong",
    "id": "2139742659",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Shikhar Bharadwaj",
    "id": "2136381352",
    "h_index": 8,
    "papers": 24
   },
   {
    "name": "Ester Hlavnova",
    "id": "2221287281",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Eli Stickgold",
    "id": "3147309",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Cordelia Schmid",
    "id": "2289833766",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "Georgi Stephanov",
    "id": "2373038697",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Zhaoqi Leng",
    "id": "2127987192",
    "h_index": 9,
    "papers": 13
   },
   {
    "name": "Frederick Liu",
    "id": "2260276550",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "L'eonard Hussenot",
    "id": "2312322565",
    "h_index": 9,
    "papers": 24
   },
   {
    "name": "Shenil Dodhia",
    "id": "87654031",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Juliana Franco",
    "id": "2275175753",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Lesley Katzen",
    "id": "2373038693",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Abhanshu Sharma",
    "id": "2275537981",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Sarah Cogan",
    "id": "2275181529",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Zuguang Yang",
    "id": "2372901168",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Aniket Ray",
    "id": "144426361",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Sergi Caelles",
    "id": "1413064976",
    "h_index": 13,
    "papers": 27
   },
   {
    "name": "Shen Yan",
    "id": "2397568104",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Ravin Kumar",
    "id": "2290629265",
    "h_index": 9,
    "papers": 27
   },
   {
    "name": "Daniel Gillick",
    "id": "2311700467",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Renee Wong",
    "id": "2288184229",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "J. Ainslie",
    "id": "2343748926",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Jonathan Hoech",
    "id": "2300094924",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "S\u00e9bastien M. R. Arnold",
    "id": "2275186656",
    "h_index": 6,
    "papers": 17
   },
   {
    "name": "Daniel A. Abolafia",
    "id": "32137535",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Anca Dragan",
    "id": "2064066935",
    "h_index": 18,
    "papers": 51
   },
   {
    "name": "B. Hora",
    "id": "2373037172",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Grace Hu",
    "id": "2114119196",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Alexey Guseynov",
    "id": "2275182203",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Yang Lu",
    "id": "2396314491",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Chas Leichner",
    "id": "108381331",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Jinmeng Rao",
    "id": "2362089818",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Abhimanyu Goyal",
    "id": "2275187402",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Nagabhushan Baddi",
    "id": "2373037623",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Daniel Hernandez Diaz",
    "id": "2077592588",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Timothy McConnell",
    "id": "48279354",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Max Bain",
    "id": "2297187552",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Jake Abernethy",
    "id": "151040960",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Qiqi Yan",
    "id": "2292838236",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Rylan Schaeffer",
    "id": "1749176844",
    "h_index": 21,
    "papers": 69
   },
   {
    "name": "Paul Vicol",
    "id": "2039154",
    "h_index": 15,
    "papers": 39
   },
   {
    "name": "W. Thompson",
    "id": "2214613274",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Montse Gonzalez Arenas",
    "id": "153134021",
    "h_index": 13,
    "papers": 17
   },
   {
    "name": "M. Bellaiche",
    "id": "2359151678",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "P. Barrio",
    "id": "2241400603",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Stefan Zinke",
    "id": "2373039361",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Riccardo Patana",
    "id": "2373039509",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Pulkit Mehta",
    "id": "2324801602",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "J. Kearns",
    "id": "2373038291",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Avraham Ruderman",
    "id": "144893251",
    "h_index": 9,
    "papers": 14
   },
   {
    "name": "Scott Pollom",
    "id": "2278855723",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "David B. D'Ambrosio",
    "id": "2352107893",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "C. Hope",
    "id": "34651833",
    "h_index": 13,
    "papers": 247
   },
   {
    "name": "Yang Yu",
    "id": "2309729944",
    "h_index": 6,
    "papers": 22
   },
   {
    "name": "Andrea Gesmundo",
    "id": "2813347",
    "h_index": 16,
    "papers": 44
   },
   {
    "name": "Kuang-Huei Lee",
    "id": "2145145412",
    "h_index": 15,
    "papers": 19
   },
   {
    "name": "Aviv Rosenberg",
    "id": "2302798001",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Yiqian Zhou",
    "id": "2373548436",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Yaoyiran Li",
    "id": "2110977488",
    "h_index": 9,
    "papers": 17
   },
   {
    "name": "D. Garmon",
    "id": "2275181432",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Yonghui Wu",
    "id": "2275892922",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Safeen Huda",
    "id": "2308034897",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Gil Fidel",
    "id": "1388062456",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "M. Baeuml",
    "id": "2066290873",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Jian Li",
    "id": "2275993901",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Phoebe Kirk",
    "id": "2314107430",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Rhys May",
    "id": "2268760156",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Tao Tu",
    "id": "2367582553",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "S. M. Carthy",
    "id": "2255299263",
    "h_index": 7,
    "papers": 18
   },
   {
    "name": "Toshiyuki Fukuzawa",
    "id": "2373035389",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Miranda Aperghis",
    "id": "2373037487",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "C. Yeh",
    "id": "2273556813",
    "h_index": 6,
    "papers": 17
   },
   {
    "name": "T. Yoshino",
    "id": "2309047173",
    "h_index": 5,
    "papers": 65
   },
   {
    "name": "Bo Li",
    "id": "2364081955",
    "h_index": 9,
    "papers": 29
   },
   {
    "name": "Austin Myers",
    "id": "2294569724",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Kaisheng Yao",
    "id": "2275191099",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Ben Limonchik",
    "id": "2373038991",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Changwan Ryu",
    "id": "2372832814",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Rohun Saxena",
    "id": "51031729",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Alex Goldin",
    "id": "40034895",
    "h_index": 7,
    "papers": 15
   },
   {
    "name": "Ruizhe Zhao",
    "id": "2275832693",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Rocky Rhodes",
    "id": "147961415",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Tao Zhu",
    "id": "2275871777",
    "h_index": 5,
    "papers": 15
   },
   {
    "name": "Divya Tyam",
    "id": "2373038669",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Heidi Howard",
    "id": "2309173933",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Nathan Byrd",
    "id": "2275185667",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Hongxu Ma",
    "id": "2249891390",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Yan Wu",
    "id": "2431555849",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Ryan Mullins",
    "id": "2291067498",
    "h_index": 10,
    "papers": 31
   },
   {
    "name": "Qingze Wang",
    "id": "2275272527",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Aida Amini",
    "id": "2288902682",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Sebastien Baur",
    "id": "2239097716",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Yiran Mao",
    "id": "2283094925",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Subhashini Venugopalan",
    "id": "46830680",
    "h_index": 11,
    "papers": 25
   },
   {
    "name": "Will Song",
    "id": "2372685872",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Wen Ding",
    "id": "2310857695",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "P. Collins",
    "id": "2289275736",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Sashank J. Reddi",
    "id": "1981186",
    "h_index": 42,
    "papers": 85
   },
   {
    "name": "Megan Shum",
    "id": "2357160173",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Andrei A. Rusu",
    "id": "2228824",
    "h_index": 21,
    "papers": 38
   },
   {
    "name": "Luisa M. Zintgraf",
    "id": "3378188",
    "h_index": 19,
    "papers": 35
   },
   {
    "name": "Kelvin C. K. Chan",
    "id": "2269731135",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Sheela Goenka",
    "id": "2373035162",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Mathieu Blondel",
    "id": "2281742464",
    "h_index": 7,
    "papers": 14
   },
   {
    "name": "Michael Collins",
    "id": "2316274102",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Renke Pan",
    "id": "2352075260",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "M. Giustina",
    "id": "6891740",
    "h_index": 45,
    "papers": 66
   },
   {
    "name": "Nikolai Chinaev",
    "id": "2239106707",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "C. Schuler",
    "id": "2070481862",
    "h_index": 5,
    "papers": 16
   },
   {
    "name": "Ce Zheng",
    "id": "2276211400",
    "h_index": 5,
    "papers": 16
   },
   {
    "name": "Jonas Valfridsson",
    "id": "2373042647",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "A. Loo",
    "id": "2201329790",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "A. Yakubovich",
    "id": "2139011202",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Jamie Smith",
    "id": "2119126396",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Tao Jiang",
    "id": "2384816695",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Rich Munoz",
    "id": "2391597124",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Gabriel Barcik",
    "id": "2319607403",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Rishabh Bansal",
    "id": "2269194652",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Ming Yang",
    "id": "2228360903",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Yilun Du",
    "id": "2377808532",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "P. Duque",
    "id": "2092912462",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Mary Phuong",
    "id": "145115235",
    "h_index": 9,
    "papers": 19
   },
   {
    "name": "Alexandra M. Belias",
    "id": "1939839468",
    "h_index": 10,
    "papers": 15
   },
   {
    "name": "Kunal Lad",
    "id": "144540164",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Zeyu Liu",
    "id": "2310603569",
    "h_index": 4,
    "papers": 15
   },
   {
    "name": "Tal Schuster",
    "id": "32303439",
    "h_index": 34,
    "papers": 62
   },
   {
    "name": "Karthik Duddu",
    "id": "52125806",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Jieru Hu",
    "id": "2311829615",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "P. Kunkle",
    "id": "1417268290",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Matthew Watson",
    "id": "2303852016",
    "h_index": 5,
    "papers": 17
   },
   {
    "name": "Jackson Tolins",
    "id": "2324799395",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Joshua R. Smith",
    "id": "2284788213",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Denis Teplyashin",
    "id": "3035073",
    "h_index": 9,
    "papers": 14
   },
   {
    "name": "G. Bingham",
    "id": "2312325703",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Marvin Ritter",
    "id": "39687627",
    "h_index": 13,
    "papers": 19
   },
   {
    "name": "Marco Andreetto",
    "id": "2612392",
    "h_index": 11,
    "papers": 18
   },
   {
    "name": "Divya Pitta",
    "id": "2373037167",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "M. Patel",
    "id": "8738254",
    "h_index": 10,
    "papers": 20
   },
   {
    "name": "S. Viswanadha",
    "id": "2260935778",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Trevor Strohman",
    "id": "2985957",
    "h_index": 25,
    "papers": 58
   },
   {
    "name": "Catalin Ionescu",
    "id": "2273228",
    "h_index": 14,
    "papers": 29
   },
   {
    "name": "Jincheng Luo",
    "id": "49811504",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Yogesh Kalley",
    "id": "2098064400",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Jeremy Wiesner",
    "id": "2275187022",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Daniel Deutsch",
    "id": "2264072172",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Derek Lockhart",
    "id": "34405222",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "Peter Choy",
    "id": "2070068655",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Rumen Dangovski",
    "id": "26916003",
    "h_index": 12,
    "papers": 53
   },
   {
    "name": "Chawin Sitawarin",
    "id": "30175233",
    "h_index": 22,
    "papers": 39
   },
   {
    "name": "Cat Graves",
    "id": "2350861697",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Tanya Lando",
    "id": "2373037060",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Joost R. van Amersfoort",
    "id": "3038326",
    "h_index": 20,
    "papers": 32
   },
   {
    "name": "Ndidi Elue",
    "id": "2373035275",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Zhouyuan Huo",
    "id": "3382735",
    "h_index": 25,
    "papers": 67
   },
   {
    "name": "Pooya Moradi",
    "id": "2427084",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Jean Tarbouriech",
    "id": "73775155",
    "h_index": 11,
    "papers": 16
   },
   {
    "name": "H. Michalewski",
    "id": "47407464",
    "h_index": 25,
    "papers": 108
   },
   {
    "name": "Wenting Ye",
    "id": "2244218504",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Eunyoung Kim",
    "id": "47056220",
    "h_index": 24,
    "papers": 295
   },
   {
    "name": "Alex Druinsky",
    "id": "3032817",
    "h_index": 9,
    "papers": 22
   },
   {
    "name": "Florent Altch'e",
    "id": "2064347514",
    "h_index": 7,
    "papers": 8
   },
   {
    "name": "Xinyi Chen",
    "id": "2303896476",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Artur Dwornik",
    "id": "2373037027",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Da-Cheng Juan",
    "id": "50270386",
    "h_index": 19,
    "papers": 46
   },
   {
    "name": "Rivka Moroshko",
    "id": "2325159311",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Horia Toma",
    "id": "2265529013",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Jarrod Kahn",
    "id": "2316579609",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Hai Qian",
    "id": "2372414543",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Maximilian Sieb",
    "id": "51039185",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Irene Cai",
    "id": "2290485789",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "R. Goldenberg",
    "id": "2349384207",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Praneeth Netrapalli",
    "id": "1751626",
    "h_index": 43,
    "papers": 110
   },
   {
    "name": "S. Raghuram",
    "id": "2291504499",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Yuan Gong",
    "id": "2328424182",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Lijie Fan",
    "id": "2355000939",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Evan Palmer",
    "id": "2275183300",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Y. Matias",
    "id": "2269148232",
    "h_index": 29,
    "papers": 118
   },
   {
    "name": "Valentin Gabeur",
    "id": "151352107",
    "h_index": 7,
    "papers": 9
   },
   {
    "name": "Shreya Pathak",
    "id": "2273651441",
    "h_index": 9,
    "papers": 32
   },
   {
    "name": "Tom Ouyang",
    "id": "2372299394",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Donald Metzler",
    "id": "2342449244",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Geoff Bacon",
    "id": "2324798872",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Srinivasan Venkatachary",
    "id": "2305877445",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Sridhar Thiagarajan",
    "id": "2311700521",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Alex A. Cullum",
    "id": "71088349",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "E. Ofek",
    "id": "69840482",
    "h_index": 86,
    "papers": 589
   },
   {
    "name": "Vytenis \u0160ak\u0117nas",
    "id": "9403532",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "M. Hammad",
    "id": "2241816117",
    "h_index": 6,
    "papers": 25
   },
   {
    "name": "C. Magalhaes",
    "id": "1900332650",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "M. Daswani",
    "id": "2842216",
    "h_index": 10,
    "papers": 18
   },
   {
    "name": "Os-car Chang",
    "id": "2275186577",
    "h_index": 8,
    "papers": 16
   },
   {
    "name": "Ashok Popat",
    "id": "2054252",
    "h_index": 22,
    "papers": 56
   },
   {
    "name": "Ruichao Li",
    "id": "2150925426",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Komal Jalan",
    "id": "153776147",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Yanhan Hou",
    "id": "13718763",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Josh Lipschultz",
    "id": "2290487597",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Antoine He",
    "id": "2290485493",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Wenhao Jia",
    "id": "2275193644",
    "h_index": 3,
    "papers": 16
   },
   {
    "name": "Pier Giuseppe Sessa",
    "id": "7281978",
    "h_index": 16,
    "papers": 36
   },
   {
    "name": "Prateek Kolhar",
    "id": "2360107",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "William Wong",
    "id": "2275542210",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Sumeet Singh",
    "id": "2109040371",
    "h_index": 15,
    "papers": 31
   },
   {
    "name": "Lukas Haas",
    "id": "2336820230",
    "h_index": 6,
    "papers": 6
   },
   {
    "name": "Jay Whang",
    "id": "21040156",
    "h_index": 11,
    "papers": 19
   },
   {
    "name": "Hanna Klimczak-Pluci'nska",
    "id": "2275187187",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Georges Rotival",
    "id": "3117981",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "G. Chung",
    "id": "2373040437",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Yiqing Hua",
    "id": "2162375",
    "h_index": 12,
    "papers": 14
   },
   {
    "name": "A. Siddiqui",
    "id": "2372408695",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Nicol\u00e1s Serrano",
    "id": "145414172",
    "h_index": 14,
    "papers": 30
   },
   {
    "name": "Dongkai Chen",
    "id": "2372717351",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Billy Porter",
    "id": "2306950876",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Libin Bai",
    "id": "2275159462",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Keshav Shivam",
    "id": "2349640034",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Sho Arora",
    "id": "2275119807",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Partha Talukdar",
    "id": "2298271272",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Tom Cobley",
    "id": "2335659259",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Sangnie Bhardwaj",
    "id": "1381645735",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "E. Gladchenko",
    "id": "1578656677",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "S. Green",
    "id": "2331874369",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Kelvin Guu",
    "id": "2091768",
    "h_index": 26,
    "papers": 47
   },
   {
    "name": "Felix Fischer",
    "id": "2143272333",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "X. Wu",
    "id": "2372597702",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Eric Wang",
    "id": "2306760317",
    "h_index": 8,
    "papers": 17
   },
   {
    "name": "Achintya Singhal",
    "id": "2275186618",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Tatiana Matejovicova",
    "id": "2166868706",
    "h_index": 8,
    "papers": 17
   },
   {
    "name": "James Martens",
    "id": "2288176078",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Hongji Li",
    "id": "2346475448",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Roma Patel",
    "id": "2330345397",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Elizabeth A. Kemp",
    "id": "2372400204",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Jiaqi Pan",
    "id": "2343786712",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "Lily Wang",
    "id": "2275203223",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Blake Jianhang Chen",
    "id": "2373552965",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Jean-Baptiste Alayrac",
    "id": "2285263",
    "h_index": 33,
    "papers": 67
   },
   {
    "name": "Navneet Potti",
    "id": "2406599",
    "h_index": 8,
    "papers": 14
   },
   {
    "name": "Erika Gemzer",
    "id": "2373042796",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Eugene Ie",
    "id": "2306992351",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "Kay McKinney",
    "id": "2274102650",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Takaaki Saeki",
    "id": "2288266495",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Edward Chou",
    "id": "2065598308",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Pascal Lamblin",
    "id": "3087941",
    "h_index": 16,
    "papers": 19
   },
   {
    "name": "SQ Mah",
    "id": "2347561427",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Zachary Fisher",
    "id": "2274936648",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Martin Chadwick",
    "id": "2159545857",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Jon Stritar",
    "id": "2373038907",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Obaid Sarvana",
    "id": "73491342",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Andrew Hogue",
    "id": "2280692075",
    "h_index": 2,
    "papers": 13
   },
   {
    "name": "A. Shtefan",
    "id": "92071644",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Hadi Hashemi",
    "id": "2070487928",
    "h_index": 8,
    "papers": 30
   },
   {
    "name": "Yang Xu",
    "id": "2275890035",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Jindong Gu",
    "id": "2345676204",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "S. Vikram",
    "id": "2425230",
    "h_index": 16,
    "papers": 34
   },
   {
    "name": "Chung-Ching Chang",
    "id": "2267391508",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Sabela Ramos",
    "id": "2253595555",
    "h_index": 8,
    "papers": 25
   },
   {
    "name": "Logan Kilpatrick",
    "id": "2314112114",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Wei Xi",
    "id": "2403066698",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Jenny Brennan",
    "id": "2275186701",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "Yinghao Sun",
    "id": "2354328389",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Ab-hishek Jindal",
    "id": "2140096847",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Ionel Gog",
    "id": "3077934",
    "h_index": 9,
    "papers": 16
   },
   {
    "name": "Dawn Chen",
    "id": "2311929488",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Felix Wu",
    "id": "24277779",
    "h_index": 22,
    "papers": 28
   },
   {
    "name": "Jason Lee",
    "id": "2328109403",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Sudhindra Kopalle",
    "id": "2373037359",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Srinadh Bhojanapalli",
    "id": "1798880",
    "h_index": 32,
    "papers": 62
   },
   {
    "name": "O. Vinyals",
    "id": "1689108",
    "h_index": 103,
    "papers": 204
   },
   {
    "name": "Natan Potikha",
    "id": "2203789376",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Burcu Karagol Ayan",
    "id": "143990191",
    "h_index": 9,
    "papers": 11
   },
   {
    "name": "Yuan Yuan",
    "id": "2372556835",
    "h_index": 3,
    "papers": 13
   },
   {
    "name": "M. Riley",
    "id": "145428168",
    "h_index": 42,
    "papers": 122
   },
   {
    "name": "P. Sta\u0144czyk",
    "id": "2067024583",
    "h_index": 13,
    "papers": 41
   },
   {
    "name": "Sergey Kishchenko",
    "id": "148074845",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Bing Wang",
    "id": "2263479572",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Dan Garrette",
    "id": "2758616",
    "h_index": 23,
    "papers": 42
   },
   {
    "name": "Antoine Yang",
    "id": "2064599701",
    "h_index": 13,
    "papers": 19
   },
   {
    "name": "Vladimir Feinberg",
    "id": "2275181199",
    "h_index": 11,
    "papers": 38
   },
   {
    "name": "Cj Carey",
    "id": "2055444975",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Javad Azizi",
    "id": "116022444",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Viral R. Shah",
    "id": "2069609721",
    "h_index": 2,
    "papers": 11
   },
   {
    "name": "Erica Moreira",
    "id": "2275185558",
    "h_index": 8,
    "papers": 29
   },
   {
    "name": "Chong-Rong Shi",
    "id": "2212034609",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Joshua Feldman",
    "id": "2286971962",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Elizabeth Salesky",
    "id": "3448427",
    "h_index": 26,
    "papers": 55
   },
   {
    "name": "Thomas Lampe",
    "id": "2066153554",
    "h_index": 22,
    "papers": 36
   },
   {
    "name": "Aneesh S. Pappu",
    "id": "50117905",
    "h_index": 10,
    "papers": 19
   },
   {
    "name": "Duhyeong Kim",
    "id": "122204358",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Jonas Adler",
    "id": "2275173231",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Avi Caciularu",
    "id": "2288816486",
    "h_index": 9,
    "papers": 17
   },
   {
    "name": "B. Walker",
    "id": "2369226463",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Yunhan Xu",
    "id": "2275191526",
    "h_index": 5,
    "papers": 16
   },
   {
    "name": "Yochai Blau",
    "id": "8347541",
    "h_index": 12,
    "papers": 23
   },
   {
    "name": "Dylan Scandinaro",
    "id": "2275186405",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Terry Huang",
    "id": "2324839712",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Sam El-Husseini",
    "id": "2373035538",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "A. Sinha",
    "id": "2261282574",
    "h_index": 3,
    "papers": 14
   },
   {
    "name": "Lijie Ren",
    "id": "2380357098",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Taylor Tobin",
    "id": "2275189014",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "P. Sundberg",
    "id": "2067152591",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "T. Sohn",
    "id": "2060371169",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Vikas Yadav",
    "id": "2252899160",
    "h_index": 4,
    "papers": 15
   },
   {
    "name": "Mimi Ly",
    "id": "39564610",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Emily Xue",
    "id": "2275188181",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Jing Xiong",
    "id": "2324168511",
    "h_index": 11,
    "papers": 39
   },
   {
    "name": "Afzal Shama Soudagar",
    "id": "2373037420",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Sneha Mondal",
    "id": "1796279288",
    "h_index": 9,
    "papers": 22
   },
   {
    "name": "Nikhil Khadke",
    "id": "2676817",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Q. Ren",
    "id": "152777224",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Ben Vargas",
    "id": "2316578365",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "S. Bileschi",
    "id": "1747918",
    "h_index": 11,
    "papers": 22
   },
   {
    "name": "Sarah Chakera",
    "id": "2292141786",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Cindy Wang",
    "id": "2275532754",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "Boyu Wang",
    "id": "2373739775",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Yoni Halpern",
    "id": "2269735987",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "J. Jiang",
    "id": "2373577487",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Vikas Sindhwani",
    "id": "1808676",
    "h_index": 52,
    "papers": 174
   },
   {
    "name": "P. Petrov",
    "id": "2353126158",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Pranavaraj Ponnuramu",
    "id": "2373035830",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Sanket Vaibhav Mehta",
    "id": "47613860",
    "h_index": 13,
    "papers": 24
   },
   {
    "name": "Yuichi Watanabe",
    "id": "2111725960",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "B. Chan",
    "id": "2275177977",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "M. Wisniewski",
    "id": "2373039908",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Trang Pham",
    "id": "2267018787",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Jingwei Zhang",
    "id": "2107958221",
    "h_index": 14,
    "papers": 22
   },
   {
    "name": "Conglong Li",
    "id": "2285843642",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "D. Cesare",
    "id": "47182967",
    "h_index": 17,
    "papers": 26
   },
   {
    "name": "Art Khurshudov",
    "id": "2373035916",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Alex Vasiloff",
    "id": "2373037890",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "M. Tan",
    "id": "114457914",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Zoe Ashwood",
    "id": "2333511945",
    "h_index": 6,
    "papers": 19
   },
   {
    "name": "Bobak Shahriari",
    "id": "2067577",
    "h_index": 15,
    "papers": 42
   },
   {
    "name": "Maryam Majzoubi",
    "id": "31393626",
    "h_index": 9,
    "papers": 13
   },
   {
    "name": "Garrett Tanzer",
    "id": "2287809580",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "Olga Kozlova",
    "id": "2304951894",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Robin Alazard",
    "id": "2373038786",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "James Lee-Thorp",
    "id": "2267341862",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Nguyet Minh Phu",
    "id": "1810717831",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "I. Tian",
    "id": "1968136057",
    "h_index": 7,
    "papers": 15
   },
   {
    "name": "Junwhan Ahn",
    "id": "2275220028",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Andy Crawford",
    "id": "2324798369",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "L. Lax",
    "id": "7889698",
    "h_index": 8,
    "papers": 21
   },
   {
    "name": "Shangguan Yuan",
    "id": "2229513388",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Iftekhar Naim",
    "id": "2373038872",
    "h_index": 10,
    "papers": 15
   },
   {
    "name": "David A. Ross",
    "id": "2257003564",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Oleksandr Ferludin",
    "id": "2175557551",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Tongfei Guo",
    "id": "2197543886",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Andrea Banino",
    "id": "4194027",
    "h_index": 17,
    "papers": 26
   },
   {
    "name": "Hubert Soyer",
    "id": "2794457",
    "h_index": 15,
    "papers": 27
   },
   {
    "name": "Xiaoen Ju",
    "id": "2331174690",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Dominika Rogozi'nska",
    "id": "2275184739",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Ishaan Malhi",
    "id": "72589171",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Marcella Valentine",
    "id": "2351907699",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Daniel Balle",
    "id": "2037843244",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Apoorv Kulshreshtha",
    "id": "1490888815",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Maciej Kula",
    "id": "2266238954",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Yiwen Song",
    "id": "2314381758",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Sophia Austin",
    "id": "2166051497",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "John Schultz",
    "id": "2335665852",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Roy Hirsch",
    "id": "2261740552",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Arthur Douillard",
    "id": "1660848177",
    "h_index": 16,
    "papers": 29
   },
   {
    "name": "A. Reddy",
    "id": "2090449480",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Michael Fink",
    "id": "2275182624",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Summer Yue",
    "id": "2107032220",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Khyatti Gupta",
    "id": "2324930005",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Ada Zhang",
    "id": "2324917703",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "N. Rink",
    "id": "2373042745",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "D. McDuff",
    "id": "2315286830",
    "h_index": 11,
    "papers": 44
   },
   {
    "name": "Lei Meng",
    "id": "2218226973",
    "h_index": 11,
    "papers": 21
   },
   {
    "name": "Andr'as Gyorgy",
    "id": "2343746240",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Yasaman Razeghi",
    "id": "2067184969",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Ricky Liang",
    "id": "2372349409",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Kazuki Osawa",
    "id": "2056855232",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Aviel Atias",
    "id": "47596161",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Matan Eyal",
    "id": "35298844",
    "h_index": 13,
    "papers": 26
   },
   {
    "name": "Tyrone Hill",
    "id": "145915215",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "N. Grigorev",
    "id": "2299027271",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Zhengdong Wang",
    "id": "2374281070",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Nitish Kulkarni",
    "id": "51115915",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Rachel Soh",
    "id": "1573666139",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Ivan Lobov",
    "id": "143985215",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Zachary Charles",
    "id": "2290913078",
    "h_index": 7,
    "papers": 9
   },
   {
    "name": "Sid Lall",
    "id": "2275182574",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Kazuma Hashimoto",
    "id": "2240536041",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Ido Kessler",
    "id": "2197529203",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "V. Gomes",
    "id": "2407056250",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Zelda E. Mariet",
    "id": "1867856",
    "h_index": 16,
    "papers": 43
   },
   {
    "name": "Danny Driess",
    "id": "2283848260",
    "h_index": 27,
    "papers": 35
   },
   {
    "name": "Alessandro Agostini",
    "id": "2353996064",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Canfer Akbulut",
    "id": "2297848348",
    "h_index": 8,
    "papers": 11
   },
   {
    "name": "Jing Hu",
    "id": "2403675896",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Marissa Ikonomidis",
    "id": "2373038673",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Emily Caveness",
    "id": "1658880345",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Kartik Audhkhasi",
    "id": "3104038",
    "h_index": 29,
    "papers": 88
   },
   {
    "name": "Saurabh Agrawal",
    "id": "2238422425",
    "h_index": 7,
    "papers": 14
   },
   {
    "name": "I. Bica",
    "id": "1751623812",
    "h_index": 14,
    "papers": 30
   },
   {
    "name": "Evan Senter",
    "id": "2268665228",
    "h_index": 9,
    "papers": 21
   },
   {
    "name": "Jayaram Mudigonda",
    "id": "2596281",
    "h_index": 13,
    "papers": 26
   },
   {
    "name": "Kelly Chen",
    "id": "2374234216",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Jingchen Ye",
    "id": "2374381305",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Xuanhui Wang",
    "id": "2261356664",
    "h_index": 11,
    "papers": 17
   },
   {
    "name": "James Svensson",
    "id": "2275188153",
    "h_index": 6,
    "papers": 18
   },
   {
    "name": "Philipp Franken",
    "id": "2237167835",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Joshua Newlan",
    "id": "2160888100",
    "h_index": 7,
    "papers": 17
   },
   {
    "name": "Li Lao",
    "id": "2290485653",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Eva Schnider",
    "id": "2280146474",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Sami Alabed",
    "id": "2377562605",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Joseph Kready",
    "id": "1845898514",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Jesse Emond",
    "id": "32757447",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Afief Halumi",
    "id": "2373037273",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "T. Zaman",
    "id": "2372485673",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Chengxi Ye",
    "id": "2296981704",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "N. Raisinghani",
    "id": "2297199371",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Vilobh Meshram",
    "id": "2314109173",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Bo-Cian Chang",
    "id": "2379767588",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "A. Rawat",
    "id": "2241094",
    "h_index": 38,
    "papers": 118
   },
   {
    "name": "Axel Stjerngren",
    "id": "2163521750",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Sergey Levi",
    "id": "2268672539",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Rui Wang",
    "id": "2374940032",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Xiang Long",
    "id": "2295733503",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "M. Rasquinha",
    "id": "2269299250",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Steven Hand",
    "id": "2275161833",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Aditi Mavalankar",
    "id": "38792754",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Lauren Agubuzu",
    "id": "2373038916",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Sudeshna Roy",
    "id": "2116470157",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Junquan Chen",
    "id": "2372286875",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Jarek Wilkiewicz",
    "id": "2224123",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Hao Zhou",
    "id": "2275577449",
    "h_index": 5,
    "papers": 16
   },
   {
    "name": "Micha\u0142 K. Jastrz\u0119bski",
    "id": "2189674684",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "A. D. Lago",
    "id": "2152469362",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Ramya Sree Boppana",
    "id": "103432693",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Wei-Jen Ko",
    "id": "2311703778",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "J. Prendki",
    "id": "41230700",
    "h_index": 27,
    "papers": 389
   },
   {
    "name": "Yao Su",
    "id": "2403572083",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Zhi Li",
    "id": "2155344114",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Eliza Rutherford",
    "id": "2143538252",
    "h_index": 9,
    "papers": 27
   },
   {
    "name": "G. Rao",
    "id": "2374181868",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "R. Comanescu",
    "id": "89066101",
    "h_index": 12,
    "papers": 35
   },
   {
    "name": "Adri\u00e0 Puigdom\u00e8nech",
    "id": "2373037415",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Qihang Chen",
    "id": "2131739425",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Dessie Petrova",
    "id": "2275189070",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Christine Chan",
    "id": "2256938625",
    "h_index": 8,
    "papers": 8
   },
   {
    "name": "V. Milutinovic",
    "id": "2372928951",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "F. Ferreira",
    "id": "2053631125",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Chin-Yi Cheng",
    "id": "2303301198",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Ming Zhang",
    "id": "2290594698",
    "h_index": 6,
    "papers": 19
   },
   {
    "name": "T. Dey",
    "id": "2392418370",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Sherry Yang",
    "id": "2336548160",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Ramesh Sampath",
    "id": "2303850366",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Quoc V. Le",
    "id": "2326311466",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Howard Zhou",
    "id": "2290352183",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Chu-cheng Lin",
    "id": "2266757860",
    "h_index": 7,
    "papers": 15
   },
   {
    "name": "Hoi Lam",
    "id": "2290486901",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Christine Kaeser-Chen",
    "id": "1403585268",
    "h_index": 9,
    "papers": 16
   },
   {
    "name": "Kai Hui",
    "id": "2294173726",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "D. Hirsch",
    "id": "2372747543",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Tom Eccles",
    "id": "2314114940",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Basil Mustafa",
    "id": "2365006141",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Shruti Rijhwani",
    "id": "7391530",
    "h_index": 18,
    "papers": 41
   },
   {
    "name": "Morgane Rivi\u00e8re",
    "id": "2275177725",
    "h_index": 6,
    "papers": 25
   },
   {
    "name": "Yuanzhong Xu",
    "id": "2145139570",
    "h_index": 18,
    "papers": 29
   },
   {
    "name": "Junjie Wang",
    "id": "2269510514",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Xinyang Geng",
    "id": "3468192",
    "h_index": 19,
    "papers": 33
   },
   {
    "name": "Xi-ance Si",
    "id": "2275182246",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Arjun Khare",
    "id": "2064906815",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Cheolmin Kim",
    "id": "2373586268",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "V. Mirrokni",
    "id": "1728881",
    "h_index": 62,
    "papers": 421
   },
   {
    "name": "Ka-Wei Lee",
    "id": "2395608995",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Khuslen Baatarsukh",
    "id": "2290486431",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "N. Braun",
    "id": "2334019224",
    "h_index": 2,
    "papers": 17
   },
   {
    "name": "Lisa Wang",
    "id": "2311736246",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "LV Pallavi",
    "id": "2324799823",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Richard Tanburn",
    "id": "1825728",
    "h_index": 10,
    "papers": 19
   },
   {
    "name": "Yuvein Zhu",
    "id": "2352040805",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Fang-fang Li",
    "id": "2146330360",
    "h_index": 9,
    "papers": 24
   },
   {
    "name": "Setareh Ariafar",
    "id": "1410478948",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Daniel W. Goldberg",
    "id": "2261280628",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "K. Burke",
    "id": "2073234525",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Daniil Mirylenka",
    "id": "1789341",
    "h_index": 7,
    "papers": 24
   },
   {
    "name": "Meiqi Guo",
    "id": "2315806814",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Olaf Ronneberger",
    "id": "2300353321",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Hadas Vogel",
    "id": "2372491857",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Liqun Cheng",
    "id": "2274812839",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Nishita Shetty",
    "id": "2066059409",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "J. Jia",
    "id": "2275694953",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Thomas Jimma",
    "id": "2373037209",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Corey Fry",
    "id": "2336816667",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Ted Xiao",
    "id": "2353415735",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "M. Sundermeyer",
    "id": "2748591",
    "h_index": 25,
    "papers": 51
   },
   {
    "name": "Ryan Burnell",
    "id": "2290484991",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Yannis Assael",
    "id": "3365565",
    "h_index": 23,
    "papers": 31
   },
   {
    "name": "Mario Pinto",
    "id": "2290662953",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "JD Chen",
    "id": "2373531424",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "R. Sathyanarayana",
    "id": "108619466",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Donghyun Cho",
    "id": "2275957667",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Jin Lu",
    "id": "2240550872",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Rishabh Agarwal",
    "id": "2253488622",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Sugato Basu",
    "id": "2329741154",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Lucas Gonzalez",
    "id": "2275585027",
    "h_index": 6,
    "papers": 16
   },
   {
    "name": "Dhruv Shah",
    "id": "2322628540",
    "h_index": 29,
    "papers": 63
   },
   {
    "name": "M. Wei",
    "id": "2260380899",
    "h_index": 7,
    "papers": 44
   },
   {
    "name": "Dre Mahaarachchi",
    "id": "2373039351",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Rohan Agrawal",
    "id": "2054897620",
    "h_index": 4,
    "papers": 14
   },
   {
    "name": "T. Rissa",
    "id": "2540625",
    "h_index": 7,
    "papers": 26
   },
   {
    "name": "Yani Donchev",
    "id": "2266468147",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Ramiro Leal-Cavazos",
    "id": "2098058631",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Adrian Hutter",
    "id": "2275176099",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "M.-T. Mircea",
    "id": "2389127544",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Alon Jacovi",
    "id": "2366159528",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Faruk Ahmed",
    "id": "2286161967",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Jiageng Zhang",
    "id": "2260185674",
    "h_index": 9,
    "papers": 28
   },
   {
    "name": "Shuguang Hu",
    "id": "2383005365",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Bo-Juen Chen",
    "id": "2373737032",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Jonni Kanerva",
    "id": "2036077",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Guillaume Desjardins",
    "id": "2319396362",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Andrew Lee",
    "id": "2275542684",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Nikos Parotsidis",
    "id": "3170658",
    "h_index": 16,
    "papers": 64
   },
   {
    "name": "Asier Mujika",
    "id": "32204498",
    "h_index": 9,
    "papers": 16
   },
   {
    "name": "Tobias Weyand",
    "id": "2284763343",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "J. Snoek",
    "id": "2339615566",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "J. Chick",
    "id": "2346362769",
    "h_index": 3,
    "papers": 16
   },
   {
    "name": "Kai Chen",
    "id": "2375092499",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Paul Chang",
    "id": "2325134691",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Ethan Mahintorabi",
    "id": "119589618",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Zi Wang",
    "id": "2352988798",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Tolly Powell",
    "id": "2275150061",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Orgad Keller",
    "id": "2275186812",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Abhirut Gupta",
    "id": "30214918",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Claire Sha",
    "id": "2373043459",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Kanav Garg",
    "id": "2289642023",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "N. Heess",
    "id": "2333926495",
    "h_index": 7,
    "papers": 18
   },
   {
    "name": "'Agoston Weisz",
    "id": "2373035287",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Cassidy Hardin",
    "id": "2275186843",
    "h_index": 9,
    "papers": 29
   },
   {
    "name": "B. Wydrowski",
    "id": "2028870",
    "h_index": 12,
    "papers": 32
   },
   {
    "name": "Benjamin Coleman",
    "id": "2240537645",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Karina Zainullina",
    "id": "2362507945",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Pankaj Joshi",
    "id": "2259168654",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Alessandro Epasto",
    "id": "2123019231",
    "h_index": 4,
    "papers": 16
   },
   {
    "name": "Terry Spitz",
    "id": "1704229781",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Binbin Xiong",
    "id": "2166057543",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Kai Zhao",
    "id": "2325142160",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Arseniy Klimovskiy",
    "id": "2275054278",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Ivy Zheng",
    "id": "2275187038",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Johan Ferret",
    "id": "151047979",
    "h_index": 18,
    "papers": 51
   },
   {
    "name": "Itay Yona",
    "id": "2283305710",
    "h_index": 9,
    "papers": 13
   },
   {
    "name": "Waleed Khawaja",
    "id": "2373038759",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Jean-Baptiste Lespiau",
    "id": "143783339",
    "h_index": 18,
    "papers": 32
   },
   {
    "name": "M. Krikun",
    "id": "2048712",
    "h_index": 23,
    "papers": 58
   },
   {
    "name": "Siamak Shakeri",
    "id": "2944868",
    "h_index": 20,
    "papers": 51
   },
   {
    "name": "Timoth\u00e9e Cour",
    "id": "2807482",
    "h_index": 13,
    "papers": 18
   },
   {
    "name": "Bonnie Li",
    "id": "2381406227",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "I. Krivokon",
    "id": "115361655",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Daniel Suh",
    "id": "49199285",
    "h_index": 7,
    "papers": 9
   },
   {
    "name": "Alex Hofer",
    "id": "2372452359",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "J. Abdallah",
    "id": "2372583738",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Nikita Putikhin",
    "id": "30473489",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Oscar Akerlund",
    "id": "2324801489",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Silvio Lattanzi",
    "id": "2303853197",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Anurag Kumar",
    "id": "2153105954",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "S. Settle",
    "id": "2577399",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Himanshu Srivastava",
    "id": "2368918433",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Folawiyo Campbell-Ajala",
    "id": "2275053606",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "E. Rosseel",
    "id": "2372728706",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Mihai Istin",
    "id": "34188940",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Nishanth Dikkala",
    "id": "2312421",
    "h_index": 17,
    "papers": 37
   },
   {
    "name": "Anand Rao",
    "id": "2314521404",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Nick Young",
    "id": "2052908092",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Kate Lin",
    "id": "2318257506",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Dhruva Bhaswar",
    "id": "2373037405",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Yiming Wang",
    "id": "2331758713",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Jaume Sanchez Elias",
    "id": "2186405649",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "K. Muralidharan",
    "id": "2358480454",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "James Keeling",
    "id": "2058168486",
    "h_index": 8,
    "papers": 22
   },
   {
    "name": "Dayou Du",
    "id": "2275121069",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Siddharth Gopal",
    "id": "2295888567",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Gregory Dibb",
    "id": "2373035737",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Charles Blundell",
    "id": "2279830503",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "M. Delakis",
    "id": "1812064",
    "h_index": 10,
    "papers": 23
   },
   {
    "name": "Jacky Liang",
    "id": "6454541",
    "h_index": 23,
    "papers": 40
   },
   {
    "name": "M. Ribeiro",
    "id": "2371106347",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "G. Karadzhov",
    "id": "2667995",
    "h_index": 11,
    "papers": 43
   },
   {
    "name": "Guillermo Garrido",
    "id": "2336103847",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Ankur Bapna",
    "id": "12295226",
    "h_index": 36,
    "papers": 71
   },
   {
    "name": "Jiawei Cao",
    "id": "2348344678",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "A. Sadovsky",
    "id": "2079448999",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "P. Tafti",
    "id": "1775270",
    "h_index": 15,
    "papers": 64
   },
   {
    "name": "A. Guez",
    "id": "35099444",
    "h_index": 27,
    "papers": 50
   },
   {
    "name": "Coline Devin",
    "id": "144373380",
    "h_index": 24,
    "papers": 39
   },
   {
    "name": "Yixian Di",
    "id": "2372521286",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Jinwei Xing",
    "id": "2275169067",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Chuqiao Xu",
    "id": "153250242",
    "h_index": 14,
    "papers": 21
   },
   {
    "name": "Hanzhao Lin",
    "id": "2275559471",
    "h_index": 4,
    "papers": 17
   },
   {
    "name": "Chun-Te Chu",
    "id": "2279021265",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Sameera S. Ponda",
    "id": "3036370",
    "h_index": 17,
    "papers": 30
   },
   {
    "name": "Wesley Helmholz",
    "id": "2050736476",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Fan Yang",
    "id": "2275801132",
    "h_index": 4,
    "papers": 19
   },
   {
    "name": "Yue Gao",
    "id": "2323519287",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Sara Javanmardi",
    "id": "2313684474",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Wael Farhan",
    "id": "2275177486",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Alex Ram\u00edrez",
    "id": "2065144612",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Ricardo Figueira",
    "id": "2324799833",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "K. Sim",
    "id": "1693612",
    "h_index": 33,
    "papers": 151
   },
   {
    "name": "Yuval Bahat",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Ashwin Vaswani",
    "id": "1824263906",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Liangzhe Yuan",
    "id": "36001694",
    "h_index": 18,
    "papers": 25
   },
   {
    "name": "Gufeng Zhang",
    "id": "2373555642",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Leland Rechis",
    "id": "147358990",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Hanjun Dai",
    "id": "2265495262",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Tayo Oguntebi",
    "id": "2814407",
    "h_index": 9,
    "papers": 13
   },
   {
    "name": "Alexandra Cordell",
    "id": "2373038797",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Eug'enie Rives",
    "id": "2373038717",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Kaan Tekelioglu",
    "id": "89478969",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Naveen Kumar",
    "id": "2116960562",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Bing Zhang",
    "id": "2396412079",
    "h_index": 8,
    "papers": 42
   },
   {
    "name": "Aurick Zhou",
    "id": "35499972",
    "h_index": 12,
    "papers": 13
   },
   {
    "name": "N. Savinov",
    "id": "2417003",
    "h_index": 17,
    "papers": 27
   },
   {
    "name": "A. Leach",
    "id": "2254412217",
    "h_index": 4,
    "papers": 17
   },
   {
    "name": "Alex Tudor",
    "id": "2290485021",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "S. Ganapathy",
    "id": "2275185831",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Yanyan Zheng",
    "id": "2321032246",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "M. Rossini",
    "id": "2061586191",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Vera Axelrod",
    "id": "82840075",
    "h_index": 9,
    "papers": 12
   },
   {
    "name": "Arnaud Autef",
    "id": "2289114513",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Yukun Zhu",
    "id": "2269885798",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Zheng Zheng",
    "id": "2342420799",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Mingda Zhang",
    "id": "2286394354",
    "h_index": 3,
    "papers": 15
   },
   {
    "name": "Baochen Sun",
    "id": "2238900563",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Jie Ren",
    "id": "2322249712",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Nenad Toma\u0161ev",
    "id": "2213266",
    "h_index": 26,
    "papers": 49
   },
   {
    "name": "Nithish Kannen",
    "id": "2147325257",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Amer Sinha",
    "id": "2309477594",
    "h_index": 11,
    "papers": 14
   },
   {
    "name": "Charlie Chen",
    "id": "2182971260",
    "h_index": 10,
    "papers": 22
   },
   {
    "name": "Louis O'Bryan",
    "id": "2373038678",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Alex Pak",
    "id": "2373039961",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Aditya Kusupati",
    "id": "52207562",
    "h_index": 14,
    "papers": 34
   },
   {
    "name": "W. Yang.",
    "id": "2402005606",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Deepak Ramachandran",
    "id": "143812128",
    "h_index": 14,
    "papers": 45
   },
   {
    "name": "Patrick Griffin",
    "id": "2261577248",
    "h_index": 1,
    "papers": 9
   },
   {
    "name": "Seokhwan Kim",
    "id": "2288403312",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "P. Neubeck",
    "id": "2414196",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Craig Schiff",
    "id": "2373039016",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Tammo Spalink",
    "id": "2493179",
    "h_index": 10,
    "papers": 26
   },
   {
    "name": "Mingyang Ling",
    "id": "2373043102",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Arun Nair",
    "id": "2285968492",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Ga-Young Joung",
    "id": "2372882551",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Linda Deng",
    "id": "2374187460",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "A. Bhoopchand",
    "id": "48262221",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "L. Aroyo",
    "id": "2257256357",
    "h_index": 9,
    "papers": 43
   },
   {
    "name": "Tom Duerig",
    "id": "2066508193",
    "h_index": 10,
    "papers": 17
   },
   {
    "name": "Jordan Griffith",
    "id": "2055443643",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Gabriel Barth-Maron",
    "id": "1403998955",
    "h_index": 15,
    "papers": 29
   },
   {
    "name": "J. Ades",
    "id": "2382351379",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Alex Haig",
    "id": "2265493853",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Ankur Taly",
    "id": "2313338667",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Yunting Song",
    "id": "2281255999",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Paul Michel",
    "id": "2275185732",
    "h_index": 9,
    "papers": 24
   },
   {
    "name": "Dave Orr",
    "id": "2353278210",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "D. Weesner",
    "id": "4696495",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Corentin Tallec",
    "id": "31803582",
    "h_index": 19,
    "papers": 25
   },
   {
    "name": "Carrie Grimes Bostock",
    "id": "2290485108",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Paul Niemczyk",
    "id": "2373037410",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Andrew Twigg",
    "id": "1889653",
    "h_index": 15,
    "papers": 36
   },
   {
    "name": "Mudit Verma",
    "id": "2379657992",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Rohith Vallu",
    "id": "13997540",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Henry Wang",
    "id": "2356860179",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Marco Gelmi",
    "id": "2219672228",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Kiranbir Sodhia",
    "id": "2303850028",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "A. Chuklin",
    "id": "2405897",
    "h_index": 17,
    "papers": 32
   },
   {
    "name": "Omer Goldman",
    "id": "40060272",
    "h_index": 12,
    "papers": 30
   },
   {
    "name": "Jasmine George",
    "id": "2372348810",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Liang Bai",
    "id": "2374382638",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Ke Zhang",
    "id": "2378116336",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Petar Sirkovic",
    "id": "2679779",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Efrat Nehoran",
    "id": "2373037785",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "G. Pundak",
    "id": "2779415",
    "h_index": 17,
    "papers": 24
   },
   {
    "name": "Jiaqi Mu",
    "id": "2243242024",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "A. Chen",
    "id": "2116402991",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Alex C. Greve",
    "id": "2373044732",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Paulo Zacchello",
    "id": "2373035699",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "David Amos",
    "id": "2064400086",
    "h_index": 7,
    "papers": 7
   },
   {
    "name": "Heming Ge",
    "id": "31856237",
    "h_index": 12,
    "papers": 24
   },
   {
    "name": "Eric Noland",
    "id": "51210148",
    "h_index": 12,
    "papers": 33
   },
   {
    "name": "Colton Bishop",
    "id": "2275183062",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Jeffrey M. Dudek",
    "id": "40511532",
    "h_index": 11,
    "papers": 19
   },
   {
    "name": "Youhei Namiki",
    "id": "1875007",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Elena Buchatskaya",
    "id": "118801223",
    "h_index": 12,
    "papers": 36
   },
   {
    "name": "Jing Li",
    "id": "2275685155",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Dorsa Sadigh",
    "id": "1779671",
    "h_index": 69,
    "papers": 226
   },
   {
    "name": "Masha Samsikova",
    "id": "2347349081",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Dan Malkin",
    "id": "2348097555",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Damien Vincent",
    "id": "2282960451",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Robert David",
    "id": "2301165761",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Rob Willoughby",
    "id": "2275189345",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Phoenix Meadowlark",
    "id": "2160711120",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Shawn Gao",
    "id": "2373587983",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Yan Li",
    "id": "2343691965",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Rajas Apte",
    "id": "2321404995",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Amit Jhindal",
    "id": "2367188103",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Stein Xudong Lin",
    "id": "2437233257",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "A. Polozov",
    "id": "144703404",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Zhicheng Wang",
    "id": "2390402457",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Tomas Mery",
    "id": "2373037434",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "G. Anirudh",
    "id": "2373037408",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "V. Yerram",
    "id": "2357188929",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "S. Stevens",
    "id": "2372443783",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Tianqi Liu",
    "id": "2275249023",
    "h_index": 7,
    "papers": 16
   },
   {
    "name": "Noah Fiedel",
    "id": "22640071",
    "h_index": 20,
    "papers": 42
   },
   {
    "name": "Charles Sutton",
    "id": "2269732453",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Matthew Johnson",
    "id": "2275221227",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Xiaodan Song",
    "id": "2295869952",
    "h_index": 2,
    "papers": 11
   },
   {
    "name": "Kate Baumli",
    "id": "1734809439",
    "h_index": 9,
    "papers": 13
   },
   {
    "name": "Nir Shabat",
    "id": "2098034814",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Muqthar Mohammad",
    "id": "2350649696",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Hao Liu",
    "id": "2364826004",
    "h_index": 10,
    "papers": 25
   },
   {
    "name": "M. Selvi",
    "id": "2269473701",
    "h_index": 8,
    "papers": 17
   },
   {
    "name": "Yichao Zhou",
    "id": "2273755922",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "M. Manshadi",
    "id": "2316326121",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Chu-Ling Ko",
    "id": "29773030",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Anthony Chen",
    "id": "2362540756",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Michael Bendersky",
    "id": "2240516450",
    "h_index": 11,
    "papers": 32
   },
   {
    "name": "J. Mendez",
    "id": "2372469102",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "N. Kothari",
    "id": "2319795274",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "A. Zandieh",
    "id": "2872461",
    "h_index": 17,
    "papers": 33
   },
   {
    "name": "Yiling Huang",
    "id": "2375085977",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "D. Andor",
    "id": "3365603",
    "h_index": 17,
    "papers": 40
   },
   {
    "name": "Ellie Pavlick",
    "id": "2260118854",
    "h_index": 11,
    "papers": 48
   },
   {
    "name": "I. Brusilovsky",
    "id": "2072258345",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Jitendra K. Harlalka",
    "id": "1685518",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Sally Goldman",
    "id": "2365127154",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "A. Lampinen",
    "id": "2270676283",
    "h_index": 8,
    "papers": 22
   },
   {
    "name": "Guowang Li",
    "id": "2406324035",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Asahi Ushio",
    "id": "27044733",
    "h_index": 15,
    "papers": 29
   },
   {
    "name": "Somit Gupta",
    "id": "2372783130",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Lei Zhang",
    "id": "2372569228",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "C. Fu",
    "id": "2345672838",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Madhavi Sewak",
    "id": "12763201",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "T. Denk",
    "id": "2262696430",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Jed Borovik",
    "id": "2373035623",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Brendan Jou",
    "id": "2301089580",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Avital Zipori",
    "id": "2302797167",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Prateek Jain",
    "id": "2277742787",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Junwen Bai",
    "id": "2279886426",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "T. Luong",
    "id": "2292086077",
    "h_index": 10,
    "papers": 43
   },
   {
    "name": "Jonathan Tompson",
    "id": "2704494",
    "h_index": 43,
    "papers": 71
   },
   {
    "name": "Alice Li",
    "id": "2373217395",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Li Liu",
    "id": "2320533036",
    "h_index": 8,
    "papers": 17
   },
   {
    "name": "George Powell",
    "id": "2191617645",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Jiajun Shen",
    "id": "2287918653",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Alex Feng",
    "id": "2351910788",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Grishma Chole",
    "id": "2373037565",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Dahai Yu",
    "id": "2377741757",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Yinlam Chow",
    "id": "2372217962",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Tongxin Yin",
    "id": "2163137975",
    "h_index": 7,
    "papers": 20
   },
   {
    "name": "Eric Malmi",
    "id": "3288074",
    "h_index": 21,
    "papers": 63
   },
   {
    "name": "Kefan Xiao",
    "id": "2268673324",
    "h_index": 6,
    "papers": 14
   },
   {
    "name": "Yash Pande",
    "id": "2373035940",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Shachi Paul",
    "id": "2324786062",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "N. D. Santo",
    "id": "41160191",
    "h_index": 7,
    "papers": 17
   },
   {
    "name": "Adil Dostmohamed",
    "id": "2218899126",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Sergio Guadarrama",
    "id": "2348540958",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Aaron B. Phillips",
    "id": "2450694",
    "h_index": 8,
    "papers": 17
   },
   {
    "name": "Thanumalayan Sankaranarayana Pillai",
    "id": "2598683",
    "h_index": 10,
    "papers": 22
   },
   {
    "name": "G. Yona",
    "id": "36825265",
    "h_index": 17,
    "papers": 35
   },
   {
    "name": "Amin Ghafouri",
    "id": "3010652",
    "h_index": 8,
    "papers": 13
   },
   {
    "name": "Preethi Lahoti",
    "id": "8777021",
    "h_index": 13,
    "papers": 29
   },
   {
    "name": "Benjamin Lee",
    "id": "2275292292",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Dhruv Madeka",
    "id": "10723295",
    "h_index": 10,
    "papers": 24
   },
   {
    "name": "Eren Sezener",
    "id": "1413718981",
    "h_index": 9,
    "papers": 10
   },
   {
    "name": "Simon Tokumine",
    "id": "148152480",
    "h_index": 6,
    "papers": 19
   },
   {
    "name": "Adrian Collister",
    "id": "69041729",
    "h_index": 11,
    "papers": 16
   },
   {
    "name": "N. Cao",
    "id": "2294876765",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "R. Shin",
    "id": "2373044428",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Uday Kalra",
    "id": "2351908147",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Parker Beak",
    "id": "2373037772",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Emily Nottage",
    "id": "2373043687",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "R. Nakashima",
    "id": "2237325781",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Ivan Jurin",
    "id": "2132326350",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Vikash Sehwag",
    "id": "3482535",
    "h_index": 23,
    "papers": 53
   },
   {
    "name": "Meenu Gaba",
    "id": "2275186023",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Junhao Zeng",
    "id": "2314347311",
    "h_index": 3,
    "papers": 13
   },
   {
    "name": "Kevin R. McKee",
    "id": "2336872661",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Fernando Pereira",
    "id": "2291062354",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Tamar Yakar",
    "id": "2373039308",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Amayika Panda",
    "id": "2409434",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Arka Dhar",
    "id": "2275244298",
    "h_index": 6,
    "papers": 19
   },
   {
    "name": "Peilin Zhong",
    "id": "2249561001",
    "h_index": 8,
    "papers": 27
   },
   {
    "name": "Daniel Sohn",
    "id": "2275175792",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "M. Brand",
    "id": "114949661",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Lars Lowe Sjoesund",
    "id": "2291065624",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Viral Carpenter",
    "id": "2315983382",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Sharon Lin",
    "id": "2292296672",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "S. Thakoor",
    "id": "41037204",
    "h_index": 16,
    "papers": 42
   },
   {
    "name": "Marcus Wainwright",
    "id": "47969243",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Ashwin Chaugule",
    "id": "2173266",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Pranesh Srinivasan",
    "id": "2300371401",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Muye Zhu",
    "id": "31913758",
    "h_index": 10,
    "papers": 25
   },
   {
    "name": "B. Orlando",
    "id": "2266077092",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Jack Weber",
    "id": "2113579153",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Ayzaan Wahid",
    "id": "88728227",
    "h_index": 21,
    "papers": 27
   },
   {
    "name": "Gilles Baechler",
    "id": "2283137677",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Apurv Suman",
    "id": "8513995",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Jovana Mitrovi'c",
    "id": "2105843960",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Gabe Taubman",
    "id": "2373042889",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Honglin Yu",
    "id": "2322457515",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Helen King",
    "id": "2298269925",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Josh Dillon",
    "id": "2275564829",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Cathy Yip",
    "id": "2349539189",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "D. Varma",
    "id": "2373040059",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "T. Izo",
    "id": "1739038",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Levent Bolelli",
    "id": "1743426",
    "h_index": 11,
    "papers": 22
   },
   {
    "name": "Borja de Balle Pigem",
    "id": "69334679",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Julia Di Trapani",
    "id": "2292140363",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Fotis Iliopoulos",
    "id": "1701211",
    "h_index": 15,
    "papers": 46
   },
   {
    "name": "Adam Paszke",
    "id": "3407277",
    "h_index": 13,
    "papers": 25
   },
   {
    "name": "Nishant Ranka",
    "id": "2373035724",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Joe Zou",
    "id": "2242887453",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "F. Pongetti",
    "id": "71464080",
    "h_index": 8,
    "papers": 32
   },
   {
    "name": "Jed N McGiffin",
    "id": "7661274",
    "h_index": 5,
    "papers": 19
   },
   {
    "name": "A. Siegman",
    "id": "2290050621",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Rich Galt",
    "id": "2334574578",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Ross Hemsley",
    "id": "2320778889",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Goran vZuvzi'c",
    "id": "2373037619",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Victor Carbune",
    "id": "2237423389",
    "h_index": 7,
    "papers": 9
   },
   {
    "name": "Tao Li",
    "id": "2372487969",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Myle Ott",
    "id": "40511414",
    "h_index": 38,
    "papers": 135
   },
   {
    "name": "F. D. C. Quitry",
    "id": "123721125",
    "h_index": 10,
    "papers": 14
   },
   {
    "name": "D. Torres",
    "id": "2058150463",
    "h_index": 2,
    "papers": 11
   },
   {
    "name": "Yuri Chervonyi",
    "id": "2344091715",
    "h_index": 8,
    "papers": 13
   },
   {
    "name": "Tomy Tsai",
    "id": "2275175372",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Prem Eruvbetine",
    "id": "2373037744",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Samuel J. Yang",
    "id": "3224268",
    "h_index": 9,
    "papers": 14
   },
   {
    "name": "Matthew Denton",
    "id": "46691989",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Jacob Walker",
    "id": "2336663153",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Slavica Andavci'c",
    "id": "2373037469",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Idan Heimlich Shtacher",
    "id": "3452452",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Vittal Premachandran",
    "id": "2806645",
    "h_index": 10,
    "papers": 23
   },
   {
    "name": "Harshal Tushar Lehri",
    "id": "10741766",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Cip Baetu",
    "id": "2324799480",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Damion Yates",
    "id": "2290486731",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "L. Lamprou",
    "id": "39521793",
    "h_index": 13,
    "papers": 28
   },
   {
    "name": "Mariko Iinuma",
    "id": "2275181534",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Ioana Mihailescu",
    "id": "2235689510",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Ben Albrecht",
    "id": "2275177777",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Shachi Dave",
    "id": "2160404",
    "h_index": 13,
    "papers": 28
   },
   {
    "name": "S. Sargsyan",
    "id": "2238856229",
    "h_index": 2,
    "papers": 16
   },
   {
    "name": "Bryan Perozzi",
    "id": "2271808",
    "h_index": 37,
    "papers": 98
   },
   {
    "name": "Lucas Manning",
    "id": "2373043955",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Chiyuan Zhang",
    "id": "2309481623",
    "h_index": 15,
    "papers": 28
   },
   {
    "name": "D. Vnukov",
    "id": "1581801720",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Igor Mordatch",
    "id": "2224543226",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Raia Hadsell Wolfgang Macherey",
    "id": "2373035612",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "R. Kappedal",
    "id": "4610543",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "J. Stephan",
    "id": "143763414",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "A. Tripathi",
    "id": "2390358651",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Klaus Macherey",
    "id": "113439369",
    "h_index": 14,
    "papers": 21
   },
   {
    "name": "J. Qian",
    "id": "2250400360",
    "h_index": 15,
    "papers": 110
   },
   {
    "name": "A. Bhowmick",
    "id": "2338932527",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Shekoofeh Azizi",
    "id": "40151244",
    "h_index": 30,
    "papers": 60
   },
   {
    "name": "R. Leblond",
    "id": "37212795",
    "h_index": 14,
    "papers": 24
   },
   {
    "name": "S. Garlapati",
    "id": "2395549375",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "T. Knight",
    "id": "2276819732",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Matt Wiethoff",
    "id": "2275252154",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Wei-Chih Hung",
    "id": "2277731985",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "A. Angelova",
    "id": "145426908",
    "h_index": 43,
    "papers": 120
   },
   {
    "name": "G. Evangelopoulos",
    "id": "2161973553",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Pawe\u0142 Janus",
    "id": "2324798644",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Dimitris Paparas",
    "id": "2275466240",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Matthew Rahtz",
    "id": "3183032",
    "h_index": 10,
    "papers": 25
   },
   {
    "name": "Ken Caluwaerts",
    "id": "2197494368",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "V. Sampathkumar",
    "id": "2089462953",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Daniel Jarrett",
    "id": "2326294679",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Shadi Noghabi",
    "id": "2292028369",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Antoine Miech",
    "id": "19200186",
    "h_index": 21,
    "papers": 42
   },
   {
    "name": "C. Yeung",
    "id": "2394703562",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "G. Clark",
    "id": "39759068",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Henry Prior",
    "id": "2360306445",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Fei Zheng",
    "id": "2373636080",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Jean Pouget-Abadie",
    "id": "1403025868",
    "h_index": 15,
    "papers": 34
   },
   {
    "name": "Indro Bhattacharya",
    "id": "2372880797",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Kalpesh Krishna",
    "id": "26161085",
    "h_index": 23,
    "papers": 35
   },
   {
    "name": "W. Bishop",
    "id": "2311700917",
    "h_index": 6,
    "papers": 14
   },
   {
    "name": "Zhe Yuan",
    "id": "2401782631",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Yunxiao Deng",
    "id": "80240628",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Ashutosh Sathe",
    "id": "2033909694",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Kacper Krasowiak",
    "id": "2373037719",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Ciprian Chelba",
    "id": "2287249574",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Cho-Jui Hsieh",
    "id": "2295690015",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Kiran Vodrahalli",
    "id": "4529644",
    "h_index": 13,
    "papers": 48
   },
   {
    "name": "Buhua Liu",
    "id": "2289863048",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "T. Koppe",
    "id": "34238827",
    "h_index": 19,
    "papers": 98
   },
   {
    "name": "Amr Khalifa",
    "id": "2289844920",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Lubo Litchev",
    "id": "2373038800",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Pichi Charoenpanit",
    "id": "2373038730",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "R. Roberts",
    "id": "2383474239",
    "h_index": 8,
    "papers": 48
   },
   {
    "name": "Sachin Yadav",
    "id": "2181288323",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Yasumasa Onoe",
    "id": "115412405",
    "h_index": 15,
    "papers": 25
   },
   {
    "name": "Desislav Ivanov",
    "id": "2324002990",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Megha Mohabey",
    "id": "2288401955",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Vighnesh Birodkar",
    "id": "3468723",
    "h_index": 11,
    "papers": 19
   },
   {
    "name": "Nemanja Raki'cevi'c",
    "id": "2185348949",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "P. Sermanet",
    "id": "3142556",
    "h_index": 39,
    "papers": 77
   },
   {
    "name": "Vaibhav Mehta",
    "id": "2193625847",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "K. Subudhi",
    "id": "2043231778",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Travis Choma",
    "id": "1927701",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Willa Ng",
    "id": "46222567",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Luheng He",
    "id": "2253917827",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Kathie Wang",
    "id": "2324838603",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Tasos Kementsietsidis",
    "id": "2248083688",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Shane Gu",
    "id": "2275182501",
    "h_index": 5,
    "papers": 22
   },
   {
    "name": "Mansi Gupta",
    "id": "48232761",
    "h_index": 8,
    "papers": 36
   },
   {
    "name": "A. Nystrom",
    "id": "2064161903",
    "h_index": 7,
    "papers": 12
   },
   {
    "name": "Mehran Kazemi",
    "id": "2317010095",
    "h_index": 4,
    "papers": 15
   },
   {
    "name": "Timothy Chung",
    "id": "2275188933",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "N. Cano",
    "id": "2058272837",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Nikhil Dhawan",
    "id": "2300598371",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Yufei Wang",
    "id": "2331254214",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Jiawei Xia",
    "id": "2275552322",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Trevor Yacovone",
    "id": "2351909612",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Eric Jia",
    "id": "2346169779",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Mingqing Chen",
    "id": "2372316676",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "S. Ivanov",
    "id": "2054725958",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Ashrith Sheshan",
    "id": "1409248531",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Sid Dalmia",
    "id": "2290487862",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Pawe\u0142 Stradomski",
    "id": "2314120085",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Pengcheng Yin",
    "id": "2372276822",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "S. Haykal",
    "id": "40269586",
    "h_index": 8,
    "papers": 14
   },
   {
    "name": "Congchao Wang",
    "id": "2304519281",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Dennis Duan",
    "id": "2275178061",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Neslihan Bulut",
    "id": "2066993156",
    "h_index": 7,
    "papers": 9
   },
   {
    "name": "Greg Kochanski",
    "id": "2316558509",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Liam MacDermed",
    "id": "2964338",
    "h_index": 8,
    "papers": 11
   },
   {
    "name": "Namrata Godbole",
    "id": "33760399",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Shitao Weng",
    "id": "2372284576",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Jingjing Chen",
    "id": "2323513308",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Rachana Fellinger",
    "id": "2203789378",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Ramin Mehran",
    "id": "2358452695",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Dan Suo",
    "id": "34632047",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Hisham Husain",
    "id": "47287134",
    "h_index": 8,
    "papers": 20
   },
   {
    "name": "Tong He",
    "id": "2367641165",
    "h_index": 8,
    "papers": 10
   },
   {
    "name": "Kaushal Patel",
    "id": "2304562761",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Joshua Howland",
    "id": "113828976",
    "h_index": 7,
    "papers": 14
   },
   {
    "name": "R. Parker",
    "id": "49966267",
    "h_index": 10,
    "papers": 28
   },
   {
    "name": "K. Nguyen",
    "id": "2314888051",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Sharath Maddineni",
    "id": "33019858",
    "h_index": 10,
    "papers": 16
   },
   {
    "name": "Christopher Rawles",
    "id": "2223952385",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Mina Khan",
    "id": "2258793616",
    "h_index": 6,
    "papers": 14
   },
   {
    "name": "Shlomi Cohen-Ganor",
    "id": "2215482449",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Amol Mandhane",
    "id": "2063800905",
    "h_index": 7,
    "papers": 8
   },
   {
    "name": "Xinyi Wu",
    "id": "2370064138",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Chenkai Kuang",
    "id": "2275189563",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Iulia Comcsa",
    "id": "2373039313",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "R. Ganeshan",
    "id": "2311956972",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Hanie Sedghi",
    "id": "2812848",
    "h_index": 25,
    "papers": 52
   },
   {
    "name": "Adam E. Bloniarz",
    "id": "2728564",
    "h_index": 11,
    "papers": 19
   },
   {
    "name": "Nuo Wang Pierse",
    "id": "1491551860",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Anton Briukhov",
    "id": "2275185833",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Petr Mitrichev",
    "id": "27742155",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Anita Gergely",
    "id": "2105841261",
    "h_index": 6,
    "papers": 14
   },
   {
    "name": "Serena Zhan",
    "id": "2372525710",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Allan Zhou",
    "id": "2352113923",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Nikita Saxena",
    "id": "2330077514",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Eva Lu",
    "id": "2387866738",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Josef Dean",
    "id": "2312753279",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Ashish Gupta",
    "id": "2257350329",
    "h_index": 4,
    "papers": 25
   },
   {
    "name": "Nicolas Perez-Nieves",
    "id": "1412926100",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Renjie Wu",
    "id": "2324223418",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Cory Y. McLean",
    "id": "2249285461",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Wei Liang",
    "id": "2317041087",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Disha Jindal",
    "id": "2347352643",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Anton Tsitsulin",
    "id": "40900939",
    "h_index": 16,
    "papers": 39
   },
   {
    "name": "Wenhao Yu",
    "id": "2265397442",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Kaiz Alarakyia",
    "id": "2336869502",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Tom Schaul",
    "id": "2332363386",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Piyush A. Patil",
    "id": "2352762124",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Peter Sung",
    "id": "98490326",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Eli Peake",
    "id": "2631855",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Hongkun Yu",
    "id": "2316652407",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Feryal M. P. Behbahani",
    "id": "145124447",
    "h_index": 14,
    "papers": 24
   },
   {
    "name": "JD Co-Reyes",
    "id": "2266464831",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "A. Ansell",
    "id": "2364684499",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Sean Sun",
    "id": "2336258846",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "C. Barbu",
    "id": "2397531787",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Jonathan Lee",
    "id": "2390403071",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Seb Noury",
    "id": "30155667",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "J. Allingham",
    "id": "1491706991",
    "h_index": 10,
    "papers": 18
   },
   {
    "name": "Bilal Piot",
    "id": "1808897",
    "h_index": 44,
    "papers": 80
   },
   {
    "name": "Mohit Sharma",
    "id": "145467103",
    "h_index": 17,
    "papers": 35
   },
   {
    "name": "Christo-pher Yew",
    "id": "2275187959",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "I. Korotkov",
    "id": "15853387",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Bibo Xu",
    "id": "2290664586",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "D. Brady",
    "id": "2250386449",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "G. Petrovic",
    "id": "2339161023",
    "h_index": 3,
    "papers": 14
   },
   {
    "name": "Shibl Mourad",
    "id": "49501871",
    "h_index": 11,
    "papers": 18
   },
   {
    "name": "Claire Cui",
    "id": "2352111750",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Aditya Gupta",
    "id": "2316245173",
    "h_index": 3,
    "papers": 16
   },
   {
    "name": "P. Schuh",
    "id": "2620528",
    "h_index": 9,
    "papers": 25
   },
   {
    "name": "Saarthak Khanna",
    "id": "30752094",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Anna Goldie",
    "id": "46684455",
    "h_index": 20,
    "papers": 48
   },
   {
    "name": "Abhinav Arora",
    "id": "50788711",
    "h_index": 9,
    "papers": 10
   },
   {
    "name": "V. Zubov",
    "id": "2373043258",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "A. Stuart",
    "id": "29783629",
    "h_index": 21,
    "papers": 78
   },
   {
    "name": "Mark Epstein",
    "id": "2333514574",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Yun Zhu",
    "id": "2372315858",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Jianqiao Liu",
    "id": "2275992651",
    "h_index": 6,
    "papers": 20
   },
   {
    "name": "Yury Stuken",
    "id": "2373044640",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Ziyue Wang",
    "id": "2359126970",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "K. Misiunas",
    "id": "4654715",
    "h_index": 12,
    "papers": 19
   },
   {
    "name": "Dee Guo",
    "id": "2301410236",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "A. Gill",
    "id": "2285391319",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "A. Hartman",
    "id": "2275184113",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Zaid Nabulsi",
    "id": "2000791833",
    "h_index": 8,
    "papers": 16
   },
   {
    "name": "Au-rko Roy",
    "id": "2275277736",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Aleksandra Faust",
    "id": "2268757423",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Jason Riesa",
    "id": "2373039291",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Ben Withbroe",
    "id": "2373039165",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Mengchao Wang",
    "id": "2355403637",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Marco Tagliasacchi",
    "id": "2296783751",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Andreea Marzoca",
    "id": "2279548162",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "James Noraky",
    "id": "3875055",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "S. Toropov",
    "id": "93368078",
    "h_index": 4,
    "papers": 22
   },
   {
    "name": "Malika Mehrotra",
    "id": "2372684688",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Bahram Raad",
    "id": "2373037727",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "S. Deur",
    "id": "1394497938",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Steve Xu",
    "id": "2278975991",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Marianne Monteiro",
    "id": "2275188033",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Zhongru Wu",
    "id": "2372387862",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Yi Luan",
    "id": "2266466809",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Sam Ritter",
    "id": "2294720392",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Nick Li",
    "id": "2120050195",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Haavard Garnes",
    "id": "2373037626",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Yanzhang He",
    "id": "2145999837",
    "h_index": 14,
    "papers": 32
   },
   {
    "name": "Martin Zlocha",
    "id": "115571937",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Jifan Zhu",
    "id": "2402450621",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Matteo Hessel",
    "id": "39357484",
    "h_index": 25,
    "papers": 39
   },
   {
    "name": "W. Wu",
    "id": "2373550898",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Spandana Raj Babbula",
    "id": "2864359",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Chizuko Kawamoto",
    "id": "72645105",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Yuanzhen Li",
    "id": "2340566519",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Mehadi Hassen",
    "id": "36224485",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Yan Wang",
    "id": "2408624437",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Brian Wieder",
    "id": "2373044733",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Jim Freedman",
    "id": "2270009258",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Yin Zhang",
    "id": "2273604007",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Xinyi Bai",
    "id": "2325240045",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Tianli Yu",
    "id": "2374304849",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "D. Reitter",
    "id": "2257286979",
    "h_index": 6,
    "papers": 14
   },
   {
    "name": "XiangHai Sheng",
    "id": "2275154497",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Mateo Wirth",
    "id": "2275185968",
    "h_index": 8,
    "papers": 16
   },
   {
    "name": "Aditya Kini",
    "id": "2372474472",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "D. Damen",
    "id": "145089978",
    "h_index": 45,
    "papers": 201
   },
   {
    "name": "Mingcen Gao",
    "id": "2374314978",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Rachel Hornung",
    "id": "2275616325",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Michael Voznesensky",
    "id": "2297942646",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Brian Roark",
    "id": "2288362372",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Adhi Kuncoro",
    "id": "2410154738",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Yuxiang Zhou",
    "id": "2373556746",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Rushin Shah",
    "id": "2316587442",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Anthony Brohan",
    "id": "118025075",
    "h_index": 12,
    "papers": 19
   },
   {
    "name": "Kuan-Chia Chen",
    "id": "2411422220",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "James G Wendt",
    "id": "2372387556",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "David Rim",
    "id": "2274102773",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "P. Rubenstein",
    "id": "2249760524",
    "h_index": 6,
    "papers": 14
   },
   {
    "name": "Jonathan J. Halcrow",
    "id": "101461191",
    "h_index": 10,
    "papers": 26
   },
   {
    "name": "Michelle Liu",
    "id": "2336829596",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Ty Geri",
    "id": "2373037966",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Yunhsuan Sung",
    "id": "2294570198",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "J. Shapiro",
    "id": "2061221754",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Shaan Bijwadia",
    "id": "2189479741",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Chris Duvarney",
    "id": "2290852776",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "C. Sorokin",
    "id": "2275186804",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Paul Natsev",
    "id": "122704930",
    "h_index": 6,
    "papers": 14
   },
   {
    "name": "R. Ingle",
    "id": "2077591551",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Pramod Gupta",
    "id": "2087120906",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Youngill Maeng",
    "id": "2405355722",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Ndaba Ndebele",
    "id": "2365132971",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Kexin Zhu",
    "id": "2342510785",
    "h_index": 5,
    "papers": 24
   },
   {
    "name": "Valentin Anklin",
    "id": "2051891614",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Katherine Lee",
    "id": "2374238527",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Yuan Liu",
    "id": "2276035024",
    "h_index": 4,
    "papers": 18
   },
   {
    "name": "Yaroslav Akulov",
    "id": "2373038957",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Shaleen Gupta",
    "id": "2174538749",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Guolong Su",
    "id": "2275186285",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Flavien Prost",
    "id": "151502827",
    "h_index": 10,
    "papers": 18
   },
   {
    "name": "Tianlin Liu",
    "id": "2308056474",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "V. Kovalev",
    "id": "134839182",
    "h_index": 2,
    "papers": 32
   },
   {
    "name": "Pol Moreno",
    "id": "2258719067",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "M. Scholz",
    "id": "2251313434",
    "h_index": 5,
    "papers": 21
   },
   {
    "name": "Sam Redmond",
    "id": "48565485",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Zongwei Zhou",
    "id": "2344954690",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Alex Castro-Ros",
    "id": "2275175484",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Andr\u00e9 Susano Pinto",
    "id": "1809220",
    "h_index": 16,
    "papers": 23
   },
   {
    "name": "Dia Kharrat",
    "id": "2373038071",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Michal Yarom",
    "id": "2065978700",
    "h_index": 8,
    "papers": 15
   },
   {
    "name": "Rachel Saputro",
    "id": "2275182030",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Jannis Bulian",
    "id": "2362210",
    "h_index": 12,
    "papers": 20
   },
   {
    "name": "Ben Caine",
    "id": "2290484287",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Ji Liu",
    "id": "2275527831",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "A. Abdolmaleki",
    "id": "2799799",
    "h_index": 29,
    "papers": 95
   },
   {
    "name": "Shariq Iqbal",
    "id": "2286681855",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Tautvydas Misiunas",
    "id": "2303841988",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Mikhail Sirotenko",
    "id": "89903811",
    "h_index": 9,
    "papers": 12
   },
   {
    "name": "Shefali Garg",
    "id": "51007421",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "G. Bensky",
    "id": "3403192",
    "h_index": 11,
    "papers": 23
   },
   {
    "name": "Huan Gui",
    "id": "2062880513",
    "h_index": 15,
    "papers": 23
   },
   {
    "name": "Xuezhi Wang",
    "id": "2309895228",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Raphael Koster",
    "id": "2297844734",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Mike Bernico",
    "id": "2373037755",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Da Huang",
    "id": "2388655343",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "R. Thoppilan",
    "id": "9501591",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Trevor Cohn",
    "id": "2285651494",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Ben Golan",
    "id": "2373042782",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Wenlei Zhou",
    "id": "2303481222",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Andrew Rosenberg",
    "id": "2278792227",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Markus Freitag",
    "id": "2284596453",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Tynan Gangwani",
    "id": "2121253335",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "V. Tsang",
    "id": "2372698411",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Anand Shukla",
    "id": "2055375193",
    "h_index": 5,
    "papers": 15
   },
   {
    "name": "Xiaoqi Ren",
    "id": "2294342423",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Minh Giang",
    "id": "2275187490",
    "h_index": 8,
    "papers": 25
   },
   {
    "name": "C. Zou",
    "id": "113487968",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "A. Elisseeff",
    "id": "2288791213",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Charline Le Lan",
    "id": "153892869",
    "h_index": 16,
    "papers": 37
   },
   {
    "name": "Dheeru Dua",
    "id": "33546336",
    "h_index": 12,
    "papers": 28
   },
   {
    "name": "S. Lall",
    "id": "2372956012",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Pranav Shyam",
    "id": "67311962",
    "h_index": 18,
    "papers": 93
   },
   {
    "name": "Frankie Garcia",
    "id": "2292183049",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Sarah Nguyen",
    "id": "2135138509",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Michael Guzm\u00e1n",
    "id": "123816685",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "AJ Maschinot",
    "id": "2199119286",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "M. Maggioni",
    "id": "2090812426",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Ming-Wei Chang",
    "id": "2277809314",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Karol Gregor",
    "id": "2292141416",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "L. Weerts",
    "id": "101496339",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "K. Venkatesan",
    "id": "2285172466",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "Bogdan Damoc",
    "id": "2143374656",
    "h_index": 7,
    "papers": 17
   },
   {
    "name": "Leo Liu",
    "id": "2335562396",
    "h_index": 5,
    "papers": 16
   },
   {
    "name": "Jan Wassenberg",
    "id": "2366076736",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Lewis Ho",
    "id": "2353278718",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Becca Roelofs",
    "id": "2080504963",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Majid Hadian",
    "id": "2303849542",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Franccois-Xavier Aubet",
    "id": "150098869",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Yu Liang",
    "id": "2333305303",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Sami Lachgar",
    "id": "2301063374",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Danny Karmon",
    "id": "2322450976",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Yong Cheng",
    "id": "2275287219",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Amelio V'azquez-Reina",
    "id": "2373038653",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Angie Chen",
    "id": "2045298",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Zhuyun Dai",
    "id": "2287053658",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Andy Brock",
    "id": "2065040422",
    "h_index": 9,
    "papers": 17
   },
   {
    "name": "Shubham Agrawal",
    "id": "2275163561",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Chenxi Pang",
    "id": "2275183348",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "P. Garst",
    "id": "48012039",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Mariella Sanchez-Vargas",
    "id": "2373035864",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Ivor Rendulic",
    "id": "2389134",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Aditya Ayyar",
    "id": "2373043748",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Andrija Ravznatovi'c",
    "id": "2373037752",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Olivia Ma",
    "id": "2373040934",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Roopali Vij",
    "id": "71667257",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "N. Sharma",
    "id": "2113523986",
    "h_index": 8,
    "papers": 100
   },
   {
    "name": "Ashwin Balakrishna",
    "id": "2348256109",
    "h_index": 8,
    "papers": 11
   },
   {
    "name": "Bing Liu",
    "id": "2314014642",
    "h_index": 9,
    "papers": 46
   },
   {
    "name": "Ian Mackinnon",
    "id": "2290485798",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Sorin Baltateanu",
    "id": "2373039208",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Petra Poklukar",
    "id": "2295887372",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Gabriel Ibagon",
    "id": "67058278",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Colin Ji",
    "id": "2325104497",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Hongyan Jiao",
    "id": "2131571455",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Isaac Noble",
    "id": "2286155698",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Wojciech Stokowiec",
    "id": "3448463",
    "h_index": 9,
    "papers": 27
   },
   {
    "name": "Zhihao Li",
    "id": "2374280762",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Jeffrey Dean",
    "id": "2265529729",
    "h_index": 8,
    "papers": 16
   },
   {
    "name": "David Lindner",
    "id": "2292262356",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Mark Omernick",
    "id": "3175815",
    "h_index": 8,
    "papers": 29
   },
   {
    "name": "Kristen Chiafullo",
    "id": "2169579399",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "M. Dimarco",
    "id": "2283736215",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Vitor Rodrigues",
    "id": "2407031637",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Vittorio Selo",
    "id": "1394635460",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Garrett Honke",
    "id": "2301015067",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Xintian Wu",
    "id": "2352621664",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Wei He",
    "id": "2398719605",
    "h_index": 5,
    "papers": 17
   },
   {
    "name": "A. Hillier",
    "id": "102435329",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Anhad Mohananey",
    "id": "51228345",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Vihari Piratla",
    "id": "2748067",
    "h_index": 9,
    "papers": 28
   },
   {
    "name": "Chang Ye",
    "id": "2147352423",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Chase Malik",
    "id": "2372736440",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Sebastian Riedel",
    "id": "2287841795",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Samuel Albanie",
    "id": "2310231640",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Zi Yang",
    "id": "2288776189",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Kenny K. Vassigh",
    "id": "103257418",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Maria Bauz\u00e1",
    "id": "2284694443",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Sheng Li",
    "id": "2316974967",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Yiqing Tao",
    "id": "2163748478",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Nevan Wichers",
    "id": "50981270",
    "h_index": 10,
    "papers": 20
   },
   {
    "name": "Andrii Maksai",
    "id": "1954782",
    "h_index": 11,
    "papers": 24
   },
   {
    "name": "Abe Ittycheriah",
    "id": "3407108",
    "h_index": 9,
    "papers": 19
   },
   {
    "name": "Ross Mcilroy",
    "id": "2275176102",
    "h_index": 7,
    "papers": 14
   },
   {
    "name": "Bryan Seybold",
    "id": "2535887",
    "h_index": 11,
    "papers": 16
   },
   {
    "name": "Noah Goodman",
    "id": "2353279551",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Romina Datta",
    "id": "2275187147",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "S. Hernandez",
    "id": "2069845285",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Tianze Shi",
    "id": "2372754776",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Yony Kochinski",
    "id": "2368453221",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Anna Bulanova",
    "id": "2265580818",
    "h_index": 7,
    "papers": 18
   },
   {
    "name": "Ken Franko",
    "id": "2118834006",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Mikita Sazanovich",
    "id": "1750948530",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Nicholas Fitzgerald",
    "id": "2316578867",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Praneeth Kacham",
    "id": "2311506957",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Shubha Raghvendra",
    "id": "5002178",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Vincent J. Hellendoorn",
    "id": "2297847302",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Alexander Grushetsky",
    "id": "27693808",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Julian Salazar",
    "id": "2342275370",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "A. Lazaridou",
    "id": "2268870905",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Jason Chang",
    "id": "2278192190",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Jan-Thorsten Peter",
    "id": "2265752718",
    "h_index": 6,
    "papers": 14
   },
   {
    "name": "Sushant Kafle",
    "id": "31707321",
    "h_index": 11,
    "papers": 21
   },
   {
    "name": "Yann Dauphin",
    "id": "2285592910",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Abhishek Rao",
    "id": "1484043592",
    "h_index": 8,
    "papers": 25
   },
   {
    "name": "Filippo Graziano",
    "id": "2373038289",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Izhak Shafran",
    "id": "1697494",
    "h_index": 34,
    "papers": 127
   },
   {
    "name": "Yuguo Liao",
    "id": "2114119592",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Tianli Ding",
    "id": "95691186",
    "h_index": 15,
    "papers": 24
   },
   {
    "name": "Geng Yan",
    "id": "2239111166",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "Grace Chu",
    "id": "2321403253",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Z. Fu",
    "id": "2329139986",
    "h_index": 14,
    "papers": 150
   },
   {
    "name": "Vincent Roulet",
    "id": "2292406464",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Gabriel Rasskin",
    "id": "2303849717",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Duncan Williams",
    "id": "2109422326",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Shahar Drath",
    "id": "2316577678",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Alexander Mossin",
    "id": "1397332827",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Raphael Hoffmann",
    "id": "2275187974",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Jordi Orbay",
    "id": "2142424210",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Francesco Bertolini",
    "id": "2324801512",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "H. Sheftel",
    "id": "1926117",
    "h_index": 10,
    "papers": 19
   },
   {
    "name": "Justin Chiu",
    "id": "2273650801",
    "h_index": 7,
    "papers": 17
   },
   {
    "name": "Siyan Xue",
    "id": "2372281640",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Yuheng Kuang",
    "id": "2161342687",
    "h_index": 9,
    "papers": 14
   },
   {
    "name": "Ferjad Naeem",
    "id": "2372806568",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Swaroop Nath",
    "id": "2268673650",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "N. Nti",
    "id": "2203545488",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Phil Culliton",
    "id": "40579094",
    "h_index": 6,
    "papers": 16
   },
   {
    "name": "Kashyap Krishnakumar",
    "id": "2290485355",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "M. Isard",
    "id": "2090818",
    "h_index": 57,
    "papers": 125
   },
   {
    "name": "Pei Sun",
    "id": "2275825026",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Ayan Chakrabarti",
    "id": "2279822092",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Nathan Clement",
    "id": "2248816648",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Regev Cohen",
    "id": "2352230348",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Arissa Wongpanich",
    "id": "1470793133",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Gs Oh",
    "id": "153190862",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Ashwin Murthy",
    "id": "1381993027",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Hao Zheng",
    "id": "2397638311",
    "h_index": 3,
    "papers": 18
   },
   {
    "name": "Jessica B. Hamrick",
    "id": "2158860",
    "h_index": 25,
    "papers": 63
   },
   {
    "name": "Oskar Bunyan",
    "id": "2275177720",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Suhas Ganesh",
    "id": "150296012",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Nitish Gupta",
    "id": "2285178",
    "h_index": 14,
    "papers": 28
   },
   {
    "name": "Roy Frostig",
    "id": "2285543397",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "J. Michael Wieting",
    "id": "2250625143",
    "h_index": 6,
    "papers": 16
   },
   {
    "name": "Yury Malkov",
    "id": "2328086974",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Pierre Marcenac",
    "id": "2293723556",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Zhixin Lai",
    "id": "2372463128",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Xiaodan Tang",
    "id": "2368437773",
    "h_index": 2,
    "papers": 16
   },
   {
    "name": "Mohammad Saleh",
    "id": "2316576401",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "Fedir Zubach",
    "id": "2185410945",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Chinmay Kulkarni",
    "id": "2349238746",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Huanjie Zhou",
    "id": "2324896603",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Vicky Zayats",
    "id": "2303651624",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Nan Ding",
    "id": "2372361786",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "A. Tripathi",
    "id": "47458672",
    "h_index": 3,
    "papers": 14
   },
   {
    "name": "Arijit Pramanik",
    "id": "1469310857",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Patrik Zochbauer",
    "id": "2373035913",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "H. Ganapathy",
    "id": "2579743",
    "h_index": 14,
    "papers": 46
   },
   {
    "name": "Vedant Misra",
    "id": "40055795",
    "h_index": 13,
    "papers": 39
   },
   {
    "name": "Zach Behrman",
    "id": "2373039790",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "H. Vallet",
    "id": "2188954757",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Mingyang Zhang",
    "id": "2275284322",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "M. Sridhar",
    "id": "2353996818",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Ye Jin",
    "id": "2246107033",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Samuel Gehman",
    "id": "1962694751",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Mohammad Babaeizadeh",
    "id": "2112897310",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Siim P\u00f5der",
    "id": "2275190197",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Megha Goel",
    "id": "2275186741",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "D. Jain",
    "id": "2261050121",
    "h_index": 3,
    "papers": 23
   },
   {
    "name": "Tajwar Nasir",
    "id": "8353740",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Shubham Mittal",
    "id": "2304764843",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Tim Dozat",
    "id": "2285738836",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Diego Ardila",
    "id": "2239099139",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "A. Severyn",
    "id": "3091861",
    "h_index": 35,
    "papers": 81
   },
   {
    "name": "Fabio Pardo",
    "id": "2274107421",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Sammy Jerome",
    "id": "2287843663",
    "h_index": 6,
    "papers": 17
   },
   {
    "name": "Siyang Qin",
    "id": "2333872078",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Louis Rouillard",
    "id": "2351906592",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Amir Yazdanbakhsh",
    "id": "2303406300",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Zizhao Zhang",
    "id": "2128158461",
    "h_index": 18,
    "papers": 61
   },
   {
    "name": "Shivani Agrawal",
    "id": "2275163563",
    "h_index": 6,
    "papers": 17
   },
   {
    "name": "K. Shivakumar",
    "id": "2272718153",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Caden Lu",
    "id": "2373572201",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Praveen Kallakuri",
    "id": "2561675",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Rachita Chhaparia",
    "id": "2104677959",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "K. Rao",
    "id": "2287242220",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Charles Kwong",
    "id": "2373032749",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Anastasiia Fadeeva",
    "id": "2296784375",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "S. Nigam",
    "id": "2358757202",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Y. Virin",
    "id": "1395899773",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Yuan Zhang",
    "id": "2370738195",
    "h_index": 6,
    "papers": 18
   },
   {
    "name": "Balaji Venkatraman",
    "id": "2290488408",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Beliz Gunel",
    "id": "2120395428",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Marc Wilson",
    "id": "2324778722",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Huiyu Wang",
    "id": "2319175885",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Abhinav Gupta",
    "id": "2285425614",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Abhinav Gupta",
    "id": "48668899",
    "h_index": 9,
    "papers": 17
   },
   {
    "name": "Adrien Ali Taiga",
    "id": "1583061741",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Kareem Mohamed",
    "id": "2314113987",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Doug Fritz",
    "id": "2275187305",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Daniel Rodr\u00edguez",
    "id": "2374090499",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Z. Ghahramani",
    "id": "1983575",
    "h_index": 22,
    "papers": 61
   },
   {
    "name": "Harry Askham",
    "id": "2792386",
    "h_index": 9,
    "papers": 18
   },
   {
    "name": "Lior Belenki",
    "id": "2316558829",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "J. Zhao",
    "id": "2364080800",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Rahul Gupta",
    "id": "2374944073",
    "h_index": 4,
    "papers": 14
   },
   {
    "name": "Krzysztof Jastrzkebski",
    "id": "2373037837",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Takahiro Kosakai",
    "id": "48805178",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "K. Katircioglu",
    "id": "2897089",
    "h_index": 14,
    "papers": 48
   },
   {
    "name": "Jon Schneider",
    "id": "2284180258",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Rina Panigrahy",
    "id": "2259931036",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Konstantinos Bousmalis",
    "id": "2732737",
    "h_index": 21,
    "papers": 37
   },
   {
    "name": "Peter Grabowski",
    "id": "2275186011",
    "h_index": 5,
    "papers": 15
   },
   {
    "name": "Prajit Ramachandran",
    "id": "3377142",
    "h_index": 13,
    "papers": 17
   },
   {
    "name": "Chaitra Hegde",
    "id": "2269668438",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Mihaela Ro\u0219ca",
    "id": "2269541835",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Angelo Scorza Scarpati",
    "id": "2190750190",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "K. Axiotis",
    "id": "2372417127",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Ying Xu",
    "id": "2256016502",
    "h_index": 10,
    "papers": 25
   },
   {
    "name": "Zach Gleicher",
    "id": "2275185661",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "A. Michaely",
    "id": "9030348",
    "h_index": 8,
    "papers": 10
   },
   {
    "name": "Mandar Sharma",
    "id": "1928141171",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Sanil Jain",
    "id": "2321910347",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Christoph Hirnschall",
    "id": "9546819",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Tal Marian",
    "id": "32639838",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Xuhui Jia",
    "id": "2269764175",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "K. Mather",
    "id": "2262406802",
    "h_index": 2,
    "papers": 11
   },
   {
    "name": "Kilol Gupta",
    "id": "2280334518",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Linhai Qiu",
    "id": "2344051713",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Nigamaa Nayakanti",
    "id": "100557961",
    "h_index": 9,
    "papers": 14
   },
   {
    "name": "Lucian Ionita",
    "id": "2275190313",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Steven Zheng",
    "id": "2275574241",
    "h_index": 6,
    "papers": 17
   },
   {
    "name": "Lucia Loher",
    "id": "2275188993",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Kurt Shuster",
    "id": "35752280",
    "h_index": 28,
    "papers": 60
   },
   {
    "name": "Igor Petrovski",
    "id": "2308030897",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Roshan Sharma",
    "id": "2254027750",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "R. Chaabouni",
    "id": "1706980",
    "h_index": 17,
    "papers": 47
   },
   {
    "name": "A. Yeh",
    "id": "2373040340",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "J. An",
    "id": "2311770592",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Arushi Gupta",
    "id": "2255710048",
    "h_index": 4,
    "papers": 26
   },
   {
    "name": "S. Schwarcz",
    "id": "52023459",
    "h_index": 23,
    "papers": 247
   },
   {
    "name": "S. Ellis",
    "id": "2068670578",
    "h_index": 18,
    "papers": 87
   },
   {
    "name": "Sam Conway-Rahman",
    "id": "2373038265",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Javier Snaider",
    "id": "2265527968",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "A. Zhai",
    "id": "144069120",
    "h_index": 11,
    "papers": 27
   },
   {
    "name": "James Atwood",
    "id": "2308030098",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "D. Golovin",
    "id": "2316558767",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Liqian Peng",
    "id": "2325464488",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "I. Te",
    "id": "2373039713",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Vivian Xia",
    "id": "2373038192",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Salvatore Scellato",
    "id": "2098849619",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Mahan Malihi",
    "id": "1907200",
    "h_index": 7,
    "papers": 9
   },
   {
    "name": "Arthur Bravzinskas",
    "id": "2373037809",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Vl\u0103duc\u0103 Ion",
    "id": "117845979",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Younghoon Jun",
    "id": "2372778568",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "James Swirhun",
    "id": "2336819438",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Soroosh Mariooryad",
    "id": "2947439",
    "h_index": 18,
    "papers": 30
   },
   {
    "name": "Jiao Sun",
    "id": "2305741851",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Steve A. Chien",
    "id": "2257251957",
    "h_index": 4,
    "papers": 15
   },
   {
    "name": "Rey Coaguila",
    "id": "3391381",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "A. Brand",
    "id": "78456415",
    "h_index": 4,
    "papers": 28
   },
   {
    "name": "Yi Gao",
    "id": "2352066300",
    "h_index": 5,
    "papers": 15
   },
   {
    "name": "T. Kwiatkowski",
    "id": "2261492907",
    "h_index": 5,
    "papers": 18
   },
   {
    "name": "Roee Aharoni",
    "id": "2335771",
    "h_index": 26,
    "papers": 45
   },
   {
    "name": "Cheng-Chun Lee",
    "id": "2324808167",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Mislav vZani'c",
    "id": "2373039220",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Yichi Zhang",
    "id": "2287296296",
    "h_index": 8,
    "papers": 24
   },
   {
    "name": "D. Ethier",
    "id": "47820383",
    "h_index": 6,
    "papers": 16
   },
   {
    "name": "Vitaly Nikolaev",
    "id": "48942032",
    "h_index": 14,
    "papers": 27
   },
   {
    "name": "Pranav Ajit Nair",
    "id": "83623712",
    "h_index": 9,
    "papers": 28
   },
   {
    "name": "Yoav Ben Shalom",
    "id": "4008273",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "H. Fitoussi",
    "id": "146885676",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "J. Gupta",
    "id": "2285891933",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Hongbin Liu",
    "id": "2369852077",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Dee Cattle",
    "id": "2351906644",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Tolga Bolukbasi",
    "id": "2843215",
    "h_index": 17,
    "papers": 44
   },
   {
    "name": "Ben Murdoch",
    "id": "40154683",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Fantine Huot",
    "id": "2174667321",
    "h_index": 11,
    "papers": 26
   },
   {
    "name": "Yin Li",
    "id": "2354575439",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "C. Hahn",
    "id": "47809006",
    "h_index": 2,
    "papers": 12
   },
   {
    "name": "Urvashi Khandelwal",
    "id": "3030219",
    "h_index": 15,
    "papers": 32
   },
   {
    "name": "Frederik Benzing",
    "id": "69940384",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "Arthur Conmy",
    "id": "2131632310",
    "h_index": 21,
    "papers": 30
   },
   {
    "name": "Andrey E. Simanovsky",
    "id": "2225136651",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Franccoise Beaufays",
    "id": "146687622",
    "h_index": 10,
    "papers": 14
   },
   {
    "name": "E. Weinstein",
    "id": "40161292",
    "h_index": 18,
    "papers": 33
   },
   {
    "name": "Tongzhou Chen",
    "id": "2110557723",
    "h_index": 9,
    "papers": 20
   },
   {
    "name": "Luke Leonhard",
    "id": "2373035994",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "B. Ramabhadran",
    "id": "1720857",
    "h_index": 52,
    "papers": 284
   }
  ],
  "comment": "72 pages, 17 figures",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2507.06261v6",
  "pdf_url": "https://arxiv.org/pdf/2507.06261v6",
  "html_url": "https://arxiv.org/html/2507.06261v6",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.0
 },
 {
  "id": "2507.05331",
  "slug": "a-careful-examination-of-large-behavior-models-for-multitask-dexterous",
  "title": "A Careful Examination of Large Behavior Models for Multitask Dexterous Manipulation",
  "abstract": "Robot manipulation has seen tremendous progress in recent years, with imitation learning policies enabling successful performance of dexterous and hard-to-model tasks. Concurrently, scaling data and model size has led to the development of capable language and vision foundation models, motivating large-scale efforts to create general-purpose robot foundation models. While these models have garnered significant enthusiasm and investment, meaningful evaluation of real-world performance remains a challenge, limiting both the pace of development and inhibiting a nuanced understanding of current capabilities. In this paper, we rigorously evaluate multitask robot manipulation policies, referred to as Large Behavior Models (LBMs), by extending the Diffusion Policy paradigm across a corpus of simulated and real-world robot data. We propose and validate an evaluation pipeline to rigorously analyze the capabilities of these models with statistical confidence. We compare against single-task baselines through blind, randomized trials in a controlled setting, using both simulation and real-world experiments. We find that multi-task pretraining makes the policies more successful and robust, and enables teaching complex new tasks more quickly, using a fraction of the data when compared to single-task baselines. Moreover, performance predictably increases as pretraining scale and diversity grows. Project page: https://toyotaresearchinstitute.github.io/lbm1/",
  "published": "2025-07-07",
  "updated": "2025-07-07",
  "year": "2025",
  "authors": [
   " TRI LBM Team",
   "Jose Barreiros",
   "Andrew Beaulieu",
   "Aditya Bhat",
   "Rick Cory",
   "Eric Cousineau",
   "Hongkai Dai",
   "Ching-Hsin Fang",
   "Kunimatsu Hashimoto",
   "Muhammad Zubair Irshad",
   "Masha Itkina",
   "Naveen Kuppuswamy",
   "Kuan-Hui Lee",
   "Katherine Liu",
   "Dale McConachie",
   "Ian McMahon",
   "Haruki Nishimura",
   "Calder Phillips-Grafflin",
   "Charles Richter",
   "Paarth Shah",
   "Krishnan Srinivasan",
   "Blake Wulfe",
   "Chen Xu",
   "Mengchao Zhang",
   "Alex Alspach",
   "Maya Angeles",
   "Kushal Arora",
   "Vitor Campagnolo Guizilini",
   "Alejandro Castro",
   "Dian Chen",
   "Ting-Sheng Chu",
   "Sam Creasey",
   "Sean Curtis",
   "Richard Denitto",
   "Emma Dixon",
   "Eric Dusel",
   "Matthew Ferreira",
   "Aimee Goncalves",
   "Grant Gould",
   "Damrong Guoy",
   "Swati Gupta",
   "Xuchen Han",
   "Kyle Hatch",
   "Brendan Hathaway",
   "Allison Henry",
   "Hillel Hochsztein",
   "Phoebe Horgan",
   "Shun Iwase",
   "Donovon Jackson",
   "Siddharth Karamcheti",
   "Sedrick Keh",
   "Joseph Masterjohn",
   "Jean Mercat",
   "Patrick Miller",
   "Paul Mitiguy",
   "Tony Nguyen",
   "Jeremy Nimmer",
   "Yuki Noguchi",
   "Reko Ong",
   "Aykut Onol",
   "Owen Pfannenstiehl",
   "Richard Poyner",
   "Leticia Priebe Mendes Rocha",
   "Gordon Richardson",
   "Christopher Rodriguez",
   "Derick Seale",
   "Michael Sherman",
   "Mariah Smith-Jones",
   "David Tago",
   "Pavel Tokmakov",
   "Matthew Tran",
   "Basile Van Hoorick",
   "Igor Vasiljevic",
   "Sergey Zakharov",
   "Mark Zolotas",
   "Rares Ambrus",
   "Kerri Fetzer-Borelli",
   "Benjamin Burchfiel",
   "Hadas Kress-Gazit",
   "Siyuan Feng",
   "Stacie Ford",
   "Russ Tedrake"
  ],
  "author_count": 82,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 162,
  "influential_citations": 7,
  "tldr": "This paper rigorously evaluates multitask robot manipulation policies, referred to as Large Behavior Models (LBMs), by extending the Diffusion Policy paradigm across a corpus of simulated and real-world robot data, and finds that multi-task pretraining makes the policies more successful and robust, and enables teaching complex new tasks more quickly.",
  "doi": "10.48550/arXiv.2507.05331",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Tri Lbm Team",
    "id": "2372326305",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Jose A. Barreiros",
    "id": "2260336204",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Andrew Beaulieu",
    "id": "82375749",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Aditya Bhat",
    "id": "2372269786",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Rick Cory",
    "id": "2372917669",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Eric Cousineau",
    "id": "2090529",
    "h_index": 12,
    "papers": 16
   },
   {
    "name": "Hongkai Dai",
    "id": "2319811942",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Ching Fang",
    "id": "2181687798",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Kunimatsu Hashimoto",
    "id": "2321435610",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Muhammad Zubair Irshad",
    "id": "147495445",
    "h_index": 17,
    "papers": 37
   },
   {
    "name": "Masha Itkina",
    "id": "30112153",
    "h_index": 16,
    "papers": 31
   },
   {
    "name": "Naveen Kuppuswamy",
    "id": "2275353318",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Kuan-Hui Lee",
    "id": "2402903611",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Katherine Liu",
    "id": "2268798781",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "D. Mcconachie",
    "id": "9926216",
    "h_index": 9,
    "papers": 12
   },
   {
    "name": "I. McMahon",
    "id": "2034746",
    "h_index": 4,
    "papers": 20
   },
   {
    "name": "Haruki Nishimura",
    "id": "2349476640",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Calder Phillips-Grafflin",
    "id": "1405637919",
    "h_index": 8,
    "papers": 16
   },
   {
    "name": "Charles Richter",
    "id": "145160705",
    "h_index": 9,
    "papers": 15
   },
   {
    "name": "Paarth Shah",
    "id": "2300432426",
    "h_index": 7,
    "papers": 15
   },
   {
    "name": "K. Srinivasan",
    "id": "2093939303",
    "h_index": 16,
    "papers": 29
   },
   {
    "name": "Blake Wulfe",
    "id": "9414028",
    "h_index": 12,
    "papers": 17
   },
   {
    "name": "Cheng Xu",
    "id": "2334916617",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Mengchao Zhang",
    "id": "2260345497",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "A. Alspach",
    "id": "2980876",
    "h_index": 15,
    "papers": 25
   },
   {
    "name": "Maya Angeles",
    "id": "2372919634",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "K. Arora",
    "id": "2284685268",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "V. Guizilini",
    "id": "2171907555",
    "h_index": 11,
    "papers": 26
   },
   {
    "name": "Alejandro M. Castro",
    "id": "2323354847",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Dian Chen",
    "id": "2281177642",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Ting-Sheng Chu",
    "id": "3381208",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "S. Creasey",
    "id": "1620487062",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Sean Curtis",
    "id": "144564863",
    "h_index": 23,
    "papers": 35
   },
   {
    "name": "Richard Denitto",
    "id": "2372922391",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Emma Dixon",
    "id": "2349538819",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Eric Dusel",
    "id": "2181797508",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "M. Ferreira",
    "id": "2107623258",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Aimee Goncalves",
    "id": "2373647794",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Grant E. Gould",
    "id": "152564676",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Damrong Guoy",
    "id": "1790468",
    "h_index": 14,
    "papers": 23
   },
   {
    "name": "Swati Gupta",
    "id": "2373581765",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Xuchen Han",
    "id": "2373052843",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Kyle Hatch",
    "id": "2328014254",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Brendan Hathaway",
    "id": "2372917723",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "A. Henry",
    "id": "2321766255",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Hillel Hochsztein",
    "id": "2163784051",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Phoebe Horgan",
    "id": "2321406547",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Shun Iwase",
    "id": "2359630378",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Donovon Jackson",
    "id": "2292188317",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Siddharth Karamcheti",
    "id": "10737060",
    "h_index": 21,
    "papers": 38
   },
   {
    "name": "Sedrick Scott Keh",
    "id": "150299584",
    "h_index": 9,
    "papers": 26
   },
   {
    "name": "Joseph Masterjohn",
    "id": "2038107",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Jean-Pierre Mercat",
    "id": "72847120",
    "h_index": 10,
    "papers": 33
   },
   {
    "name": "Patrick Miller",
    "id": "2292161974",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "P. Mitiguy",
    "id": "3292997",
    "h_index": 6,
    "papers": 16
   },
   {
    "name": "Tony Nguyen",
    "id": "2349547163",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Jeremy W. Nimmer",
    "id": "1706891",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Yuki Noguchi",
    "id": "2292782886",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Reko Ong",
    "id": "2372510057",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "A. Onol",
    "id": "2782339",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Owen Pfannenstiehl",
    "id": "2372918868",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "R. Poyner",
    "id": "80666264",
    "h_index": 5,
    "papers": 17
   },
   {
    "name": "L. Rocha",
    "id": "2373650656",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Gordon Richardson",
    "id": "2321409651",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Christopher Rodriguez",
    "id": "2349602626",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Derick Seale",
    "id": "2292183089",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Michael Sherman",
    "id": "2106101345",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Mariah Smith-Jones",
    "id": "2372918819",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "David Tago",
    "id": "2372918858",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "P. Tokmakov",
    "id": "2931554",
    "h_index": 24,
    "papers": 72
   },
   {
    "name": "M. Tran",
    "id": "2054039152",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Basile Van Hoorick",
    "id": "1470838102",
    "h_index": 8,
    "papers": 16
   },
   {
    "name": "Igor Vasiljevic",
    "id": "2291068185",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Sergey Zakharov",
    "id": "2331626375",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Mark Zolotas",
    "id": "10708100",
    "h_index": 7,
    "papers": 29
   },
   {
    "name": "Rares Ambrus",
    "id": "1829964",
    "h_index": 32,
    "papers": 90
   },
   {
    "name": "Kerri Fetzer-Borelli",
    "id": "2372917572",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Benjamin Burchfiel",
    "id": "2319412766",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "H. Kress-Gazit",
    "id": "1387890355",
    "h_index": 35,
    "papers": 154
   },
   {
    "name": "Siyuan Feng",
    "id": "2284620540",
    "h_index": 8,
    "papers": 10
   },
   {
    "name": "Stacie Ford",
    "id": "2372401572",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Russ Tedrake",
    "id": "2263905014",
    "h_index": 14,
    "papers": 36
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "imitation-diffusion",
   "foundation-pretraining"
  ],
  "orgs": [
   "Toyota Research Institute"
  ],
  "abs_url": "https://arxiv.org/abs/2507.05331v1",
  "pdf_url": "https://arxiv.org/pdf/2507.05331v1",
  "html_url": "https://arxiv.org/html/2507.05331v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.71
 },
 {
  "id": "2507.04447",
  "slug": "dreamvla-a-vision-language-action-model-dreamed-with-comprehensive-wor",
  "title": "DreamVLA: A Vision-Language-Action Model Dreamed with Comprehensive World Knowledge",
  "abstract": "Recent advances in vision-language-action (VLA) models have shown promise in integrating image generation with action prediction to improve generalization and reasoning in robot manipulation. However, existing methods are limited to challenging image-based forecasting, which suffers from redundant information and lacks comprehensive and critical world knowledge, including dynamic, spatial and semantic information. To address these limitations, we propose DreamVLA, a novel VLA framework that integrates comprehensive world knowledge forecasting to enable inverse dynamics modeling, thereby establishing a perception-prediction-action loop for manipulation tasks. Specifically, DreamVLA introduces a dynamic-region-guided world knowledge prediction, integrated with the spatial and semantic cues, which provide compact yet comprehensive representations for action planning. This design aligns with how humans interact with the world by first forming abstract multimodal reasoning chains before acting. To mitigate interference among the dynamic, spatial and semantic information during training, we adopt a block-wise structured attention mechanism that masks their mutual attention, preventing information leakage and keeping each representation clean and disentangled. Moreover, to model the conditional distribution over future actions, we employ a diffusion-based transformer that disentangles action representations from shared latent features. Extensive experiments on both real-world and simulation environments demonstrate that DreamVLA achieves 76.7% success rate on real robot tasks and 4.44 average length on the CALVIN ABC-D benchmarks.",
  "published": "2025-07-06",
  "updated": "2025-08-26",
  "year": "2025",
  "authors": [
   "Wenyao Zhang",
   "Hongsi Liu",
   "Zekun Qi",
   "Yunnan Wang",
   "Xinqiang Yu",
   "Jiazhao Zhang",
   "Runpei Dong",
   "Jiawei He",
   "Fan Lu",
   "He Wang",
   "Zhizheng Zhang",
   "Li Yi",
   "Wenjun Zeng",
   "Xin Jin"
  ],
  "author_count": 14,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 200,
  "influential_citations": 19,
  "tldr": "This work proposes DreamVLA, a novel VLA framework that integrates comprehensive world knowledge forecasting to enable inverse dynamics modeling, thereby establishing a perception-prediction-action loop for manipulation tasks and adopting a block-wise structured attention mechanism that masks their mutual attention, preventing information leakage and keeping each representation clean and disentangled.",
  "doi": "10.48550/arXiv.2507.04447",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Wenyao Zhang",
    "id": "2282545418",
    "h_index": 10,
    "papers": 26
   },
   {
    "name": "Hongsi Liu",
    "id": "2331868768",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "Zekun Qi",
    "id": "3424017",
    "h_index": 13,
    "papers": 14
   },
   {
    "name": "Yunnan Wang",
    "id": "2314085002",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Xinqiang Yu",
    "id": "2328936584",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Jiazhao Zhang",
    "id": "2107990526",
    "h_index": 19,
    "papers": 43
   },
   {
    "name": "Runpei Dong",
    "id": "2056965063",
    "h_index": 17,
    "papers": 27
   },
   {
    "name": "Jiawei He",
    "id": "2153103015",
    "h_index": 21,
    "papers": 56
   },
   {
    "name": "He Wang",
    "id": "2334325487",
    "h_index": 10,
    "papers": 21
   },
   {
    "name": "Zhizheng Zhang",
    "id": "2287041015",
    "h_index": 10,
    "papers": 29
   },
   {
    "name": "Li Yi",
    "id": "2242612318",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Wenjun Zeng",
    "id": "2247938835",
    "h_index": 13,
    "papers": 69
   },
   {
    "name": "Xin Jin",
    "id": "2346309936",
    "h_index": 5,
    "papers": 7
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2507.04447v3",
  "pdf_url": "https://arxiv.org/pdf/2507.04447v3",
  "html_url": "https://arxiv.org/html/2507.04447v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.8
 },
 {
  "id": "2507.03745",
  "slug": "streamdit-real-time-streaming-text-to-video-generation",
  "title": "StreamDiT: Real-Time Streaming Text-to-Video Generation",
  "abstract": "Recently, great progress has been achieved in text-to-video (T2V) generation by scaling transformer-based diffusion models to billions of parameters, which can generate high-quality videos. However, existing models typically produce only short clips offline, restricting their use cases in interactive and real-time applications. This paper addresses these challenges by proposing StreamDiT, a streaming video generation model. StreamDiT training is based on flow matching by adding a moving buffer. We design mixed training with different partitioning schemes of buffered frames to boost both content consistency and visual quality. StreamDiT modeling is based on adaLN DiT with varying time embedding and window attention. To practice the proposed method, we train a StreamDiT model with 4B parameters. In addition, we propose a multistep distillation method tailored for StreamDiT. Sampling distillation is performed in each segment of a chosen partitioning scheme. After distillation, the total number of function evaluations (NFEs) is reduced to the number of chunks in a buffer. Finally, our distilled model reaches real-time performance at 16 FPS on one GPU, which can generate video streams at 512p resolution. We evaluate our method through both quantitative metrics and human evaluation. Our model enables real-time applications, e.g. streaming generation, interactive generation, and video-to-video. We provide video results and more examples in our project website: https://cumulo-autumn.github.io/StreamDiT/",
  "published": "2025-07-04",
  "updated": "2026-03-27",
  "year": "2025",
  "authors": [
   "Akio Kodaira",
   "Tingbo Hou",
   "Ji Hou",
   "Markos Georgopoulos",
   "Felix Juefei-Xu",
   "Masayoshi Tomizuka",
   "Yue Zhao"
  ],
  "author_count": 7,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.LG",
   "eess.IV"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR 2026",
  "venue_source": "arxiv-comment",
  "citations": 40,
  "influential_citations": 2,
  "tldr": "This paper proposes StreamDiT, a streaming video generation model based on adaLN DiT with varying time embedding and window attention that enables real-time applications, e.g. streaming generation, interactive generation, and video-to-video.",
  "doi": "10.48550/arXiv.2507.03745",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Akio Kodaira",
    "id": "2275352414",
    "h_index": 6,
    "papers": 6
   },
   {
    "name": "Tingbo Hou",
    "id": "2326294705",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Ji Hou",
    "id": "2249723114",
    "h_index": 11,
    "papers": 22
   },
   {
    "name": "Masayoshi Tomizuka",
    "id": "2322446616",
    "h_index": 3,
    "papers": 15
   },
   {
    "name": "Yue Zhao",
    "id": "2270809396",
    "h_index": 2,
    "papers": 5
   }
  ],
  "comment": "CVPR 2026",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2507.03745v4",
  "pdf_url": "https://arxiv.org/pdf/2507.03745v4",
  "html_url": "https://arxiv.org/html/2507.03745v4",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.11
 },
 {
  "id": "2507.00990",
  "slug": "robotic-manipulation-by-imitating-generated-videos-without-physical-de",
  "title": "Robotic Manipulation by Imitating Generated Videos Without Physical Demonstrations",
  "abstract": "This work introduces Robots Imitating Generated Videos (RIGVid), a system that enables robots to perform complex manipulation tasks--such as pouring, wiping, and mixing--purely by imitating AI-generated videos, without requiring any physical demonstrations or robot-specific training. Given a language command and an initial scene image, a video diffusion model generates potential demonstration videos, and a vision-language model (VLM) automatically filters out results that do not follow the command. A 6D pose tracker then extracts object trajectories from the video, and the trajectories are retargeted to the robot in an embodiment-agnostic fashion. Through extensive real-world evaluations, we show that filtered generated videos are as effective as real demonstrations, and that performance improves with generation quality. We also show that relying on generated videos outperforms more compact alternatives such as keypoint prediction using VLMs, and that strong 6D pose tracking outperforms other ways to extract trajectories, such as dense feature point tracking. These findings suggest that videos produced by a state-of-the-art off-the-shelf model can offer an effective source of supervision for robotic manipulation.",
  "published": "2025-07-01",
  "updated": "2026-05-13",
  "year": "2025",
  "authors": [
   "Shivansh Patel",
   "Shraddhaa Mohan",
   "Hanlin Mai",
   "Unnat Jain",
   "Svetlana Lazebnik",
   "Yunzhu Li"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "ICLR 2026",
  "venue_source": "arxiv-comment",
  "citations": 43,
  "influential_citations": 6,
  "tldr": "",
  "doi": "10.48550/arXiv.2507.00990",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Shivansh Patel",
    "id": "152264213",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Shraddhaa Mohan",
    "id": "2280706631",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Hanlin Mai",
    "id": "2268673391",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Unnat Jain",
    "id": "10680632",
    "h_index": 23,
    "papers": 33
   },
   {
    "name": "Svetlana Lazebnik",
    "id": "2267723274",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Yunzhu Li",
    "id": "2319504878",
    "h_index": 2,
    "papers": 5
   }
  ],
  "comment": "In ICLR 2026. Website: https://rigvid-robot.github.io/",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2507.00990v3",
  "pdf_url": "https://arxiv.org/pdf/2507.00990v3",
  "html_url": "https://arxiv.org/html/2507.00990v3",
  "code_url": "https://rigvid-robot.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.14
 },
 {
  "id": "2506.21552",
  "slug": "whole-body-conditioned-egocentric-video-prediction",
  "title": "Whole-Body Conditioned Egocentric Video Prediction",
  "abstract": "We train models to Predict Ego-centric Video from human Actions (PEVA), given the past video and an action represented by the relative 3D body pose. By conditioning on kinematic pose trajectories, structured by the joint hierarchy of the body, our model learns to simulate how physical human actions shape the environment from a first-person point of view. We train an auto-regressive conditional diffusion transformer on Nymeria, a large-scale dataset of real-world egocentric video and body pose capture. We further design a hierarchical evaluation protocol with increasingly challenging tasks, enabling a comprehensive analysis of the model's embodied prediction and control abilities. Our work represents an initial attempt to tackle the challenges of modeling complex real-world environments and embodied agent behaviors with video prediction from the perspective of a human.",
  "published": "2025-06-26",
  "updated": "2025-06-26",
  "year": "2025",
  "authors": [
   "Yutong Bai",
   "Danny Tran",
   "Amir Bar",
   "Yann LeCun",
   "Trevor Darrell",
   "Jitendra Malik"
  ],
  "author_count": 6,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.LG",
   "cs.MM",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 32,
  "influential_citations": 2,
  "tldr": "This work trains models to Predict Ego-centric Video from human Actions (PEVA), given the past video and an action represented by the relative 3D body pose, and designs a hierarchical evaluation protocol with increasingly challenging tasks, enabling a comprehensive analysis of the model's embodied prediction and control abilities.",
  "doi": "10.48550/arXiv.2506.21552",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yutong Bai",
    "id": "2349426965",
    "h_index": 6,
    "papers": 20
   },
   {
    "name": "Danny Tran",
    "id": "2296703884",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Amir Bar",
    "id": "2063958674",
    "h_index": 13,
    "papers": 23
   },
   {
    "name": "Yann LeCun",
    "id": "2252537770",
    "h_index": 8,
    "papers": 19
   },
   {
    "name": "Trevor Darrell",
    "id": "2295666869",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "Jitendra Malik",
    "id": "2286436946",
    "h_index": 2,
    "papers": 7
   }
  ],
  "comment": "Project Page: https://dannytran123.github.io/PEVA",
  "topics": [
   "world-models",
   "humanoids",
   "egocentric-data",
   "sim2real"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2506.21552v1",
  "pdf_url": "https://arxiv.org/pdf/2506.21552v1",
  "html_url": "https://arxiv.org/html/2506.21552v1",
  "code_url": "https://dannytran123.github.io/PEVA",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.02
 },
 {
  "id": "2506.20668",
  "slug": "demodiffusion-one-shot-human-imitation-using-pre-trained-diffusion-pol",
  "title": "DemoDiffusion: One-Shot Human Imitation using pre-trained Diffusion Policy",
  "abstract": "We propose DemoDiffusion, a simple method for enabling robots to perform manipulation tasks by imitating a single human demonstration, without requiring task-specific training or paired human-robot data. Our approach is based on two insights. First, the hand motion in a human demonstration provides a useful prior for the robot's end-effector trajectory, which we can convert into a rough open-loop robot motion trajectory via kinematic retargeting. Second, while this retargeted motion captures the overall structure of the task, it may not align well with plausible robot actions in-context. To address this, we leverage a pre-trained generalist diffusion policy to modify the trajectory, ensuring it both follows the human motion and remains within the distribution of plausible robot actions. Unlike approaches based on online reinforcement learning or paired human-robot data, our method enables robust adaptation to new tasks and scenes with minimal effort. In real-world experiments across 8 diverse manipulation tasks, DemoDiffusion achieves 83.8\\% average success rate, compared to 13.8\\% for the pre-trained policy and 52.5\\% for kinematic retargeting, succeeding even on tasks where the pre-trained generalist policy fails entirely. Project page: https://demodiffusion.github.io/",
  "published": "2025-06-25",
  "updated": "2026-06-12",
  "year": "2025",
  "authors": [
   "Sungjae Park",
   "Homanga Bharadhwaj",
   "Shubham Tulsiani"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA 2026",
  "venue_source": "arxiv-comment",
  "citations": 17,
  "influential_citations": 2,
  "tldr": "",
  "doi": "10.48550/arXiv.2506.20668",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Sungjae Park",
    "id": "2371412504",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Homanga Bharadhwaj",
    "id": "51113848",
    "h_index": 23,
    "papers": 59
   },
   {
    "name": "Shubham Tulsiani",
    "id": "2757335",
    "h_index": 45,
    "papers": 98
   }
  ],
  "comment": "11 pages. Published at ICRA 2026",
  "topics": [
   "egocentric-data",
   "imitation-diffusion",
   "rl-control",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2506.20668v3",
  "pdf_url": "https://arxiv.org/pdf/2506.20668v3",
  "html_url": "https://arxiv.org/html/2506.20668v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.76
 },
 {
  "id": "2506.19851",
  "slug": "animax-animating-the-inanimate-in-3d-with-joint-video-pose-diffusion-m",
  "title": "AnimaX: Animating the Inanimate in 3D with Joint Video-Pose Diffusion Models",
  "abstract": "We present AnimaX, a feed-forward 3D animation framework that bridges the motion priors of video diffusion models with the controllable structure of skeleton-based animation. Traditional motion synthesis methods are either restricted to fixed skeletal topologies or require costly optimization in high-dimensional deformation spaces. In contrast, AnimaX effectively transfers video-based motion knowledge to the 3D domain, supporting diverse articulated meshes with arbitrary skeletons. Our method represents 3D motion as multi-view, multi-frame 2D pose maps, and enables joint video-pose diffusion conditioned on template renderings and a textual motion prompt. We introduce shared positional encodings and modality-aware embeddings to ensure spatial-temporal alignment between video and pose sequences, effectively transferring video priors to motion generation task. The resulting multi-view pose sequences are triangulated into 3D joint positions and converted into mesh animation via inverse kinematics. Trained on a newly curated dataset of 160,000 rigged sequences, AnimaX achieves state-of-the-art results on VBench in generalization, motion fidelity, and efficiency, offering a scalable solution for category-agnostic 3D animation. Project page: \\href{https://anima-x.github.io/}{https://anima-x.github.io/}.",
  "published": "2025-06-24",
  "updated": "2025-06-24",
  "year": "2025",
  "authors": [
   "Zehuan Huang",
   "Haoran Feng",
   "Yangtian Sun",
   "Yuanchen Guo",
   "Yanpei Cao",
   "Lu Sheng"
  ],
  "author_count": 6,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "SIGGRAPH",
  "venue_source": "semantic-scholar",
  "citations": 17,
  "influential_citations": 1,
  "tldr": "AnimaX is presented, a feed-forward 3D animation framework that bridges the motion priors of video diffusion models with the controllable structure of skeleton-based animation, offering a scalable solution for category-agnostic 3D animation.",
  "doi": "10.1145/3757377.3763885",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zehuan Huang",
    "id": "2273905755",
    "h_index": 11,
    "papers": 24
   },
   {
    "name": "Hao-li Feng",
    "id": "2335459729",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Yang-Tian Sun",
    "id": "2371476170",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Yuan-Chen Guo",
    "id": "2330513790",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Yan-Pei Cao",
    "id": "2276490175",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Lu Sheng",
    "id": "2297850106",
    "h_index": 7,
    "papers": 16
   }
  ],
  "comment": "Project page: https://anima-x.github.io/",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2506.19851v1",
  "pdf_url": "https://arxiv.org/pdf/2506.19851v1",
  "html_url": "https://arxiv.org/html/2506.19851v1",
  "code_url": "https://anima-x.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.76
 },
 {
  "id": "2506.18901",
  "slug": "from-virtual-games-to-real-world-play",
  "title": "From Virtual Games to Real-World Play",
  "abstract": "We introduce RealPlay, a neural network-based real-world game engine that enables interactive video generation from user control signals. Unlike prior works focused on game-style visuals, RealPlay aims to produce photorealistic, temporally consistent video sequences that resemble real-world footage. It operates in an interactive loop: users observe a generated scene, issue a control command, and receive a short video chunk in response. To enable such realistic and responsive generation, we address key challenges including iterative chunk-wise prediction for low-latency feedback, temporal consistency across iterations, and accurate control response. RealPlay is trained on a combination of labeled game data and unlabeled real-world videos, without requiring real-world action annotations. Notably, we observe two forms of generalization: (1) control transfer-RealPlay effectively maps control signals from virtual to real-world scenarios; and (2) entity transfer-although training labels originate solely from a car racing game, RealPlay generalizes to control diverse real-world entities, including bicycles and pedestrians, beyond vehicles. Project page can be found: https://wenqsun.github.io/RealPlay/",
  "published": "2025-06-23",
  "updated": "2025-06-23",
  "year": "2025",
  "authors": [
   "Wenqiang Sun",
   "Fangyun Wei",
   "Jinjing Zhao",
   "Xi Chen",
   "Zilong Chen",
   "Hongyang Zhang",
   "Jun Zhang",
   "Yan Lu"
  ],
  "author_count": 8,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 9,
  "influential_citations": 0,
  "tldr": "RealPlay aims to produce photorealistic, temporally consistent video sequences that resemble real-world footage, and addresses key challenges including iterative chunk-wise prediction for low-latency feedback, temporal consistency across iterations, and accurate control response.",
  "doi": "10.48550/arXiv.2506.18901",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Wenqiang Sun",
    "id": "2215875229",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "Fangyun Wei",
    "id": "2239197291",
    "h_index": 11,
    "papers": 15
   },
   {
    "name": "Jinjing Zhao",
    "id": "2256929424",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Xi Chen",
    "id": "2333485561",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Zilong Chen",
    "id": "2389218871",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Hongyang Zhang",
    "id": "2281685288",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Jun Zhang",
    "id": "2304515490",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Yan Lu",
    "id": "2333479454",
    "h_index": 4,
    "papers": 7
   }
  ],
  "comment": "Project page: https://wenqsun.github.io/RealPlay/",
  "topics": [
   "data-teleop",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2506.18901v1",
  "pdf_url": "https://arxiv.org/pdf/2506.18901v1",
  "html_url": "https://arxiv.org/html/2506.18901v1",
  "code_url": "https://wenqsun.github.io/RealPlay/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.0
 },
 {
  "id": "2506.18890",
  "slug": "4d-lrm-large-space-time-reconstruction-model-from-and-to-any-view-at-a",
  "title": "4D-LRM: Large Space-Time Reconstruction Model From and To Any View at Any Time",
  "abstract": "Can we scale 4D pretraining to learn general space-time representations that reconstruct an object from a few views at some times to any view at any time? We provide an affirmative answer with 4D-LRM, the first large-scale 4D reconstruction model that takes input from unconstrained views and timestamps and renders arbitrary novel view-time combinations. Unlike prior 4D approaches, e.g., optimization-based, geometry-based, or generative, that struggle with efficiency, generalization, or faithfulness, 4D-LRM learns a unified space-time representation and directly predicts per-pixel 4D Gaussian primitives from posed image tokens across time, enabling fast, high-quality rendering at, in principle, infinite frame rate. Our results demonstrate that scaling spatiotemporal pretraining enables accurate and efficient 4D reconstruction. We show that 4D-LRM generalizes to novel objects, interpolates across time, and handles diverse camera setups. It reconstructs 24-frame sequences in one forward pass with less than 1.5 seconds on a single A100 GPU.",
  "published": "2025-06-23",
  "updated": "2025-06-23",
  "year": "2025",
  "authors": [
   "Ziqiao Ma",
   "Xuweiyi Chen",
   "Shoubin Yu",
   "Sai Bi",
   "Kai Zhang",
   "Chen Ziwen",
   "Sihan Xu",
   "Jianing Yang",
   "Zexiang Xu",
   "Kalyan Sunkavalli",
   "Mohit Bansal",
   "Joyce Chai",
   "Hao Tan"
  ],
  "author_count": 13,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 10,
  "influential_citations": 0,
  "tldr": "4D-LRM is the first large-scale 4D reconstruction model that takes input from unconstrained views and timestamps and renders arbitrary novel view-time combinations, enabling fast, high-quality rendering at, in principle, infinite frame rate.",
  "doi": "10.48550/arXiv.2506.18890",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ziqiao Ma",
    "id": "2151006930",
    "h_index": 17,
    "papers": 39
   },
   {
    "name": "Xuweiyi Chen",
    "id": "2290026490",
    "h_index": 8,
    "papers": 16
   },
   {
    "name": "Shoubin Yu",
    "id": "2164249715",
    "h_index": 12,
    "papers": 29
   },
   {
    "name": "Sai Bi",
    "id": "2265648463",
    "h_index": 18,
    "papers": 33
   },
   {
    "name": "Kai Zhang",
    "id": "2265847064",
    "h_index": 18,
    "papers": 31
   },
   {
    "name": "Ziwen Chen",
    "id": "2332538419",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Sihan Xu",
    "id": "2261187041",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Jianing Yang",
    "id": "2344742509",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Zexiang Xu",
    "id": "2297828138",
    "h_index": 10,
    "papers": 15
   },
   {
    "name": "Kalyan Sunkavalli",
    "id": "2123318412",
    "h_index": 20,
    "papers": 42
   },
   {
    "name": "Mohit Bansal",
    "id": "2257237631",
    "h_index": 7,
    "papers": 14
   },
   {
    "name": "Joyce Chai",
    "id": "2244741764",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Hao Tan",
    "id": "2265922019",
    "h_index": 17,
    "papers": 31
   }
  ],
  "comment": "Project page: https://4dlrm.github.io/",
  "topics": [
   "spatial-3d",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2506.18890v1",
  "pdf_url": "https://arxiv.org/pdf/2506.18890v1",
  "html_url": "https://arxiv.org/html/2506.18890v1",
  "code_url": "https://4dlrm.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.54
 },
 {
  "id": "2506.15666",
  "slug": "vision-in-action-learning-active-perception-from-human-demonstrations",
  "title": "Vision in Action: Learning Active Perception from Human Demonstrations",
  "abstract": "We present Vision in Action (ViA), an active perception system for bimanual robot manipulation. ViA learns task-relevant active perceptual strategies (e.g., searching, tracking, and focusing) directly from human demonstrations. On the hardware side, ViA employs a simple yet effective 6-DoF robotic neck to enable flexible, human-like head movements. To capture human active perception strategies, we design a VR-based teleoperation interface that creates a shared observation space between the robot and the human operator. To mitigate VR motion sickness caused by latency in the robot's physical movements, the interface uses an intermediate 3D scene representation, enabling real-time view rendering on the operator side while asynchronously updating the scene with the robot's latest observations. Together, these design elements enable the learning of robust visuomotor policies for three complex, multi-stage bimanual manipulation tasks involving visual occlusions, significantly outperforming baseline systems.",
  "published": "2025-06-18",
  "updated": "2025-06-18",
  "year": "2025",
  "authors": [
   "Haoyu Xiong",
   "Xiaomeng Xu",
   "Jimmy Wu",
   "Yifan Hou",
   "Jeannette Bohg",
   "Shuran Song"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 45,
  "influential_citations": 5,
  "tldr": "A VR-based teleoperation interface that creates a shared observation space between the robot and the human operator and the learning of robust visuomotor policies for three complex, multi-stage bimanual manipulation tasks involving visual occlusions, significantly outperforming baseline systems.",
  "doi": "10.48550/arXiv.2506.15666",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Haoyu Xiong",
    "id": "2281036863",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Xiaomeng Xu",
    "id": "2286521452",
    "h_index": 7,
    "papers": 9
   },
   {
    "name": "Jimmy Wu",
    "id": "2155142153",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Yifan Hou",
    "id": "2327832445",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Jeannette Bohg",
    "id": "2323565347",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Shuran Song",
    "id": "2364257433",
    "h_index": 6,
    "papers": 17
   }
  ],
  "comment": "",
  "topics": [
   "egocentric-data",
   "spatial-3d",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2506.15666v1",
  "pdf_url": "https://arxiv.org/pdf/2506.15666v1",
  "html_url": "https://arxiv.org/html/2506.15666v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.66
 },
 {
  "id": "2506.14770",
  "slug": "gmt-general-motion-tracking-for-humanoid-whole-body-control",
  "title": "GMT: General Motion Tracking for Humanoid Whole-Body Control",
  "abstract": "The ability to track general whole-body motions in the real world is a useful way to build general-purpose humanoid robots. However, achieving this can be challenging due to the temporal and kinematic diversity of the motions, the policy's capability, and the difficulty of coordination of the upper and lower bodies. To address these issues, we propose GMT, a general and scalable motion-tracking framework that trains a single unified policy to enable humanoid robots to track diverse motions in the real world. GMT is built upon two core components: an Adaptive Sampling strategy and a Motion Mixture-of-Experts (MoE) architecture. The Adaptive Sampling automatically balances easy and difficult motions during training. The MoE ensures better specialization of different regions of the motion manifold. We show through extensive experiments in both simulation and the real world the effectiveness of GMT, achieving state-of-the-art performance across a broad spectrum of motions using a unified general policy. Videos and additional information can be found at https://gmt-humanoid.github.io.",
  "published": "2025-06-17",
  "updated": "2025-09-04",
  "year": "2025",
  "authors": [
   "Zixuan Chen",
   "Mazeyu Ji",
   "Xuxin Cheng",
   "Xuanbin Peng",
   "Xue Bin Peng",
   "Xiaolong Wang"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 134,
  "influential_citations": 20,
  "tldr": "GMT, a general and scalable motion-tracking framework that trains a single unified policy to enable humanoid robots to track diverse motions in the real world, is proposed.",
  "doi": "10.48550/arXiv.2506.14770",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zixuan Chen",
    "id": "2326326961",
    "h_index": 7,
    "papers": 9
   },
   {
    "name": "Mazeyu Ji",
    "id": "2319410473",
    "h_index": 7,
    "papers": 8
   },
   {
    "name": "Xuxin Cheng",
    "id": "2287822264",
    "h_index": 8,
    "papers": 9
   },
   {
    "name": "Xuanbin Peng",
    "id": "2334527659",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Xue Bin Peng",
    "id": "2326248112",
    "h_index": 6,
    "papers": 6
   },
   {
    "name": "Xiaolong Wang",
    "id": "2294782536",
    "h_index": 12,
    "papers": 17
   }
  ],
  "comment": "",
  "topics": [
   "humanoids"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2506.14770v2",
  "pdf_url": "https://arxiv.org/pdf/2506.14770v2",
  "html_url": "https://arxiv.org/html/2506.14770v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.13
 },
 {
  "id": "2506.14754",
  "slug": "tactile-beyond-pixels-multisensory-touch-representations-for-robot-man",
  "title": "Tactile Beyond Pixels: Multisensory Touch Representations for Robot Manipulation",
  "abstract": "We present Sparsh-X, the first multisensory touch representations across four tactile modalities: image, audio, motion, and pressure. Trained on ~1M contact-rich interactions collected with the Digit 360 sensor, Sparsh-X captures complementary touch signals at diverse temporal and spatial scales. By leveraging self-supervised learning, Sparsh-X fuses these modalities into a unified representation that captures physical properties useful for robot manipulation tasks. We study how to effectively integrate real-world touch representations for both imitation learning and tactile adaptation of sim-trained policies, showing that Sparsh-X boosts policy success rates by 63% over an end-to-end model using tactile images and improves robustness by 90% in recovering object states from touch. Finally, we benchmark Sparsh-X ability to make inferences about physical properties, such as object-action identification, material-quantity estimation, and force estimation. Sparsh-X improves accuracy in characterizing physical properties by 48% compared to end-to-end approaches, demonstrating the advantages of multisensory pretraining for capturing features essential for dexterous manipulation.",
  "published": "2025-06-17",
  "updated": "2025-06-17",
  "year": "2025",
  "authors": [
   "Carolina Higuera",
   "Akash Sharma",
   "Taosha Fan",
   "Chaithanya Krishna Bodduluri",
   "Byron Boots",
   "Michael Kaess",
   "Mike Lambeta",
   "Tingfan Wu",
   "Zixi Liu",
   "Francois Robert Hogan",
   "Mustafa Mukadam"
  ],
  "author_count": 11,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 20,
  "influential_citations": 0,
  "tldr": "Sparsh-X improves accuracy in characterizing physical properties by 48% compared to end-to-end approaches, demonstrating the advantages of multisensory pretraining for capturing features essential for dexterous manipulation.",
  "doi": "10.48550/arXiv.2506.14754",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Carolina Higuera",
    "id": "2248215923",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Akash Sharma",
    "id": "2109364933",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "Taosha Fan",
    "id": "2275595472",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Chaithanya Krishna Bodduluri",
    "id": "2328409907",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Byron Boots",
    "id": "2276429486",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Michael Kaess",
    "id": "2279716032",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Mike Lambeta",
    "id": "3427691",
    "h_index": 13,
    "papers": 20
   },
   {
    "name": "Tingfan Wu",
    "id": "2254158966",
    "h_index": 12,
    "papers": 21
   },
   {
    "name": "Zixi Liu",
    "id": "2362147041",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Francois Hogan",
    "id": "2344089432",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Mustafa Mukadam",
    "id": "2874057",
    "h_index": 30,
    "papers": 68
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "imitation-diffusion",
   "foundation-pretraining",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2506.14754v1",
  "pdf_url": "https://arxiv.org/pdf/2506.14754v1",
  "html_url": "https://arxiv.org/html/2506.14754v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.32
 },
 {
  "id": "2506.13922",
  "slug": "dynaguide-steering-diffusion-polices-with-active-dynamic-guidance",
  "title": "DynaGuide: Steering Diffusion Polices with Active Dynamic Guidance",
  "abstract": "Deploying large, complex policies in the real world requires the ability to steer them to fit the needs of a situation. Most common steering approaches, like goal-conditioning, require training the robot policy with a distribution of test-time objectives in mind. To overcome this limitation, we present DynaGuide, a steering method for diffusion policies using guidance from an external dynamics model during the diffusion denoising process. DynaGuide separates the dynamics model from the base policy, which gives it multiple advantages, including the ability to steer towards multiple objectives, enhance underrepresented base policy behaviors, and maintain robustness on low-quality objectives. The separate guidance signal also allows DynaGuide to work with off-the-shelf pretrained diffusion policies. We demonstrate the performance and features of DynaGuide against other steering approaches in a series of simulated and real experiments, showing an average steering success of 70% on a set of articulated CALVIN tasks and outperforming goal-conditioning by 5.4x when steered with low-quality objectives. We also successfully steer an off-the-shelf real robot policy to express preference for particular objects and even create novel behavior. Videos and more can be found on the project website: https://dynaguide.github.io",
  "published": "2025-06-16",
  "updated": "2025-11-09",
  "year": "2025",
  "authors": [
   "Maximilian Du",
   "Shuran Song"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 46,
  "influential_citations": 6,
  "tldr": "DynaGuide is presented, a steering method for diffusion policies using guidance from an external dynamics model during the diffusion denoising process, which gives it multiple advantages, including the ability to steer towards multiple objectives, enhance underrepresented base policy behaviors, and maintain robustness on low-quality objectives.",
  "doi": "10.48550/arXiv.2506.13922",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Maximilian Du",
    "id": "117791840",
    "h_index": 10,
    "papers": 17
   },
   {
    "name": "Shuran Song",
    "id": "2257831707",
    "h_index": 4,
    "papers": 7
   }
  ],
  "comment": "9 pages main, 21 pages with appendix and citations. 9 figures. Presented at Neurips 2025",
  "topics": [
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2506.13922v2",
  "pdf_url": "https://arxiv.org/pdf/2506.13922v2",
  "html_url": "https://arxiv.org/html/2506.13922v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.17
 },
 {
  "id": "2506.13751",
  "slug": "leverb-humanoid-whole-body-control-with-latent-vision-language-instruc",
  "title": "LeVERB: Humanoid Whole-Body Control with Latent Vision-Language Instruction",
  "abstract": "Vision-language-action (VLA) models have demonstrated strong semantic understanding and zero-shot generalization, yet most existing systems assume an accurate low-level controller with hand-crafted action \"vocabulary\" such as end-effector pose or root velocity. This assumption confines prior work to quasi-static tasks and precludes the agile, whole-body behaviors required by humanoid whole-body control (WBC) tasks. To capture this gap in the literature, we start by introducing the first sim-to-real-ready, vision-language, closed-loop benchmark for humanoid WBC, comprising over 150 tasks from 10 categories. We then propose LeVERB: Latent Vision-Language-Encoded Robot Behavior, a hierarchical latent instruction-following framework for humanoid vision-language WBC, the first of its kind. At the top level, a vision-language policy learns a latent action vocabulary from synthetically rendered kinematic demonstrations; at the low level, a reinforcement-learned WBC policy consumes these latent verbs to generate dynamics-level commands. In our benchmark, LeVERB can zero-shot attain a 80% success rate on simple visual navigation tasks, and 58.5% success rate overall, outperforming naive hierarchical whole-body VLA implementation by 7.8 times.",
  "published": "2025-06-16",
  "updated": "2025-09-25",
  "year": "2025",
  "authors": [
   "Haoru Xue",
   "Xiaoyu Huang",
   "Dantong Niu",
   "Qiayuan Liao",
   "Thomas Kragerud",
   "Jan Tommy Gravdahl",
   "Xue Bin Peng",
   "Guanya Shi",
   "Trevor Darrell",
   "Koushil Sreenath",
   "Shankar Sastry"
  ],
  "author_count": 11,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 54,
  "influential_citations": 2,
  "tldr": "This work introduces the first sim-to-real-ready, vision-language, closed-loop benchmark for humanoid WBC, and proposes LeVERB: Latent Vision-Language-Encoded Robot Behavior, a hierarchical latent instruction-following framework for humanoid vision-language WBC, the first of its kind.",
  "doi": "10.48550/arXiv.2506.13751",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Haoru Xue",
    "id": "2347874052",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Xiaoyu Huang",
    "id": "2293553057",
    "h_index": 9,
    "papers": 10
   },
   {
    "name": "Dantong Niu",
    "id": "2268757542",
    "h_index": 8,
    "papers": 17
   },
   {
    "name": "Qiayuan Liao",
    "id": "1713616371",
    "h_index": 13,
    "papers": 20
   },
   {
    "name": "Thomas Kragerud",
    "id": "2367274338",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "J. Gravdahl",
    "id": "1739946",
    "h_index": 45,
    "papers": 385
   },
   {
    "name": "Xue Bin Peng",
    "id": "2300358388",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Guanya Shi",
    "id": "2249759531",
    "h_index": 20,
    "papers": 31
   },
   {
    "name": "Trevor Darrell",
    "id": "2257973285",
    "h_index": 11,
    "papers": 27
   },
   {
    "name": "K. Sreenath",
    "id": "144116765",
    "h_index": 55,
    "papers": 231
   },
   {
    "name": "S. Sastry",
    "id": "2325954829",
    "h_index": 4,
    "papers": 5
   }
  ],
  "comment": "https://ember-lab-berkeley.github.io/LeVERB-Website/",
  "topics": [
   "vla",
   "humanoids",
   "sim2real",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2506.13751v3",
  "pdf_url": "https://arxiv.org/pdf/2506.13751v3",
  "html_url": "https://arxiv.org/html/2506.13751v3",
  "code_url": "https://ember-lab-berkeley.github.io/LeVERB-Website/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.74
 },
 {
  "id": "2506.13536",
  "slug": "what-matters-in-learning-from-large-scale-datasets-for-robot-manipulat",
  "title": "What Matters in Learning from Large-Scale Datasets for Robot Manipulation",
  "abstract": "Imitation learning from large multi-task demonstration datasets has emerged as a promising path for building generally-capable robots. As a result, 1000s of hours have been spent on building such large-scale datasets around the globe. Despite the continuous growth of such efforts, we still lack a systematic understanding of what data should be collected to improve the utility of a robotics dataset and facilitate downstream policy learning. In this work, we conduct a large-scale dataset composition study to answer this question. We develop a data generation framework to procedurally emulate common sources of diversity in existing datasets (such as sensor placements and object types and arrangements), and use it to generate large-scale robot datasets with controlled compositions, enabling a suite of dataset composition studies that would be prohibitively expensive in the real world. We focus on two practical settings: (1) what types of diversity should be emphasized when future researchers collect large-scale datasets for robotics, and (2) how should current practitioners retrieve relevant demonstrations from existing datasets to maximize downstream policy performance on tasks of interest. Our study yields several critical insights -- for example, we find that camera poses and spatial arrangements are crucial dimensions for both diversity in collection and alignment in retrieval. In real-world robot learning settings, we find that not only do our insights from simulation carry over, but our retrieval strategies on existing datasets such as DROID allow us to consistently outperform existing training strategies by up to 70%. More results at https://robo-mimiclabs.github.io/",
  "published": "2025-06-16",
  "updated": "2025-06-16",
  "year": "2025",
  "authors": [
   "Vaibhav Saxena",
   "Matthew Bronars",
   "Nadun Ranawaka Arachchige",
   "Kuancheng Wang",
   "Woo Chul Shin",
   "Soroush Nasiriany",
   "Ajay Mandlekar",
   "Danfei Xu"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 40,
  "influential_citations": 0,
  "tldr": "This work develops a data generation framework to procedurally emulate common sources of diversity in existing datasets, and uses it to generate large-scale robot datasets with controlled compositions, enabling a suite of dataset composition studies that would be prohibitively expensive in the real world.",
  "doi": "10.48550/arXiv.2506.13536",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Vaibhav Saxena",
    "id": "2311885269",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Matthew Bronars",
    "id": "2260124870",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "N. R. Arachchige",
    "id": "2141127741",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Kuan Wang",
    "id": "2284955273",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Woo-Chul Shin",
    "id": "2360757506",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Soroush Nasiriany",
    "id": "3457048",
    "h_index": 18,
    "papers": 24
   },
   {
    "name": "A. Mandlekar",
    "id": "49686756",
    "h_index": 36,
    "papers": 67
   },
   {
    "name": "Danfei Xu",
    "id": "2260291195",
    "h_index": 7,
    "papers": 14
   }
  ],
  "comment": "",
  "topics": [
   "imitation-diffusion"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2506.13536v1",
  "pdf_url": "https://arxiv.org/pdf/2506.13536v1",
  "html_url": "https://arxiv.org/html/2506.13536v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.11
 },
 {
  "id": "2506.09995",
  "slug": "playerone-egocentric-world-simulator",
  "title": "PlayerOne: Egocentric World Simulator",
  "abstract": "We introduce PlayerOne, the first egocentric realistic world simulator, facilitating immersive and unrestricted exploration within vividly dynamic environments. Given an egocentric scene image from the user, PlayerOne can accurately construct the corresponding world and generate egocentric videos that are strictly aligned with the real scene human motion of the user captured by an exocentric camera. PlayerOne is trained in a coarse-to-fine pipeline that first performs pretraining on large-scale egocentric text-video pairs for coarse-level egocentric understanding, followed by finetuning on synchronous motion-video data extracted from egocentric-exocentric video datasets with our automatic construction pipeline. Besides, considering the varying importance of different components, we design a part-disentangled motion injection scheme, enabling precise control of part-level movements. In addition, we devise a joint reconstruction framework that progressively models both the 4D scene and video frames, ensuring scene consistency in the long-form video generation. Experimental results demonstrate its great generalization ability in precise control of varying human movements and worldconsistent modeling of diverse scenarios. It marks the first endeavor into egocentric real-world simulation and can pave the way for the community to delve into fresh frontiers of world modeling and its diverse applications.",
  "published": "2025-06-11",
  "updated": "2025-12-10",
  "year": "2025",
  "authors": [
   "Yuanpeng Tu",
   "Hao Luo",
   "Xi Chen",
   "Xiang Bai",
   "Fan Wang",
   "Hengshuang Zhao"
  ],
  "author_count": 6,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 20,
  "influential_citations": 3,
  "tldr": "PlayerOne is introduced, the first egocentric realistic world simulator, facilitating immersive and unrestricted exploration within vividly dynamic environments and can pave the way for the community to delve into fresh frontiers of world modeling and its diverse applications.",
  "doi": "10.48550/arXiv.2506.09995",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yuanpeng Tu",
    "id": "2338275715",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Hao Luo",
    "id": "2336741205",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Xi Chen",
    "id": "2334721679",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Xiang Bai",
    "id": "2367597034",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Fan Wang",
    "id": "2274912982",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Hengshuang Zhao",
    "id": "2293765898",
    "h_index": 10,
    "papers": 41
   }
  ],
  "comment": "Project page: https://playerone-hku.github.io/",
  "topics": [
   "world-models",
   "egocentric-data",
   "sim2real",
   "navigation",
   "foundation-pretraining",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2506.09995v3",
  "pdf_url": "https://arxiv.org/pdf/2506.09995v3",
  "html_url": "https://arxiv.org/html/2506.09995v3",
  "code_url": "https://playerone-hku.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.82
 },
 {
  "id": "2506.09985",
  "slug": "v-jepa-2-self-supervised-video-models-enable-understanding-prediction",
  "title": "V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning",
  "abstract": "A major challenge for modern AI is to learn to understand the world and learn to act largely by observation. This paper explores a self-supervised approach that combines internet-scale video data with a small amount of interaction data (robot trajectories), to develop models capable of understanding, predicting, and planning in the physical world. We first pre-train an action-free joint-embedding-predictive architecture, V-JEPA 2, on a video and image dataset comprising over 1 million hours of internet video. V-JEPA 2 achieves strong performance on motion understanding (77.3 top-1 accuracy on Something-Something v2) and state-of-the-art performance on human action anticipation (39.7 recall-at-5 on Epic-Kitchens-100) surpassing previous task-specific models. Additionally, after aligning V-JEPA 2 with a large language model, we demonstrate state-of-the-art performance on multiple video question-answering tasks at the 8 billion parameter scale (e.g., 84.0 on PerceptionTest, 76.9 on TempCompass). Finally, we show how self-supervised learning can be applied to robotic planning tasks by post-training a latent action-conditioned world model, V-JEPA 2-AC, using less than 62 hours of unlabeled robot videos from the Droid dataset. We deploy V-JEPA 2-AC zero-shot on Franka arms in two different labs and enable picking and placing of objects using planning with image goals. Notably, this is achieved without collecting any data from the robots in these environments, and without any task-specific training or reward. This work demonstrates how self-supervised learning from web-scale data and a small amount of robot interaction data can yield a world model capable of planning in the physical world.",
  "published": "2025-06-11",
  "updated": "2025-06-11",
  "year": "2025",
  "authors": [
   "Mido Assran",
   "Adrien Bardes",
   "David Fan",
   "Quentin Garrido",
   "Russell Howes",
   " Mojtaba",
   " Komeili",
   "Matthew Muckley",
   "Ammar Rizvi",
   "Claire Roberts",
   "Koustuv Sinha",
   "Artem Zholus",
   "Sergio Arnaud",
   "Abha Gejji",
   "Ada Martin",
   "Francois Robert Hogan",
   "Daniel Dugas",
   "Piotr Bojanowski",
   "Vasil Khalidov",
   "Patrick Labatut",
   "Francisco Massa",
   "Marc Szafraniec",
   "Kapil Krishnakumar",
   "Yong Li",
   "Xiaodong Ma",
   "Sarath Chandar",
   "Franziska Meier",
   "Yann LeCun",
   "Michael Rabbat",
   "Nicolas Ballas"
  ],
  "author_count": 30,
  "categories": [
   "cs.AI",
   "cs.CV",
   "cs.LG",
   "cs.RO"
  ],
  "primary_category": "cs.AI",
  "venue": "",
  "venue_source": "",
  "citations": 630,
  "influential_citations": 92,
  "tldr": "This work demonstrates how self-supervised learning from web-scale data and a small amount of robot interaction data can yield a world model capable of planning in the physical world.",
  "doi": "10.48550/arXiv.2506.09985",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Mahmoud Assran",
    "id": "38698856",
    "h_index": 15,
    "papers": 21
   },
   {
    "name": "Adrien Bardes",
    "id": "1453740540",
    "h_index": 16,
    "papers": 23
   },
   {
    "name": "David Fan",
    "id": "2335871751",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Q. Garrido",
    "id": "2048163343",
    "h_index": 16,
    "papers": 22
   },
   {
    "name": "Russell Howes",
    "id": "2366426723",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Mojtaba Komeili",
    "id": "2324568313",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Matthew Muckley",
    "id": "2954796",
    "h_index": 19,
    "papers": 45
   },
   {
    "name": "Ammar Rizvi",
    "id": "2366426182",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Claire Roberts",
    "id": "2366431895",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Koustuv Sinha",
    "id": "2266467601",
    "h_index": 11,
    "papers": 16
   },
   {
    "name": "Artem Zholus",
    "id": "1416761815",
    "h_index": 12,
    "papers": 26
   },
   {
    "name": "Sergio Arnaud",
    "id": "2213251873",
    "h_index": 9,
    "papers": 15
   },
   {
    "name": "Abha Gejji",
    "id": "1752829922",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Ada Martin",
    "id": "2347686350",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Francois Hogan",
    "id": "2344089432",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Daniel Dugas",
    "id": "2356547509",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Piotr Bojanowski",
    "id": "2329288",
    "h_index": 38,
    "papers": 77
   },
   {
    "name": "Vasil Khalidov",
    "id": "2182694",
    "h_index": 16,
    "papers": 28
   },
   {
    "name": "Patrick Labatut",
    "id": "1744868",
    "h_index": 20,
    "papers": 38
   },
   {
    "name": "Francisco Massa",
    "id": "2366427448",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Marc Szafraniec",
    "id": "23994377",
    "h_index": 11,
    "papers": 17
   },
   {
    "name": "K. Krishnakumar",
    "id": "2278303846",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Yong Li",
    "id": "2383231005",
    "h_index": 4,
    "papers": 14
   },
   {
    "name": "Xiaodong Ma",
    "id": "2391499701",
    "h_index": 11,
    "papers": 60
   },
   {
    "name": "Sarath Chandar",
    "id": "123607932",
    "h_index": 22,
    "papers": 128
   },
   {
    "name": "Franziska Meier",
    "id": "153145615",
    "h_index": 31,
    "papers": 77
   },
   {
    "name": "Yann LeCun",
    "id": "2265899558",
    "h_index": 22,
    "papers": 47
   },
   {
    "name": "Michael Rabbat",
    "id": "2284991448",
    "h_index": 12,
    "papers": 18
   },
   {
    "name": "Nicolas Ballas",
    "id": "2289844757",
    "h_index": 13,
    "papers": 19
   },
   {
    "name": "Fair at Meta",
    "id": "2364685176",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "M. Qu\u00e9bec",
    "id": "2312779111",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "AI Institute",
    "id": "2312747323",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "P. Montr'eal",
    "id": "2102129911",
    "h_index": 4,
    "papers": 37
   }
  ],
  "comment": "48 pages, 19 figures",
  "topics": [
   "world-models"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2506.09985v1",
  "pdf_url": "https://arxiv.org/pdf/2506.09985v1",
  "html_url": "https://arxiv.org/html/2506.09985v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.8
 },
 {
  "id": "2506.09982",
  "slug": "animateanymesh-a-feed-forward-4d-foundation-model-for-text-driven-univ",
  "title": "AnimateAnyMesh: A Feed-Forward 4D Foundation Model for Text-Driven Universal Mesh Animation",
  "abstract": "Recent advances in 4D content generation have attracted increasing attention, yet creating high-quality animated 3D models remains challenging due to the complexity of modeling spatio-temporal distributions and the scarcity of 4D training data. In this paper, we present AnimateAnyMesh, the first feed-forward framework that enables efficient text-driven animation of arbitrary 3D meshes. Our approach leverages a novel DyMeshVAE architecture that effectively compresses and reconstructs dynamic mesh sequences by disentangling spatial and temporal features while preserving local topological structures. To enable high-quality text-conditional generation, we employ a Rectified Flow-based training strategy in the compressed latent space. Additionally, we contribute the DyMesh Dataset, containing over 4M diverse dynamic mesh sequences with text annotations. Experimental results demonstrate that our method generates semantically accurate and temporally coherent mesh animations in a few seconds, significantly outperforming existing approaches in both quality and efficiency. Our work marks a substantial step forward in making 4D content creation more accessible and practical. All the data, code, and models will be open-released.",
  "published": "2025-06-11",
  "updated": "2025-06-11",
  "year": "2025",
  "authors": [
   "Zijie Wu",
   "Chaohui Yu",
   "Fan Wang",
   "Xiang Bai"
  ],
  "author_count": 4,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "ICCV",
  "venue_source": "semantic-scholar",
  "citations": 38,
  "influential_citations": 11,
  "tldr": "AnimateAnyMesh is presented, the first feed-forward framework that enables efficient textdriven animation of arbitrary 3D meshes, and leverages a novel DyMeshVAE architecture that effectively compresses and reconstructs dynamic mesh sequences by disentangling spatial and temporal features while preserving local topological structures.",
  "doi": "10.1109/ICCV51701.2025.01259",
  "oa_pdf": "https://arxiv.org/pdf/2506.09982",
  "s2_authors": [
   {
    "name": "Zijie Wu",
    "id": "2187780453",
    "h_index": 4,
    "papers": 14
   },
   {
    "name": "Chaohui Yu",
    "id": "2110961040",
    "h_index": 16,
    "papers": 45
   },
   {
    "name": "Fan Wang",
    "id": "2257894784",
    "h_index": 8,
    "papers": 18
   },
   {
    "name": "Xiang Bai",
    "id": "2367597034",
    "h_index": 2,
    "papers": 5
   }
  ],
  "comment": "Project Page: https://animateanymesh.github.io/AnimateAnyMesh/",
  "topics": [
   "foundation-pretraining",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2506.09982v1",
  "pdf_url": "https://arxiv.org/pdf/2506.09982v1",
  "html_url": "https://arxiv.org/html/2506.09982v1",
  "code_url": "https://animateanymesh.github.io/AnimateAnyMesh/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.09
 },
 {
  "id": "2506.09981",
  "slug": "resim-reliable-world-simulation-for-autonomous-driving",
  "title": "ReSim: Reliable World Simulation for Autonomous Driving",
  "abstract": "How can we reliably simulate future driving scenarios under a wide range of ego driving behaviors? Recent driving world models, developed exclusively on real-world driving data composed mainly of safe expert trajectories, struggle to follow hazardous or non-expert behaviors, which are rare in such data. This limitation restricts their applicability to tasks such as policy evaluation. In this work, we address this challenge by enriching real-world human demonstrations with diverse non-expert data collected from a driving simulator (e.g., CARLA), and building a controllable world model trained on this heterogeneous corpus. Starting with a video generator featuring a diffusion transformer architecture, we devise several strategies to effectively integrate conditioning signals and improve prediction controllability and fidelity. The resulting model, ReSim, enables Reliable Simulation of diverse open-world driving scenarios under various actions, including hazardous non-expert ones. To close the gap between high-fidelity simulation and applications that require reward signals to judge different actions, we introduce a Video2Reward module that estimates a reward from ReSim's simulated future. Our ReSim paradigm achieves up to 44% higher visual fidelity, improves controllability for both expert and non-expert actions by over 50%, and boosts planning and policy selection performance on NAVSIM by 2% and 25%, respectively.",
  "published": "2025-06-11",
  "updated": "2026-04-28",
  "year": "2025",
  "authors": [
   "Jiazhi Yang",
   "Kashyap Chitta",
   "Shenyuan Gao",
   "Long Chen",
   "Yuqian Shao",
   "Xiaosong Jia",
   "Hongyang Li",
   "Andreas Geiger",
   "Xiangyu Yue",
   "Li Chen"
  ],
  "author_count": 10,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 47,
  "influential_citations": 3,
  "tldr": "This work enrichs real-world human demonstrations with diverse non-expert data collected from a driving simulator, and builds a controllable world model trained on this heterogeneous corpus, enabling Reliable Simulation of diverse open-world driving scenarios under various actions, including hazardous non-expert ones.",
  "doi": "10.48550/arXiv.2506.09981",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jiazhi Yang",
    "id": "2184753565",
    "h_index": 9,
    "papers": 13
   },
   {
    "name": "Kashyap Chitta",
    "id": "31352445",
    "h_index": 26,
    "papers": 53
   },
   {
    "name": "Shenyuan Gao",
    "id": "2350220976",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Long Chen",
    "id": "2304614426",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Yuqian Shao",
    "id": "2364366288",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Xiaosong Jia",
    "id": "1958998899",
    "h_index": 23,
    "papers": 53
   },
   {
    "name": "Hongyang Li",
    "id": "2303413508",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Andreas Geiger",
    "id": "2265966462",
    "h_index": 11,
    "papers": 19
   },
   {
    "name": "Xiangyu Yue",
    "id": "2366354906",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Li Chen",
    "id": "2170764701",
    "h_index": 22,
    "papers": 31
   }
  ],
  "comment": "NeurIPS 2025 Spotlight. Project page: https://opendrivelab.com/ReSim",
  "topics": [
   "world-models",
   "egocentric-data",
   "sim2real",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2506.09981v2",
  "pdf_url": "https://arxiv.org/pdf/2506.09981v2",
  "html_url": "https://arxiv.org/html/2506.09981v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.18
 },
 {
  "id": "2506.09366",
  "slug": "skillblender-towards-versatile-humanoid-whole-body-loco-manipulation-v",
  "title": "SkillBlender: Towards Versatile Humanoid Whole-Body Loco-Manipulation via Skill Blending",
  "abstract": "Humanoid robots hold significant potential in accomplishing daily tasks across diverse environments thanks to their flexibility and human-like morphology. Recent works have made significant progress in humanoid whole-body control and loco-manipulation leveraging optimal control or reinforcement learning. However, these methods require tedious task-specific tuning for each task to achieve satisfactory behaviors, limiting their versatility and scalability to diverse tasks in daily scenarios. To that end, we introduce SkillBlender, a novel hierarchical reinforcement learning framework for versatile humanoid loco-manipulation. SkillBlender first pretrains goal-conditioned task-agnostic primitive skills, and then dynamically blends these skills to accomplish complex loco-manipulation tasks with minimal task-specific reward engineering. We also introduce SkillBench, a parallel, cross-embodiment, and diverse simulated benchmark containing three embodiments, four primitive skills, and eight challenging loco-manipulation tasks, accompanied by a set of scientific evaluation metrics balancing accuracy and feasibility. Extensive simulated experiments show that our method significantly outperforms all baselines, while naturally regularizing behaviors to avoid reward hacking, resulting in more accurate and feasible movements for diverse loco-manipulation tasks in our daily scenarios. Our code and benchmark will be open-sourced to the community to facilitate future research. Project page: https://usc-gvl.github.io/SkillBlender-web/.",
  "published": "2025-06-11",
  "updated": "2025-06-11",
  "year": "2025",
  "authors": [
   "Yuxuan Kuang",
   "Haoran Geng",
   "Amine Elhafsi",
   "Tan-Dzung Do",
   "Pieter Abbeel",
   "Jitendra Malik",
   "Marco Pavone",
   "Yue Wang"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 17,
  "influential_citations": 0,
  "tldr": "SkillBlender is introduced, a novel hierarchical reinforcement learning framework for versatile humanoid loco-manipulation that significantly outperforms all baselines, while naturally regularizing behaviors to avoid reward hacking, resulting in more accurate and feasible movements for diverse loco-manipulation tasks in daily scenarios.",
  "doi": "10.48550/arXiv.2506.09366",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yuxuan Kuang",
    "id": "2256991055",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Haoran Geng",
    "id": "2287929608",
    "h_index": 7,
    "papers": 15
   },
   {
    "name": "Amine Elhafsi",
    "id": "71578349",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Tan-Dzung Do",
    "id": "2366398986",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Pieter Abbeel",
    "id": "2350871358",
    "h_index": 8,
    "papers": 22
   },
   {
    "name": "Jitendra Malik",
    "id": "2362090500",
    "h_index": 8,
    "papers": 18
   },
   {
    "name": "Marco Pavone",
    "id": "2328979520",
    "h_index": 10,
    "papers": 34
   },
   {
    "name": "Yue Wang",
    "id": "2291101194",
    "h_index": 12,
    "papers": 19
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "rl-control",
   "foundation-pretraining",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2506.09366v1",
  "pdf_url": "https://arxiv.org/pdf/2506.09366v1",
  "html_url": "https://arxiv.org/html/2506.09366v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.26
 },
 {
  "id": "2506.09350",
  "slug": "autoregressive-adversarial-post-training-for-real-time-interactive-vid",
  "title": "Autoregressive Adversarial Post-Training for Real-Time Interactive Video Generation",
  "abstract": "Existing large-scale video generation models are computationally intensive, preventing adoption in real-time and interactive applications. In this work, we propose autoregressive adversarial post-training (AAPT) to transform a pre-trained latent video diffusion model into a real-time, interactive video generator. Our model autoregressively generates a latent frame at a time using a single neural function evaluation (1NFE). The model can stream the result to the user in real time and receive interactive responses as controls to generate the next latent frame. Unlike existing approaches, our method explores adversarial training as an effective paradigm for autoregressive generation. This not only allows us to design an architecture that is more efficient for one-step generation while fully utilizing the KV cache, but also enables training the model in a student-forcing manner that proves to be effective in reducing error accumulation during long video generation. Our experiments demonstrate that our 8B model achieves real-time, 24fps, streaming video generation at 736x416 resolution on a single H100, or 1280x720 on 8xH100 up to a minute long (1440 frames). Visit our research website at https://seaweed-apt.com/2",
  "published": "2025-06-11",
  "updated": "2025-10-01",
  "year": "2025",
  "authors": [
   "Shanchuan Lin",
   "Ceyuan Yang",
   "Hao He",
   "Jianwen Jiang",
   "Yuxi Ren",
   "Xin Xia",
   "Yang Zhao",
   "Xuefeng Xiao",
   "Lu Jiang"
  ],
  "author_count": 9,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.CV",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 77,
  "influential_citations": 2,
  "tldr": "This work proposes autoregressive adversarial post-training (AAPT) to transform a pre-trained latent video diffusion model into a real-time, interactive video generator that autoregressively generates a latent frame at a time using a single neural function evaluation (1NFE).",
  "doi": "10.48550/arXiv.2506.09350",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Shanchuan Lin",
    "id": "32370203",
    "h_index": 19,
    "papers": 31
   },
   {
    "name": "Ceyuan Yang",
    "id": "49984891",
    "h_index": 37,
    "papers": 72
   },
   {
    "name": "Hao He",
    "id": "2294676293",
    "h_index": 9,
    "papers": 14
   },
   {
    "name": "Jianwen Jiang",
    "id": "2293669359",
    "h_index": 9,
    "papers": 16
   },
   {
    "name": "Yuxi Ren",
    "id": "2238397045",
    "h_index": 13,
    "papers": 28
   },
   {
    "name": "X. Xia",
    "id": "2290134117",
    "h_index": 15,
    "papers": 33
   },
   {
    "name": "Yang Zhao",
    "id": "2366209055",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "Xuefeng Xiao",
    "id": "2118724465",
    "h_index": 27,
    "papers": 58
   },
   {
    "name": "Lu Jiang",
    "id": "2338350990",
    "h_index": 8,
    "papers": 11
   }
  ],
  "comment": "NeurIPS 2025",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2506.09350v2",
  "pdf_url": "https://arxiv.org/pdf/2506.09350v2",
  "html_url": "https://arxiv.org/html/2506.09350v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.39
 },
 {
  "id": "2506.09284",
  "slug": "uad-unsupervised-affordance-distillation-for-generalization-in-robotic",
  "title": "UAD: Unsupervised Affordance Distillation for Generalization in Robotic Manipulation",
  "abstract": "Understanding fine-grained object affordances is imperative for robots to manipulate objects in unstructured environments given open-ended task instructions. However, existing methods of visual affordance predictions often rely on manually annotated data or conditions only on a predefined set of tasks. We introduce UAD (Unsupervised Affordance Distillation), a method for distilling affordance knowledge from foundation models into a task-conditioned affordance model without any manual annotations. By leveraging the complementary strengths of large vision models and vision-language models, UAD automatically annotates a large-scale dataset with detailed $<$instruction, visual affordance$>$ pairs. Training only a lightweight task-conditioned decoder atop frozen features, UAD exhibits notable generalization to in-the-wild robotic scenes and to various human activities, despite only being trained on rendered objects in simulation. Using affordance provided by UAD as the observation space, we show an imitation learning policy that demonstrates promising generalization to unseen object instances, object categories, and even variations in task instructions after training on as few as 10 demonstrations. Project website: https://unsup-affordance.github.io/",
  "published": "2025-06-10",
  "updated": "2025-08-25",
  "year": "2025",
  "authors": [
   "Yihe Tang",
   "Wenlong Huang",
   "Yingke Wang",
   "Chengshu Li",
   "Roy Yuan",
   "Ruohan Zhang",
   "Jiajun Wu",
   "Li Fei-Fei"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA",
  "venue_source": "semantic-scholar",
  "citations": 37,
  "influential_citations": 6,
  "tldr": "Unsupervised Affordance Distillation is introduced, a method for distilling affordance knowledge from foundation models into a task-conditioned affordance model without any manual annotations that demonstrates promising generalization to unseen object instances, object categories, and even variations in task instructions after training on as few as 10 demonstrations.",
  "doi": "10.1109/ICRA55743.2025.11128868",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yihe Tang",
    "id": "2301351607",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Wenlong Huang",
    "id": "2319777572",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Yingke Wang",
    "id": "2328260603",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Chengshu Li",
    "id": "2328502402",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Roy Yuan",
    "id": "2366426819",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Ruohan Zhang",
    "id": "2328111135",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Jiajun Wu",
    "id": "2326444500",
    "h_index": 35,
    "papers": 214
   },
   {
    "name": "Fei-Fei Li",
    "id": "2330589126",
    "h_index": 3,
    "papers": 15
   }
  ],
  "comment": "",
  "topics": [
   "imitation-diffusion",
   "foundation-pretraining",
   "data-teleop",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2506.09284v2",
  "pdf_url": "https://arxiv.org/pdf/2506.09284v2",
  "html_url": "https://arxiv.org/html/2506.09284v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.08
 },
 {
  "id": "2506.08931",
  "slug": "clone-closed-loop-whole-body-humanoid-teleoperation-for-long-horizon-t",
  "title": "CLONE: Closed-Loop Whole-Body Humanoid Teleoperation for Long-Horizon Tasks",
  "abstract": "Humanoid teleoperation plays a vital role in demonstrating and collecting data for complex humanoid-scene interactions. However, current teleoperation systems face critical limitations: they decouple upper- and lower-body control to maintain stability, restricting natural coordination, and operate open-loop without real-time position feedback, leading to accumulated drift. The fundamental challenge is achieving precise, coordinated whole-body teleoperation over extended durations while maintaining accurate global positioning. Here we show that an MoE-based teleoperation system, CLONE, with closed-loop error correction enables unprecedented whole-body teleoperation fidelity, maintaining minimal positional drift over long-range trajectories using only head and hand tracking from an MR headset. Unlike previous methods that either sacrifice coordination for stability or suffer from unbounded drift, CLONE learns diverse motion skills while preventing tracking error accumulation through real-time feedback, enabling complex coordinated movements such as ``picking up objects from the ground.'' These results establish a new milestone for whole-body humanoid teleoperation for long-horizon humanoid-scene interaction tasks.",
  "published": "2025-06-10",
  "updated": "2025-08-30",
  "year": "2025",
  "authors": [
   "Yixuan Li",
   "Yutang Lin",
   "Jieming Cui",
   "Tengyu Liu",
   "Wei Liang",
   "Yixin Zhu",
   "Siyuan Huang"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 84,
  "influential_citations": 3,
  "tldr": "This work shows that an MoE-based teleoperation system, CLONE, with closed-loop error correction enables unprecedented whole-body teleoperation fidelity, maintaining minimal positional drift over long-range trajectories using only head and hand tracking from an MR headset.",
  "doi": "10.48550/arXiv.2506.08931",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yixuan Li",
    "id": "2317040291",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Yutang Lin",
    "id": "2327171621",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Jieming Cui",
    "id": "2217941702",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Tengyu Liu",
    "id": "2110032600",
    "h_index": 21,
    "papers": 32
   },
   {
    "name": "Wei Liang",
    "id": "2150326083",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Yixin Zhu",
    "id": "2261513442",
    "h_index": 11,
    "papers": 20
   },
   {
    "name": "Siyuan Huang",
    "id": "2264375840",
    "h_index": 13,
    "papers": 27
   }
  ],
  "comment": "18 pages, 13 figures",
  "topics": [
   "humanoids",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2506.08931v2",
  "pdf_url": "https://arxiv.org/pdf/2506.08931v2",
  "html_url": "https://arxiv.org/html/2506.08931v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.93
 },
 {
  "id": "2506.08009",
  "slug": "self-forcing-bridging-the-train-test-gap-in-autoregressive-video-diffu",
  "title": "Self Forcing: Bridging the Train-Test Gap in Autoregressive Video Diffusion",
  "abstract": "We introduce Self Forcing, a novel training paradigm for autoregressive video diffusion models. It addresses the longstanding issue of exposure bias, where models trained on ground-truth context must generate sequences conditioned on their own imperfect outputs during inference. Unlike prior methods that denoise future frames based on ground-truth context frames, Self Forcing conditions each frame's generation on previously self-generated outputs by performing autoregressive rollout with key-value (KV) caching during training. This strategy enables supervision through a holistic loss at the video level that directly evaluates the quality of the entire generated sequence, rather than relying solely on traditional frame-wise objectives. To ensure training efficiency, we employ a few-step diffusion model along with a stochastic gradient truncation strategy, effectively balancing computational cost and performance. We further introduce a rolling KV cache mechanism that enables efficient autoregressive video extrapolation. Extensive experiments demonstrate that our approach achieves real-time streaming video generation with sub-second latency on a single GPU, while matching or even surpassing the generation quality of significantly slower and non-causal diffusion models. Project website: http://self-forcing.github.io/",
  "published": "2025-06-09",
  "updated": "2025-11-10",
  "year": "2025",
  "authors": [
   "Xun Huang",
   "Zhengqi Li",
   "Guande He",
   "Mingyuan Zhou",
   "Eli Shechtman"
  ],
  "author_count": 5,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.CV",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 507,
  "influential_citations": 166,
  "tldr": "Self Forcing is introduced, a novel training paradigm for autoregressive video diffusion models that addresses the longstanding issue of exposure bias, and employs a few-step diffusion model along with a stochastic gradient truncation strategy, effectively balancing computational cost and performance.",
  "doi": "10.48550/arXiv.2506.08009",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xun Huang",
    "id": "2334843806",
    "h_index": 8,
    "papers": 14
   },
   {
    "name": "Zhengqi Li",
    "id": "2367076348",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Guande He",
    "id": "2368698002",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Mingyuan Zhou",
    "id": "2366102366",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Eli Shechtman",
    "id": "2268760396",
    "h_index": 16,
    "papers": 33
   }
  ],
  "comment": "NeurIPS 2025 spotlight. Project website: http://self-forcing.github.io/",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2506.08009v2",
  "pdf_url": "https://arxiv.org/pdf/2506.08009v2",
  "html_url": "https://arxiv.org/html/2506.08009v2",
  "code_url": "https://self-forcing.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.21
 },
 {
  "id": "2506.06199",
  "slug": "3dflowaction-learning-cross-embodiment-manipulation-from-3d-flow-world",
  "title": "3DFlowAction: Learning Cross-Embodiment Manipulation from 3D Flow World Model",
  "abstract": "Manipulation has long been a challenging task for robots, while humans can effortlessly perform complex interactions with objects, such as hanging a cup on the mug rack. A key reason is the lack of a large and uniform dataset for teaching robots manipulation skills. Current robot datasets often record robot action in different action spaces within a simple scene. This hinders the robot to learn a unified and robust action representation for different robots within diverse scenes. Observing how humans understand a manipulation task, we find that understanding how the objects should move in the 3D space is a critical clue for guiding actions. This clue is embodiment-agnostic and suitable for both humans and different robots. Motivated by this, we aim to learn a 3D flow world model from both human and robot manipulation data. This model predicts the future movement of the interacting objects in 3D space, guiding action planning for manipulation. Specifically, we synthesize a large-scale 3D optical flow dataset, named ManiFlow-110k, through a moving object auto-detect pipeline. A video diffusion-based world model then learns manipulation physics from these data, generating 3D optical flow trajectories conditioned on language instructions. With the generated 3D object optical flow, we propose a flow-guided rendering mechanism, which renders the predicted final state and leverages GPT-4o to assess whether the predicted flow aligns with the task description. This equips the robot with a closed-loop planning ability. Finally, we consider the predicted 3D optical flow as constraints for an optimization policy to determine a chunk of robot actions for manipulation. Extensive experiments demonstrate strong generalization across diverse robotic manipulation tasks and reliable cross-embodiment adaptation without hardware-specific training.",
  "published": "2025-06-06",
  "updated": "2025-06-06",
  "year": "2025",
  "authors": [
   "Hongyan Zhi",
   "Peihao Chen",
   "Siyuan Zhou",
   "Yubo Dong",
   "Quanxi Wu",
   "Lei Han",
   "Mingkui Tan"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 36,
  "influential_citations": 3,
  "tldr": "This work synthesizes a large-scale 3D optical flow dataset, named ManiFlow-110k, through a moving object auto-detect pipeline, and proposes a flow-guided rendering mechanism, which renders the predicted final state and leverages GPT-4o to assess whether the predicted flow aligns with the task description.",
  "doi": "10.48550/arXiv.2506.06199",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hongyan Zhi",
    "id": "2231606704",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Peihao Chen",
    "id": "2158502526",
    "h_index": 21,
    "papers": 34
   },
   {
    "name": "Siyuan Zhou",
    "id": "2258920872",
    "h_index": 9,
    "papers": 19
   },
   {
    "name": "Dongjie Yu",
    "id": "2307759380",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Quanxi Wu",
    "id": "2365990709",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Lei Han",
    "id": "2362318590",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Mingkui Tan",
    "id": "2257488640",
    "h_index": 7,
    "papers": 15
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "foundation-pretraining",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2506.06199v1",
  "pdf_url": "https://arxiv.org/pdf/2506.06199v1",
  "html_url": "https://arxiv.org/html/2506.06199v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.57
 },
 {
  "id": "2506.05294",
  "slug": "a-smooth-sea-never-made-a-skilled-sailor-robust-imitation-via-learning",
  "title": "A Smooth Sea Never Made a Skilled SAILOR: Robust Imitation via Learning to Search",
  "abstract": "The fundamental limitation of the behavioral cloning (BC) approach to imitation learning is that it only teaches an agent what the expert did at states the expert visited. This means that when a BC agent makes a mistake which takes them out of the support of the demonstrations, they often don't know how to recover from it. In this sense, BC is akin to giving the agent the fish -- giving them dense supervision across a narrow set of states -- rather than teaching them to fish: to be able to reason independently about achieving the expert's outcome even when faced with unseen situations at test-time. In response, we explore learning to search (L2S) from expert demonstrations, i.e. learning the components required to, at test time, plan to match expert outcomes, even after making a mistake. These include (1) a world model and (2) a reward model. We carefully ablate the set of algorithmic and design decisions required to combine these and other components for stable and sample/interaction-efficient learning of recovery behavior without additional human corrections. Across a dozen visual manipulation tasks from three benchmarks, our approach SAILOR consistently out-performs state-of-the-art Diffusion Policies trained via BC on the same data. Furthermore, scaling up the amount of demonstrations used for BC by 5-10x still leaves a performance gap. We find that SAILOR can identify nuanced failures and is robust to reward hacking. Our code is available at https://github.com/arnavkj1995/SAILOR .",
  "published": "2025-06-05",
  "updated": "2025-10-24",
  "year": "2025",
  "authors": [
   "Arnav Kumar Jain",
   "Vibhakar Mohta",
   "Subin Kim",
   "Atiksh Bhardwaj",
   "Juntao Ren",
   "Yunhai Feng",
   "Sanjiban Choudhury",
   "Gokul Swamy"
  ],
  "author_count": 8,
  "categories": [
   "cs.LG"
  ],
  "primary_category": "cs.LG",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 25,
  "influential_citations": 2,
  "tldr": "Across a dozen visual manipulation tasks from three benchmarks, the approach SAILOR consistently out-performs state-of-the-art Diffusion Policies trained via BC on the same data and is robust to reward hacking.",
  "doi": "10.48550/arXiv.2506.05294",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "A. Jain",
    "id": "2319764306",
    "h_index": 6,
    "papers": 16
   },
   {
    "name": "Vibhakar Mohta",
    "id": "1471685177",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Subin Kim",
    "id": "2281283346",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Atiksh Bhardwaj",
    "id": "2261084253",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Juntao Ren",
    "id": "2284191057",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Yunhai Feng",
    "id": "2348093839",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Sanjiban Choudhury",
    "id": "2319670026",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Gokul Swamy",
    "id": "2073365488",
    "h_index": 14,
    "papers": 28
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "imitation-diffusion",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2506.05294v2",
  "pdf_url": "https://arxiv.org/pdf/2506.05294v2",
  "html_url": "https://arxiv.org/html/2506.05294v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.91
 },
 {
  "id": "2506.04227",
  "slug": "object-centric-3d-motion-field-for-robot-learning-from-human-videos",
  "title": "Object-centric 3D Motion Field for Robot Learning from Human Videos",
  "abstract": "Learning robot control policies from human videos is a promising direction for scaling up robot learning. However, how to extract action knowledge (or action representations) from videos for policy learning remains a key challenge. Existing action representations such as video frames, pixelflow, and pointcloud flow have inherent limitations such as modeling complexity or loss of information. In this paper, we propose to use object-centric 3D motion field to represent actions for robot learning from human videos, and present a novel framework for extracting this representation from videos for zero-shot control. We introduce two novel components in its implementation. First, a novel training pipeline for training a ''denoising'' 3D motion field estimator to extract fine object 3D motions from human videos with noisy depth robustly. Second, a dense object-centric 3D motion field prediction architecture that favors both cross-embodiment transfer and policy generalization to background. We evaluate the system in real world setups. Experiments show that our method reduces 3D motion estimation error by over 50% compared to the latest method, achieve 55% average success rate in diverse tasks where prior approaches fail~($\\lesssim 10$\\%), and can even acquire fine-grained manipulation skills like insertion.",
  "published": "2025-06-04",
  "updated": "2025-06-04",
  "year": "2025",
  "authors": [
   "Zhao-Heng Yin",
   "Sherry Yang",
   "Pieter Abbeel"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 15,
  "influential_citations": 0,
  "tldr": "Experiments show that the method reduces 3D motion estimation error by over 50% compared to the latest method, achieve 55% average success rate in diverse tasks where prior approaches fail, and can even acquire fine-grained manipulation skills like insertion.",
  "doi": "10.48550/arXiv.2506.04227",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhao-Heng Yin",
    "id": "2290035529",
    "h_index": 7,
    "papers": 12
   },
   {
    "name": "Sherry Yang",
    "id": "2336548160",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Pieter Abbeel",
    "id": "2257003229",
    "h_index": 11,
    "papers": 17
   }
  ],
  "comment": "Project: https://zhaohengyin.github.io/3DMF",
  "topics": [
   "egocentric-data",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2506.04227v1",
  "pdf_url": "https://arxiv.org/pdf/2506.04227v1",
  "html_url": "https://arxiv.org/html/2506.04227v1",
  "code_url": "https://zhaohengyin.github.io/3DMF",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.7
 },
 {
  "id": "2506.03517",
  "slug": "densedpo-fine-grained-temporal-preference-optimization-for-video-diffu",
  "title": "DenseDPO: Fine-Grained Temporal Preference Optimization for Video Diffusion Models",
  "abstract": "Direct Preference Optimization (DPO) has recently been applied as a post-training technique for text-to-video diffusion models. To obtain training data, annotators are asked to provide preferences between two videos generated from independent noise. However, this approach prohibits fine-grained comparisons, and we point out that it biases the annotators towards low-motion clips as they often contain fewer visual artifacts. In this work, we introduce DenseDPO, a method that addresses these shortcomings by making three contributions. First, we create each video pair for DPO by denoising corrupted copies of a ground truth video. This results in aligned pairs with similar motion structures while differing in local details, effectively neutralizing the motion bias. Second, we leverage the resulting temporal alignment to label preferences on short segments rather than entire clips, yielding a denser and more precise learning signal. With only one-third of the labeled data, DenseDPO greatly improves motion generation over vanilla DPO, while matching it in text alignment, visual quality, and temporal consistency. Finally, we show that DenseDPO unlocks automatic preference annotation using off-the-shelf Vision Language Models (VLMs): GPT accurately predicts segment-level preferences similar to task-specifically fine-tuned video reward models, and DenseDPO trained on these labels achieves performance close to using human labels.",
  "published": "2025-06-04",
  "updated": "2025-10-10",
  "year": "2025",
  "authors": [
   "Ziyi Wu",
   "Anil Kag",
   "Ivan Skorokhodov",
   "Willi Menapace",
   "Ashkan Mirzaei",
   "Igor Gilitschenski",
   "Sergey Tulyakov",
   "Aliaksandr Siarohin"
  ],
  "author_count": 8,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 34,
  "influential_citations": 3,
  "tldr": "DenseDPO is introduced, a method that greatly improves motion generation over vanilla DPO, while matching it in text alignment, visual quality, and temporal consistency and unlocks automatic preference annotation using off-the-shelf Vision Language Models (VLMs).",
  "doi": "10.48550/arXiv.2506.03517",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ziyi Wu",
    "id": "2253894469",
    "h_index": 7,
    "papers": 14
   },
   {
    "name": "Anil Kag",
    "id": "2284982329",
    "h_index": 9,
    "papers": 23
   },
   {
    "name": "Ivan Skorokhodov",
    "id": "51118864",
    "h_index": 22,
    "papers": 45
   },
   {
    "name": "W. Menapace",
    "id": "1698103472",
    "h_index": 22,
    "papers": 53
   },
   {
    "name": "Ashkan Mirzaei",
    "id": "2174737496",
    "h_index": 13,
    "papers": 25
   },
   {
    "name": "Igor Gilitschenski",
    "id": "2262216913",
    "h_index": 13,
    "papers": 24
   },
   {
    "name": "Sergey Tulyakov",
    "id": "2292401534",
    "h_index": 17,
    "papers": 63
   },
   {
    "name": "Aliaksandr Siarohin",
    "id": "10753214",
    "h_index": 31,
    "papers": 76
   }
  ],
  "comment": "NeurIPS 2025 Spotlight. Project page: https://snap-research.github.io/DenseDPO/",
  "topics": [
   "rl-control",
   "data-teleop",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2506.03517v2",
  "pdf_url": "https://arxiv.org/pdf/2506.03517v2",
  "html_url": "https://arxiv.org/html/2506.03517v2",
  "code_url": "https://snap-research.github.io/DenseDPO/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.04
 },
 {
  "id": "2506.01955",
  "slug": "dual-process-image-generation",
  "title": "Dual-Process Image Generation",
  "abstract": "Prior methods for controlling image generation are limited in their ability to be taught new tasks. In contrast, vision-language models, or VLMs, can learn tasks in-context and produce the correct outputs for a given input. We propose a dual-process distillation scheme that allows feed-forward image generators to learn new tasks from deliberative VLMs. Our scheme uses a VLM to rate the generated images and backpropagates this gradient to update the weights of the image generator. Our general framework enables a wide variety of new control tasks through the same text-and-image based interface. We showcase a handful of applications of this technique for different types of control signals, such as commonsense inferences and visual prompts. With our method, users can implement multimodal controls for properties such as color palette, line weight, horizon position, and relative depth within a matter of minutes. Project page: https://dual-process.github.io.",
  "published": "2025-06-02",
  "updated": "2025-06-02",
  "year": "2025",
  "authors": [
   "Grace Luo",
   "Jonathan Granskog",
   "Aleksander Holynski",
   "Trevor Darrell"
  ],
  "author_count": 4,
  "categories": [
   "cs.CV",
   "cs.CL",
   "cs.LG"
  ],
  "primary_category": "cs.CV",
  "venue": "ICCV",
  "venue_source": "semantic-scholar",
  "citations": 15,
  "influential_citations": 1,
  "tldr": "A dual-process distillation scheme that allows feed-forward image generators to learn new tasks from deliberative VLMs, which uses a VLM to rate the generated images and backpropagates this gradient to update the weights of the image generator.",
  "doi": "10.1109/ICCV51701.2025.01670",
  "oa_pdf": "https://arxiv.org/pdf/2506.01955",
  "s2_authors": [
   {
    "name": "Grace Luo",
    "id": "2074105688",
    "h_index": 9,
    "papers": 13
   },
   {
    "name": "Jonathan Granskog",
    "id": "104323907",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Aleksander Holynski",
    "id": "2333897728",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Trevor Darrell",
    "id": "2295666869",
    "h_index": 8,
    "papers": 12
   }
  ],
  "comment": "",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2506.01955v1",
  "pdf_url": "https://arxiv.org/pdf/2506.01955v1",
  "html_url": "https://arxiv.org/html/2506.01955v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.7
 },
 {
  "id": "2506.01944",
  "slug": "feel-the-force-contact-driven-learning-from-humans",
  "title": "Feel the Force: Contact-Driven Learning from Humans",
  "abstract": "Controlling fine-grained forces during manipulation remains a core challenge in robotics. While robot policies learned from robot-collected data or simulation show promise, they struggle to generalize across the diverse range of real-world interactions. Learning directly from humans offers a scalable solution, enabling demonstrators to perform skills in their natural embodiment and in everyday environments. However, visual demonstrations alone lack the information needed to infer precise contact forces. We present FeelTheForce (FTF): a robot learning system that models human tactile behavior to learn force-sensitive manipulation. Using a tactile glove to measure contact forces and a vision-based model to estimate hand pose, we train a closed-loop policy that continuously predicts the forces needed for manipulation. This policy is re-targeted to a Franka Panda robot with tactile gripper sensors using shared visual and action representations. At execution, a PD controller modulates gripper closure to track predicted forces-enabling precise, force-aware control. Our approach grounds robust low-level force control in scalable human supervision, achieving a 77% success rate across 5 force-sensitive manipulation tasks. Code and videos are available at https://feel-the-force-ftf.github.io.",
  "published": "2025-06-02",
  "updated": "2025-06-02",
  "year": "2025",
  "authors": [
   "Ademi Adeniji",
   "Zhuoran Chen",
   "Vincent Liu",
   "Venkatesh Pattabiraman",
   "Raunaq Bhirangi",
   "Siddhant Haldar",
   "Pieter Abbeel",
   "Lerrel Pinto"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 18,
  "influential_citations": 1,
  "tldr": "FeelTheForce is a robot learning system that models human tactile behavior to learn force-sensitive manipulation and grounds robust low-level force control in scalable human supervision, achieving a 77% success rate across 5 force-sensitive manipulation tasks.",
  "doi": "10.48550/arXiv.2506.01944",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ademi Adeniji",
    "id": "67338486",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Zhuoran Chen",
    "id": "2304293044",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Vincent Liu",
    "id": "2363579642",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Venkatesh Pattabiraman",
    "id": "2284217920",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Raunaq M. Bhirangi",
    "id": "152466940",
    "h_index": 11,
    "papers": 17
   },
   {
    "name": "Siddhant Haldar",
    "id": "51445278",
    "h_index": 13,
    "papers": 21
   },
   {
    "name": "Pieter Abbeel",
    "id": "2363582403",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Lerrel Pinto",
    "id": "2320806817",
    "h_index": 10,
    "papers": 12
   }
  ],
  "comment": "",
  "topics": [
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2506.01944v1",
  "pdf_url": "https://arxiv.org/pdf/2506.01944v1",
  "html_url": "https://arxiv.org/html/2506.01944v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.28
 },
 {
  "id": "2505.24853",
  "slug": "dexmachina-functional-retargeting-for-bimanual-dexterous-manipulation",
  "title": "DexMachina: Functional Retargeting for Bimanual Dexterous Manipulation",
  "abstract": "We study the problem of functional retargeting: learning dexterous manipulation policies to track object states from human hand-object demonstrations. We focus on long-horizon, bimanual tasks with articulated objects, which is challenging due to large action space, spatiotemporal discontinuities, and embodiment gap between human and robot hands. We propose DexMachina, a novel curriculum-based algorithm: the key idea is to use virtual object controllers with decaying strength: an object is first driven automatically towards its target states, such that the policy can gradually learn to take over under motion and contact guidance. We release a simulation benchmark with a diverse set of tasks and dexterous hands, and show that DexMachina significantly outperforms baseline methods. Our algorithm and benchmark enable a functional comparison for hardware designs, and we present key findings informed by quantitative and qualitative results. With the recent surge in dexterous hand development, we hope this work will provide a useful platform for identifying desirable hardware capabilities and lower the barrier for contributing to future research. Videos and more at https://project-dexmachina.github.io/",
  "published": "2025-05-30",
  "updated": "2025-05-30",
  "year": "2025",
  "authors": [
   "Zhao Mandi",
   "Yifan Hou",
   "Dieter Fox",
   "Yashraj Narang",
   "Ajay Mandlekar",
   "Shuran Song"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 55,
  "influential_citations": 6,
  "tldr": "This work proposes DexMachina, a novel curriculum-based algorithm, to use virtual object controllers with decaying strength: an object is first driven automatically towards its target states, such that the policy can gradually learn to take over under motion and contact guidance.",
  "doi": "10.48550/arXiv.2505.24853",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhao Mandi",
    "id": "2126966292",
    "h_index": 6,
    "papers": 6
   },
   {
    "name": "Yifan Hou",
    "id": "2327832445",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Dieter Fox",
    "id": "2258436157",
    "h_index": 12,
    "papers": 14
   },
   {
    "name": "Yashraj S. Narang",
    "id": "5046361",
    "h_index": 20,
    "papers": 36
   },
   {
    "name": "A. Mandlekar",
    "id": "49686756",
    "h_index": 36,
    "papers": 67
   },
   {
    "name": "Shuran Song",
    "id": "2364257433",
    "h_index": 6,
    "papers": 17
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "sim2real",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2505.24853v1",
  "pdf_url": "https://arxiv.org/pdf/2505.24853v1",
  "html_url": "https://arxiv.org/html/2505.24853v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.75
 },
 {
  "id": "2505.23705",
  "slug": "knowledge-insulating-vision-language-action-models-train-fast-run-fast",
  "title": "Knowledge Insulating Vision-Language-Action Models: Train Fast, Run Fast, Generalize Better",
  "abstract": "Vision-language-action (VLA) models provide a powerful approach to training control policies for physical systems, such as robots, by combining end-to-end learning with transfer of semantic knowledge from web-scale vision-language model (VLM) training. However, the constraints of real-time control are often at odds with the design of VLMs: the most powerful VLMs have tens or hundreds of billions of parameters, presenting an obstacle to real-time inference, and operate on discrete tokens rather than the continuous-valued outputs that are required for controlling robots. To address this challenge, recent VLA models have used specialized modules for efficient continuous control, such as action experts or continuous output heads, which typically require adding new untrained parameters to the pretrained VLM backbone. While these modules improve real-time and control capabilities, it remains an open question whether they preserve or degrade the semantic knowledge contained in the pretrained VLM, and what effect they have on the VLA training dynamics. In this paper, we study this question in the context of VLAs that include a continuous diffusion or flow matching action expert, showing that naively including such experts significantly harms both training speed and knowledge transfer. We provide an extensive analysis of various design choices, their impact on performance and knowledge transfer, and propose a technique for insulating the VLM backbone during VLA training that mitigates this issue. Videos are available at https://pi.website/research/knowledge_insulation.",
  "published": "2025-05-29",
  "updated": "2025-05-29",
  "year": "2025",
  "authors": [
   "Danny Driess",
   "Jost Tobias Springenberg",
   "Brian Ichter",
   "Lili Yu",
   "Adrian Li-Bell",
   "Karl Pertsch",
   "Allen Z. Ren",
   "Homer Walke",
   "Quan Vuong",
   "Lucy Xiaoyang Shi",
   "Sergey Levine"
  ],
  "author_count": 11,
  "categories": [
   "cs.LG",
   "cs.RO"
  ],
  "primary_category": "cs.LG",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 131,
  "influential_citations": 16,
  "tldr": "This paper provides an extensive analysis of various design choices, their impact on performance and knowledge transfer, and proposes a technique for insulating the VLM backbone during VLA training that mitigates this issue.",
  "doi": "10.48550/arXiv.2505.23705",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Danny Driess",
    "id": "2283848260",
    "h_index": 27,
    "papers": 35
   },
   {
    "name": "Jost Tobias Springenberg",
    "id": "2060551",
    "h_index": 44,
    "papers": 93
   },
   {
    "name": "Brian Ichter",
    "id": "2704814",
    "h_index": 37,
    "papers": 60
   },
   {
    "name": "Lili Yu",
    "id": "2356801221",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Adrian Li-Bell",
    "id": "2332927424",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Karl Pertsch",
    "id": "31719101",
    "h_index": 32,
    "papers": 56
   },
   {
    "name": "Allen Z. Ren",
    "id": "2356677628",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "H. Walke",
    "id": "2029241116",
    "h_index": 17,
    "papers": 23
   },
   {
    "name": "Quan Vuong",
    "id": "2288210223",
    "h_index": 19,
    "papers": 29
   },
   {
    "name": "L. Shi",
    "id": "2292341452",
    "h_index": 12,
    "papers": 17
   },
   {
    "name": "Sergey Levine",
    "id": "2249615151",
    "h_index": 32,
    "papers": 53
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2505.23705v1",
  "pdf_url": "https://arxiv.org/pdf/2505.23705v1",
  "html_url": "https://arxiv.org/html/2505.23705v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.62
 },
 {
  "id": "2505.22159",
  "slug": "forcevla-enhancing-vla-models-with-a-force-aware-moe-for-contact-rich",
  "title": "ForceVLA: Enhancing VLA Models with a Force-aware MoE for Contact-rich Manipulation",
  "abstract": "Vision-Language-Action (VLA) models have advanced general-purpose robotic manipulation by leveraging pretrained visual and linguistic representations. However, they struggle with contact-rich tasks that require fine-grained control involving force, especially under visual occlusion or dynamic uncertainty. To address these limitations, we propose ForceVLA, a novel end-to-end manipulation framework that treats external force sensing as a first-class modality within VLA systems. ForceVLA introduces FVLMoE, a force-aware Mixture-of-Experts fusion module that dynamically integrates pretrained visual-language embeddings with real-time 6-axis force feedback during action decoding. This enables context-aware routing across modality-specific experts, enhancing the robot's ability to adapt to subtle contact dynamics. We also introduce \\textbf{ForceVLA-Data}, a new dataset comprising synchronized vision, proprioception, and force-torque signals across five contact-rich manipulation tasks. ForceVLA improves average task success by 23.2% over strong pi_0-based baselines, achieving up to 80% success in tasks such as plug insertion. Our approach highlights the importance of multimodal integration for dexterous manipulation and sets a new benchmark for physically intelligent robotic control. Code and data will be released at https://sites.google.com/view/forcevla2025.",
  "published": "2025-05-28",
  "updated": "2025-09-18",
  "year": "2025",
  "authors": [
   "Jiawen Yu",
   "Hairuo Liu",
   "Qiaojun Yu",
   "Jieji Ren",
   "Ce Hao",
   "Haitong Ding",
   "Guangyu Huang",
   "Guofan Huang",
   "Yan Song",
   "Panpan Cai",
   "Cewu Lu",
   "Wenqiang Zhang"
  ],
  "author_count": 12,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 112,
  "influential_citations": 5,
  "tldr": "This work proposes ForceVLA, a novel end-to-end manipulation framework that treats external force sensing as a first-class modality within VLA systems and introduces FVLMoE, a force-aware Mixture-of-Experts fusion module that dynamically integrates pretrained visual-language embeddings with real-time 6-axis force feedback during action decoding.",
  "doi": "10.48550/arXiv.2505.22159",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jiawen Yu",
    "id": "2277144624",
    "h_index": 6,
    "papers": 16
   },
   {
    "name": "Hairuo Liu",
    "id": "2364015850",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Qiaojun Yu",
    "id": "2153795608",
    "h_index": 10,
    "papers": 23
   },
   {
    "name": "Jieji Ren",
    "id": "2364840092",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Ce Hao",
    "id": "2247914643",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Haitong Ding",
    "id": "2381803017",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Guangyu Huang",
    "id": "2456345962",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Guofan Huang",
    "id": "2328259146",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Yan Song",
    "id": "2364074345",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Panpan Cai",
    "id": "2323565504",
    "h_index": 4,
    "papers": 14
   },
   {
    "name": "Cewu Lu",
    "id": "2323718989",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Wenqiang Zhang",
    "id": "2363877464",
    "h_index": 1,
    "papers": 2
   }
  ],
  "comment": "NeurIPS 2025",
  "topics": [
   "vla",
   "dexterous-manipulation",
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2505.22159v3",
  "pdf_url": "https://arxiv.org/pdf/2505.22159v3",
  "html_url": "https://arxiv.org/html/2505.22159v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.55
 },
 {
  "id": "2505.21864",
  "slug": "dexumi-using-human-hand-as-the-universal-manipulation-interface-for-de",
  "title": "DexUMI: Using Human Hand as the Universal Manipulation Interface for Dexterous Manipulation",
  "abstract": "We present DexUMI - a data collection and policy learning framework that uses the human hand as the natural interface to transfer dexterous manipulation skills to various robot hands. DexUMI includes hardware and software adaptations to minimize the embodiment gap between the human hand and various robot hands. The hardware adaptation bridges the kinematics gap using a wearable hand exoskeleton. It allows direct haptic feedback in manipulation data collection and adapts human motion to feasible robot hand motion. The software adaptation bridges the visual gap by replacing the human hand in video data with high-fidelity robot hand inpainting. We demonstrate DexUMI's capabilities through comprehensive real-world experiments on two different dexterous robot hand hardware platforms, achieving an average task success rate of 86%.",
  "published": "2025-05-28",
  "updated": "2025-10-02",
  "year": "2025",
  "authors": [
   "Mengda Xu",
   "Han Zhang",
   "Yifan Hou",
   "Zhenjia Xu",
   "Linxi Fan",
   "Manuela Veloso",
   "Shuran Song"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 111,
  "influential_citations": 5,
  "tldr": "",
  "doi": "10.48550/arXiv.2505.21864",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Mengda Xu",
    "id": "2110683030",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "H. Zhang",
    "id": "2183907239",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Yifan Hou",
    "id": "2327832445",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Zhenjia Xu",
    "id": "2329068450",
    "h_index": 6,
    "papers": 6
   },
   {
    "name": "L. Fan",
    "id": "2257381161",
    "h_index": 18,
    "papers": 25
   },
   {
    "name": "Manuela Veloso",
    "id": "2312325086",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Shuran Song",
    "id": "2364257433",
    "h_index": 6,
    "papers": 17
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2505.21864v3",
  "pdf_url": "https://arxiv.org/pdf/2505.21864v3",
  "html_url": "https://arxiv.org/html/2505.21864v3",
  "code_url": null,
  "presented": [
   {
    "club": "saturday-robotics",
    "club_name": "Saturday Robotics",
    "session_number": 8,
    "session_title": "Robotics & World Models Reading Club 08: Embodied Human Data as the \u201cInternet of Motion and Behavior\u201d \u2014 San Francisco 0516",
    "date_text": "Saturday, May 16, 2026 | 2:00 PM \u2013 5:00 PM",
    "city": "San Francisco, CA",
    "url": "https://lu.ma/qoxioge7",
    "listed_as": "Ego-Exo Transfer: Learning Action from First- and Third-Person Data"
   }
  ],
  "club_note": "Learns view-invariant action representations by jointly leveraging egocentric and exocentric video data.",
  "featured": true,
  "signal": 5.05
 },
 {
  "id": "2506.15691",
  "slug": "what-do-latent-action-models-actually-learn",
  "title": "What Do Latent Action Models Actually Learn?",
  "abstract": "Latent action models (LAMs) aim to learn action-relevant changes from unlabeled videos by compressing changes between frames as latents. However, differences between video frames can be caused by controllable changes as well as exogenous noise, leading to an important concern -- do latents capture the changes caused by actions or irrelevant noise? This paper studies this issue analytically, presenting a linear model that encapsulates the essence of LAM learning, while being tractable.This provides several insights, including connections between LAM and principal component analysis (PCA), desiderata of the data-generating policy, and justification of strategies to encourage learning controllable changes using data augmentation, data cleaning, and auxiliary action-prediction. We also provide illustrative results based on numerical simulation, shedding light on the specific structure of observations, actions, and noise in data that influence LAM learning.",
  "published": "2025-05-27",
  "updated": "2025-11-12",
  "year": "2025",
  "authors": [
   "Chuheng Zhang",
   "Tim Pearce",
   "Pushi Zhang",
   "Kaixin Wang",
   "Xiaoyu Chen",
   "Wei Shen",
   "Li Zhao",
   "Jiang Bian"
  ],
  "author_count": 8,
  "categories": [
   "cs.LG",
   "cs.AI"
  ],
  "primary_category": "cs.LG",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 31,
  "influential_citations": 2,
  "tldr": "This paper studies the issue analytically, presenting a linear model that encapsulates the essence of LAM learning, while being tractable, and provides several insights, including connections between LAM and principal component analysis (PCA), desiderata of the data-generating policy, and justification of strategies to encourage learning controllable changes using data augmentation, data cleaning, and auxiliary action-prediction.",
  "doi": "10.48550/arXiv.2506.15691",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Chuheng Zhang",
    "id": "144114271",
    "h_index": 15,
    "papers": 36
   },
   {
    "name": "Tim Pearce",
    "id": "2350757236",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Pushi Zhang",
    "id": "1570021289",
    "h_index": 9,
    "papers": 19
   },
   {
    "name": "Kaixin Wang",
    "id": "2367933865",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "Xiaoyu Chen",
    "id": "2329208858",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Wei Shen",
    "id": "2319630583",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Li Zhao",
    "id": "2218154011",
    "h_index": 7,
    "papers": 25
   },
   {
    "name": "Jiang Bian",
    "id": "2287806843",
    "h_index": 7,
    "papers": 16
   }
  ],
  "comment": "Accepted by NeurIPS-25",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2506.15691v3",
  "pdf_url": "https://arxiv.org/pdf/2506.15691v3",
  "html_url": "https://arxiv.org/html/2506.15691v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.01
 },
 {
  "id": "2505.20290",
  "slug": "egozero-robot-learning-from-smart-glasses",
  "title": "EgoZero: Robot Learning from Smart Glasses",
  "abstract": "Despite recent progress in general purpose robotics, robot policies still lag far behind basic human capabilities in the real world. Humans interact constantly with the physical world, yet this rich data resource remains largely untapped in robot learning. We propose EgoZero, a minimal system that learns robust manipulation policies from human demonstrations captured with Project Aria smart glasses, $\\textbf{and zero robot data}$. EgoZero enables: (1) extraction of complete, robot-executable actions from in-the-wild, egocentric, human demonstrations, (2) compression of human visual observations into morphology-agnostic state representations, and (3) closed-loop policy learning that generalizes morphologically, spatially, and semantically. We deploy EgoZero policies on a gripper Franka Panda robot and demonstrate zero-shot transfer with 70% success rate over 7 manipulation tasks and only 20 minutes of data collection per task. Our results suggest that in-the-wild human data can serve as a scalable foundation for real-world robot learning - paving the way toward a future of abundant, diverse, and naturalistic training data for robots. Code and videos are available at https://egozero-robot.github.io.",
  "published": "2025-05-26",
  "updated": "2025-06-03",
  "year": "2025",
  "authors": [
   "Vincent Liu",
   "Ademi Adeniji",
   "Haotian Zhan",
   "Siddhant Haldar",
   "Raunaq Bhirangi",
   "Pieter Abbeel",
   "Lerrel Pinto"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 41,
  "influential_citations": 1,
  "tldr": "EgoZero is a minimal system that learns robust manipulation policies from human demonstrations captured with Project Aria smart glasses, and suggests that in-the-wild human data can serve as a scalable foundation for real-world robot learning.",
  "doi": "10.48550/arXiv.2505.20290",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Vincent Liu",
    "id": "2363579642",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Ademi Adeniji",
    "id": "67338486",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Haotian Zhan",
    "id": "2363565779",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Raunaq M. Bhirangi",
    "id": "152466940",
    "h_index": 11,
    "papers": 17
   },
   {
    "name": "Pieter Abbeel",
    "id": "2363582403",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Lerrel Pinto",
    "id": "2320806817",
    "h_index": 10,
    "papers": 12
   }
  ],
  "comment": "",
  "topics": [
   "egocentric-data",
   "data-teleop",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2505.20290v2",
  "pdf_url": "https://arxiv.org/pdf/2505.20290v2",
  "html_url": "https://arxiv.org/html/2505.20290v2",
  "code_url": null,
  "presented": [
   {
    "club": "saturday-robotics",
    "club_name": "Saturday Robotics",
    "session_number": 13,
    "session_title": "Robotics & World Models Reading Club 13: HumanEgo: Train Robot Policy from 30 min Egocentric Videos \u2014 SF 0620",
    "date_text": "Saturday, June 20, 2026 | 2:00 PM \u2013 5:00 PM",
    "city": "San Francisco, CA",
    "url": "https://lu.ma/6vkhxnum",
    "listed_as": ""
   }
  ],
  "club_note": "",
  "featured": true,
  "signal": 4.62
 },
 {
  "id": "2505.20283",
  "slug": "category-agnostic-neural-object-rigging",
  "title": "Category-Agnostic Neural Object Rigging",
  "abstract": "The motion of deformable 4D objects lies in a low-dimensional manifold. To better capture the low dimensionality and enable better controllability, traditional methods have devised several heuristic-based methods, i.e., rigging, for manipulating dynamic objects in an intuitive fashion. However, such representations are not scalable due to the need for expert knowledge of specific categories. Instead, we study the automatic exploration of such low-dimensional structures in a purely data-driven manner. Specifically, we design a novel representation that encodes deformable 4D objects into a sparse set of spatially grounded blobs and an instance-aware feature volume to disentangle the pose and instance information of the 3D shape. With such a representation, we can manipulate the pose of 3D objects intuitively by modifying the parameters of the blobs, while preserving rich instance-specific information. We evaluate the proposed method on a variety of object categories and demonstrate the effectiveness of the proposed framework. Project page: https://guangzhaohe.com/canor",
  "published": "2025-05-26",
  "updated": "2025-05-26",
  "year": "2025",
  "authors": [
   "Guangzhao He",
   "Chen Geng",
   "Shangzhe Wu",
   "Jiajun Wu"
  ],
  "author_count": 4,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 8,
  "influential_citations": 0,
  "tldr": "This work designs a novel representation that encodes deformable 4D objects into a sparse set of spatially grounded blobs and an instance-aware feature volume to disentangle the pose and instance information of the 3D shape and can manipulate the pose of 3D objects intuitively by modifying the parameters of the blobs, while preserving rich instance-specific information.",
  "doi": "10.1109/CVPR52734.2025.02056",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Guangzhao He",
    "id": "2258959915",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Chen Geng",
    "id": "2334353317",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Shangzhe Wu",
    "id": "2112538311",
    "h_index": 11,
    "papers": 21
   },
   {
    "name": "Jiajun Wu",
    "id": "2326444500",
    "h_index": 35,
    "papers": 214
   }
  ],
  "comment": "Accepted to CVPR 2025. Project Page: https://guangzhaohe.com/canor",
  "topics": [
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2505.20283v1",
  "pdf_url": "https://arxiv.org/pdf/2505.20283v1",
  "html_url": "https://arxiv.org/html/2505.20283v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.45
 },
 {
  "id": "2505.19017",
  "slug": "worldeval-world-model-as-real-world-robot-policies-evaluator",
  "title": "WorldEval: World Model as Real-World Robot Policies Evaluator",
  "abstract": "The field of robotics has made significant strides toward developing generalist robot manipulation policies. However, evaluating these policies in real-world scenarios remains time-consuming and challenging, particularly as the number of tasks scales and environmental conditions change. In this work, we demonstrate that world models can serve as a scalable, reproducible, and reliable proxy for real-world robot policy evaluation. A key challenge is generating accurate policy videos from world models that faithfully reflect the robot actions. We observe that directly inputting robot actions or using high-dimensional encoding methods often fails to generate action-following videos. To address this, we propose Policy2Vec, a simple yet effective approach to turn a video generation model into a world simulator that follows latent action to generate the robot video. We then introduce WorldEval, an automated pipeline designed to evaluate real-world robot policies entirely online. WorldEval effectively ranks various robot policies and individual checkpoints within a single policy, and functions as a safety detector to prevent dangerous actions by newly developed robot models. Through comprehensive paired evaluations of manipulation policies in real-world environments, we demonstrate a strong correlation between policy performance in WorldEval and real-world scenarios. Furthermore, our method significantly outperforms popular methods such as real-to-sim approach.",
  "published": "2025-05-25",
  "updated": "2025-05-25",
  "year": "2025",
  "authors": [
   "Yaxuan Li",
   "Yichen Zhu",
   "Junjie Wen",
   "Chaomin Shen",
   "Yi Xu"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 56,
  "influential_citations": 5,
  "tldr": "This work proposes Policy2Vec, a simple yet effective approach to turn a video generation model into a world simulator that follows latent action to generate the robot video, and introduces WorldEval, an automated pipeline designed to evaluate real-world robot policies entirely online.",
  "doi": "10.48550/arXiv.2505.19017",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yaxuan Li",
    "id": "2363501851",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Yichen Zhu",
    "id": "2275531481",
    "h_index": 21,
    "papers": 36
   },
   {
    "name": "Junjie Wen",
    "id": "2278247833",
    "h_index": 17,
    "papers": 22
   },
   {
    "name": "Chaomin Shen",
    "id": "2335417053",
    "h_index": 9,
    "papers": 11
   },
   {
    "name": "Yi Xu",
    "id": "2363513002",
    "h_index": 5,
    "papers": 9
   }
  ],
  "comment": "The project page is available at https://worldeval.github.io",
  "topics": [
   "world-models",
   "sim2real",
   "data-teleop",
   "video-generation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2505.19017v1",
  "pdf_url": "https://arxiv.org/pdf/2505.19017v1",
  "html_url": "https://arxiv.org/html/2505.19017v1",
  "code_url": "https://worldeval.github.io",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.76
 },
 {
  "id": "2505.18472",
  "slug": "manifeel-benchmarking-and-understanding-visuotactile-manipulation-poli",
  "title": "ManiFeel: Benchmarking and Understanding Visuotactile Manipulation Policy Learning",
  "abstract": "Supervised visuomotor policies have shown strong performance in robotic manipulation but often struggle in tasks with limited visual inputs, such as operations in confined spaces and dimly lit environments, or tasks requiring precise perception of object properties and environmental interactions. In such cases, tactile feedback becomes essential for manipulation. While the rapid progress of supervised visuomotor policies has benefited greatly from high-quality, reproducible simulation benchmarks in visual imitation, the visuotactile domain still lacks a similarly comprehensive and reliable benchmark for large-scale and rigorous evaluation. To address this, we introduce ManiFeel, a reproducible and scalable simulation benchmark designed to systematically study supervised visuotactile policy learning. ManiFeel offers a diverse suite of contact-rich and visually challenging manipulation tasks, a modular evaluation pipeline spanning sensing modalities, tactile representations, and policy architectures, as well as real-world validation. Through extensive experiments, ManiFeel demonstrates how tactile sensing enhances policy performance across diverse manipulation scenarios, ranging from precise contact-driven operations to visually constrained settings. In addition, the results reveal task-dependent strengths of different tactile modalities and identify key design principles and open challenges for robust visuotactile policy learning. Real-world evaluations further confirm that ManiFeel provides a reliable and meaningful foundation for benchmarking and future visuotactile policy development. To foster reproducibility and future research, we will release our codebase, datasets, training logs, and pretrained checkpoints, aiming to accelerate progress toward generalizable visuotactile policy learning and manipulation.",
  "published": "2025-05-24",
  "updated": "2026-01-12",
  "year": "2025",
  "authors": [
   "Quan Khanh Luu",
   "Pokuang Zhou",
   "Zhengtong Xu",
   "Zhiyuan Zhang",
   "Qiang Qiu",
   "Yu She"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 21,
  "influential_citations": 0,
  "tldr": "ManiFeel is introduced, a reproducible and scalable simulation benchmark designed to systematically study supervised visuotactile policy learning and manipulation that demonstrates how tactile sensing enhances policy performance across diverse manipulation scenarios, ranging from precise contact-driven operations to visually constrained settings.",
  "doi": "10.48550/arXiv.2505.18472",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Q. Luu",
    "id": "2406884310",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Pokuang Zhou",
    "id": "2314146911",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Zhengtong Xu",
    "id": "2238437162",
    "h_index": 8,
    "papers": 26
   },
   {
    "name": "Zhiyuan Zhang",
    "id": "2363492892",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Qiang Qiu",
    "id": "2238332103",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Yu She",
    "id": "2238342225",
    "h_index": 7,
    "papers": 24
   }
  ],
  "comment": "",
  "topics": [
   "tactile",
   "sim2real"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2505.18472v2",
  "pdf_url": "https://arxiv.org/pdf/2505.18472v2",
  "html_url": "https://arxiv.org/html/2505.18472v2",
  "code_url": null,
  "presented": [
   {
    "club": "saturday-robotics",
    "club_name": "Saturday Robotics",
    "session_number": 17,
    "session_title": "Robotics & World Models Reading Club 17: Soft Tactile-Centric Multimodal Intelligence Toward Safe and Dexterous Manipulation. SF 07/11",
    "date_text": "Saturday, July 11, 2026 | 2:00 PM \u2013 5:00 PM",
    "city": "San Francisco, CA",
    "url": "https://lu.ma/e53zawq2",
    "listed_as": "ManiFeel preprint and project website:"
   }
  ],
  "club_note": "Website: https://zhengtongxu.github.io/manifeel-website/",
  "featured": true,
  "signal": 4.34
 },
 {
  "id": "2505.18151",
  "slug": "wonderplay-dynamic-3d-scene-generation-from-a-single-image-and-actions",
  "title": "WonderPlay: Dynamic 3D Scene Generation from a Single Image and Actions",
  "abstract": "WonderPlay is a novel framework integrating physics simulation with video generation for generating action-conditioned dynamic 3D scenes from a single image. While prior works are restricted to rigid body or simple elastic dynamics, WonderPlay features a hybrid generative simulator to synthesize a wide range of 3D dynamics. The hybrid generative simulator first uses a physics solver to simulate coarse 3D dynamics, which subsequently conditions a video generator to produce a video with finer, more realistic motion. The generated video is then used to update the simulated dynamic 3D scene, closing the loop between the physics solver and the video generator. This approach enables intuitive user control to be combined with the accurate dynamics of physics-based simulators and the expressivity of diffusion-based video generators. Experimental results demonstrate that WonderPlay enables users to interact with various scenes of diverse content, including cloth, sand, snow, liquid, smoke, elastic, and rigid bodies -- all using a single image input. Code will be made public. Project website: https://kyleleey.github.io/WonderPlay/",
  "published": "2025-05-23",
  "updated": "2025-11-29",
  "year": "2025",
  "authors": [
   "Zizhang Li",
   "Hong-Xing Yu",
   "Wei Liu",
   "Yin Yang",
   "Charles Herrmann",
   "Gordon Wetzstein",
   "Jiajun Wu"
  ],
  "author_count": 7,
  "categories": [
   "cs.GR",
   "cs.AI",
   "cs.CV"
  ],
  "primary_category": "cs.GR",
  "venue": "ICCV",
  "venue_source": "semantic-scholar",
  "citations": 45,
  "influential_citations": 2,
  "tldr": "Experimental results demonstrate that WonderPlay enables users to interact with various scenes of diverse content, including cloth, sand, snow, liquid, smoke, elastic, and rigid bodies - all using a single image input.",
  "doi": "10.1109/ICCV51701.2025.00849",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zizhang Li",
    "id": "2118275555",
    "h_index": 11,
    "papers": 23
   },
   {
    "name": "Hong-Xing Yu",
    "id": "2239448099",
    "h_index": 18,
    "papers": 32
   },
   {
    "name": "Wei Liu",
    "id": "2351750467",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Yin Yang",
    "id": "2363456425",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Charles Herrmann",
    "id": "2363348424",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Gordon Wetzstein",
    "id": "2297763521",
    "h_index": 7,
    "papers": 21
   },
   {
    "name": "Jiajun Wu",
    "id": "2326444500",
    "h_index": 35,
    "papers": 214
   }
  ],
  "comment": "ICCV 2025 (Highlight). The first two authors contributed equally. Project website: https://kyleleey.github.io/WonderPlay/",
  "topics": [
   "sim2real",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2505.18151v2",
  "pdf_url": "https://arxiv.org/pdf/2505.18151v2",
  "html_url": "https://arxiv.org/html/2505.18151v2",
  "code_url": "https://kyleleey.github.io/WonderPlay/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.16
 },
 {
  "id": "2505.17006",
  "slug": "como-learning-continuous-latent-motion-from-internet-videos-for-scalab",
  "title": "CoMo: Learning Continuous Latent Motion from Internet Videos for Scalable Robot Learning",
  "abstract": "Unsupervised learning of latent motion from Internet videos is crucial for robot learning. Existing discrete methods generally mitigate the shortcut learning caused by extracting excessive static backgrounds through vector quantization with a small codebook size. However, they suffer from information loss and struggle to capture more complex and fine-grained dynamics. Moreover, there is an inherent gap between the distribution of discrete latent motion and continuous robot action, which hinders the joint learning of a unified policy. We propose CoMo, which aims to learn more precise continuous latent motion from internet-scale videos. CoMo employs an early temporal difference (Td) mechanism to increase the shortcut learning difficulty and explicitly enhance motion cues. Additionally, to ensure latent motion better captures meaningful foregrounds, we further propose a temporal contrastive learning (Tcl) scheme. Specifically, positive pairs are constructed with a small future frame temporal offset, while negative pairs are formed by directly reversing the temporal direction. The proposed Td and Tcl work synergistically and effectively ensure that the latent motion focuses better on the foreground and reinforces motion cues. Critically, CoMo exhibits strong zeroshot generalization, enabling it to generate effective pseudo action labels for unseen videos. Extensive simulated and real-world experiments show that policies co-trained with CoMo pseudo action labels achieve superior performance with both diffusion and auto-regressive architectures.",
  "published": "2025-05-22",
  "updated": "2026-06-18",
  "year": "2025",
  "authors": [
   "Jiange Yang",
   "Yansong Shi",
   "Haoyi Zhu",
   "Mingyu Liu",
   "Kaijing Ma",
   "Yating Wang",
   "Gangshan Wu",
   "Tong He",
   "Limin Wang"
  ],
  "author_count": 9,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR 2026",
  "venue_source": "arxiv-comment",
  "citations": 40,
  "influential_citations": 2,
  "tldr": "The proposed CoMo aims to learn more precise continuous latent motion from internet-scale videos by employing an early temporal difference mechanism to increase the shortcut learning difficulty and explicitly enhance motion cues, and proposes a temporal contrastive learning (Tcl) scheme.",
  "doi": "10.48550/arXiv.2505.17006",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jiange Yang",
    "id": "2220590629",
    "h_index": 9,
    "papers": 12
   },
   {
    "name": "Yansong Shi",
    "id": "2293684856",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Haoyi Zhu",
    "id": "2331578393",
    "h_index": 9,
    "papers": 17
   },
   {
    "name": "Mingyu Liu",
    "id": "2363320659",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Kai Ma",
    "id": "2310573301",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Yating Wang",
    "id": "2282545307",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Gangshan Wu",
    "id": "2249716237",
    "h_index": 11,
    "papers": 34
   },
   {
    "name": "Tong He",
    "id": "2325202989",
    "h_index": 7,
    "papers": 12
   },
   {
    "name": "Limin Wang",
    "id": "2290519814",
    "h_index": 8,
    "papers": 9
   }
  ],
  "comment": "CVPR 2026",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2505.17006v3",
  "pdf_url": "https://arxiv.org/pdf/2505.17006v3",
  "html_url": "https://arxiv.org/html/2505.17006v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.11
 },
 {
  "id": "2505.16602",
  "slug": "megohand-multimodal-egocentric-hand-object-interaction-motion-generati",
  "title": "MEgoHand: Multimodal Egocentric Hand-Object Interaction Motion Generation",
  "abstract": "Egocentric hand-object motion generation is crucial for immersive AR/VR and robotic imitation but remains challenging due to unstable viewpoints, self-occlusions, perspective distortion, and noisy ego-motion. Existing methods rely on predefined 3D object priors, limiting generalization to novel objects, which restricts their generalizability to novel objects. Meanwhile, recent multimodal approaches suffer from ambiguous generation from abstract textual cues, intricate pipelines for modeling 3D hand-object correlation, and compounding errors in open-loop prediction. We propose MEgoHand, a multimodal framework that synthesizes physically plausible hand-object interactions from egocentric RGB, text, and initial hand pose. MEgoHand introduces a bi-level architecture: a high-level \"cerebrum\" leverages a vision language model (VLM) to infer motion priors from visual-textual context and a monocular depth estimator for object-agnostic spatial reasoning, while a low-level DiT-based flow-matching policy generates fine-grained trajectories with temporal orthogonal filtering to enhance stability. To address dataset inconsistency, we design a dataset curation paradigm with an Inverse MANO Retargeting Network and Virtual RGB-D Renderer, curating a unified dataset of 3.35M RGB-D frames, 24K interactions, and 1.2K objects. Extensive experiments across five in-domain and two cross-domain datasets demonstrate the effectiveness of MEgoHand, achieving substantial reductions in wrist translation error (86.9%) and joint rotation error (34.1%), highlighting its capacity to accurately model fine-grained hand joint structures and generalize robustly across diverse scenarios.",
  "published": "2025-05-22",
  "updated": "2025-05-22",
  "year": "2025",
  "authors": [
   "Bohan Zhou",
   "Yi Zhan",
   "Zhongbin Zhang",
   "Zongqing Lu"
  ],
  "author_count": 4,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 12,
  "influential_citations": 0,
  "tldr": "MEgoHand is proposed, a multimodal framework that synthesizes physically plausible hand-object interactions from egocentric RGB, text, and initial hand pose and introduces a bi-level architecture, highlighting its capacity to accurately model fine-grained hand joint structures and generalize robustly across diverse scenarios.",
  "doi": "10.48550/arXiv.2505.16602",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Bohan Zhou",
    "id": "2290134533",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Yi Zhan",
    "id": "2350514215",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Zhongbin Zhang",
    "id": "2264806961",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Zongqing Lu",
    "id": "2258676670",
    "h_index": 29,
    "papers": 144
   }
  ],
  "comment": "",
  "topics": [
   "egocentric-data",
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2505.16602v1",
  "pdf_url": "https://arxiv.org/pdf/2505.16602v1",
  "html_url": "https://arxiv.org/html/2505.16602v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.61
 },
 {
  "id": "2505.15659",
  "slug": "flare-robot-learning-with-implicit-world-modeling",
  "title": "FLARE: Robot Learning with Implicit World Modeling",
  "abstract": "We introduce $\\textbf{F}$uture $\\textbf{LA}$tent $\\textbf{RE}$presentation Alignment ($\\textbf{FLARE}$), a novel framework that integrates predictive latent world modeling into robot policy learning. By aligning features from a diffusion transformer with latent embeddings of future observations, $\\textbf{FLARE}$ enables a diffusion transformer policy to anticipate latent representations of future observations, allowing it to reason about long-term consequences while generating actions. Remarkably lightweight, $\\textbf{FLARE}$ requires only minimal architectural modifications -- adding a few tokens to standard vision-language-action (VLA) models -- yet delivers substantial performance gains. Across two challenging multitask simulation imitation learning benchmarks spanning single-arm and humanoid tabletop manipulation, $\\textbf{FLARE}$ achieves state-of-the-art performance, outperforming prior policy learning baselines by up to 26%. Moreover, $\\textbf{FLARE}$ unlocks the ability to co-train with human egocentric video demonstrations without action labels, significantly boosting policy generalization to a novel object with unseen geometry with as few as a single robot demonstration. Our results establish $\\textbf{FLARE}$ as a general and scalable approach for combining implicit world modeling with high-frequency robotic control.",
  "published": "2025-05-21",
  "updated": "2025-05-21",
  "year": "2025",
  "authors": [
   "Ruijie Zheng",
   "Jing Wang",
   "Scott Reed",
   "Johan Bjorck",
   "Yu Fang",
   "Fengyuan Hu",
   "Joel Jang",
   "Kaushil Kundalia",
   "Zongyu Lin",
   "Loic Magne",
   "Avnish Narayan",
   "You Liang Tan",
   "Guanzhi Wang",
   "Qi Wang",
   "Jiannan Xiang",
   "Yinzhen Xu",
   "Seonghyeon Ye",
   "Jan Kautz",
   "Furong Huang",
   "Yuke Zhu",
   "Linxi Fan"
  ],
  "author_count": 21,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 84,
  "influential_citations": 9,
  "tldr": "The results establish the novel framework FLARE as a general and scalable approach for combining implicit world modeling with high-frequency robotic control.",
  "doi": "10.48550/arXiv.2505.15659",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ruijie Zheng",
    "id": "2345931905",
    "h_index": 9,
    "papers": 15
   },
   {
    "name": "Jing Wang",
    "id": "2350827994",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Scott Reed",
    "id": "2344615904",
    "h_index": 8,
    "papers": 9
   },
   {
    "name": "Johan Bjorck",
    "id": "2362299097",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Yu Fang",
    "id": "2351241177",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Fengyuan Hu",
    "id": "2352992614",
    "h_index": 8,
    "papers": 11
   },
   {
    "name": "J. Jang",
    "id": "2333419032",
    "h_index": 11,
    "papers": 13
   },
   {
    "name": "Kaushil Kundalia",
    "id": "1471804481",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Zongyu Lin",
    "id": "2350991916",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Loic Magne",
    "id": "2350862986",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Avnish Narayan",
    "id": "2014184266",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Y. Tan",
    "id": "2350869215",
    "h_index": 7,
    "papers": 9
   },
   {
    "name": "Guanzhi Wang",
    "id": "96374437",
    "h_index": 16,
    "papers": 22
   },
   {
    "name": "Qi Wang",
    "id": "2326830174",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Jiannan Xiang",
    "id": "2362827946",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Yinzhen Xu",
    "id": "2351665438",
    "h_index": 7,
    "papers": 8
   },
   {
    "name": "Seonghyeon Ye",
    "id": "2152111477",
    "h_index": 23,
    "papers": 30
   },
   {
    "name": "Jan Kautz",
    "id": "2364684748",
    "h_index": 21,
    "papers": 30
   },
   {
    "name": "Furong Huang",
    "id": "2238405926",
    "h_index": 10,
    "papers": 19
   },
   {
    "name": "Yuke Zhu",
    "id": "2258068214",
    "h_index": 13,
    "papers": 21
   },
   {
    "name": "L. Fan",
    "id": "2257381161",
    "h_index": 18,
    "papers": 25
   }
  ],
  "comment": "Project Webpage / Blogpost: https://research.nvidia.com/labs/gear/flare",
  "topics": [
   "world-models",
   "vla",
   "humanoids",
   "egocentric-data",
   "imitation-diffusion"
  ],
  "orgs": [
   "NVIDIA"
  ],
  "abs_url": "https://arxiv.org/abs/2505.15659v1",
  "pdf_url": "https://arxiv.org/pdf/2505.15659v1",
  "html_url": "https://arxiv.org/html/2505.15659v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.43
 },
 {
  "id": "2505.14357",
  "slug": "vid2world-crafting-video-diffusion-models-to-interactive-world-models",
  "title": "Vid2World: Crafting Video Diffusion Models to Interactive World Models",
  "abstract": "World models, which predict future transitions from past observation and action sequences, have shown great promise for improving data efficiency in sequential decision-making. However, existing world models often require extensive domain-specific training and still produce low-fidelity, coarse predictions, limiting their usefulness in complex environments. In contrast, video diffusion models trained on large-scale internet data have demonstrated impressive capabilities in generating high-quality videos that capture diverse real-world dynamics. In this work, we present Vid2World, a general approach for leveraging and transferring pre-trained video diffusion models into interactive world models. To bridge the gap, Vid2World systematically explores video diffusion causalization, reshaping both the architecture and training objective of pre-trained models to enable autoregressive generation. Additionally, it incorporates a causal action guidance mechanism to enhance action controllability in the resulting interactive world models. Extensive experiments across multiple domains, including robot manipulation, 3D game simulation, and open-world navigation, demonstrate that our method offers a scalable and effective pathway for repurposing highly capable video diffusion models into interactive world models.",
  "published": "2025-05-20",
  "updated": "2026-03-08",
  "year": "2025",
  "authors": [
   "Siqiao Huang",
   "Jialong Wu",
   "Qixing Zhou",
   "Shangchen Miao",
   "Mingsheng Long"
  ],
  "author_count": 5,
  "categories": [
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 59,
  "influential_citations": 8,
  "tldr": "Vid2World systematically explores video diffusion causalization, reshaping both the architecture and training objective of pre-trained models to enable autoregressive generation, and incorporates a causal action guidance mechanism to enhance action controllability in the resulting interactive world models.",
  "doi": "10.48550/arXiv.2505.14357",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Siqiao Huang",
    "id": "2343778101",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Jialong Wu",
    "id": "2154707054",
    "h_index": 10,
    "papers": 21
   },
   {
    "name": "Qixing Zhou",
    "id": "2383423134",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Shangchen Miao",
    "id": "2362405671",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Mingsheng Long",
    "id": "2324065346",
    "h_index": 5,
    "papers": 8
   }
  ],
  "comment": "Project page: http://knightnemo.github.io/vid2world/",
  "topics": [
   "world-models",
   "navigation",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2505.14357v3",
  "pdf_url": "https://arxiv.org/pdf/2505.14357v3",
  "html_url": "https://arxiv.org/html/2505.14357v3",
  "code_url": "https://knightnemo.github.io/vid2world/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.78
 },
 {
  "id": "2505.13447",
  "slug": "mean-flows-for-one-step-generative-modeling",
  "title": "Mean Flows for One-step Generative Modeling",
  "abstract": "We propose a principled and effective framework for one-step generative modeling. We introduce the notion of average velocity to characterize flow fields, in contrast to instantaneous velocity modeled by Flow Matching methods. A well-defined identity between average and instantaneous velocities is derived and used to guide neural network training. Our method, termed the MeanFlow model, is self-contained and requires no pre-training, distillation, or curriculum learning. MeanFlow demonstrates strong empirical performance: it achieves an FID of 3.43 with a single function evaluation (1-NFE) on ImageNet 256x256 trained from scratch, significantly outperforming previous state-of-the-art one-step diffusion/flow models. Our study substantially narrows the gap between one-step diffusion/flow models and their multi-step predecessors, and we hope it will motivate future research to revisit the foundations of these powerful models.",
  "published": "2025-05-19",
  "updated": "2025-05-19",
  "year": "2025",
  "authors": [
   "Zhengyang Geng",
   "Mingyang Deng",
   "Xingjian Bai",
   "J. Zico Kolter",
   "Kaiming He"
  ],
  "author_count": 5,
  "categories": [
   "cs.LG",
   "cs.CV"
  ],
  "primary_category": "cs.LG",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 515,
  "influential_citations": 129,
  "tldr": "This study substantially narrows the gap between one-step diffusion/flow models and their multi-step predecessors, and it is hoped it will motivate future research to revisit the foundations of these powerful models.",
  "doi": "10.48550/arXiv.2505.13447",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhengyang Geng",
    "id": "2056612063",
    "h_index": 13,
    "papers": 17
   },
   {
    "name": "Mingyang Deng",
    "id": "2306970309",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Xingjian Bai",
    "id": "2264676687",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Zico Kolter",
    "id": "117539586",
    "h_index": 32,
    "papers": 68
   },
   {
    "name": "Kaiming He",
    "id": "2270025109",
    "h_index": 6,
    "papers": 8
   }
  ],
  "comment": "Tech report",
  "topics": [
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2505.13447v1",
  "pdf_url": "https://arxiv.org/pdf/2505.13447v1",
  "html_url": "https://arxiv.org/html/2505.13447v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.21
 },
 {
  "id": "2505.13389",
  "slug": "vsa-faster-video-diffusion-with-trainable-sparse-attention",
  "title": "VSA: Faster Video Diffusion with Trainable Sparse Attention",
  "abstract": "Scaling video diffusion transformers (DiTs) is limited by their quadratic 3D attention, even though most of the attention mass concentrates on a small subset of positions. We turn this observation into VSA, a trainable, hardware-efficient sparse attention that replaces full attention at \\emph{both} training and inference. In VSA, a lightweight coarse stage pools tokens into tiles and identifies high-weight \\emph{critical tokens}; a fine stage computes token-level attention only inside those tiles subjecting to block computing layout to ensure hard efficiency. This leads to a single differentiable kernel that trains end-to-end, requires no post-hoc profiling, and sustains 85\\% of FlashAttention3 MFU. We perform a large sweep of ablation studies and scaling-law experiments by pretraining DiTs from 60M to 1.4B parameters. VSA reaches a Pareto point that cuts training FLOPS by 2.53$\\times$ with no drop in diffusion loss. Retrofitting the open-source Wan-2.1 model speeds up attention time by 6$\\times$ and lowers end-to-end generation time from 31s to 18s with comparable quality. These results establish trainable sparse attention as a practical alternative to full attention and a key enabler for further scaling of video diffusion models. Code will be available at https://github.com/hao-ai-lab/FastVideo.",
  "published": "2025-05-19",
  "updated": "2025-10-28",
  "year": "2025",
  "authors": [
   "Peiyuan Zhang",
   "Yongqi Chen",
   "Haofeng Huang",
   "Will Lin",
   "Zhengzhong Liu",
   "Ion Stoica",
   "Eric Xing",
   "Hao Zhang"
  ],
  "author_count": 8,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 95,
  "influential_citations": 9,
  "tldr": "VSA is established, a trainable, hardware-efficient sparse attention that replaces full attention at \\emph{both} training and inference, and a key enabler for further scaling of video diffusion models.",
  "doi": "10.48550/arXiv.2505.13389",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Peiyuan Zhang",
    "id": "2345061814",
    "h_index": 6,
    "papers": 19
   },
   {
    "name": "Haofeng Huang",
    "id": "2363577296",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Yongqi Chen",
    "id": "2345261421",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Will Lin",
    "id": "2404252789",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Zhengzhong Liu",
    "id": "100468503",
    "h_index": 15,
    "papers": 29
   },
   {
    "name": "Ion Stoica",
    "id": "2344601177",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Eric P. Xing",
    "id": "2243336934",
    "h_index": 7,
    "papers": 14
   },
   {
    "name": "Hao Zhang",
    "id": "2345362123",
    "h_index": 4,
    "papers": 4
   }
  ],
  "comment": "Accepted by Neurips 2025",
  "topics": [
   "foundation-pretraining",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2505.13389v5",
  "pdf_url": "https://arxiv.org/pdf/2505.13389v5",
  "html_url": "https://arxiv.org/html/2505.13389v5",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.48
 },
 {
  "id": "2505.12705",
  "slug": "dreamgen-unlocking-generalization-in-robot-learning-through-video-worl",
  "title": "DreamGen: Unlocking Generalization in Robot Learning through Video World Models",
  "abstract": "We introduce DreamGen, a simple yet highly effective 4-stage pipeline for training robot policies that generalize across behaviors and environments through neural trajectories - synthetic robot data generated from video world models. DreamGen leverages state-of-the-art image-to-video generative models, adapting them to the target robot embodiment to produce photorealistic synthetic videos of familiar or novel tasks in diverse environments. Since these models generate only videos, we recover pseudo-action sequences using either a latent action model or an inverse-dynamics model (IDM). Despite its simplicity, DreamGen unlocks strong behavior and environment generalization: a humanoid robot can perform 22 new behaviors in both seen and unseen environments, while requiring teleoperation data from only a single pick-and-place task in one environment. To evaluate the pipeline systematically, we introduce DreamGen Bench, a video generation benchmark that shows a strong correlation between benchmark performance and downstream policy success. Our work establishes a promising new axis for scaling robot learning well beyond manual data collection. Code available at https://github.com/NVIDIA/GR00T-Dreams.",
  "published": "2025-05-19",
  "updated": "2025-06-17",
  "year": "2025",
  "authors": [
   "Joel Jang",
   "Seonghyeon Ye",
   "Zongyu Lin",
   "Jiannan Xiang",
   "Johan Bjorck",
   "Yu Fang",
   "Fengyuan Hu",
   "Spencer Huang",
   "Kaushil Kundalia",
   "Yen-Chen Lin",
   "Loic Magne",
   "Ajay Mandlekar",
   "Avnish Narayan",
   "You Liang Tan",
   "Guanzhi Wang",
   "Jing Wang",
   "Qi Wang",
   "Yinzhen Xu",
   "Xiaohui Zeng",
   "Kaiyuan Zheng",
   "Ruijie Zheng",
   "Ming-Yu Liu",
   "Luke Zettlemoyer",
   "Dieter Fox",
   "Jan Kautz",
   "Scott Reed",
   "Yuke Zhu",
   "Linxi Fan"
  ],
  "author_count": 28,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 139,
  "influential_citations": 18,
  "tldr": "This work introduces DreamGen, a simple yet highly effective 4-stage pipeline for training robot policies that generalize across behaviors and environments through neural trajectories - synthetic robot data generated from video world models and establishes a promising new axis for scaling robot learning well beyond manual data collection.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "J. Jang",
    "id": "2333419032",
    "h_index": 11,
    "papers": 13
   },
   {
    "name": "Seonghyeon Ye",
    "id": "2152111477",
    "h_index": 23,
    "papers": 30
   },
   {
    "name": "Zongyu Lin",
    "id": "2350991916",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Jiannan Xiang",
    "id": "2362827946",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Johan Bjorck",
    "id": "2362299097",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Yu Fang",
    "id": "2351241177",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Fengyuan Hu",
    "id": "2352992614",
    "h_index": 8,
    "papers": 11
   },
   {
    "name": "Spencer Huang",
    "id": "2350990085",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Kaushil Kundalia",
    "id": "1471804481",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Yen-Chen Lin",
    "id": "2313179579",
    "h_index": 8,
    "papers": 8
   },
   {
    "name": "Loic Magne",
    "id": "2350862986",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "A. Mandlekar",
    "id": "49686756",
    "h_index": 36,
    "papers": 67
   },
   {
    "name": "Avnish Narayan",
    "id": "2014184266",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Y. Tan",
    "id": "2350869215",
    "h_index": 7,
    "papers": 9
   },
   {
    "name": "Guanzhi Wang",
    "id": "96374437",
    "h_index": 16,
    "papers": 22
   },
   {
    "name": "Jing Wang",
    "id": "2350827994",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Qi Wang",
    "id": "2326830174",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Yinzhen Xu",
    "id": "2351665438",
    "h_index": 7,
    "papers": 8
   },
   {
    "name": "Xi Zeng",
    "id": "2316600217",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Kaiyuan Zheng",
    "id": "2333236365",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Ruijie Zheng",
    "id": "2345931905",
    "h_index": 9,
    "papers": 15
   },
   {
    "name": "Ming-Yu Liu",
    "id": "2385747822",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "Luke S. Zettlemoyer",
    "id": "2325955099",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Dieter Fox",
    "id": "2258436157",
    "h_index": 12,
    "papers": 14
   },
   {
    "name": "Jan Kautz",
    "id": "2364684748",
    "h_index": 21,
    "papers": 30
   },
   {
    "name": "Scott Reed",
    "id": "2344615904",
    "h_index": 8,
    "papers": 9
   },
   {
    "name": "Yuke Zhu",
    "id": "2258068214",
    "h_index": 13,
    "papers": 21
   },
   {
    "name": "L. Fan",
    "id": "2257381161",
    "h_index": 18,
    "papers": 25
   }
  ],
  "comment": "See website for videos: https://research.nvidia.com/labs/gear/dreamgen",
  "topics": [
   "world-models",
   "humanoids",
   "data-teleop",
   "video-generation"
  ],
  "orgs": [
   "NVIDIA"
  ],
  "abs_url": "https://arxiv.org/abs/2505.12705v2",
  "pdf_url": "https://arxiv.org/pdf/2505.12705v2",
  "html_url": "https://arxiv.org/html/2505.12705v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.65
 },
 {
  "id": "2505.11920",
  "slug": "h2r-a-human-to-robot-data-augmentation-for-robot-pre-training-from-vid",
  "title": "H2R: A Human-to-Robot Data Augmentation for Robot Pre-training from Videos",
  "abstract": "Large-scale pre-training using egocentric human videos has proven effective for robot learning. However, the models pre-trained on such data can be suboptimal for robot learning due to the significant visual gap between human hands and those of different robots. To remedy this, we propose H2R, a human-to-robot data augmentation pipeline that converts egocentric human videos into robot-centric visual data. H2R estimates human hand pose from videos, retargets the motion to simulated robotic arms, removes human limbs via segmentation and inpainting, and composites rendered robot embodiments into the original frames with camera-aligned geometry. This process explicitly bridges the visual gap between human and robot embodiments during pre-training. We apply H2R to augment large-scale egocentric human video datasets such as Ego4D and SSv2. To verify the effectiveness of the augmentation pipeline, we introduce a CLIP-based image-text similarity metric that quantitatively evaluates the semantic fidelity of robot-rendered frames to the original human actions. We evaluate H2R through comprehensive experiments in both simulation and real-world settings. In simulation, H2R consistently improves downstream success rates across four benchmark suites-Robomimic, RLBench, PushT, and CortexBench-yielding gains of 1.3%-10.2% across different visual encoders and policy learning methods. In real-world experiments, H2R improves performance on UR5 and dual-arm Franka/UR5 manipulation platforms, achieving 3.3%-23.3% success rate gains across gripper-based, dexterous, and bimanual tasks. We further demonstrate the potential of H2R in cross-embodiment generalization and its compatibility with vision-language-action models. These results indicate that H2R improves the generalization ability of robotic policies by mitigating the visual discrepancies between human and robot domains.",
  "published": "2025-05-17",
  "updated": "2026-03-16",
  "year": "2025",
  "authors": [
   "Guangrun Li",
   "Yaoxu Lyu",
   "Zhuoyang Liu",
   "Chengkai Hou",
   "Jieyu Zhang",
   "Shanghang Zhang"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 24,
  "influential_citations": 2,
  "tldr": "H2R improves the generalization ability of robotic policies by mitigating the visual discrepancies between human and robot domains, and demonstrates the potential of H2R in cross-embodiment generalization and its compatibility with vision-language-action models.",
  "doi": "10.48550/arXiv.2505.11920",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Guangrun Li",
    "id": "2363550799",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Yaoxu Lyu",
    "id": "2335858007",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Zhuoyang Liu",
    "id": "2349957939",
    "h_index": 8,
    "papers": 18
   },
   {
    "name": "Chengkai Hou",
    "id": "2220734428",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Jieyu Zhang",
    "id": "2362292377",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Shanghang Zhang",
    "id": "2242835846",
    "h_index": 20,
    "papers": 59
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "dexterous-manipulation",
   "egocentric-data",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2505.11920v4",
  "pdf_url": "https://arxiv.org/pdf/2505.11920v4",
  "html_url": "https://arxiv.org/html/2505.11920v4",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.4
 },
 {
  "id": "2505.11709",
  "slug": "egodex-learning-dexterous-manipulation-from-large-scale-egocentric-vid",
  "title": "EgoDex: Learning Dexterous Manipulation from Large-Scale Egocentric Video",
  "abstract": "Imitation learning for manipulation has a well-known data scarcity problem. Unlike natural language and 2D computer vision, there is no Internet-scale corpus of data for dexterous manipulation. One appealing option is egocentric human video, a passively scalable data source. However, existing large-scale datasets such as Ego4D do not have native hand pose annotations and do not focus on object manipulation. To this end, we use Apple Vision Pro to collect EgoDex: the largest and most diverse dataset of dexterous human manipulation to date. EgoDex has 829 hours of egocentric video with paired 3D hand and finger tracking data collected at the time of recording, where multiple calibrated cameras and on-device SLAM can be used to precisely track the pose of every joint of each hand. The dataset covers a wide range of diverse manipulation behaviors with everyday household objects in 194 different tabletop tasks ranging from tying shoelaces to folding laundry. Furthermore, we train and systematically evaluate imitation learning policies for hand trajectory prediction on the dataset, introducing metrics and benchmarks for measuring progress in this increasingly important area. By releasing this large-scale dataset, we hope to push the frontier of robotics, computer vision, and foundation models. EgoDex is publicly available for download at https://github.com/apple/ml-egodex.",
  "published": "2025-05-16",
  "updated": "2026-03-09",
  "year": "2025",
  "authors": [
   "Ryan Hoque",
   "Peide Huang",
   "David J. Yoon",
   "Mouli Sivapurapu",
   "Jian Zhang"
  ],
  "author_count": 5,
  "categories": [
   "cs.CV",
   "cs.LG",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "ICLR 2026",
  "venue_source": "arxiv-comment",
  "citations": 193,
  "influential_citations": 24,
  "tldr": "This work uses Apple Vision Pro to collect EgoDex: the largest and most diverse dataset of dexterous human manipulation to date and train and systematically evaluate imitation learning policies for hand trajectory prediction on the dataset, introducing metrics and benchmarks for measuring progress in this increasingly important area.",
  "doi": "10.48550/arXiv.2505.11709",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ryan Hoque",
    "id": "2335570611",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Peide Huang",
    "id": "2328617675",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "David J. Yoon",
    "id": "2351774576",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Mouli Sivapurapu",
    "id": "3317431",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Jian Zhang",
    "id": "2335574774",
    "h_index": 5,
    "papers": 7
   }
  ],
  "comment": "ICLR 2026",
  "topics": [
   "dexterous-manipulation",
   "egocentric-data",
   "imitation-diffusion",
   "spatial-3d",
   "foundation-pretraining",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2505.11709v3",
  "pdf_url": "https://arxiv.org/pdf/2505.11709v3",
  "html_url": "https://arxiv.org/html/2505.11709v3",
  "code_url": null,
  "presented": [
   {
    "club": "saturday-robotics",
    "club_name": "Saturday Robotics",
    "session_number": 13,
    "session_title": "Robotics & World Models Reading Club 13: HumanEgo: Train Robot Policy from 30 min Egocentric Videos \u2014 SF 0620",
    "date_text": "Saturday, June 20, 2026 | 2:00 PM \u2013 5:00 PM",
    "city": "San Francisco, CA",
    "url": "https://lu.ma/6vkhxnum",
    "listed_as": ""
   }
  ],
  "club_note": "",
  "featured": true,
  "signal": 6.79
 },
 {
  "id": "2505.11420",
  "slug": "self-supervised-perception-for-tactile-skin-covered-dexterous-hands",
  "title": "Self-supervised perception for tactile skin covered dexterous hands",
  "abstract": "We present Sparsh-skin, a pre-trained encoder for magnetic skin sensors distributed across the fingertips, phalanges, and palm of a dexterous robot hand. Magnetic tactile skins offer a flexible form factor for hand-wide coverage with fast response times, in contrast to vision-based tactile sensors that are restricted to the fingertips and limited by bandwidth. Full hand tactile perception is crucial for robot dexterity. However, a lack of general-purpose models, challenges with interpreting magnetic flux and calibration have limited the adoption of these sensors. Sparsh-skin, given a history of kinematic and tactile sensing across a hand, outputs a latent tactile embedding that can be used in any downstream task. The encoder is self-supervised via self-distillation on a variety of unlabeled hand-object interactions using an Allegro hand sensorized with Xela uSkin. In experiments across several benchmark tasks, from state estimation to policy learning, we find that pretrained Sparsh-skin representations are both sample efficient in learning downstream tasks and improve task performance by over 41% compared to prior work and over 56% compared to end-to-end learning.",
  "published": "2025-05-16",
  "updated": "2025-05-16",
  "year": "2025",
  "authors": [
   "Akash Sharma",
   "Carolina Higuera",
   "Chaithanya Krishna Bodduluri",
   "Zixi Liu",
   "Taosha Fan",
   "Tess Hellebrekers",
   "Mike Lambeta",
   "Byron Boots",
   "Michael Kaess",
   "Tingfan Wu",
   "Francois Robert Hogan",
   "Mustafa Mukadam"
  ],
  "author_count": 12,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 8,
  "influential_citations": 0,
  "tldr": "In experiments across several benchmark tasks, it is found that pretrained Sparsh-skin representations are both sample efficient in learning downstream tasks and improve task performance by over 41% compared to prior work and over 56% compared to end-to-end learning.",
  "doi": "10.48550/arXiv.2505.11420",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Akash Sharma",
    "id": "2109364933",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "Carolina Higuera",
    "id": "2248215923",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Chaithanya Krishna Bodduluri",
    "id": "2328409907",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Zixi Liu",
    "id": "2362147041",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Taosha Fan",
    "id": "2275595472",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "T. Hellebrekers",
    "id": "2576308",
    "h_index": 18,
    "papers": 31
   },
   {
    "name": "Mike Lambeta",
    "id": "3427691",
    "h_index": 13,
    "papers": 20
   },
   {
    "name": "Byron Boots",
    "id": "2276429486",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Michael Kaess",
    "id": "2279716032",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Tingfan Wu",
    "id": "2254158966",
    "h_index": 12,
    "papers": 21
   },
   {
    "name": "Francois Hogan",
    "id": "2344089432",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Mustafa Mukadam",
    "id": "2874057",
    "h_index": 30,
    "papers": 68
   }
  ],
  "comment": "18 pages, 15 figures",
  "topics": [
   "dexterous-manipulation",
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2505.11420v1",
  "pdf_url": "https://arxiv.org/pdf/2505.11420v1",
  "html_url": "https://arxiv.org/html/2505.11420v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.95
 },
 {
  "id": "2505.10075",
  "slug": "flowdreamer-a-rgb-d-world-model-with-flow-based-motion-representations",
  "title": "FlowDreamer: A RGB-D World Model with Flow-based Motion Representations for Robot Manipulation",
  "abstract": "This paper investigates training better visual world models for robot manipulation, i.e., models that can predict future visual observations by conditioning on past frames and robot actions. Specifically, we consider world models that operate on RGB-D frames (RGB-D world models). As opposed to canonical approaches that handle dynamics prediction mostly implicitly and reconcile it with visual rendering in a single model, we introduce FlowDreamer, which adopts 3D scene flow as explicit motion representations. FlowDreamer first predicts 3D scene flow from past frame and action conditions with a U-Net, and then a diffusion model will predict the future frame utilizing the scene flow. FlowDreamer is trained end-to-end despite its modularized nature. We conduct experiments on 4 different benchmarks, covering both video prediction and visual planning tasks. The results demonstrate that FlowDreamer achieves better performance compared to other baseline RGB-D world models by 7% on semantic similarity, 11% on pixel quality, and 6% on success rate in various robot manipulation domains.",
  "published": "2025-05-15",
  "updated": "2025-05-15",
  "year": "2025",
  "authors": [
   "Jun Guo",
   "Xiaojian Ma",
   "Yikai Wang",
   "Min Yang",
   "Huaping Liu",
   "Qing Li"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "RA-L",
  "venue_source": "semantic-scholar",
  "citations": 23,
  "influential_citations": 3,
  "tldr": "This paper introduces FlowDreamer, which adopts 3D scene flow as explicit motion representations and achieves better performance compared to other baseline RGB-D world models by 7% on semantic similarity, 11% on pixel quality, and 6% on success rate in various robot manipulation domains.",
  "doi": "10.1109/LRA.2026.3653273",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jun Guo",
    "id": "2293357006",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Xiaojian Ma",
    "id": "2241105586",
    "h_index": 12,
    "papers": 25
   },
   {
    "name": "Yikai Wang",
    "id": "2322069387",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Min Yang",
    "id": "2110951471",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Huaping Liu",
    "id": "2293552267",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Qing Li",
    "id": "2282731856",
    "h_index": 6,
    "papers": 12
   }
  ],
  "comment": "Project page: see https://sharinka0715.github.io/FlowDreamer/",
  "topics": [
   "world-models",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2505.10075v1",
  "pdf_url": "https://arxiv.org/pdf/2505.10075v1",
  "html_url": "https://arxiv.org/html/2505.10075v1",
  "code_url": "https://sharinka0715.github.io/FlowDreamer/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.88
 },
 {
  "id": "2505.09723",
  "slug": "enerverse-ac-envisioning-embodied-environments-with-action-condition",
  "title": "EnerVerse-AC: Envisioning Embodied Environments with Action Condition",
  "abstract": "Robotic imitation learning has advanced from solving static tasks to addressing dynamic interaction scenarios, but testing and evaluation remain costly and challenging due to the need for real-time interaction with dynamic environments. We propose EnerVerse-AC (EVAC), an action-conditional world model that generates future visual observations based on an agent's predicted actions, enabling realistic and controllable robotic inference. Building on prior architectures, EVAC introduces a multi-level action-conditioning mechanism and ray map encoding for dynamic multi-view image generation while expanding training data with diverse failure trajectories to improve generalization. As both a data engine and evaluator, EVAC augments human-collected trajectories into diverse datasets and generates realistic, action-conditioned video observations for policy testing, eliminating the need for physical robots or complex simulations. This approach significantly reduces costs while maintaining high fidelity in robotic manipulation evaluation. Extensive experiments validate the effectiveness of our method. Code, checkpoints, and datasets can be found at <https://annaj2178.github.io/EnerverseAC.github.io>.",
  "published": "2025-05-14",
  "updated": "2025-05-14",
  "year": "2025",
  "authors": [
   "Yuxin Jiang",
   "Shengcong Chen",
   "Siyuan Huang",
   "Liliang Chen",
   "Pengfei Zhou",
   "Yue Liao",
   "Xindong He",
   "Chiming Liu",
   "Hongsheng Li",
   "Maoqing Yao",
   "Guanghui Ren"
  ],
  "author_count": 11,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 45,
  "influential_citations": 4,
  "tldr": "EnerVerse-AC is proposed, an action-conditional world model that generates future visual observations based on an agent's predicted actions, enabling realistic and controllable robotic inference and significantly reduces costs while maintaining high fidelity in robotic manipulation evaluation.",
  "doi": "10.48550/arXiv.2505.09723",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yuxin Jiang",
    "id": "2361654632",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Shengcong Chen",
    "id": "2339699627",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Siyuan Huang",
    "id": "2361492506",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Liliang Chen",
    "id": "2338767348",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Pengfei Zhou",
    "id": "2338818174",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Yue Liao",
    "id": "2362056474",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Xindong He",
    "id": "2349803110",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Chiming Liu",
    "id": "2349426158",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Hongsheng Li",
    "id": "2285017444",
    "h_index": 9,
    "papers": 11
   },
   {
    "name": "Maoqing Yao",
    "id": "2338695140",
    "h_index": 12,
    "papers": 22
   },
   {
    "name": "Guanghui Ren",
    "id": "2338694069",
    "h_index": 13,
    "papers": 30
   }
  ],
  "comment": "Website: https://annaj2178.github.io/EnerverseAC.github.io",
  "topics": [
   "world-models",
   "imitation-diffusion",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2505.09723v1",
  "pdf_url": "https://arxiv.org/pdf/2505.09723v1",
  "html_url": "https://arxiv.org/html/2505.09723v1",
  "code_url": "https://annaj2178.github.io/EnerverseAC.github.io",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.66
 },
 {
  "id": "2505.09577",
  "slug": "vtla-vision-tactile-language-action-model-with-preference-learning-for",
  "title": "VTLA: Vision-Tactile-Language-Action Model with Preference Learning for Insertion Manipulation",
  "abstract": "While vision-language models have advanced significantly, their application in language-conditioned robotic manipulation is still underexplored, especially for contact-rich tasks that extend beyond visually dominant pick-and-place scenarios. To bridge this gap, we introduce Vision-Tactile-Language-Action model, a novel framework that enables robust policy generation in contact-intensive scenarios by effectively integrating visual and tactile inputs through cross-modal language grounding. A low-cost, multi-modal dataset has been constructed in a simulation environment, containing vision-tactile-action-instruction pairs specifically designed for the fingertip insertion task. Furthermore, we introduce Direct Preference Optimization (DPO) to offer regression-like supervision for the VTLA model, effectively bridging the gap between classification-based next token prediction loss and continuous robotic tasks. Experimental results show that the VTLA model outperforms traditional imitation learning methods (e.g., diffusion policies) and existing multi-modal baselines (TLA/VLA), achieving over 90% success rates on unseen peg shapes. Finally, we conduct real-world peg-in-hole experiments to demonstrate the exceptional Sim2Real performance of the proposed VTLA model. For supplementary videos and results, please visit our project website: https://sites.google.com/view/vtla",
  "published": "2025-05-14",
  "updated": "2025-05-14",
  "year": "2025",
  "authors": [
   "Chaofan Zhang",
   "Peng Hao",
   "Xiaoge Cao",
   "Xiaoshuai Hao",
   "Shaowei Cui",
   "Shuo Wang"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 82,
  "influential_citations": 2,
  "tldr": "Experimental results show that the VTLA model outperforms traditional imitation learning methods and existing multi-modal baselines (TLA/VLA) and existing multi-modal baselines (TLA/VLA), achieving over 90% success rates on unseen peg shapes.",
  "doi": "10.48550/arXiv.2505.09577",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Chaofan Zhang",
    "id": "2256775583",
    "h_index": 7,
    "papers": 20
   },
   {
    "name": "Peng Hao",
    "id": "2298907411",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Xiaoge Cao",
    "id": "66819025",
    "h_index": 6,
    "papers": 14
   },
   {
    "name": "Xiaoshuai Hao",
    "id": "2313750556",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Shaowei Cui",
    "id": "1853836031",
    "h_index": 17,
    "papers": 58
   },
   {
    "name": "Shuo Wang",
    "id": "2360886366",
    "h_index": 3,
    "papers": 14
   }
  ],
  "comment": "",
  "topics": [
   "tactile",
   "sim2real",
   "imitation-diffusion",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2505.09577v1",
  "pdf_url": "https://arxiv.org/pdf/2505.09577v1",
  "html_url": "https://arxiv.org/html/2505.09577v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.92
 },
 {
  "id": "2505.08787",
  "slug": "uniskill-imitating-human-videos-via-cross-embodiment-skill-representat",
  "title": "UniSkill: Imitating Human Videos via Cross-Embodiment Skill Representations",
  "abstract": "Mimicry is a fundamental learning mechanism in humans, enabling individuals to learn new tasks by observing and imitating experts. However, applying this ability to robots presents significant challenges due to the inherent differences between human and robot embodiments in both their visual appearance and physical capabilities. While previous methods bridge this gap using cross-embodiment datasets with shared scenes and tasks, collecting such aligned data between humans and robots at scale is not trivial. In this paper, we propose UniSkill, a novel framework that learns embodiment-agnostic skill representations from large-scale cross-embodiment video data without any labels, enabling skills extracted from human video prompts to effectively transfer to robot policies trained only on robot data. Our experiments in both simulation and real-world environments show that our cross-embodiment skills successfully guide robots in selecting appropriate actions, even with unseen video prompts. The project website can be found at: https://kimhanjung.github.io/UniSkill.",
  "published": "2025-05-13",
  "updated": "2025-09-20",
  "year": "2025",
  "authors": [
   "Hanjung Kim",
   "Jaehyun Kang",
   "Hyolim Kang",
   "Meedeum Cho",
   "Seon Joo Kim",
   "Youngwoon Lee"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL 2025",
  "venue_source": "arxiv-comment",
  "citations": 38,
  "influential_citations": 4,
  "tldr": "UniSkill is a novel framework that learns embodiment-agnostic skill representations from large-scale cross-embodiment video data without any labels, enabling skills extracted from human video prompts to effectively transfer to robot policies trained only on robot data.",
  "doi": "10.48550/arXiv.2505.08787",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hanjung Kim",
    "id": "2118625998",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Jaehyun Kang",
    "id": "2258679044",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Hyolim Kang",
    "id": "2115457558",
    "h_index": 8,
    "papers": 15
   },
   {
    "name": "Meedeum Cho",
    "id": "2361871840",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Seon Joo Kim",
    "id": "2248551871",
    "h_index": 10,
    "papers": 21
   },
   {
    "name": "Youngwoon Lee",
    "id": "2360616539",
    "h_index": 2,
    "papers": 2
   }
  ],
  "comment": "CoRL 2025. Project Page: https://kimhanjung.github.io/UniSkill/",
  "topics": [
   "egocentric-data",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2505.08787v4",
  "pdf_url": "https://arxiv.org/pdf/2505.08787v4",
  "html_url": "https://arxiv.org/html/2505.08787v4",
  "code_url": "https://kimhanjung.github.io/UniSkill/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.09
 },
 {
  "id": "2505.07818",
  "slug": "dancegrpo-unleashing-grpo-on-visual-generation",
  "title": "DanceGRPO: Unleashing GRPO on Visual Generation",
  "abstract": "Recent advances in generative AI have revolutionized visual content creation, yet aligning model outputs with human preferences remains a critical challenge. While Reinforcement Learning (RL) has emerged as a promising approach for fine-tuning generative models, existing methods like DDPO and DPOK face fundamental limitations - particularly their inability to maintain stable optimization when scaling to large and diverse prompt sets, severely restricting their practical utility. This paper presents DanceGRPO, a framework that addresses these limitations through an innovative adaptation of Group Relative Policy Optimization (GRPO) for visual generation tasks. Our key insight is that GRPO's inherent stability mechanisms uniquely position it to overcome the optimization challenges that plague prior RL-based approaches on visual generation. DanceGRPO establishes several significant advances: First, it demonstrates consistent and stable policy optimization across multiple modern generative paradigms, including both diffusion models and rectified flows. Second, it maintains robust performance when scaling to complex, real-world scenarios encompassing three key tasks and four foundation models. Third, it shows remarkable versatility in optimizing for diverse human preferences as captured by five distinct reward models assessing image/video aesthetics, text-image alignment, video motion quality, and binary feedback. Our comprehensive experiments reveal that DanceGRPO outperforms baseline methods by up to 181\\% across multiple established benchmarks, including HPS-v2.1, CLIP Score, VideoAlign, and GenEval. Our results establish DanceGRPO as a robust and versatile solution for scaling Reinforcement Learning from Human Feedback (RLHF) tasks in visual generation, offering new insights into harmonizing reinforcement learning and visual synthesis.",
  "published": "2025-05-12",
  "updated": "2025-08-28",
  "year": "2025",
  "authors": [
   "Zeyue Xue",
   "Jie Wu",
   "Yu Gao",
   "Fangyuan Kong",
   "Lingting Zhu",
   "Mengzhao Chen",
   "Zhiheng Liu",
   "Wei Liu",
   "Qiushan Guo",
   "Weilin Huang",
   "Ping Luo"
  ],
  "author_count": 11,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 350,
  "influential_citations": 68,
  "tldr": "The results establish DanceGRPO as a robust and versatile solution for scaling Reinforcement Learning from Human Feedback tasks in visual generation, offering new insights into harmonizing reinforcement learning and visual synthesis.",
  "doi": "10.48550/arXiv.2505.07818",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zeyue Xue",
    "id": "2292486781",
    "h_index": 6,
    "papers": 21
   },
   {
    "name": "Jie Wu",
    "id": "2295685180",
    "h_index": 12,
    "papers": 22
   },
   {
    "name": "Yu Gao",
    "id": "2296727303",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Fangyuan Kong",
    "id": "2360838945",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Lingting Zhu",
    "id": "2352404831",
    "h_index": 4,
    "papers": 15
   },
   {
    "name": "Mengzhao Chen",
    "id": "2287768783",
    "h_index": 10,
    "papers": 16
   },
   {
    "name": "Zhiheng Liu",
    "id": "2360881881",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Wei Liu",
    "id": "2335767700",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Qiushan Guo",
    "id": "2356856580",
    "h_index": 6,
    "papers": 16
   },
   {
    "name": "Weilin Huang",
    "id": "2295680253",
    "h_index": 9,
    "papers": 12
   },
   {
    "name": "Ping Luo",
    "id": "2360710127",
    "h_index": 4,
    "papers": 6
   }
  ],
  "comment": "Project Page: https://dancegrpo.github.io/",
  "topics": [
   "rl-control",
   "foundation-pretraining",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2505.07818v4",
  "pdf_url": "https://arxiv.org/pdf/2505.07818v4",
  "html_url": "https://arxiv.org/html/2505.07818v4",
  "code_url": "https://dancegrpo.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.55
 },
 {
  "id": "2505.07813",
  "slug": "dexwild-dexterous-human-interactions-for-in-the-wild-robot-policies",
  "title": "DexWild: Dexterous Human Interactions for In-the-Wild Robot Policies",
  "abstract": "Large-scale, diverse robot datasets have emerged as a promising path toward enabling dexterous manipulation policies to generalize to novel environments, but acquiring such datasets presents many challenges. While teleoperation provides high-fidelity datasets, its high cost limits its scalability. Instead, what if people could use their own hands, just as they do in everyday life, to collect data? In DexWild, a diverse team of data collectors uses their hands to collect hours of interactions across a multitude of environments and objects. To record this data, we create DexWild-System, a low-cost, mobile, and easy-to-use device. The DexWild learning framework co-trains on both human and robot demonstrations, leading to improved performance compared to training on each dataset individually. This combination results in robust robot policies capable of generalizing to novel environments, tasks, and embodiments with minimal additional robot-specific data. Experimental results demonstrate that DexWild significantly improves performance, achieving a 68.5% success rate in unseen environments-nearly four times higher than policies trained with robot data only-and offering 5.8x better cross-embodiment generalization. Video results, codebases, and instructions at https://dexwild.github.io",
  "published": "2025-05-12",
  "updated": "2026-05-17",
  "year": "2025",
  "authors": [
   "Tony Tao",
   "Mohan Kumar Srirama",
   "Jason Jingzhou Liu",
   "Kenneth Shaw",
   "Deepak Pathak"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "Robotics",
  "venue_source": "semantic-scholar",
  "citations": 52,
  "influential_citations": 1,
  "tldr": "Experimental results demonstrate that DexWild significantly improves performance, achieving a 68.5% success rate in unseen environments-nearly four times higher than policies trained with robot data only-and offering 5.8x better cross-embodiment generalization.",
  "doi": "10.48550/arXiv.2505.07813",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Tony Tao",
    "id": "2346975492",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "M. K. Srirama",
    "id": "2193493900",
    "h_index": 12,
    "papers": 14
   },
   {
    "name": "J. Liu",
    "id": "2346996723",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Kenneth Shaw",
    "id": "2263541750",
    "h_index": 10,
    "papers": 18
   },
   {
    "name": "Deepak Pathak",
    "id": "2269734979",
    "h_index": 11,
    "papers": 17
   }
  ],
  "comment": "In RSS 2025. Website at https://dexwild.github.io",
  "topics": [
   "dexterous-manipulation",
   "foundation-pretraining",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2505.07813v2",
  "pdf_url": "https://arxiv.org/pdf/2505.07813v2",
  "html_url": "https://arxiv.org/html/2505.07813v2",
  "code_url": "https://dexwild.github.io",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.22
 },
 {
  "id": "2505.07728",
  "slug": "guiding-data-collection-via-factored-scaling-curves",
  "title": "Guiding Data Collection via Factored Scaling Curves",
  "abstract": "Generalist imitation learning policies trained on large datasets show great promise for solving diverse manipulation tasks. However, to ensure generalization to different conditions, policies need to be trained with data collected across a large set of environmental factor variations (e.g., camera pose, table height, distractors) $-$ a prohibitively expensive undertaking, if done exhaustively. We introduce a principled method for deciding what data to collect and how much to collect for each factor by constructing factored scaling curves (FSC), which quantify how policy performance varies as data scales along individual or paired factors. These curves enable targeted data acquisition for the most influential factor combinations within a given budget. We evaluate the proposed method through extensive simulated and real-world experiments, across both training-from-scratch and fine-tuning settings, and show that it boosts success rates in real-world tasks in new environments by up to 26% over existing data-collection strategies. We further demonstrate how factored scaling curves can effectively guide data collection using an offline metric, without requiring real-world evaluation at scale.",
  "published": "2025-05-12",
  "updated": "2025-05-12",
  "year": "2025",
  "authors": [
   "Lihan Zha",
   "Apurva Badithela",
   "Michael Zhang",
   "Justin Lidard",
   "Jeremy Bao",
   "Emily Zhou",
   "David Snyder",
   "Allen Z. Ren",
   "Dhruv Shah",
   "Anirudha Majumdar"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 12,
  "influential_citations": 0,
  "tldr": "This work introduces a principled method for deciding what data to collect and how much to collect for each factor by constructing factored scaling curves (FSC), which quantify how policy performance varies as data scales along individual or paired factors.",
  "doi": "10.48550/arXiv.2505.07728",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Lihan Zha",
    "id": "2360691960",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Apurva Badithela",
    "id": "100709217",
    "h_index": 9,
    "papers": 24
   },
   {
    "name": "Michael Zhang",
    "id": "2367090063",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Justin Lidard",
    "id": "2284993276",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Jeremy Bao",
    "id": "2362294276",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Emily Zhou",
    "id": "2360693310",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "David Snyder",
    "id": "2350351251",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Allen Z. Ren",
    "id": "1861425270",
    "h_index": 12,
    "papers": 18
   },
   {
    "name": "Dhruv Shah",
    "id": "2322628540",
    "h_index": 29,
    "papers": 63
   },
   {
    "name": "Anirudha Majumdar",
    "id": "2237986850",
    "h_index": 10,
    "papers": 25
   }
  ],
  "comment": "Project website: https://factored-data-scaling.github.io",
  "topics": [
   "imitation-diffusion",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2505.07728v1",
  "pdf_url": "https://arxiv.org/pdf/2505.07728v1",
  "html_url": "https://arxiv.org/html/2505.07728v1",
  "code_url": "https://factored-data-scaling.github.io",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.11
 },
 {
  "id": "2505.06227",
  "slug": "anymate-a-dataset-and-baselines-for-learning-3d-object-rigging",
  "title": "Anymate: A Dataset and Baselines for Learning 3D Object Rigging",
  "abstract": "Rigging and skinning are essential steps to create realistic 3D animations, often requiring significant expertise and manual effort. Traditional attempts at automating these processes rely heavily on geometric heuristics and often struggle with objects of complex geometry. Recent data-driven approaches show potential for better generality, but are often constrained by limited training data. We present the Anymate Dataset, a large-scale dataset of 230K 3D assets paired with expert-crafted rigging and skinning information -- 70 times larger than existing datasets. Using this dataset, we propose a learning-based auto-rigging framework with three sequential modules for joint, connectivity, and skinning weight prediction. We systematically design and experiment with various architectures as baselines for each module and conduct comprehensive evaluations on our dataset to compare their performance. Our models significantly outperform existing methods, providing a foundation for comparing future methods in automated rigging and skinning. Code and dataset can be found at https://anymate3d.github.io/.",
  "published": "2025-05-09",
  "updated": "2025-07-04",
  "year": "2025",
  "authors": [
   "Yufan Deng",
   "Yuhao Zhang",
   "Chen Geng",
   "Shangzhe Wu",
   "Jiajun Wu"
  ],
  "author_count": 5,
  "categories": [
   "cs.GR",
   "cs.CV"
  ],
  "primary_category": "cs.GR",
  "venue": "SIGGRAPH 2025",
  "venue_source": "arxiv-comment",
  "citations": 29,
  "influential_citations": 5,
  "tldr": "This work proposes a learning-based auto-rigging framework with three sequential modules for joint, connectivity, and skinning weight prediction, and significantly outperform existing methods, providing a foundation for comparing future methods in automated rigging and skinning.",
  "doi": "10.1145/3721238.3730743",
  "oa_pdf": "https://doi.org/10.1145/3721238.3730743",
  "s2_authors": [
   {
    "name": "Yufan Deng",
    "id": "2269764462",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Yuhao Zhang",
    "id": "2269797462",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Chen Geng",
    "id": "2334353317",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Shangzhe Wu",
    "id": "2112538311",
    "h_index": 11,
    "papers": 21
   },
   {
    "name": "Jiajun Wu",
    "id": "2326444500",
    "h_index": 35,
    "papers": 214
   }
  ],
  "comment": "SIGGRAPH 2025. Project page: https://anymate3d.github.io/",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2505.06227v2",
  "pdf_url": "https://arxiv.org/pdf/2505.06227v2",
  "html_url": "https://arxiv.org/html/2505.06227v2",
  "code_url": "https://anymate3d.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.98
 },
 {
  "id": "2505.06111",
  "slug": "univla-learning-to-act-anywhere-with-task-centric-latent-actions",
  "title": "UniVLA: Learning to Act Anywhere with Task-centric Latent Actions",
  "abstract": "A generalist robot should perform effectively across various environments. However, most existing approaches heavily rely on scaling action-annotated data to enhance their capabilities. Consequently, they are often limited to single physical specification and struggle to learn transferable knowledge across different embodiments and environments. To confront these limitations, we propose UniVLA, a new framework for learning cross-embodiment vision-language-action (VLA) policies. Our key innovation is to derive task-centric action representations from videos with a latent action model. This enables us to exploit extensive data across a wide spectrum of embodiments and perspectives. To mitigate the effect of task-irrelevant dynamics, we incorporate language instructions and establish a latent action model within the DINO feature space. Learned from internet-scale videos, the generalist policy can be deployed to various robots through efficient latent action decoding. We obtain state-of-the-art results across multiple manipulation and navigation benchmarks, as well as real-robot deployments. UniVLA achieves superior performance over OpenVLA with less than 1/20 of pretraining compute and 1/10 of downstream data. Continuous performance improvements are observed as heterogeneous data, even including human videos, are incorporated into the training pipeline. The results underscore UniVLA's potential to facilitate scalable and efficient robot policy learning.",
  "published": "2025-05-09",
  "updated": "2025-11-03",
  "year": "2025",
  "authors": [
   "Qingwen Bu",
   "Yanting Yang",
   "Jisong Cai",
   "Shenyuan Gao",
   "Guanghui Ren",
   "Maoqing Yao",
   "Ping Luo",
   "Hongyang Li"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "Robotics",
  "venue_source": "semantic-scholar",
  "citations": 400,
  "influential_citations": 66,
  "tldr": "This work proposes UniVLA, a new framework for learning cross-embodiment vision-language-action (VLA) policies that achieves superior performance over OpenVLA with less than 1/20 of pretraining compute and 1/10 of downstream data.",
  "doi": "10.48550/arXiv.2505.06111",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Qingwen Bu",
    "id": "2290184536",
    "h_index": 13,
    "papers": 19
   },
   {
    "name": "Yanting Yang",
    "id": "2303415296",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Jisong Cai",
    "id": "2327001484",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Shenyuan Gao",
    "id": "2350220976",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Guanghui Ren",
    "id": "2338694069",
    "h_index": 13,
    "papers": 30
   },
   {
    "name": "Maoqing Yao",
    "id": "2338695140",
    "h_index": 12,
    "papers": 22
   },
   {
    "name": "Ping Luo",
    "id": "2262515628",
    "h_index": 14,
    "papers": 21
   },
   {
    "name": "Hongyang Li",
    "id": "2290243003",
    "h_index": 11,
    "papers": 19
   }
  ],
  "comment": "Accepted to RSS 2025. Code is available at https://github.com/OpenDriveLab/UniVLA",
  "topics": [
   "vla",
   "egocentric-data",
   "navigation",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2505.06111v3",
  "pdf_url": "https://arxiv.org/pdf/2505.06111v3",
  "html_url": "https://arxiv.org/html/2505.06111v3",
  "code_url": "https://github.com/OpenDriveLab/UniVLA",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.1
 },
 {
  "id": "2505.05470",
  "slug": "flow-grpo-training-flow-matching-models-via-online-rl",
  "title": "Flow-GRPO: Training Flow Matching Models via Online RL",
  "abstract": "We propose Flow-GRPO, the first method to integrate online policy gradient reinforcement learning (RL) into flow matching models. Our approach uses two key strategies: (1) an ODE-to-SDE conversion that transforms a deterministic Ordinary Differential Equation (ODE) into an equivalent Stochastic Differential Equation (SDE) that matches the original model's marginal distribution at all timesteps, enabling statistical sampling for RL exploration; and (2) a Denoising Reduction strategy that reduces training denoising steps while retaining the original number of inference steps, significantly improving sampling efficiency without sacrificing performance. Empirically, Flow-GRPO is effective across multiple text-to-image tasks. For compositional generation, RL-tuned SD3.5-M generates nearly perfect object counts, spatial relations, and fine-grained attributes, increasing GenEval accuracy from $63\\%$ to $95\\%$. In visual text rendering, accuracy improves from $59\\%$ to $92\\%$, greatly enhancing text generation. Flow-GRPO also achieves substantial gains in human preference alignment. Notably, very little reward hacking occurred, meaning rewards did not increase at the cost of appreciable image quality or diversity degradation.",
  "published": "2025-05-08",
  "updated": "2025-10-27",
  "year": "2025",
  "authors": [
   "Jie Liu",
   "Gongye Liu",
   "Jiajun Liang",
   "Yangguang Li",
   "Jiaheng Liu",
   "Xintao Wang",
   "Pengfei Wan",
   "Di Zhang",
   "Wanli Ouyang"
  ],
  "author_count": 9,
  "categories": [
   "cs.CV",
   "cs.AI"
  ],
  "primary_category": "cs.CV",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 554,
  "influential_citations": 151,
  "tldr": "Flow-GRPO, the first method to integrate online policy gradient reinforcement learning (RL) into flow matching models, achieves substantial gains in human preference alignment and very little reward hacking occurred.",
  "doi": "10.48550/arXiv.2505.05470",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jie Liu",
    "id": "2285060791",
    "h_index": 9,
    "papers": 19
   },
   {
    "name": "Gongye Liu",
    "id": "2269171464",
    "h_index": 9,
    "papers": 18
   },
   {
    "name": "Jiajun Liang",
    "id": "2342646262",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Yangguang Li",
    "id": "2357852532",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Jiaheng Liu",
    "id": "2359774530",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Xintao Wang",
    "id": "2305033532",
    "h_index": 21,
    "papers": 67
   },
   {
    "name": "Pengfei Wan",
    "id": "2276606835",
    "h_index": 27,
    "papers": 84
   },
   {
    "name": "Di Zhang",
    "id": "2332361648",
    "h_index": 18,
    "papers": 28
   },
   {
    "name": "Wanli Ouyang",
    "id": "2336864797",
    "h_index": 8,
    "papers": 11
   }
  ],
  "comment": "Code: https://github.com/yifan123/flow_grpo",
  "topics": [
   "rl-control",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2505.05470v5",
  "pdf_url": "https://arxiv.org/pdf/2505.05470v5",
  "html_url": "https://arxiv.org/html/2505.05470v5",
  "code_url": "https://github.com/yifan123/flow_grpo",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.24
 },
 {
  "id": "2505.04999",
  "slug": "clam-continuous-latent-action-models-for-robot-learning-from-unlabeled",
  "title": "CLAM: Continuous Latent Action Models for Robot Learning from Unlabeled Demonstrations",
  "abstract": "Learning robot control policies from demonstrations typically requires action-labeled expert data, which is expensive to collect through teleoperation. We study a more practical setting in which expert demonstrations are available only as observation sequences without action labels, and only task-agnostic play data contains actions. We introduce continuous latent action models (CLAM), a framework that infers continuous latent actions between consecutive observations using self-supervised dynamics prediction. To ground these latent actions into executable motor commands, CLAM jointly trains an action decoder using a small amount of task-agnostic play data. We show that continuous latent actions combined with this joint training are essential for high-dimensional continuous control. Across DMControl locomotion, MetaWorld manipulation, and real-world WidowX robot tasks, CLAM improves average task success rates by 2-3x over prior latent-action baselines and approaches behavior cloning trained with privileged expert action labels. Our results demonstrate that effective robot policies can be learned from unlabeled demonstrations and deployed on real hardware without collecting expert action-labeled data. Videos and code are available at clamrobot.github.io.",
  "published": "2025-05-08",
  "updated": "2026-07-30",
  "year": "2025",
  "authors": [
   "Anthony Liang",
   "Pavel Czempin",
   "Matthew M. Hong",
   "Yutai Zhou",
   "Jingzhen Wang",
   "Erdem Biyik",
   "Stephen Tu"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 34,
  "influential_citations": 2,
  "tldr": "This work introduces continuous latent action models (CLAM), a framework that infers continuous latent actions between consecutive observations using self-supervised dynamics prediction, and shows that continuous latent actions combined with this joint training are essential for high-dimensional continuous control.",
  "doi": "10.48550/arXiv.2505.04999",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Anthony Liang",
    "id": "2286891988",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Pavel Czempin",
    "id": "2114875351",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Matthew Hong",
    "id": "2359819518",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Yutai Zhou",
    "id": "2359697245",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Erdem Biyik",
    "id": "8307674",
    "h_index": 24,
    "papers": 78
   },
   {
    "name": "Stephen Tu",
    "id": "2321787050",
    "h_index": 5,
    "papers": 9
   }
  ],
  "comment": "Latent Action Models, Self-supervised Pretraining, Learning from Videos",
  "topics": [
   "humanoids",
   "imitation-diffusion",
   "foundation-pretraining",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2505.04999v2",
  "pdf_url": "https://arxiv.org/pdf/2505.04999v2",
  "html_url": "https://arxiv.org/html/2505.04999v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.54
 },
 {
  "id": "2505.05517",
  "slug": "web2grasp-learning-functional-grasps-from-web-images-of-hand-object-in",
  "title": "Web2Grasp: Learning Functional Grasps from Web Images of Hand-Object Interactions",
  "abstract": "Functional grasping is essential for enabling dexterous multi-finger robot hands to manipulate objects effectively. Prior work largely focuses on power grasps, which only involve holding an object, or relies on in-domain demonstrations for specific objects. We propose leveraging human grasp information extracted from web images, which capture natural and functional hand-object interactions (HOI). Using a pretrained 3D reconstruction model, we recover 3D human HOI meshes from RGB images. To train on these noisy HOI data, we propose to use: (1) an interaction-centric model to learn the functional interaction pattern between hand and object, and (2) geometry-based filtering to remove the infeasible grasps and physical simulation to retain grasps who can resist disturbance. In IssacGym simulation, our model trained on reconstructed HOI grasps achieves a 75.8% success rate on objects from the web dataset and generalizes to unseen objects, outperforming baseline methods in both grasp success and functional quality. In real-world experiments with the LEAP hand and Inspire hand, it attains a 77.5% success rate across 12 objects, including challenging ones such as a syringe, spray bottle, knife, and tongs. Project website is at: https://web2grasp.github.io/.",
  "published": "2025-05-07",
  "updated": "2026-06-26",
  "year": "2025",
  "authors": [
   "Hongyi Chen",
   "Yunchao Yao",
   "Yufei Ye",
   "Zhixuan Xu",
   "Homanga Bharadhwaj",
   "Jiashun Wang",
   "Arthur Jakobsson",
   "Ruihan Zhao",
   "Shubham Tulsiani",
   "Zackory Erickson",
   "Jeffrey Ichnowski"
  ],
  "author_count": 11,
  "categories": [
   "cs.CV",
   "cs.LG",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 12,
  "influential_citations": 0,
  "tldr": "This work uses a interaction-centric model to learn the functional interaction pattern between hand and object, and geometry-based filtering to remove the infeasible grasps and physical simulation to retain grasps who can resist disturbance to train on noisy HOI data.",
  "doi": "10.48550/arXiv.2505.05517",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hongyi Chen",
    "id": "2309203483",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Yunchao Yao",
    "id": "2269410520",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Yufei Ye",
    "id": "9653518",
    "h_index": 10,
    "papers": 18
   },
   {
    "name": "Zhixuan Xu",
    "id": "2284845139",
    "h_index": 8,
    "papers": 24
   },
   {
    "name": "Homanga Bharadhwaj",
    "id": "51113848",
    "h_index": 23,
    "papers": 59
   },
   {
    "name": "Jiashun Wang",
    "id": "2360235274",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Shubham Tulsiani",
    "id": "2757335",
    "h_index": 45,
    "papers": 98
   },
   {
    "name": "Zackory Erickson",
    "id": "2360174150",
    "h_index": 4,
    "papers": 20
   },
   {
    "name": "Jeffrey Ichnowski",
    "id": "2269146110",
    "h_index": 9,
    "papers": 29
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2505.05517v3",
  "pdf_url": "https://arxiv.org/pdf/2505.05517v3",
  "html_url": "https://arxiv.org/html/2505.05517v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.11
 },
 {
  "id": "2505.03738",
  "slug": "amo-adaptive-motion-optimization-for-hyper-dexterous-humanoid-whole-bo",
  "title": "AMO: Adaptive Motion Optimization for Hyper-Dexterous Humanoid Whole-Body Control",
  "abstract": "Humanoid robots derive much of their dexterity from hyper-dexterous whole-body movements, enabling tasks that require a large operational workspace: such as picking objects off the ground. However, achieving these capabilities on real humanoids remains challenging due to their high degrees of freedom (DoF) and nonlinear dynamics. We propose Adaptive Motion Optimization (AMO), a framework that integrates sim-to-real reinforcement learning (RL) with trajectory optimization for real-time, adaptive whole-body control. To mitigate distribution bias in motion imitation RL, we construct a hybrid AMO dataset and train a network capable of robust, on-demand adaptation to potentially O.O.D. commands. We validate AMO in simulation and on a 29-DoF Unitree G1 humanoid robot, demonstrating superior stability and an expanded workspace compared to strong baselines. Finally, we show that AMO's consistent performance supports autonomous task execution via imitation learning, underscoring the system's versatility and robustness.",
  "published": "2025-05-06",
  "updated": "2025-05-06",
  "year": "2025",
  "authors": [
   "Jialong Li",
   "Xuxin Cheng",
   "Tianshu Huang",
   "Shiqi Yang",
   "Ri-Zhao Qiu",
   "Xiaolong Wang"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "Robotics",
  "venue_source": "semantic-scholar",
  "citations": 101,
  "influential_citations": 17,
  "tldr": "This work proposes Adaptive Motion Optimization (AMO), a framework that integrates sim-to-real reinforcement learning (RL) with trajectory optimization for real-time, adaptive whole-body control for real-time, adaptive whole-body control in humanoid robots.",
  "doi": "10.48550/arXiv.2505.03738",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jialong Li",
    "id": "2309196968",
    "h_index": 7,
    "papers": 7
   },
   {
    "name": "Xuxin Cheng",
    "id": "2287822264",
    "h_index": 8,
    "papers": 9
   },
   {
    "name": "Tianshu Huang",
    "id": "2359285165",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Shiqi Yang",
    "id": "2309666838",
    "h_index": 8,
    "papers": 9
   },
   {
    "name": "Ri-Zhao Qiu",
    "id": "2290904526",
    "h_index": 11,
    "papers": 19
   },
   {
    "name": "Xiaolong Wang",
    "id": "2294782536",
    "h_index": 12,
    "papers": 17
   }
  ],
  "comment": "website: https://amo-humanoid.github.io",
  "topics": [
   "dexterous-manipulation",
   "humanoids",
   "sim2real",
   "imitation-diffusion",
   "rl-control",
   "safety-eval"
  ],
  "orgs": [
   "Unitree"
  ],
  "abs_url": "https://arxiv.org/abs/2505.03738v1",
  "pdf_url": "https://arxiv.org/pdf/2505.03738v1",
  "html_url": "https://arxiv.org/html/2505.03738v1",
  "code_url": "https://amo-humanoid.github.io",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.01
 },
 {
  "id": "2505.03728",
  "slug": "pyroki-a-modular-toolkit-for-robot-kinematic-optimization",
  "title": "PyRoki: A Modular Toolkit for Robot Kinematic Optimization",
  "abstract": "Robot motion can have many goals. Depending on the task, we might optimize for pose error, speed, collision, or similarity to a human demonstration. Motivated by this, we present PyRoki: a modular, extensible, and cross-platform toolkit for solving kinematic optimization problems. PyRoki couples an interface for specifying kinematic variables and costs with an efficient nonlinear least squares optimizer. Unlike existing tools, it is also cross-platform: optimization runs natively on CPU, GPU, and TPU. In this paper, we present (i) the design and implementation of PyRoki, (ii) motion retargeting and planning case studies that highlight the advantages of PyRoki's modularity, and (iii) optimization benchmarking, where PyRoki can be 1.4-1.7x faster and converges to lower errors than cuRobo, an existing GPU-accelerated inverse kinematics library.",
  "published": "2025-05-06",
  "updated": "2025-05-06",
  "year": "2025",
  "authors": [
   "Chung Min Kim",
   "Brent Yi",
   "Hongsuk Choi",
   "Yi Ma",
   "Ken Goldberg",
   "Angjoo Kanazawa"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS",
  "venue_source": "semantic-scholar",
  "citations": 41,
  "influential_citations": 4,
  "tldr": "The design and implementation of PyRoki are presented, and motion retargeting and planning case studies that highlight the advantages of PyRoki\u2019s modularity are presented, and optimization benchmarking is presented, where PyRoki can be 1.4-1.7x faster and converges to lower errors than cuRobo, an existing GPU-accelerated inverse kinematics library.",
  "doi": "10.1109/IROS60139.2025.11246651",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Chung Min Kim",
    "id": "2359652583",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Brent Yi",
    "id": "2242880086",
    "h_index": 18,
    "papers": 27
   },
   {
    "name": "Hongsuk Choi",
    "id": "2337083924",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Yi Ma",
    "id": "2324935696",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Kenneth Y. Goldberg",
    "id": "2394971519",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Angjoo Kanazawa",
    "id": "20615377",
    "h_index": 60,
    "papers": 126
   }
  ],
  "comment": "First two authors contributed equally. Code is available at https://pyroki-toolkit.github.io",
  "topics": [
   "egocentric-data"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2505.03728v1",
  "pdf_url": "https://arxiv.org/pdf/2505.03728v1",
  "html_url": "https://arxiv.org/html/2505.03728v1",
  "code_url": "https://pyroki-toolkit.github.io",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.12
 },
 {
  "id": "2504.21853",
  "slug": "a-survey-of-interactive-generative-video",
  "title": "A Survey of Interactive Generative Video",
  "abstract": "Interactive Generative Video (IGV) has emerged as a crucial technology in response to the growing demand for high-quality, interactive video content across various domains. In this paper, we define IGV as a technology that combines generative capabilities to produce diverse high-quality video content with interactive features that enable user engagement through control signals and responsive feedback. We survey the current landscape of IGV applications, focusing on three major domains: 1) gaming, where IGV enables infinite exploration in virtual worlds; 2) embodied AI, where IGV serves as a physics-aware environment synthesizer for training agents in multimodal interaction with dynamically evolving scenes; and 3) autonomous driving, where IGV provides closed-loop simulation capabilities for safety-critical testing and validation. To guide future development, we propose a comprehensive framework that decomposes an ideal IGV system into five essential modules: Generation, Control, Memory, Dynamics, and Intelligence. Furthermore, we systematically analyze the technical challenges and future directions in realizing each component for an ideal IGV system, such as achieving real-time generation, enabling open-domain control, maintaining long-term coherence, simulating accurate physics, and integrating causal reasoning. We believe that this systematic analysis will facilitate future research and development in the field of IGV, ultimately advancing the technology toward more sophisticated and practical applications.",
  "published": "2025-04-30",
  "updated": "2025-04-30",
  "year": "2025",
  "authors": [
   "Jiwen Yu",
   "Yiran Qin",
   "Haoxuan Che",
   "Quande Liu",
   "Xintao Wang",
   "Pengfei Wan",
   "Di Zhang",
   "Kun Gai",
   "Hao Chen",
   "Xihui Liu"
  ],
  "author_count": 10,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 27,
  "influential_citations": 1,
  "tldr": "This paper defines IGV as a technology that combines generative capabilities to produce diverse high-quality video content with interactive features that enable user engagement through control signals and responsive feedback, and proposes a comprehensive framework that decomposes an ideal IGV system into five essential modules.",
  "doi": "10.48550/arXiv.2504.21853",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jiwen Yu",
    "id": "2116420946",
    "h_index": 16,
    "papers": 28
   },
   {
    "name": "Yiran Qin",
    "id": "2342256177",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Haoxuan Che",
    "id": "2351605641",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Quande Liu",
    "id": "2339261137",
    "h_index": 9,
    "papers": 19
   },
   {
    "name": "Xintao Wang",
    "id": "2305033532",
    "h_index": 21,
    "papers": 67
   },
   {
    "name": "Pengfei Wan",
    "id": "2276606835",
    "h_index": 27,
    "papers": 84
   },
   {
    "name": "Di Zhang",
    "id": "2332361648",
    "h_index": 18,
    "papers": 28
   },
   {
    "name": "Kun Gai",
    "id": "2238953242",
    "h_index": 17,
    "papers": 39
   },
   {
    "name": "Hao Chen",
    "id": "2358586443",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Xihui Liu",
    "id": "2340184176",
    "h_index": 6,
    "papers": 12
   }
  ],
  "comment": "",
  "topics": [
   "navigation",
   "video-generation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2504.21853v1",
  "pdf_url": "https://arxiv.org/pdf/2504.21853v1",
  "html_url": "https://arxiv.org/html/2504.21853v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.45
 },
 {
  "id": "2504.21738",
  "slug": "langwbc-language-directed-humanoid-whole-body-control-via-end-to-end-l",
  "title": "LangWBC: Language-directed Humanoid Whole-Body Control via End-to-end Learning",
  "abstract": "General-purpose humanoid robots are expected to interact intuitively with humans, enabling seamless integration into daily life. Natural language provides the most accessible medium for this purpose. However, translating language into humanoid whole-body motion remains a significant challenge, primarily due to the gap between linguistic understanding and physical actions. In this work, we present an end-to-end, language-directed policy for real-world humanoid whole-body control. Our approach combines reinforcement learning with policy distillation, allowing a single neural network to interpret language commands and execute corresponding physical actions directly. To enhance motion diversity and compositionality, we incorporate a Conditional Variational Autoencoder (CVAE) structure. The resulting policy achieves agile and versatile whole-body behaviors conditioned on language inputs, with smooth transitions between various motions, enabling adaptation to linguistic variations and the emergence of novel motions. We validate the efficacy and generalizability of our method through extensive simulations and real-world experiments, demonstrating robust whole-body control. Please see our website at LangWBC.github.io for more information.",
  "published": "2025-04-30",
  "updated": "2025-04-30",
  "year": "2025",
  "authors": [
   "Yiyang Shao",
   "Xiaoyu Huang",
   "Bike Zhang",
   "Qiayuan Liao",
   "Yuman Gao",
   "Yufeng Chi",
   "Zhongyu Li",
   "Sophia Shao",
   "Koushil Sreenath"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "Robotics",
  "venue_source": "semantic-scholar",
  "citations": 44,
  "influential_citations": 1,
  "tldr": "This work presents an end-to-end, language-directed policy for real-world humanoid whole-body control, allowing a single neural network to interpret language commands and execute corresponding physical actions directly, and incorporates a Conditional Variational Autoencoder structure.",
  "doi": "10.48550/arXiv.2504.21738",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yiyang Shao",
    "id": "2362323779",
    "h_index": 3,
    "papers": 13
   },
   {
    "name": "Xiaoyu Huang",
    "id": "2293553057",
    "h_index": 9,
    "papers": 10
   },
   {
    "name": "Bike Zhang",
    "id": "4848974",
    "h_index": 16,
    "papers": 24
   },
   {
    "name": "Qiayuan Liao",
    "id": "1713616371",
    "h_index": 13,
    "papers": 20
   },
   {
    "name": "Yuman Gao",
    "id": "2358305183",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Yufeng Chi",
    "id": "2143834260",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Zhongyu Li",
    "id": "1491078398",
    "h_index": 24,
    "papers": 50
   },
   {
    "name": "Sophia Shao",
    "id": "2298968272",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "K. Sreenath",
    "id": "144116765",
    "h_index": 55,
    "papers": 231
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2504.21738v1",
  "pdf_url": "https://arxiv.org/pdf/2504.21738v1",
  "html_url": "https://arxiv.org/html/2504.21738v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.15
 },
 {
  "id": "2504.16054",
  "slug": "0-5-a-vision-language-action-model-with-open-world-generalization",
  "title": "$\u03c0_{0.5}$: a Vision-Language-Action Model with Open-World Generalization",
  "abstract": "In order for robots to be useful, they must perform practically relevant tasks in the real world, outside of the lab. While vision-language-action (VLA) models have demonstrated impressive results for end-to-end robot control, it remains an open question how far such models can generalize in the wild. We describe $\u03c0_{0.5}$, a new model based on $\u03c0_{0}$ that uses co-training on heterogeneous tasks to enable broad generalization. $\u03c0_{0.5}$\\ uses data from multiple robots, high-level semantic prediction, web data, and other sources to enable broadly generalizable real-world robotic manipulation. Our system uses a combination of co-training and hybrid multi-modal examples that combine image observations, language commands, object detections, semantic subtask prediction, and low-level actions. Our experiments show that this kind of knowledge transfer is essential for effective generalization, and we demonstrate for the first time that an end-to-end learning-enabled robotic system can perform long-horizon and dexterous manipulation skills, such as cleaning a kitchen or bedroom, in entirely new homes.",
  "published": "2025-04-22",
  "updated": "2025-04-22",
  "year": "2025",
  "authors": [
   "Physical Intelligence",
   "Kevin Black",
   "Noah Brown",
   "James Darpinian",
   "Karan Dhabalia",
   "Danny Driess",
   "Adnan Esmail",
   "Michael Equi",
   "Chelsea Finn",
   "Niccolo Fusai",
   "Manuel Y. Galliker",
   "Dibya Ghosh",
   "Lachy Groom",
   "Karol Hausman",
   "Brian Ichter",
   "Szymon Jakubczak",
   "Tim Jones",
   "Liyiming Ke",
   "Devin LeBlanc",
   "Sergey Levine",
   "Adrian Li-Bell",
   "Mohith Mothukuri",
   "Suraj Nair",
   "Karl Pertsch",
   "Allen Z. Ren",
   "Lucy Xiaoyang Shi",
   "Laura Smith",
   "Jost Tobias Springenberg",
   "Kyle Stachowicz",
   "James Tanner",
   "Quan Vuong",
   "Homer Walke",
   "Anna Walling",
   "Haohuan Wang",
   "Lili Yu",
   "Ury Zhilinsky"
  ],
  "author_count": 36,
  "categories": [
   "cs.LG",
   "cs.RO"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 1625,
  "influential_citations": 309,
  "tldr": "A new model based onpi 0.5 is described that uses co-training on heterogeneous tasks to enable broad generalization and is demonstrated for the first time that an end-to-end learning-enabled robotic system can perform long-horizon and dexterous manipulation skills, such as cleaning a kitchen or bedroom, in entirely new homes.",
  "doi": "10.48550/arXiv.2504.16054",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Physical Intelligence",
    "id": "2356784273",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Kevin Black",
    "id": "2258959388",
    "h_index": 13,
    "papers": 15
   },
   {
    "name": "Noah Brown",
    "id": "2161343011",
    "h_index": 10,
    "papers": 16
   },
   {
    "name": "James Darpinian",
    "id": "2356784408",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Karan Dhabalia",
    "id": "2356782366",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Danny Driess",
    "id": "2283848260",
    "h_index": 27,
    "papers": 35
   },
   {
    "name": "A. Esmail",
    "id": "2332926590",
    "h_index": 7,
    "papers": 12
   },
   {
    "name": "Michael Equi",
    "id": "2298901898",
    "h_index": 10,
    "papers": 15
   },
   {
    "name": "Chelsea Finn",
    "id": "2257346440",
    "h_index": 23,
    "papers": 32
   },
   {
    "name": "Niccolo Fusai",
    "id": "2332926006",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Manuel Y. Galliker",
    "id": "1580069611",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Dibya Ghosh",
    "id": "8021910",
    "h_index": 18,
    "papers": 24
   },
   {
    "name": "Lachy Groom",
    "id": "2332926792",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Karol Hausman",
    "id": "1944801",
    "h_index": 47,
    "papers": 122
   },
   {
    "name": "Brian Ichter",
    "id": "2704814",
    "h_index": 37,
    "papers": 60
   },
   {
    "name": "S. Jakubczak",
    "id": "2332926523",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Tim Jones",
    "id": "2333409218",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Liyiming Ke",
    "id": "2332976956",
    "h_index": 7,
    "papers": 8
   },
   {
    "name": "Devin LeBlanc",
    "id": "2356787296",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Sergey Levine",
    "id": "2249615151",
    "h_index": 32,
    "papers": 53
   },
   {
    "name": "Adrian Li-Bell",
    "id": "2332927424",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Mohith Mothukuri",
    "id": "2332926533",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Suraj Nair",
    "id": "2286638954",
    "h_index": 12,
    "papers": 19
   },
   {
    "name": "Karl Pertsch",
    "id": "31719101",
    "h_index": 32,
    "papers": 56
   },
   {
    "name": "Allen Z. Ren",
    "id": "2356677628",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "L. Shi",
    "id": "2292341452",
    "h_index": 12,
    "papers": 17
   },
   {
    "name": "Laura Smith",
    "id": "2302181645",
    "h_index": 8,
    "papers": 8
   },
   {
    "name": "Jost Tobias Springenberg",
    "id": "2060551",
    "h_index": 44,
    "papers": 93
   },
   {
    "name": "Kyle Stachowicz",
    "id": "2106415427",
    "h_index": 12,
    "papers": 21
   },
   {
    "name": "James Tanner",
    "id": "2332926892",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Quan Vuong",
    "id": "2288210223",
    "h_index": 19,
    "papers": 29
   },
   {
    "name": "H. Walke",
    "id": "2029241116",
    "h_index": 17,
    "papers": 23
   },
   {
    "name": "Anna Walling",
    "id": "2333982746",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Haohuan Wang",
    "id": "2332952253",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Lili Yu",
    "id": "2356801221",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Ury Zhilinsky",
    "id": "3187915",
    "h_index": 5,
    "papers": 7
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "dexterous-manipulation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2504.16054v1",
  "pdf_url": "https://arxiv.org/pdf/2504.16054v1",
  "html_url": "https://arxiv.org/html/2504.16054v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.0
 },
 {
  "id": "2504.13165",
  "slug": "ruka-rethinking-the-design-of-humanoid-hands-with-learning",
  "title": "RUKA: Rethinking the Design of Humanoid Hands with Learning",
  "abstract": "Dexterous manipulation is a fundamental capability for robotic systems, yet progress has been limited by hardware trade-offs between precision, compactness, strength, and affordability. Existing control methods impose compromises on hand designs and applications. However, learning-based approaches present opportunities to rethink these trade-offs, particularly to address challenges with tendon-driven actuation and low-cost materials. This work presents RUKA, a tendon-driven humanoid hand that is compact, affordable, and capable. Made from 3D-printed parts and off-the-shelf components, RUKA has 5 fingers with 15 underactuated degrees of freedom enabling diverse human-like grasps. Its tendon-driven actuation allows powerful grasping in a compact, human-sized form factor. To address control challenges, we learn joint-to-actuator and fingertip-to-actuator models from motion-capture data collected by the MANUS glove, leveraging the hand's morphological accuracy. Extensive evaluations demonstrate RUKA's superior reachability, durability, and strength compared to other robotic hands. Teleoperation tasks further showcase RUKA's dexterous movements. The open-source design and assembly instructions of RUKA, code, and data are available at https://ruka-hand.github.io/.",
  "published": "2025-04-17",
  "updated": "2025-04-17",
  "year": "2025",
  "authors": [
   "Anya Zorin",
   "Irmak Guzey",
   "Billy Yan",
   "Aadhithya Iyer",
   "Lisa Kondrich",
   "Nikhil X. Bhattasali",
   "Lerrel Pinto"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "Robotics",
  "venue_source": "semantic-scholar",
  "citations": 21,
  "influential_citations": 1,
  "tldr": "To address control challenges, RUKA is presented, a tendon-driven humanoid hand that is compact, affordable, and capable, and joint-to-actuator and fingertip-to-actuator models from motion-capture data collected by the MANUS glove are learned, leveraging the hand's morphological accuracy.",
  "doi": "10.48550/arXiv.2504.13165",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Anya Zorin",
    "id": "2355867631",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Irmak G\u00fczey",
    "id": "2212471096",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Billy Yan",
    "id": "2356233381",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Aadhithya Iyer",
    "id": "2203910567",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Lisa Kondrich",
    "id": "2355874542",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Nikhil X. Bhattasali",
    "id": "2104521429",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Lerrel Pinto",
    "id": "2320806817",
    "h_index": 10,
    "papers": 12
   }
  ],
  "comment": "Website at https://ruka-hand.github.io/",
  "topics": [
   "dexterous-manipulation",
   "humanoids",
   "data-teleop",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2504.13165v1",
  "pdf_url": "https://arxiv.org/pdf/2504.13165v1",
  "html_url": "https://arxiv.org/html/2504.13165v1",
  "code_url": "https://ruka-hand.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.84
 },
 {
  "id": "2504.13059",
  "slug": "robotwin-dual-arm-robot-benchmark-with-generative-digital-twins",
  "title": "RoboTwin: Dual-Arm Robot Benchmark with Generative Digital Twins",
  "abstract": "In the rapidly advancing field of robotics, dual-arm coordination and complex object manipulation are essential capabilities for developing advanced autonomous systems. However, the scarcity of diverse, high-quality demonstration data and real-world-aligned evaluation benchmarks severely limits such development. To address this, we introduce RoboTwin, a generative digital twin framework that uses 3D generative foundation models and large language models to produce diverse expert datasets and provide a real-world-aligned evaluation platform for dual-arm robotic tasks. Specifically, RoboTwin creates varied digital twins of objects from single 2D images, generating realistic and interactive scenarios. It also introduces a spatial relation-aware code generation framework that combines object annotations with large language models to break down tasks, determine spatial constraints, and generate precise robotic movement code. Our framework offers a comprehensive benchmark with both simulated and real-world data, enabling standardized evaluation and better alignment between simulated training and real-world performance. We validated our approach using the open-source COBOT Magic Robot platform. Policies pre-trained on RoboTwin-generated data and fine-tuned with limited real-world samples demonstrate significant potential for enhancing dual-arm robotic manipulation systems by improving success rates by over 70% for single-arm tasks and over 40% for dual-arm tasks compared to models trained solely on real-world data.",
  "published": "2025-04-17",
  "updated": "2025-04-17",
  "year": "2025",
  "authors": [
   "Yao Mu",
   "Tianxing Chen",
   "Zanxin Chen",
   "Shijia Peng",
   "Zhiqian Lan",
   "Zeyu Gao",
   "Zhixuan Liang",
   "Qiaojun Yu",
   "Yude Zou",
   "Mingkun Xu",
   "Lunkai Lin",
   "Zhiqiang Xie",
   "Mingyu Ding",
   "Ping Luo"
  ],
  "author_count": 14,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CL"
  ],
  "primary_category": "cs.RO",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 153,
  "influential_citations": 30,
  "tldr": "RoboTwin is introduced, a generative digital twin framework that uses 3D generative foundation models and large language models to produce diverse expert datasets and provide a real-world-aligned evaluation platform for dual-arm robotic tasks and offers a comprehensive benchmark with both simulated and real-world data.",
  "doi": "10.1109/CVPR52734.2025.02575",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yao Mu",
    "id": "2248348669",
    "h_index": 12,
    "papers": 16
   },
   {
    "name": "Tianxing Chen",
    "id": "2316455829",
    "h_index": 10,
    "papers": 31
   },
   {
    "name": "Zanxin Chen",
    "id": "2319612137",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Shijia Peng",
    "id": "2319812486",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Zhiqian Lan",
    "id": "2355872613",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Zeyu Gao",
    "id": "2262087748",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Zhixuan Liang",
    "id": "2257485304",
    "h_index": 10,
    "papers": 35
   },
   {
    "name": "Qiaojun Yu",
    "id": "2287935820",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Yude Zou",
    "id": "2320101926",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Min Xu",
    "id": "2273795815",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Lunkai Lin",
    "id": "2319589609",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Zhiqiang Xie",
    "id": "2319898037",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Mingyu Ding",
    "id": "2284988821",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Ping Luo",
    "id": "2257349788",
    "h_index": 8,
    "papers": 23
   }
  ],
  "comment": "CVPR 2025 Highlight. 22 pages. Project page: https://robotwin-benchmark.github.io/",
  "topics": [
   "foundation-pretraining",
   "data-teleop",
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2504.13059v1",
  "pdf_url": "https://arxiv.org/pdf/2504.13059v1",
  "html_url": "https://arxiv.org/html/2504.13059v1",
  "code_url": "https://robotwin-benchmark.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.69
 },
 {
  "id": "2504.10567",
  "slug": "h3ae-high-compression-high-speed-and-high-quality-autoencoder-for-vide",
  "title": "H3AE: High Compression, High Speed, and High Quality AutoEncoder for Video Diffusion Models",
  "abstract": "Autoencoder (AE) is the key to the success of latent diffusion models for image and video generation, reducing the denoising resolution and improving efficiency. However, the power of AE has long been underexplored in terms of network design, compression ratio, and training strategy. In this work, we systematically examine the architecture design choices and optimize the computation distribution to obtain a series of efficient and high-compression video AEs that can decode in real time even on mobile devices. We also propose an omni-training objective to unify the design of plain Autoencoder and image-conditioned I2V VAE, achieving multifunctionality in a single VAE network but with enhanced quality. In addition, we propose a novel latent consistency loss that provides stable improvements in reconstruction quality. Latent consistency loss outperforms prior auxiliary losses including LPIPS, GAN and DWT in terms of both quality improvements and simplicity. H3AE achieves ultra-high compression ratios and real-time decoding speed on GPU and mobile, and outperforms prior arts in terms of reconstruction metrics by a large margin. We finally validate our AE by training a DiT on its latent space and demonstrate fast, high-quality text-to-video generation capability.",
  "published": "2025-04-14",
  "updated": "2025-10-01",
  "year": "2025",
  "authors": [
   "Yushu Wu",
   "Yanyu Li",
   "Ivan Skorokhodov",
   "Anil Kag",
   "Willi Menapace",
   "Sharath Girish",
   "Aliaksandr Siarohin",
   "Yanzhi Wang",
   "Sergey Tulyakov"
  ],
  "author_count": 9,
  "categories": [
   "cs.CV",
   "eess.IV"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 12,
  "influential_citations": 2,
  "tldr": "H3AE achieves ultra-high compression ratios and real-time decoding speed on GPU and mobile, and outperforms prior arts in terms of reconstruction metrics by a large margin, and is validated by training a DiT on its latent space and demonstrates fast, high-quality text-to-video generation capability.",
  "doi": "10.48550/arXiv.2504.10567",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yushu Wu",
    "id": "2048040696",
    "h_index": 11,
    "papers": 28
   },
   {
    "name": "Yanyu Li",
    "id": "2257369142",
    "h_index": 8,
    "papers": 23
   },
   {
    "name": "Ivan Skorokhodov",
    "id": "51118864",
    "h_index": 22,
    "papers": 45
   },
   {
    "name": "Anil Kag",
    "id": "2284982329",
    "h_index": 9,
    "papers": 23
   },
   {
    "name": "W. Menapace",
    "id": "1698103472",
    "h_index": 22,
    "papers": 53
   },
   {
    "name": "Sharath Girish",
    "id": "143720888",
    "h_index": 11,
    "papers": 18
   },
   {
    "name": "Aliaksandr Siarohin",
    "id": "10753214",
    "h_index": 31,
    "papers": 76
   },
   {
    "name": "Yanzhi Wang",
    "id": "2257348403",
    "h_index": 7,
    "papers": 16
   },
   {
    "name": "Sergey Tulyakov",
    "id": "2292401534",
    "h_index": 17,
    "papers": 63
   }
  ],
  "comment": "17 pages, 6 figures, 9 tables",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2504.10567v2",
  "pdf_url": "https://arxiv.org/pdf/2504.10567v2",
  "html_url": "https://arxiv.org/html/2504.10567v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.11
 },
 {
  "id": "2504.10479",
  "slug": "internvl3-exploring-advanced-training-and-test-time-recipes-for-open-s",
  "title": "InternVL3: Exploring Advanced Training and Test-Time Recipes for Open-Source Multimodal Models",
  "abstract": "We introduce InternVL3, a significant advancement in the InternVL series featuring a native multimodal pre-training paradigm. Rather than adapting a text-only large language model (LLM) into a multimodal large language model (MLLM) that supports visual inputs, InternVL3 jointly acquires multimodal and linguistic capabilities from both diverse multimodal data and pure-text corpora during a single pre-training stage. This unified training paradigm effectively addresses the complexities and alignment challenges commonly encountered in conventional post-hoc training pipelines for MLLMs. To further improve performance and scalability, InternVL3 incorporates variable visual position encoding (V2PE) to support extended multimodal contexts, employs advanced post-training techniques such as supervised fine-tuning (SFT) and mixed preference optimization (MPO), and adopts test-time scaling strategies alongside an optimized training infrastructure. Extensive empirical evaluations demonstrate that InternVL3 delivers superior performance across a wide range of multi-modal tasks. In particular, InternVL3-78B achieves a score of 72.2 on the MMMU benchmark, setting a new state-of-the-art among open-source MLLMs. Its capabilities remain highly competitive with leading proprietary models, including ChatGPT-4o, Claude 3.5 Sonnet, and Gemini 2.5 Pro, while also maintaining strong pure-language proficiency. In pursuit of open-science principles, we will publicly release both the training data and model weights to foster further research and development in next-generation MLLMs.",
  "published": "2025-04-14",
  "updated": "2025-04-19",
  "year": "2025",
  "authors": [
   "Jinguo Zhu",
   "Weiyun Wang",
   "Zhe Chen",
   "Zhaoyang Liu",
   "Shenglong Ye",
   "Lixin Gu",
   "Hao Tian",
   "Yuchen Duan",
   "Weijie Su",
   "Jie Shao",
   "Zhangwei Gao",
   "Erfei Cui",
   "Xuehui Wang",
   "Yue Cao",
   "Yangzhou Liu",
   "Xingguang Wei",
   "Hongjie Zhang",
   "Haomin Wang",
   "Weiye Xu",
   "Hao Li",
   "Jiahao Wang",
   "Nianchen Deng",
   "Songze Li",
   "Yinan He",
   "Tan Jiang",
   "Jiapeng Luo",
   "Yi Wang",
   "Conghui He",
   "Botian Shi",
   "Xingcheng Zhang",
   "Wenqi Shao",
   "Junjun He",
   "Yingtong Xiong",
   "Wenwen Qu",
   "Peng Sun",
   "Penglong Jiao",
   "Han Lv",
   "Lijun Wu",
   "Kaipeng Zhang",
   "Huipeng Deng",
   "Jiaye Ge",
   "Kai Chen",
   "Limin Wang",
   "Min Dou",
   "Lewei Lu",
   "Xizhou Zhu",
   "Tong Lu",
   "Dahua Lin",
   "Yu Qiao",
   "Jifeng Dai",
   "Wenhai Wang"
  ],
  "author_count": 51,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 1718,
  "influential_citations": 196,
  "tldr": "InternVL3 incorporates variable visual position encoding (V2PE) to support extended multimodal contexts, employs advanced post-training techniques such as supervised fine-tuning (SFT) and mixed preference optimization (MPO), and adopts test-time scaling strategies alongside an optimized training infrastructure.",
  "doi": "10.48550/arXiv.2504.10479",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jinguo Zhu",
    "id": "2327111699",
    "h_index": 9,
    "papers": 16
   },
   {
    "name": "Weiyun Wang",
    "id": "2190474418",
    "h_index": 23,
    "papers": 36
   },
   {
    "name": "Zhe Chen",
    "id": "2305731793",
    "h_index": 19,
    "papers": 31
   },
   {
    "name": "Zhaoyang Liu",
    "id": "46270766",
    "h_index": 20,
    "papers": 30
   },
   {
    "name": "Shenglong Ye",
    "id": "2298577575",
    "h_index": 13,
    "papers": 24
   },
   {
    "name": "Lixin Gu",
    "id": "2334465171",
    "h_index": 6,
    "papers": 16
   },
   {
    "name": "Yuchen Duan",
    "id": "2290131661",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Hao Tian",
    "id": "2251049221",
    "h_index": 9,
    "papers": 12
   },
   {
    "name": "Weijie Su",
    "id": "2276207009",
    "h_index": 10,
    "papers": 18
   },
   {
    "name": "Jie Shao",
    "id": "2335567013",
    "h_index": 5,
    "papers": 18
   },
   {
    "name": "Zhangwei Gao",
    "id": "2262400432",
    "h_index": 11,
    "papers": 18
   },
   {
    "name": "Erfei Cui",
    "id": "2262217755",
    "h_index": 10,
    "papers": 20
   },
   {
    "name": "Yue Cao",
    "id": "2337506544",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Yangzhou Liu",
    "id": "2312345209",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Haomin Wang",
    "id": "2301056324",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Weiye Xu",
    "id": "2355336617",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Hao Li",
    "id": "2274232642",
    "h_index": 10,
    "papers": 20
   },
   {
    "name": "Jiahao Wang",
    "id": "2335071973",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Han Lv",
    "id": "2335196335",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "De-Hua Chen",
    "id": "2358688187",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Songze Li",
    "id": "2293357052",
    "h_index": 8,
    "papers": 17
   },
   {
    "name": "Yinan He",
    "id": "2118918324",
    "h_index": 28,
    "papers": 48
   },
   {
    "name": "Tan Jiang",
    "id": "2114746694",
    "h_index": 1,
    "papers": 7
   },
   {
    "name": "Jiapeng Luo",
    "id": "2279419424",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Jiapeng Luo",
    "id": "2002008",
    "h_index": 8,
    "papers": 15
   },
   {
    "name": "Conghui He",
    "id": "2267889334",
    "h_index": 34,
    "papers": 77
   },
   {
    "name": "Botian Shi",
    "id": "2297846008",
    "h_index": 12,
    "papers": 21
   },
   {
    "name": "Xingcheng Zhang",
    "id": "2298585927",
    "h_index": 9,
    "papers": 16
   },
   {
    "name": "Wenqi Shao",
    "id": "2283133523",
    "h_index": 22,
    "papers": 59
   },
   {
    "name": "Junjun He",
    "id": "2329901826",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Ying Xiong",
    "id": "2279899097",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Wenwen Qu",
    "id": "2313633991",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Peng Sun",
    "id": "2323108535",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Penglong Jiao",
    "id": "2293475034",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Lijun Wu",
    "id": "2325195689",
    "h_index": 10,
    "papers": 41
   },
   {
    "name": "Kai Zhang",
    "id": "2256442212",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Hui Deng",
    "id": "2347000377",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Jiaye Ge",
    "id": "2293426655",
    "h_index": 7,
    "papers": 17
   },
   {
    "name": "Kaiming Chen",
    "id": "2298717051",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Limin Wang",
    "id": "2189093874",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Min Dou",
    "id": "2197075911",
    "h_index": 18,
    "papers": 27
   },
   {
    "name": "Lewei Lu",
    "id": "152309485",
    "h_index": 32,
    "papers": 53
   },
   {
    "name": "Xizhou Zhu",
    "id": "2578924",
    "h_index": 48,
    "papers": 82
   },
   {
    "name": "Tong Lu",
    "id": "2336552315",
    "h_index": 7,
    "papers": 14
   },
   {
    "name": "Dahua Lin",
    "id": "2303596290",
    "h_index": 10,
    "papers": 23
   },
   {
    "name": "Yu Qiao",
    "id": "2342101236",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "Jifeng Dai",
    "id": "2292283383",
    "h_index": 27,
    "papers": 55
   },
   {
    "name": "Wenhai Wang",
    "id": "2257133501",
    "h_index": 29,
    "papers": 64
   }
  ],
  "comment": "Technical Report",
  "topics": [
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2504.10479v3",
  "pdf_url": "https://arxiv.org/pdf/2504.10479v3",
  "html_url": "https://arxiv.org/html/2504.10479v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.0
 },
 {
  "id": "2504.08388",
  "slug": "mineworld-a-real-time-and-open-source-interactive-world-model-on-minec",
  "title": "MineWorld: a Real-Time and Open-Source Interactive World Model on Minecraft",
  "abstract": "World modeling is a crucial task for enabling intelligent agents to effectively interact with humans and operate in dynamic environments. In this work, we propose MineWorld, a real-time interactive world model on Minecraft, an open-ended sandbox game which has been utilized as a common testbed for world modeling. MineWorld is driven by a visual-action autoregressive Transformer, which takes paired game scenes and corresponding actions as input, and generates consequent new scenes following the actions. Specifically, by transforming visual game scenes and actions into discrete token ids with an image tokenizer and an action tokenizer correspondingly, we consist the model input with the concatenation of the two kinds of ids interleaved. The model is then trained with next token prediction to learn rich representations of game states as well as the conditions between states and actions simultaneously. In inference, we develop a novel parallel decoding algorithm that predicts the spatial redundant tokens in each frame at the same time, letting models in different scales generate $4$ to $7$ frames per second and enabling real-time interactions with game players. In evaluation, we propose new metrics to assess not only visual quality but also the action following capacity when generating new scenes, which is crucial for a world model. Our comprehensive evaluation shows the efficacy of MineWorld, outperforming SoTA open-sourced diffusion based world models significantly. The code and model have been released.",
  "published": "2025-04-11",
  "updated": "2025-04-11",
  "year": "2025",
  "authors": [
   "Junliang Guo",
   "Yang Ye",
   "Tianyu He",
   "Haoyu Wu",
   "Yushu Jiang",
   "Tim Pearce",
   "Jiang Bian"
  ],
  "author_count": 7,
  "categories": [
   "cs.CV",
   "cs.AI"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 84,
  "influential_citations": 11,
  "tldr": "MineWorld, a real-time interactive world model on Minecraft, outperforming SoTA open-sourced diffusion based world models significantly is proposed, and a novel parallel decoding algorithm that predicts the spatial redundant tokens in each frame at the same time is developed.",
  "doi": "10.48550/arXiv.2504.08388",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Junliang Guo",
    "id": "2284789660",
    "h_index": 11,
    "papers": 18
   },
   {
    "name": "Yang Ye",
    "id": "2350759471",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Tianyu He",
    "id": "2259935939",
    "h_index": 13,
    "papers": 30
   },
   {
    "name": "Haoyu Wu",
    "id": "2351231497",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Yushu Jiang",
    "id": "2355244065",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Tim Pearce",
    "id": "2350757236",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Jiang Bian",
    "id": "2304555579",
    "h_index": 10,
    "papers": 18
   }
  ],
  "comment": "Technical report. Project page https://aka.ms/mineworld",
  "topics": [
   "world-models",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2504.08388v1",
  "pdf_url": "https://arxiv.org/pdf/2504.08388v1",
  "html_url": "https://arxiv.org/html/2504.08388v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.93
 },
 {
  "id": "2504.06156",
  "slug": "vitamin-learning-contact-rich-tasks-through-robot-free-visuo-tactile-m",
  "title": "ViTaMIn: Learning Contact-Rich Tasks Through Robot-Free Visuo-Tactile Manipulation Interface",
  "abstract": "Tactile information plays a crucial role for humans and robots to interact effectively with their environment, particularly for tasks requiring the understanding of contact properties. Solving such dexterous manipulation tasks often relies on imitation learning from demonstration datasets, which are typically collected via teleoperation systems and often demand substantial time and effort. To address these challenges, we present ViTaMIn, an embodiment-free manipulation interface that seamlessly integrates visual and tactile sensing into a hand-held gripper, enabling data collection without the need for teleoperation. Our design employs a compliant Fin Ray gripper with tactile sensing, allowing operators to perceive force feedback during manipulation for more intuitive operation. Additionally, we propose a multimodal representation learning strategy to obtain pre-trained tactile representations, improving data efficiency and policy robustness. Experiments on seven contact-rich manipulation tasks demonstrate that ViTaMIn significantly outperforms baseline methods, demonstrating its effectiveness for complex manipulation tasks.",
  "published": "2025-04-08",
  "updated": "2025-09-01",
  "year": "2025",
  "authors": [
   "Fangchen Liu",
   "Chuanyu Li",
   "Yihua Qin",
   "Jing Xu",
   "Pieter Abbeel",
   "Rui Chen"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 46,
  "influential_citations": 1,
  "tldr": "This work presents ViTaMIn, an embodiment-free manipulation interface that seamlessly integrates visual and tactile sensing into a hand-held gripper, enabling data collection without the need for teleoperation and proposes a multimodal representation learning strategy to obtain pre-trained tactile representations, improving data efficiency and policy robustness.",
  "doi": "10.48550/arXiv.2504.06156",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Fangchen Liu",
    "id": "2279745979",
    "h_index": 8,
    "papers": 10
   },
   {
    "name": "Chuanyu Li",
    "id": "2310944952",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Yihua Qin",
    "id": "2355365055",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Ankit Shaw",
    "id": "2332090898",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Jing Xu",
    "id": "2321185480",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Pieter Abbeel",
    "id": "2257184474",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Rui Chen",
    "id": "2332011508",
    "h_index": 4,
    "papers": 11
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "imitation-diffusion",
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2504.06156v2",
  "pdf_url": "https://arxiv.org/pdf/2504.06156v2",
  "html_url": "https://arxiv.org/html/2504.06156v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.67
 },
 {
  "id": "2504.06084",
  "slug": "maple-encoding-dexterous-robotic-manipulation-priors-learned-from-egoc",
  "title": "MAPLE: Encoding Dexterous Robotic Manipulation Priors Learned From Egocentric Videos",
  "abstract": "Large-scale egocentric video datasets capture diverse human activities across a wide range of scenarios, offering rich and detailed insights into how humans interact with objects, especially those that require fine-grained dexterous control. Such complex, dexterous skills with precise controls are crucial for many robotic manipulation tasks, yet are often insufficiently addressed by traditional data-driven approaches to robotic manipulation. To address this gap, we leverage manipulation priors learned from large-scale egocentric video datasets to improve policy learning for dexterous robotic manipulation tasks. We present MAPLE, a novel method for dexterous robotic manipulation that learns features to predict object contact points and detailed hand poses at the moment of contact from egocentric images. We then use the learned features to train policies for downstream manipulation tasks. Experimental results demonstrate the effectiveness of MAPLE across 4 existing simulation benchmarks, as well as a newly designed set of 4 challenging simulation tasks requiring fine-grained object control and complex dexterous skills. The benefits of MAPLE are further highlighted in real-world experiments using a 17 DoF dexterous robotic hand, whereas the simultaneous evaluation across both simulation and real-world experiments has remained underexplored in prior work. We additionally showcase the efficacy of our model on an egocentric contact point prediction task, validating its usefulness beyond dexterous manipulation policy learning.",
  "published": "2025-04-08",
  "updated": "2025-12-08",
  "year": "2025",
  "authors": [
   "Alexey Gavryushin",
   "Xi Wang",
   "Robert J. S. Malate",
   "Chenyu Yang",
   "Davide Liconti",
   "Ren\u00e9 Zurbr\u00fcgg",
   "Robert K. Katzschmann",
   "Marc Pollefeys"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 14,
  "influential_citations": 0,
  "tldr": "MAPLE is presented, a novel method for dexterous robotic manipulation that learns features to predict object contact points and detailed hand poses at the moment of contact from egocentric images, and uses the learned features to train policies for downstream manipulation tasks.",
  "doi": "10.48550/arXiv.2504.06084",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Alexey Gavryushin",
    "id": "2202228606",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Xi Wang",
    "id": "2352959391",
    "h_index": 2,
    "papers": 11
   },
   {
    "name": "R. J. Malate",
    "id": "2354260083",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Chenyu Yang",
    "id": "2329514965",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Xiangyi Jia",
    "id": "2354297532",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Shubh Goel",
    "id": "2313833258",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Davide Liconti",
    "id": "2298269450",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Ren'e Zurbrugg",
    "id": "2297668196",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Robert K. Katzschmann",
    "id": "50191333",
    "h_index": 26,
    "papers": 87
   },
   {
    "name": "Marc Pollefeys",
    "id": "2242552467",
    "h_index": 13,
    "papers": 34
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "egocentric-data",
   "sim2real"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2504.06084v2",
  "pdf_url": "https://arxiv.org/pdf/2504.06084v2",
  "html_url": "https://arxiv.org/html/2504.06084v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.18
 },
 {
  "id": "2504.02792",
  "slug": "unified-world-models-coupling-video-and-action-diffusion-for-pretraini",
  "title": "Unified World Models: Coupling Video and Action Diffusion for Pretraining on Large Robotic Datasets",
  "abstract": "Imitation learning has emerged as a promising approach towards building generalist robots. However, scaling imitation learning for large robot foundation models remains challenging due to its reliance on high-quality expert demonstrations. Meanwhile, large amounts of video data depicting a wide range of environments and diverse behaviors are readily available. This data provides a rich source of information about real-world dynamics and agent-environment interactions. Leveraging this data directly for imitation learning, however, has proven difficult due to the lack of action annotation. In this work, we present Unified World Models (UWM), a framework that allows for leveraging both video and action data for policy learning. Specifically, a UWM integrates an action diffusion process and a video diffusion process within a unified transformer architecture, where independent diffusion timesteps govern each modality. By controlling each diffusion timestep, UWM can flexibly represent a policy, a forward dynamics, an inverse dynamics, and a video generator. Through simulated and real-world experiments, we show that: (1) UWM enables effective pretraining on large-scale multitask robot datasets with both dynamics and action predictions, resulting in more generalizable and robust policies than imitation learning, (2) UWM naturally facilitates learning from action-free video data through independent control of modality-specific diffusion timesteps, further improving the performance of finetuned policies. Our results suggest that UWM offers a promising step toward harnessing large, heterogeneous datasets for scalable robot learning, and provides a simple unification between the often disparate paradigms of imitation learning and world modeling. Videos and code are available at https://weirdlabuw.github.io/uwm/.",
  "published": "2025-04-03",
  "updated": "2025-05-23",
  "year": "2025",
  "authors": [
   "Chuning Zhu",
   "Raymond Yu",
   "Siyuan Feng",
   "Benjamin Burchfiel",
   "Paarth Shah",
   "Abhishek Gupta"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "Robotics",
  "venue_source": "semantic-scholar",
  "citations": 199,
  "influential_citations": 17,
  "tldr": "UWM enables effective pretraining on large-scale multitask robot datasets with both dynamics and action predictions, resulting in more generalizable and robust policies than imitation learning, and naturally facilitates learning from action-free video data through independent control of modality-specific diffusion timesteps, further improving the performance of finetuned policies.",
  "doi": "10.48550/arXiv.2504.02792",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Chuning Zhu",
    "id": "2118160513",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "Raymond Yu",
    "id": "2357457189",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Siyuan Feng",
    "id": "2284620540",
    "h_index": 8,
    "papers": 10
   },
   {
    "name": "B. Burchfiel",
    "id": "2302757",
    "h_index": 18,
    "papers": 30
   },
   {
    "name": "Paarth Shah",
    "id": "2300432426",
    "h_index": 7,
    "papers": 15
   },
   {
    "name": "Abhishek Gupta",
    "id": "2264049829",
    "h_index": 5,
    "papers": 10
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "imitation-diffusion",
   "foundation-pretraining",
   "data-teleop",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2504.02792v3",
  "pdf_url": "https://arxiv.org/pdf/2504.02792v3",
  "html_url": "https://arxiv.org/html/2504.02792v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.8
 },
 {
  "id": "2504.02439",
  "slug": "estimating-scene-flow-in-robot-surroundings-with-distributed-miniaturi",
  "title": "Estimating Scene Flow in Robot Surroundings with Distributed Miniaturized Time-of-Flight Sensors",
  "abstract": "Tracking motions of humans or objects in the surroundings of the robot is essential to improve safe robot motions and reactions. In this work, we present an approach for scene flow estimation from low-density and noisy point clouds acquired from miniaturized Time of Flight (ToF) sensors distributed on the robot body. The proposed method clusters points from consecutive frames and applies Iterative Closest Point (ICP) to estimate a dense motion flow, with additional steps introduced to mitigate the impact of sensor noise and low-density data points. Specifically, we employ a fitness-based classification to distinguish between stationary and moving points and an inlier removal strategy to refine geometric correspondences. The proposed approach is validated in an experimental setup where 24 ToF are used to estimate the velocity of an object moving at different controlled speeds. Experimental results show that the method consistently approximates the direction of the motion and its magnitude with an error which is in line with sensor noise.",
  "published": "2025-04-03",
  "updated": "2025-07-31",
  "year": "2025",
  "authors": [
   "Jack Sander",
   "Giammarco Caroleo",
   "Alessandro Albini",
   "Perla Maiolino"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 4,
  "influential_citations": 0,
  "tldr": "This work presents an approach for scene flow estimation from low-density and noisy point clouds acquired from miniaturised Time-of-Flight sensors distributed across the robot\u2019s body, which employs a fitness-based classification to distinguish between stationary and moving points and an inlier removal strategy to refine geometric correspondences.",
  "doi": "10.1109/RO-MAN63969.2025.11217813",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "J. Sander",
    "id": "2355351719",
    "h_index": 5,
    "papers": 26
   },
   {
    "name": "G. Caroleo",
    "id": "2337117303",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "A. Albini",
    "id": "32226685",
    "h_index": 11,
    "papers": 51
   },
   {
    "name": "P. Maiolino",
    "id": "40414175",
    "h_index": 19,
    "papers": 112
   }
  ],
  "comment": "7 pages, 5 figures, 2 tables, 1 algorithm, IEEE RO-MAN 2025 accepted paper",
  "topics": [
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2504.02439v2",
  "pdf_url": "https://arxiv.org/pdf/2504.02439v2",
  "html_url": "https://arxiv.org/html/2504.02439v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.7
 },
 {
  "id": "2503.23877",
  "slug": "zeromimic-distilling-robotic-manipulation-skills-from-web-videos",
  "title": "ZeroMimic: Distilling Robotic Manipulation Skills from Web Videos",
  "abstract": "Many recent advances in robotic manipulation have come through imitation learning, yet these rely largely on mimicking a particularly hard-to-acquire form of demonstrations: those collected on the same robot in the same room with the same objects as the trained policy must handle at test time. In contrast, large pre-recorded human video datasets demonstrating manipulation skills in-the-wild already exist, which contain valuable information for robots. Is it possible to distill a repository of useful robotic skill policies out of such data without any additional requirements on robot-specific demonstrations or exploration? We present the first such system ZeroMimic, that generates immediately deployable image goal-conditioned skill policies for several common categories of manipulation tasks (opening, closing, pouring, pick&place, cutting, and stirring) each capable of acting upon diverse objects and across diverse unseen task setups. ZeroMimic is carefully designed to exploit recent advances in semantic and geometric visual understanding of human videos, together with modern grasp affordance detectors and imitation policy classes. After training ZeroMimic on the popular EpicKitchens dataset of ego-centric human videos, we evaluate its out-of-the-box performance in varied real-world and simulated kitchen settings with two different robot embodiments, demonstrating its impressive abilities to handle these varied tasks. To enable plug-and-play reuse of ZeroMimic policies on other task setups and robots, we release software and policy checkpoints of our skill policies.",
  "published": "2025-03-31",
  "updated": "2025-03-31",
  "year": "2025",
  "authors": [
   "Junyao Shi",
   "Zhuolun Zhao",
   "Tianyou Wang",
   "Ian Pedroza",
   "Amy Luo",
   "Jie Wang",
   "Jason Ma",
   "Dinesh Jayaraman"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA",
  "venue_source": "semantic-scholar",
  "citations": 38,
  "influential_citations": 2,
  "tldr": "ZeroMimic is carefully designed to exploit recent advances in semantic and geometric visual understanding of human videos, together with modern grasp affordance detectors and imitation policy classes, and to enable plug-and-play reuse of ZeroMimic policies on other task setups and robots.",
  "doi": "10.1109/ICRA55743.2025.11128283",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Junyao Shi",
    "id": "2297813255",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Zhuolun Zhao",
    "id": "2352993777",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Tianyou Wang",
    "id": "2293019501",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Ian Pedroza",
    "id": "2352943595",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Amy Luo",
    "id": "13005443",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Jie Wang",
    "id": "2353064089",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Jason Ma",
    "id": "102232319",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Dinesh Jayaraman",
    "id": "2352942932",
    "h_index": 4,
    "papers": 8
   }
  ],
  "comment": "ICRA 2025. Project website: https://zeromimic.github.io/",
  "topics": [
   "dexterous-manipulation",
   "egocentric-data",
   "imitation-diffusion",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2503.23877v1",
  "pdf_url": "https://arxiv.org/pdf/2503.23877v1",
  "html_url": "https://arxiv.org/html/2503.23877v1",
  "code_url": "https://zeromimic.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.09
 },
 {
  "id": "2503.23452",
  "slug": "videogen-eval-agent-based-system-for-video-generation-evaluation",
  "title": "VideoGen-Eval: Agent-based System for Video Generation Evaluation",
  "abstract": "The rapid advancement of video generation has rendered existing evaluation systems inadequate for assessing state-of-the-art models, primarily due to simple prompts that cannot showcase the model's capabilities, fixed evaluation operators struggling with Out-of-Distribution (OOD) cases, and misalignment between computed metrics and human preferences. To bridge the gap, we propose VideoGen-Eval, an agent evaluation system that integrates LLM-based content structuring, MLLM-based content judgment, and patch tools designed for temporal-dense dimensions, to achieve a dynamic, flexible, and expandable video generation evaluation. Additionally, we introduce a video generation benchmark to evaluate existing cutting-edge models and verify the effectiveness of our evaluation system. It comprises 700 structured, content-rich prompts (both T2V and I2V) and over 12,000 videos generated by 20+ models, among them, 8 cutting-edge models are selected as quantitative evaluation for the agent and human. Extensive experiments validate that our proposed agent-based evaluation system demonstrates strong alignment with human preferences and reliably completes the evaluation, as well as the diversity and richness of the benchmark.",
  "published": "2025-03-30",
  "updated": "2025-04-27",
  "year": "2025",
  "authors": [
   "Yuhang Yang",
   "Ke Fan",
   "Shangkun Sun",
   "Hongxiang Li",
   "Ailing Zeng",
   "FeiLin Han",
   "Wei Zhai",
   "Wei Liu",
   "Yang Cao",
   "Zheng-Jun Zha"
  ],
  "author_count": 10,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 24,
  "influential_citations": 0,
  "tldr": "VideoGen-Eval is proposed, an agent evaluation system that integrates LLM-based content structuring, MLLM-based content judgment, and patch tools designed for temporal-dense dimensions to achieve a dynamic, flexible, and expandable video generation evaluation.",
  "doi": "10.48550/arXiv.2503.23452",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yuhang Yang",
    "id": "2274200074",
    "h_index": 10,
    "papers": 22
   },
   {
    "name": "Ke Fan",
    "id": "2352937809",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Shangkun Sun",
    "id": "2167034767",
    "h_index": 8,
    "papers": 16
   },
   {
    "name": "Hongxiang Li",
    "id": "2350141695",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Ailing Zeng",
    "id": "2304555389",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "Feilin Han",
    "id": "2264128793",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Wei-hao Zhai",
    "id": "2204179910",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Wei Liu",
    "id": "2157222547",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Yang Cao",
    "id": "2352939527",
    "h_index": 3,
    "papers": 14
   },
   {
    "name": "Zhengjun Zha",
    "id": "2289018296",
    "h_index": 14,
    "papers": 37
   }
  ],
  "comment": "project:https://github.com/AILab-CVC/VideoGen-Eval",
  "topics": [
   "video-generation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2503.23452v2",
  "pdf_url": "https://arxiv.org/pdf/2503.23452v2",
  "html_url": "https://arxiv.org/html/2503.23452v2",
  "code_url": "https://github.com/AILab-CVC/VideoGen-Eval",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.4
 },
 {
  "id": "2503.22634",
  "slug": "empirical-analysis-of-sim-and-real-cotraining-of-diffusion-policies-fo",
  "title": "Empirical Analysis of Sim-and-Real Cotraining of Diffusion Policies for Planar Pushing from Pixels",
  "abstract": "Cotraining with demonstration data generated both in simulation and on real hardware has emerged as a promising recipe for scaling imitation learning in robotics. This work seeks to elucidate basic principles of this sim-and-real cotraining to inform simulation design, sim-and-real dataset creation, and policy training. Our experiments confirm that cotraining with simulated data can dramatically improve performance, especially when real data is limited. We show that these performance gains scale with additional simulated data up to a plateau; adding more real-world data increases this performance ceiling. The results also suggest that reducing physical domain gaps may be more impactful than visual fidelity for non-prehensile or contact-rich tasks. Perhaps surprisingly, we find that some visual gap can help cotraining -- binary probes reveal that high-performing policies must learn to distinguish simulated domains from real. We conclude by investigating this nuance and mechanisms that facilitate positive transfer between sim-and-real. Focusing narrowly on the canonical task of planar pushing from pixels allows us to be thorough in our study. In total, our experiments span 50+ real-world policies (evaluated on 1000+ trials) and 250 simulated policies (evaluated on 50,000+ trials). Videos and code can be found at https://sim-and-real-cotraining.github.io/.",
  "published": "2025-03-28",
  "updated": "2025-08-05",
  "year": "2025",
  "authors": [
   "Adam Wei",
   "Abhinav Agarwal",
   "Boyuan Chen",
   "Rohan Bosworth",
   "Nicholas Pfaff",
   "Russ Tedrake"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS",
  "venue_source": "semantic-scholar",
  "citations": 25,
  "influential_citations": 1,
  "tldr": "",
  "doi": "10.1109/IROS60139.2025.11246304",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Adam Wei",
    "id": "2352792737",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Abhinav Agarwal",
    "id": "2387770128",
    "h_index": 2,
    "papers": 14
   },
   {
    "name": "Boyuan Chen",
    "id": "2352939822",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Rohan Bosworth",
    "id": "2275274269",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Nicholas Pfaff",
    "id": "2348265706",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Russ Tedrake",
    "id": "2263905014",
    "h_index": 14,
    "papers": 36
   }
  ],
  "comment": "11 pages, 17 figures, IROS 2025 Aug 5, 2025 update: Included new experiments in Sections V and VII. Updated abstract and minor changes to text",
  "topics": [
   "tactile",
   "imitation-diffusion"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2503.22634v2",
  "pdf_url": "https://arxiv.org/pdf/2503.22634v2",
  "html_url": "https://arxiv.org/html/2503.22634v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.91
 },
 {
  "id": "2503.21755",
  "slug": "vbench-2-0-advancing-video-generation-benchmark-suite-for-intrinsic-fa",
  "title": "VBench-2.0: Advancing Video Generation Benchmark Suite for Intrinsic Faithfulness",
  "abstract": "Video generation has advanced significantly, evolving from producing unrealistic outputs to generating videos that appear visually convincing and temporally coherent. To evaluate these video generative models, benchmarks such as VBench have been developed to assess their faithfulness, measuring factors like per-frame aesthetics, temporal consistency, and basic prompt adherence. However, these aspects mainly represent superficial faithfulness, which focus on whether the video appears visually convincing rather than whether it adheres to real-world principles. While recent models perform increasingly well on these metrics, they still struggle to generate videos that are not just visually plausible but fundamentally realistic. To achieve real \"world models\" through video generation, the next frontier lies in intrinsic faithfulness to ensure that generated videos adhere to physical laws, commonsense reasoning, anatomical correctness, and compositional integrity. Achieving this level of realism is essential for applications such as AI-assisted filmmaking and simulated world modeling. To bridge this gap, we introduce VBench-2.0, a next-generation benchmark designed to automatically evaluate video generative models for their intrinsic faithfulness. VBench-2.0 assesses five key dimensions: Human Fidelity, Controllability, Creativity, Physics, and Commonsense, each further broken down into fine-grained capabilities. Tailored to individual dimensions, our evaluation framework integrates generalists such as SOTA VLMs and LLMs, and specialists, including anomaly detection methods proposed for video generation. We conduct extensive human annotations to ensure evaluation alignment with human judgment. By pushing beyond superficial faithfulness toward intrinsic faithfulness, VBench-2.0 aims to set a new standard for the next generation of video generative models in pursuit of intrinsic faithfulness.",
  "published": "2025-03-27",
  "updated": "2025-08-20",
  "year": "2025",
  "authors": [
   "Dian Zheng",
   "Ziqi Huang",
   "Hongbo Liu",
   "Kai Zou",
   "Yinan He",
   "Fan Zhang",
   "Lulu Gu",
   "Yuanhan Zhang",
   "Jingwen He",
   "Wei-Shi Zheng",
   "Yu Qiao",
   "Ziwei Liu"
  ],
  "author_count": 12,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 248,
  "influential_citations": 25,
  "tldr": "VBench-2.0 aims to set a new standard for the next generation of video generative models in pursuit of intrinsic faithfulness, a next-generation benchmark designed to automatically evaluate video generative models for their intrinsic faithfulness.",
  "doi": "10.48550/arXiv.2503.21755",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Dian Zheng",
    "id": "2352408090",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Ziqi Huang",
    "id": "2243375536",
    "h_index": 13,
    "papers": 28
   },
   {
    "name": "Hongbo Liu",
    "id": "2352405612",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Kai Zou",
    "id": "2352277890",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Yinan He",
    "id": "2118918324",
    "h_index": 28,
    "papers": 48
   },
   {
    "name": "Fan Zhang",
    "id": "2268852317",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Yuanhan Zhang",
    "id": "2145784327",
    "h_index": 22,
    "papers": 47
   },
   {
    "name": "Jingwen He",
    "id": "2334349148",
    "h_index": 8,
    "papers": 8
   },
   {
    "name": "Wei-Shi Zheng",
    "id": "2352671384",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Yu Qiao",
    "id": "2327935778",
    "h_index": 11,
    "papers": 15
   },
   {
    "name": "Ziwei Liu",
    "id": "2243875466",
    "h_index": 18,
    "papers": 29
   }
  ],
  "comment": "Equal contributions from first two authors. Project page: https://vchitect.github.io/VBench-2.0-project/ Code: https://github.com/Vchitect/VBench",
  "topics": [
   "world-models",
   "data-teleop",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2503.21755v2",
  "pdf_url": "https://arxiv.org/pdf/2503.21755v2",
  "html_url": "https://arxiv.org/html/2503.21755v2",
  "code_url": "https://vchitect.github.io/VBench-2.0-project/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.4
 },
 {
  "id": "2503.20746",
  "slug": "physgen3d-crafting-a-miniature-interactive-world-from-a-single-image",
  "title": "PhysGen3D: Crafting a Miniature Interactive World from a Single Image",
  "abstract": "Envisioning physically plausible outcomes from a single image requires a deep understanding of the world's dynamics. To address this, we introduce PhysGen3D, a novel framework that transforms a single image into an amodal, camera-centric, interactive 3D scene. By combining advanced image-based geometric and semantic understanding with physics-based simulation, PhysGen3D creates an interactive 3D world from a static image, enabling us to \"imagine\" and simulate future scenarios based on user input. At its core, PhysGen3D estimates 3D shapes, poses, physical and lighting properties of objects, thereby capturing essential physical attributes that drive realistic object interactions. This framework allows users to specify precise initial conditions, such as object speed or material properties, for enhanced control over generated video outcomes. We evaluate PhysGen3D's performance against closed-source state-of-the-art (SOTA) image-to-video models, including Pika, Kling, and Gen-3, showing PhysGen3D's capacity to generate videos with realistic physics while offering greater flexibility and fine-grained control. Our results show that PhysGen3D achieves a unique balance of photorealism, physical plausibility, and user-driven interactivity, opening new possibilities for generating dynamic, physics-grounded video from an image.",
  "published": "2025-03-26",
  "updated": "2025-03-26",
  "year": "2025",
  "authors": [
   "Boyuan Chen",
   "Hanxiao Jiang",
   "Shaowei Liu",
   "Saurabh Gupta",
   "Yunzhu Li",
   "Hao Zhao",
   "Shenlong Wang"
  ],
  "author_count": 7,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 70,
  "influential_citations": 10,
  "tldr": "PhysGen3D achieves a unique balance of photorealism, physical plausibility, and user-driven interactivity, opening new possibilities for generating dynamic, physics-grounded video from an image.",
  "doi": "10.1109/CVPR52734.2025.00579",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Boyuan Chen",
    "id": "2295095893",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Hanxiao Jiang",
    "id": "2288026132",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Shaowei Liu",
    "id": "2254663052",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Saurabh Gupta",
    "id": "2254261428",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Yunzhu Li",
    "id": "2352183976",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Hao Zhao",
    "id": "2352622609",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Shenlong Wang",
    "id": "2288429921",
    "h_index": 4,
    "papers": 5
   }
  ],
  "comment": "CVPR 2025, Project page: https://by-luckk.github.io/PhysGen3D",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2503.20746v1",
  "pdf_url": "https://arxiv.org/pdf/2503.20746v1",
  "html_url": "https://arxiv.org/html/2503.20746v1",
  "code_url": "https://by-luckk.github.io/PhysGen3D",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.35
 },
 {
  "id": "2503.20523",
  "slug": "gaia-2-a-controllable-multi-view-generative-world-model-for-autonomous",
  "title": "GAIA-2: A Controllable Multi-View Generative World Model for Autonomous Driving",
  "abstract": "Generative models offer a scalable and flexible paradigm for simulating complex environments, yet current approaches fall short in addressing the domain-specific requirements of autonomous driving - such as multi-agent interactions, fine-grained control, and multi-camera consistency. We introduce GAIA-2, Generative AI for Autonomy, a latent diffusion world model that unifies these capabilities within a single generative framework. GAIA-2 supports controllable video generation conditioned on a rich set of structured inputs: ego-vehicle dynamics, agent configurations, environmental factors, and road semantics. It generates high-resolution, spatiotemporally consistent multi-camera videos across geographically diverse driving environments (UK, US, Germany). The model integrates both structured conditioning and external latent embeddings (e.g., from a proprietary driving model) to facilitate flexible and semantically grounded scene synthesis. Through this integration, GAIA-2 enables scalable simulation of both common and rare driving scenarios, advancing the use of generative world models as a core tool in the development of autonomous systems. Videos are available at https://wayve.ai/thinking/gaia-2.",
  "published": "2025-03-26",
  "updated": "2025-03-26",
  "year": "2025",
  "authors": [
   "Lloyd Russell",
   "Anthony Hu",
   "Lorenzo Bertoni",
   "George Fedoseev",
   "Jamie Shotton",
   "Elahe Arani",
   "Gianluca Corrado"
  ],
  "author_count": 7,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 171,
  "influential_citations": 9,
  "tldr": "GAIA-2, Generative AI for Autonomy, a latent diffusion world model that unifies capabilities of multi-agent interactions, fine-grained control, and multi-camera consistency within a single generative framework.",
  "doi": "10.48550/arXiv.2503.20523",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Lloyd Russell",
    "id": "2249537838",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Anthony Hu",
    "id": "2280065315",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Lorenzo Bertoni",
    "id": "51165585",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "George Fedoseev",
    "id": "2249528986",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Jamie Shotton",
    "id": "2249528512",
    "h_index": 8,
    "papers": 9
   },
   {
    "name": "Elahe Arani",
    "id": "2306781370",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Gianluca Corrado",
    "id": "2249536755",
    "h_index": 4,
    "papers": 4
   }
  ],
  "comment": "Technical Report",
  "topics": [
   "world-models",
   "navigation",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2503.20523v1",
  "pdf_url": "https://arxiv.org/pdf/2503.20523v1",
  "html_url": "https://arxiv.org/html/2503.20523v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.24
 },
 {
  "id": "2503.20314",
  "slug": "wan-open-and-advanced-large-scale-video-generative-models",
  "title": "Wan: Open and Advanced Large-Scale Video Generative Models",
  "abstract": "This report presents Wan, a comprehensive and open suite of video foundation models designed to push the boundaries of video generation. Built upon the mainstream diffusion transformer paradigm, Wan achieves significant advancements in generative capabilities through a series of innovations, including our novel VAE, scalable pre-training strategies, large-scale data curation, and automated evaluation metrics. These contributions collectively enhance the model's performance and versatility. Specifically, Wan is characterized by four key features: Leading Performance: The 14B model of Wan, trained on a vast dataset comprising billions of images and videos, demonstrates the scaling laws of video generation with respect to both data and model size. It consistently outperforms the existing open-source models as well as state-of-the-art commercial solutions across multiple internal and external benchmarks, demonstrating a clear and significant performance superiority. Comprehensiveness: Wan offers two capable models, i.e., 1.3B and 14B parameters, for efficiency and effectiveness respectively. It also covers multiple downstream applications, including image-to-video, instruction-guided video editing, and personal video generation, encompassing up to eight tasks. Consumer-Grade Efficiency: The 1.3B model demonstrates exceptional resource efficiency, requiring only 8.19 GB VRAM, making it compatible with a wide range of consumer-grade GPUs. Openness: We open-source the entire series of Wan, including source code and all models, with the goal of fostering the growth of the video generation community. This openness seeks to significantly expand the creative possibilities of video production in the industry and provide academia with high-quality video foundation models. All the code and models are available at https://github.com/Wan-Video/Wan2.1.",
  "published": "2025-03-26",
  "updated": "2025-04-19",
  "year": "2025",
  "authors": [
   "Team Wan",
   "Ang Wang",
   "Baole Ai",
   "Bin Wen",
   "Chaojie Mao",
   "Chen-Wei Xie",
   "Di Chen",
   "Feiwu Yu",
   "Haiming Zhao",
   "Jianxiao Yang",
   "Jianyuan Zeng",
   "Jiayu Wang",
   "Jingfeng Zhang",
   "Jingren Zhou",
   "Jinkai Wang",
   "Jixuan Chen",
   "Kai Zhu",
   "Kang Zhao",
   "Keyu Yan",
   "Lianghua Huang",
   "Mengyang Feng",
   "Ningyi Zhang",
   "Pandeng Li",
   "Pingyu Wu",
   "Ruihang Chu",
   "Ruili Feng",
   "Shiwei Zhang",
   "Siyang Sun",
   "Tao Fang",
   "Tianxing Wang",
   "Tianyi Gui",
   "Tingyu Weng",
   "Tong Shen",
   "Wei Lin",
   "Wei Wang",
   "Wei Wang",
   "Wenmeng Zhou",
   "Wente Wang",
   "Wenting Shen",
   "Wenyuan Yu",
   "Xianzhong Shi",
   "Xiaoming Huang",
   "Xin Xu",
   "Yan Kou",
   "Yangyu Lv",
   "Yifei Li",
   "Yijing Liu",
   "Yiming Wang",
   "Yingya Zhang",
   "Yitong Huang",
   "Yong Li",
   "You Wu",
   "Yu Liu",
   "Yulin Pan",
   "Yun Zheng",
   "Yuntao Hong",
   "Yupeng Shi",
   "Yutong Feng",
   "Zeyinzi Jiang",
   "Zhen Han",
   "Zhi-Fan Wu",
   "Ziyu Liu"
  ],
  "author_count": 62,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 2385,
  "influential_citations": 750,
  "tldr": "This report presents Wan, a comprehensive and open suite of video foundation models designed to push the boundaries of video generation, built upon the mainstream diffusion transformer paradigm, which consistently outperforms the existing open-source models as well as state-of-the-art commercial solutions across multiple internal and external benchmarks.",
  "doi": "10.48550/arXiv.2503.20314",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ang Wang",
    "id": "2316094898",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Baole Ai",
    "id": "2315922148",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Bin Wen",
    "id": "2346834982",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Chaojie Mao",
    "id": "2263499355",
    "h_index": 10,
    "papers": 19
   },
   {
    "name": "Chen-Wei Xie",
    "id": "2324708991",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Di Chen",
    "id": "2202464939",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Feiwu Yu",
    "id": "2347864655",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Haiming Zhao",
    "id": "2215401267",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Jianxiao Yang",
    "id": "51225204",
    "h_index": 17,
    "papers": 53
   },
   {
    "name": "Jianyuan Zeng",
    "id": "2316895193",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Jiayu Wang",
    "id": "2233240814",
    "h_index": 10,
    "papers": 14
   },
   {
    "name": "Jingfeng Zhang",
    "id": "47539929",
    "h_index": 25,
    "papers": 104
   },
   {
    "name": "Jingren Zhou",
    "id": "2351881319",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Jinkai Wang",
    "id": "2313274254",
    "h_index": 4,
    "papers": 15
   },
   {
    "name": "Jixuan Chen",
    "id": "2309669030",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Kai Zhu",
    "id": "2335438647",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Kang Zhao",
    "id": "2211097160",
    "h_index": 7,
    "papers": 14
   },
   {
    "name": "Keyu Yan",
    "id": "2309483988",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Lianghua Huang",
    "id": "2293560914",
    "h_index": 6,
    "papers": 21
   },
   {
    "name": "Xiaofeng Meng",
    "id": "2350237579",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Ningying Zhang",
    "id": "2329625237",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Pandeng Li",
    "id": "2334691790",
    "h_index": 6,
    "papers": 17
   },
   {
    "name": "Ping Wu",
    "id": "2143778284",
    "h_index": 9,
    "papers": 14
   },
   {
    "name": "Ruihang Chu",
    "id": "2334866310",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Rui Feng",
    "id": "2338126404",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Shiwei Zhang",
    "id": "2219846537",
    "h_index": 19,
    "papers": 33
   },
   {
    "name": "Siyang Sun",
    "id": "2187387762",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Tao Fang",
    "id": "2289214954",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Tianxing Wang",
    "id": "2352307538",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "T. Gui",
    "id": "2352099739",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Tingyu Weng",
    "id": "2348441108",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Tong Shen",
    "id": "2335673164",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Wei Lin",
    "id": "2342451676",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Wei Wang",
    "id": "2214591861",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Wei Wang",
    "id": "2214591861",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Wen-Chao Zhou",
    "id": "2272086377",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Wente Wang",
    "id": "2342454024",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Wen Shen",
    "id": "2088422443",
    "h_index": 6,
    "papers": 14
   },
   {
    "name": "Wenyuan Yu",
    "id": "2342570538",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Xianzhong Shi",
    "id": "2352869154",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Xiaomin Huang",
    "id": "2317137595",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Xin Xu",
    "id": "2339233642",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Yan Kou",
    "id": "2352099760",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Yan-Mei Lv",
    "id": "2188980291",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Yifei Li",
    "id": "2167506235",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Yijing Liu",
    "id": "2343117564",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Yiming Wang",
    "id": "2335225276",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Yingya Zhang",
    "id": "2244766555",
    "h_index": 17,
    "papers": 35
   },
   {
    "name": "Yitong Huang",
    "id": "2364797843",
    "h_index": 5,
    "papers": 23
   },
   {
    "name": "Yong Li",
    "id": "2348373355",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "You Wu",
    "id": "2327878412",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Yu Liu",
    "id": "2340369109",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Yulin Pan",
    "id": "2275285644",
    "h_index": 8,
    "papers": 13
   },
   {
    "name": "Yun Zheng",
    "id": "2279756190",
    "h_index": 7,
    "papers": 17
   },
   {
    "name": "Yuntao Hong",
    "id": "2330760989",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Yupeng Shi",
    "id": "2327018450",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Yutong Feng",
    "id": "2268567496",
    "h_index": 11,
    "papers": 15
   },
   {
    "name": "Zeyinzi Jiang",
    "id": "2210197520",
    "h_index": 11,
    "papers": 17
   },
   {
    "name": "Zhen Han",
    "id": "2275160141",
    "h_index": 9,
    "papers": 12
   },
   {
    "name": "Zhigang Wu",
    "id": "2286318517",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Ziyu Liu",
    "id": "2117942962",
    "h_index": 1,
    "papers": 3
   }
  ],
  "comment": "60 pages, 33 figures",
  "topics": [
   "foundation-pretraining",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2503.20314v2",
  "pdf_url": "https://arxiv.org/pdf/2503.20314v2",
  "html_url": "https://arxiv.org/html/2503.20314v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.0
 },
 {
  "id": "2503.19786",
  "slug": "gemma-3-technical-report",
  "title": "Gemma 3 Technical Report",
  "abstract": "We introduce Gemma 3, a multimodal addition to the Gemma family of lightweight open models, ranging in scale from 1 to 27 billion parameters. This version introduces vision understanding abilities, a wider coverage of languages and longer context - at least 128K tokens. We also change the architecture of the model to reduce the KV-cache memory that tends to explode with long context. This is achieved by increasing the ratio of local to global attention layers, and keeping the span on local attention short. The Gemma 3 models are trained with distillation and achieve superior performance to Gemma 2 for both pre-trained and instruction finetuned versions. In particular, our novel post-training recipe significantly improves the math, chat, instruction-following and multilingual abilities, making Gemma3-4B-IT competitive with Gemma2-27B-IT and Gemma3-27B-IT comparable to Gemini-1.5-Pro across benchmarks. We release all our models to the community.",
  "published": "2025-03-25",
  "updated": "2025-03-25",
  "year": "2025",
  "authors": [
   " Gemma Team",
   "Aishwarya Kamath",
   "Johan Ferret",
   "Shreya Pathak",
   "Nino Vieillard",
   "Ramona Merhej",
   "Sarah Perrin",
   "Tatiana Matejovicova",
   "Alexandre Ram\u00e9",
   "Morgane Rivi\u00e8re",
   "Louis Rouillard",
   "Thomas Mesnard",
   "Geoffrey Cideron",
   "Jean-bastien Grill",
   "Sabela Ramos",
   "Edouard Yvinec",
   "Michelle Casbon",
   "Etienne Pot",
   "Ivo Penchev",
   "Ga\u00ebl Liu",
   "Francesco Visin",
   "Kathleen Kenealy",
   "Lucas Beyer",
   "Xiaohai Zhai",
   "Anton Tsitsulin",
   "Robert Busa-Fekete",
   "Alex Feng",
   "Noveen Sachdeva",
   "Benjamin Coleman",
   "Yi Gao",
   "Basil Mustafa",
   "Iain Barr",
   "Emilio Parisotto",
   "David Tian",
   "Matan Eyal",
   "Colin Cherry",
   "Jan-Thorsten Peter",
   "Danila Sinopalnikov",
   "Surya Bhupatiraju",
   "Rishabh Agarwal",
   "Mehran Kazemi",
   "Dan Malkin",
   "Ravin Kumar",
   "David Vilar",
   "Idan Brusilovsky",
   "Jiaming Luo",
   "Andreas Steiner",
   "Abe Friesen",
   "Abhanshu Sharma",
   "Abheesht Sharma",
   "Adi Mayrav Gilady",
   "Adrian Goedeckemeyer",
   "Alaa Saade",
   "Alex Feng",
   "Alexander Kolesnikov",
   "Alexei Bendebury",
   "Alvin Abdagic",
   "Amit Vadi",
   "Andr\u00e1s Gy\u00f6rgy",
   "Andr\u00e9 Susano Pinto",
   "Anil Das",
   "Ankur Bapna",
   "Antoine Miech",
   "Antoine Yang",
   "Antonia Paterson",
   "Ashish Shenoy",
   "Ayan Chakrabarti",
   "Bilal Piot",
   "Bo Wu",
   "Bobak Shahriari",
   "Bryce Petrini",
   "Charlie Chen",
   "Charline Le Lan",
   "Christopher A. Choquette-Choo",
   "CJ Carey",
   "Cormac Brick",
   "Daniel Deutsch",
   "Danielle Eisenbud",
   "Dee Cattle",
   "Derek Cheng",
   "Dimitris Paparas",
   "Divyashree Shivakumar Sreepathihalli",
   "Doug Reid",
   "Dustin Tran",
   "Dustin Zelle",
   "Eric Noland",
   "Erwin Huizenga",
   "Eugene Kharitonov",
   "Frederick Liu",
   "Gagik Amirkhanyan",
   "Glenn Cameron",
   "Hadi Hashemi",
   "Hanna Klimczak-Pluci\u0144ska",
   "Harman Singh",
   "Harsh Mehta",
   "Harshal Tushar Lehri",
   "Hussein Hazimeh",
   "Ian Ballantyne",
   "Idan Szpektor",
   "Ivan Nardini",
   "Jean Pouget-Abadie",
   "Jetha Chan",
   "Joe Stanton",
   "John Wieting",
   "Jonathan Lai",
   "Jordi Orbay",
   "Joseph Fernandez",
   "Josh Newlan",
   "Ju-yeong Ji",
   "Jyotinder Singh",
   "Kat Black",
   "Kathy Yu",
   "Kevin Hui",
   "Kiran Vodrahalli",
   "Klaus Greff",
   "Linhai Qiu",
   "Marcella Valentine",
   "Marina Coelho",
   "Marvin Ritter",
   "Matt Hoffman",
   "Matthew Watson",
   "Mayank Chaturvedi",
   "Michael Moynihan",
   "Min Ma",
   "Nabila Babar",
   "Natasha Noy",
   "Nathan Byrd",
   "Nick Roy",
   "Nikola Momchev",
   "Nilay Chauhan",
   "Noveen Sachdeva",
   "Oskar Bunyan",
   "Pankil Botarda",
   "Paul Caron",
   "Paul Kishan Rubenstein",
   "Phil Culliton",
   "Philipp Schmid",
   "Pier Giuseppe Sessa",
   "Pingmei Xu",
   "Piotr Stanczyk",
   "Pouya Tafti",
   "Rakesh Shivanna",
   "Renjie Wu",
   "Renke Pan",
   "Reza Rokni",
   "Rob Willoughby",
   "Rohith Vallu",
   "Ryan Mullins",
   "Sammy Jerome",
   "Sara Smoot",
   "Sertan Girgin",
   "Shariq Iqbal",
   "Shashir Reddy",
   "Shruti Sheth",
   "Siim P\u00f5der",
   "Sijal Bhatnagar",
   "Sindhu Raghuram Panyam",
   "Sivan Eiger",
   "Susan Zhang",
   "Tianqi Liu",
   "Trevor Yacovone",
   "Tyler Liechty",
   "Uday Kalra",
   "Utku Evci",
   "Vedant Misra",
   "Vincent Roseberry",
   "Vlad Feinberg",
   "Vlad Kolesnikov",
   "Woohyun Han",
   "Woosuk Kwon",
   "Xi Chen",
   "Yinlam Chow",
   "Yuvein Zhu",
   "Zichuan Wei",
   "Zoltan Egyed",
   "Victor Cotruta",
   "Minh Giang",
   "Phoebe Kirk",
   "Anand Rao",
   "Kat Black",
   "Nabila Babar",
   "Jessica Lo",
   "Erica Moreira",
   "Luiz Gustavo Martins",
   "Omar Sanseviero",
   "Lucas Gonzalez",
   "Zach Gleicher",
   "Tris Warkentin",
   "Vahab Mirrokni",
   "Evan Senter",
   "Eli Collins",
   "Joelle Barral",
   "Zoubin Ghahramani",
   "Raia Hadsell",
   "Yossi Matias",
   "D. Sculley",
   "Slav Petrov",
   "Noah Fiedel",
   "Noam Shazeer",
   "Oriol Vinyals",
   "Jeff Dean",
   "Demis Hassabis",
   "Koray Kavukcuoglu",
   "Clement Farabet",
   "Elena Buchatskaya",
   "Jean-Baptiste Alayrac",
   "Rohan Anil",
   " Dmitry",
   " Lepikhin",
   "Sebastian Borgeaud",
   "Olivier Bachem",
   "Armand Joulin",
   "Alek Andreev",
   "Cassidy Hardin",
   "Robert Dadashi",
   "L\u00e9onard Hussenot"
  ],
  "author_count": 216,
  "categories": [
   "cs.CL",
   "cs.AI"
  ],
  "primary_category": "cs.CL",
  "venue": "",
  "venue_source": "",
  "citations": 1752,
  "influential_citations": 304,
  "tldr": "A novel post-training recipe significantly improves the math, chat, instruction-following and multilingual abilities, making Gemma3-4B-IT competitive with Gemma2-27B-IT and Gemma3-27B-IT comparable to Gemini-1.5-Pro across benchmarks.",
  "doi": "10.48550/arXiv.2503.19786",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Gemma Team Aishwarya Kamath",
    "id": "2351910857",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Johan Ferret",
    "id": "151047979",
    "h_index": 18,
    "papers": 51
   },
   {
    "name": "Shreya Pathak",
    "id": "2273651441",
    "h_index": 9,
    "papers": 32
   },
   {
    "name": "Nino Vieillard",
    "id": "2308037523",
    "h_index": 10,
    "papers": 31
   },
   {
    "name": "Ramona Merhej",
    "id": "2092094191",
    "h_index": 7,
    "papers": 23
   },
   {
    "name": "Sarah Perrin",
    "id": "2312326042",
    "h_index": 8,
    "papers": 22
   },
   {
    "name": "Tatiana Matejovicova",
    "id": "2166868706",
    "h_index": 8,
    "papers": 17
   },
   {
    "name": "Alexandre Ram'e",
    "id": "2280134846",
    "h_index": 9,
    "papers": 12
   },
   {
    "name": "Morgane Rivi\u00e8re",
    "id": "2275177725",
    "h_index": 6,
    "papers": 25
   },
   {
    "name": "Louis Rouillard",
    "id": "2351906592",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Thomas Mesnard",
    "id": "2237423540",
    "h_index": 16,
    "papers": 42
   },
   {
    "name": "Geoffrey Cideron",
    "id": "2282966842",
    "h_index": 8,
    "papers": 16
   },
   {
    "name": "Jean-Bastien Grill",
    "id": "145840757",
    "h_index": 15,
    "papers": 34
   },
   {
    "name": "Sabela Ramos",
    "id": "2253595555",
    "h_index": 8,
    "papers": 25
   },
   {
    "name": "Edouard Yvinec",
    "id": "1632928879",
    "h_index": 11,
    "papers": 34
   },
   {
    "name": "M. Casbon",
    "id": "2314115981",
    "h_index": 6,
    "papers": 17
   },
   {
    "name": "Etienne Pot",
    "id": "38627717",
    "h_index": 9,
    "papers": 18
   },
   {
    "name": "Ivo Penchev",
    "id": "2275187196",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Gael Liu",
    "id": "2352144736",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Francesco Visin",
    "id": "2077146",
    "h_index": 16,
    "papers": 28
   },
   {
    "name": "Kathleen Kenealy",
    "id": "1914502282",
    "h_index": 9,
    "papers": 26
   },
   {
    "name": "Lucas Beyer",
    "id": "39611591",
    "h_index": 39,
    "papers": 59
   },
   {
    "name": "Xiaohai Zhai",
    "id": "2351913330",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Anton Tsitsulin",
    "id": "40900939",
    "h_index": 16,
    "papers": 39
   },
   {
    "name": "R. Busa-Fekete",
    "id": "1398396696",
    "h_index": 31,
    "papers": 101
   },
   {
    "name": "Alex Feng",
    "id": "2351910788",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Noveen Sachdeva",
    "id": "40705044",
    "h_index": 15,
    "papers": 32
   },
   {
    "name": "Benjamin Coleman",
    "id": "2240537645",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Yi Gao",
    "id": "2352066300",
    "h_index": 5,
    "papers": 15
   },
   {
    "name": "Basil Mustafa",
    "id": "40608942",
    "h_index": 27,
    "papers": 40
   },
   {
    "name": "Iain Barr",
    "id": "2159207795",
    "h_index": 8,
    "papers": 21
   },
   {
    "name": "Emilio Parisotto",
    "id": "3166516",
    "h_index": 26,
    "papers": 44
   },
   {
    "name": "David Tian",
    "id": "2352238566",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Matan Eyal",
    "id": "35298844",
    "h_index": 13,
    "papers": 26
   },
   {
    "name": "Colin Cherry",
    "id": "2257002653",
    "h_index": 9,
    "papers": 21
   },
   {
    "name": "Jan-Thorsten Peter",
    "id": "2265752718",
    "h_index": 6,
    "papers": 14
   },
   {
    "name": "Danila Sinopalnikov",
    "id": "1470524352",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Surya Bhupatiraju",
    "id": "9692128",
    "h_index": 11,
    "papers": 37
   },
   {
    "name": "Rishabh Agarwal",
    "id": "2258553001",
    "h_index": 7,
    "papers": 18
   },
   {
    "name": "Mehran Kazemi",
    "id": "2283308055",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Dan Malkin",
    "id": "2348097555",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Ravin Kumar",
    "id": "2290629265",
    "h_index": 9,
    "papers": 27
   },
   {
    "name": "David Vilar",
    "id": "2257003298",
    "h_index": 6,
    "papers": 14
   },
   {
    "name": "I. Brusilovsky",
    "id": "2072258345",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Jiaming Luo",
    "id": "19236313",
    "h_index": 9,
    "papers": 17
   },
   {
    "name": "A. Steiner",
    "id": "2079614268",
    "h_index": 15,
    "papers": 23
   },
   {
    "name": "Abe Friesen",
    "id": "2312322254",
    "h_index": 6,
    "papers": 19
   },
   {
    "name": "Abhanshu Sharma",
    "id": "2275537981",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Abheesht Sharma",
    "id": "2304366556",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Adi Mayrav Gilady",
    "id": "2348098277",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Adrian Goedeckemeyer",
    "id": "2275188881",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Alaa Saade",
    "id": "16927419",
    "h_index": 13,
    "papers": 31
   },
   {
    "name": "Alexander Kolesnikov",
    "id": "2333232272",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Alexei Bendebury",
    "id": "2351911525",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Alvin Abdagic",
    "id": "3026185",
    "h_index": 6,
    "papers": 16
   },
   {
    "name": "Amit Vadi",
    "id": "2351910059",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Andr'as Gyorgy",
    "id": "2343746240",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Andr\u00e9 Susano Pinto",
    "id": "1809220",
    "h_index": 16,
    "papers": 23
   },
   {
    "name": "Anil Das",
    "id": "2352906247",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Ankur Bapna",
    "id": "12295226",
    "h_index": 36,
    "papers": 71
   },
   {
    "name": "Antoine Miech",
    "id": "19200186",
    "h_index": 21,
    "papers": 42
   },
   {
    "name": "Antoine Yang",
    "id": "2064599701",
    "h_index": 13,
    "papers": 19
   },
   {
    "name": "Antonia Paterson",
    "id": "2291068272",
    "h_index": 6,
    "papers": 18
   },
   {
    "name": "Ashish Shenoy",
    "id": "2275187275",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Ayan Chakrabarti",
    "id": "2279822092",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Bilal Piot",
    "id": "1808897",
    "h_index": 44,
    "papers": 80
   },
   {
    "name": "Boxi Wu",
    "id": "2275291909",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Bobak Shahriari",
    "id": "2067577",
    "h_index": 15,
    "papers": 42
   },
   {
    "name": "Bryce Petrini",
    "id": "2181215497",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Charlie Chen",
    "id": "2182971260",
    "h_index": 10,
    "papers": 22
   },
   {
    "name": "Charline Le Lan",
    "id": "153892869",
    "h_index": 16,
    "papers": 37
   },
   {
    "name": "Christopher A. Choquette-Choo",
    "id": "2314115870",
    "h_index": 14,
    "papers": 38
   },
   {
    "name": "Cj Carey",
    "id": "2055444975",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "C. Brick",
    "id": "34587012",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Daniel Deutsch",
    "id": "2264072172",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Danielle Eisenbud",
    "id": "2351908333",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Dee Cattle",
    "id": "2351906644",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "D. Cheng",
    "id": "2352214485",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Dimitris Paparas",
    "id": "2275466240",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Divyashree Shivakumar Sreepathihalli",
    "id": "2303850363",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Doug Reid",
    "id": "2351913146",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Dustin Tran",
    "id": "2273790995",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Dustin Zelle",
    "id": "1389613483",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "Eric Noland",
    "id": "51210148",
    "h_index": 12,
    "papers": 33
   },
   {
    "name": "Erwin Huizenga",
    "id": "2351908317",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "E. Kharitonov",
    "id": "144875326",
    "h_index": 27,
    "papers": 45
   },
   {
    "name": "Frederick Liu",
    "id": "2260276550",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "G. Amirkhanyan",
    "id": "2536021",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Glenn Cameron",
    "id": "2295989714",
    "h_index": 6,
    "papers": 16
   },
   {
    "name": "Hadi Hashemi",
    "id": "2070487928",
    "h_index": 8,
    "papers": 30
   },
   {
    "name": "Hanna Klimczak-Pluci'nska",
    "id": "2275187187",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Harman Singh",
    "id": "2119151340",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Harsh Mehta",
    "id": "2337578397",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Harshal Tushar Lehri",
    "id": "10741766",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Hussein Hazimeh",
    "id": "2284233466",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Ian Ballantyne",
    "id": "2351909424",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Idan Szpektor",
    "id": "1711977",
    "h_index": 37,
    "papers": 118
   },
   {
    "name": "Ivan Nardini",
    "id": "2314115441",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Jean Pouget-Abadie",
    "id": "1403025868",
    "h_index": 15,
    "papers": 34
   },
   {
    "name": "Jetha Chan",
    "id": "2314546523",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Joe Stanton",
    "id": "2275190309",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "J. Michael Wieting",
    "id": "2250625143",
    "h_index": 6,
    "papers": 16
   },
   {
    "name": "J. Lai",
    "id": "2276422942",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Jordi Orbay",
    "id": "2142424210",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Joe Fernandez",
    "id": "2314081561",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Joshua Newlan",
    "id": "2160888100",
    "h_index": 7,
    "papers": 17
   },
   {
    "name": "Junsong Ji",
    "id": "2299505170",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Jyotinder Singh",
    "id": "2352007848",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Kat Black",
    "id": "2314110240",
    "h_index": 7,
    "papers": 19
   },
   {
    "name": "Kathy Yu",
    "id": "2291882760",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Kevin Hui",
    "id": "2290487874",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Kiran Vodrahalli",
    "id": "4529644",
    "h_index": 13,
    "papers": 48
   },
   {
    "name": "Klaus Greff",
    "id": "3035541",
    "h_index": 28,
    "papers": 49
   },
   {
    "name": "Linhai Qiu",
    "id": "2344051713",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Marcella Valentine",
    "id": "2351907699",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Marina Coelho",
    "id": "2351909700",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Marvin Ritter",
    "id": "39687627",
    "h_index": 13,
    "papers": 19
   },
   {
    "name": "Matt Hoffman",
    "id": "2312323782",
    "h_index": 6,
    "papers": 14
   },
   {
    "name": "Matthew Watson",
    "id": "2303852016",
    "h_index": 5,
    "papers": 17
   },
   {
    "name": "Mayank Chaturvedi",
    "id": "2351907725",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Michael Moynihan",
    "id": "2314111806",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Min Ma",
    "id": "2352024723",
    "h_index": 7,
    "papers": 16
   },
   {
    "name": "Nabila Babar",
    "id": "2351909372",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Natasha Noy",
    "id": "2351907620",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Nathan Byrd",
    "id": "2275185667",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Nick Roy",
    "id": "2352018552",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Nikola Momchev",
    "id": "1470531643",
    "h_index": 12,
    "papers": 26
   },
   {
    "name": "Nilay Chauhan",
    "id": "2295989865",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "Oskar Bunyan",
    "id": "2275177720",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Pankil Botarda",
    "id": "2314113467",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Paul Caron",
    "id": "2351909416",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "P. Rubenstein",
    "id": "2249760524",
    "h_index": 6,
    "papers": 14
   },
   {
    "name": "Phil Culliton",
    "id": "40579094",
    "h_index": 6,
    "papers": 16
   },
   {
    "name": "P. Schmid",
    "id": "51016683",
    "h_index": 18,
    "papers": 104
   },
   {
    "name": "Pier Giuseppe Sessa",
    "id": "7281978",
    "h_index": 16,
    "papers": 36
   },
   {
    "name": "Ping-mei Xu",
    "id": "2352600773",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "P. Sta\u0144czyk",
    "id": "2067024583",
    "h_index": 13,
    "papers": 41
   },
   {
    "name": "P. Tafti",
    "id": "1775270",
    "h_index": 15,
    "papers": 64
   },
   {
    "name": "Rakesh Shivanna",
    "id": "2934334",
    "h_index": 9,
    "papers": 20
   },
   {
    "name": "Renjie Wu",
    "id": "2324223418",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Renke Pan",
    "id": "2352075260",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "R. Rokni",
    "id": "15170591",
    "h_index": 5,
    "papers": 18
   },
   {
    "name": "Rob Willoughby",
    "id": "2275189345",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Rohith Vallu",
    "id": "13997540",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Ryan Mullins",
    "id": "2291067498",
    "h_index": 10,
    "papers": 31
   },
   {
    "name": "Sammy Jerome",
    "id": "2287843663",
    "h_index": 6,
    "papers": 17
   },
   {
    "name": "Sara Smoot",
    "id": "2351911771",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Sertan Girgin",
    "id": "35022714",
    "h_index": 23,
    "papers": 78
   },
   {
    "name": "Shariq Iqbal",
    "id": "2899335",
    "h_index": 13,
    "papers": 21
   },
   {
    "name": "Shashir Reddy",
    "id": "2352422018",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Shruti Sheth",
    "id": "35702090",
    "h_index": 7,
    "papers": 20
   },
   {
    "name": "Siim P\u00f5der",
    "id": "2275190197",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Sijal Bhatnagar",
    "id": "2351909196",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Sindhu Raghuram Panyam",
    "id": "2093079201",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Sivan Eiger",
    "id": "2348098272",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Susan Zhang",
    "id": "2238121623",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Tianqi Liu",
    "id": "2275249023",
    "h_index": 7,
    "papers": 16
   },
   {
    "name": "Trevor Yacovone",
    "id": "2351909612",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "T. Liechty",
    "id": "2275188012",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Uday Kalra",
    "id": "2351908147",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Utku Evci",
    "id": "3399348",
    "h_index": 19,
    "papers": 33
   },
   {
    "name": "Vedant Misra",
    "id": "40055795",
    "h_index": 13,
    "papers": 39
   },
   {
    "name": "Vincent Roseberry",
    "id": "2351906611",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Vladimir Feinberg",
    "id": "2275181199",
    "h_index": 11,
    "papers": 38
   },
   {
    "name": "Vlad Kolesnikov",
    "id": "2351908597",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Woohyun Han",
    "id": "2215449616",
    "h_index": 7,
    "papers": 15
   },
   {
    "name": "Woosuk Kwon",
    "id": "2314115546",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Xi Chen",
    "id": "2275535939",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Yinlam Chow",
    "id": "1819830",
    "h_index": 29,
    "papers": 75
   },
   {
    "name": "Yuvein Zhu",
    "id": "2352040805",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Zichuan Wei",
    "id": "2314328859",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "Z. Egyed",
    "id": "41183279",
    "h_index": 22,
    "papers": 103
   },
   {
    "name": "Victor Cotruta",
    "id": "2275189218",
    "h_index": 6,
    "papers": 20
   },
   {
    "name": "Minh Giang",
    "id": "2275187490",
    "h_index": 8,
    "papers": 25
   },
   {
    "name": "Phoebe Kirk",
    "id": "2314107430",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Anand Rao",
    "id": "2314521404",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Jessica Lo",
    "id": "2351910249",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Erica Moreira",
    "id": "2275185558",
    "h_index": 8,
    "papers": 29
   },
   {
    "name": "Luiz Gustavo Martins",
    "id": "2296198494",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Omar Sanseviero",
    "id": "2186979509",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Lucas Gonzalez",
    "id": "2275585027",
    "h_index": 6,
    "papers": 16
   },
   {
    "name": "Zach Gleicher",
    "id": "2275185661",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Tris Warkentin",
    "id": "1986491804",
    "h_index": 13,
    "papers": 29
   },
   {
    "name": "V. Mirrokni",
    "id": "1728881",
    "h_index": 62,
    "papers": 421
   },
   {
    "name": "Evan Senter",
    "id": "2268665228",
    "h_index": 9,
    "papers": 21
   },
   {
    "name": "Eli Collins",
    "id": "2275181648",
    "h_index": 8,
    "papers": 23
   },
   {
    "name": "Joelle Barral",
    "id": "2254701020",
    "h_index": 6,
    "papers": 17
   },
   {
    "name": "Z. Ghahramani",
    "id": "1983575",
    "h_index": 22,
    "papers": 61
   },
   {
    "name": "R. Hadsell",
    "id": "2315504",
    "h_index": 51,
    "papers": 111
   },
   {
    "name": "Y. Matias",
    "id": "2269148232",
    "h_index": 29,
    "papers": 118
   },
   {
    "name": "D. Sculley",
    "id": "2264141229",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Slav Petrov",
    "id": "2257293575",
    "h_index": 6,
    "papers": 17
   },
   {
    "name": "Noah Fiedel",
    "id": "22640071",
    "h_index": 20,
    "papers": 42
   },
   {
    "name": "Noam Shazeer",
    "id": "1846258",
    "h_index": 40,
    "papers": 146
   },
   {
    "name": "O. Vinyals",
    "id": "1689108",
    "h_index": 103,
    "papers": 204
   },
   {
    "name": "Jeffrey Dean",
    "id": "2265529729",
    "h_index": 8,
    "papers": 16
   },
   {
    "name": "D. Hassabis",
    "id": "48987704",
    "h_index": 92,
    "papers": 160
   },
   {
    "name": "K. Kavukcuoglu",
    "id": "2645384",
    "h_index": 76,
    "papers": 124
   },
   {
    "name": "C. Farabet",
    "id": "2256269",
    "h_index": 27,
    "papers": 50
   },
   {
    "name": "Elena Buchatskaya",
    "id": "118801223",
    "h_index": 12,
    "papers": 36
   },
   {
    "name": "Jean-Baptiste Alayrac",
    "id": "2285263",
    "h_index": 33,
    "papers": 67
   },
   {
    "name": "Rohan Anil",
    "id": "1508890387",
    "h_index": 21,
    "papers": 57
   },
   {
    "name": "Dmitry Lepikhin",
    "id": "150077954",
    "h_index": 13,
    "papers": 25
   },
   {
    "name": "Sebastian Borgeaud",
    "id": "148016269",
    "h_index": 20,
    "papers": 52
   },
   {
    "name": "Olivier Bachem",
    "id": "1936951",
    "h_index": 41,
    "papers": 85
   },
   {
    "name": "Armand Joulin",
    "id": "2319608",
    "h_index": 72,
    "papers": 151
   },
   {
    "name": "Alek Andreev",
    "id": "2290741315",
    "h_index": 8,
    "papers": 20
   },
   {
    "name": "Cassidy Hardin",
    "id": "2275186843",
    "h_index": 9,
    "papers": 29
   },
   {
    "name": "Robert Dadashi",
    "id": "51914693",
    "h_index": 20,
    "papers": 40
   },
   {
    "name": "L'eonard Hussenot",
    "id": "2312322565",
    "h_index": 9,
    "papers": 24
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2503.19786v1",
  "pdf_url": "https://arxiv.org/pdf/2503.19786v1",
  "html_url": "https://arxiv.org/html/2503.19786v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.0
 },
 {
  "id": "2503.19225",
  "slug": "coinft-a-coin-sized-capacitive-6-axis-force-torque-sensor-for-robotic",
  "title": "CoinFT: A Coin-Sized, Capacitive 6-Axis Force Torque Sensor for Robotic Applications",
  "abstract": "We introduce CoinFT, a capacitive 6-axis force/torque (F/T) sensor that is compact, light, low-cost, and robust with an average root-mean-squared error of 0.16N for force and 1.08mNm for moment when the input ranges from 0~14N and 0~5N in normal and shear directions, respectively. CoinFT is a stack of two rigid PCBs with comb-shaped electrodes connected by an array of silicone rubber pillars. The microcontroller interrogates the electrodes in different subsets in order to enhance sensitivity for measuring 6-axis F/T. The combination of features of CoinFT enables various contact-rich robot interactions across different embodiment domains including drones, robot end-effectors, and wearable haptic devices. We demonstrate the utility of CoinFT through two representative applications: a multi-axial contact-probing experiment in which a CoinFT mounted beneath a hemispherical fingertip measures 6-axes of force and torque representative of manipulation scenarios, and an attitude-based force-control task on a drone. The design, fabrication, and firmware of CoinFT are open-sourced at https://coin-ft.github.io/.",
  "published": "2025-03-25",
  "updated": "2026-01-15",
  "year": "2025",
  "authors": [
   "Hojung Choi",
   "Jun En Low",
   "Tae Myung Huh",
   "Seongheon Hong",
   "Gabriela A. Uribe",
   "Kenneth A. W. Hoffmann",
   "Julia Di",
   "Tony G. Chen",
   "Andrew A. Stanley",
   "Mark R. Cutkosky"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO",
   "cs.HC"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 18,
  "influential_citations": 1,
  "tldr": "",
  "doi": "10.48550/arXiv.2503.19225",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hojung Choi",
    "id": "2284304602",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Jun En Low",
    "id": "9195524",
    "h_index": 9,
    "papers": 12
   },
   {
    "name": "Tae Myung Huh",
    "id": "9529777",
    "h_index": 13,
    "papers": 22
   },
   {
    "name": "Gabriela Uribe",
    "id": "2084764700",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Seongheon Hong",
    "id": "2352066104",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Kenneth A. W. Hoffman",
    "id": "2352013587",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Julia Di",
    "id": "2296989885",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Tony G. Chen",
    "id": "2352001958",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Andrew A. Stanley",
    "id": "2248195541",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Mark R. Cutkosky",
    "id": "2329168577",
    "h_index": 4,
    "papers": 21
   }
  ],
  "comment": "",
  "topics": [
   "tactile",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2503.19225v3",
  "pdf_url": "https://arxiv.org/pdf/2503.19225v3",
  "html_url": "https://arxiv.org/html/2503.19225v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.28
 },
 {
  "id": "2503.18738",
  "slug": "roboengine-plug-and-play-robot-data-augmentation-with-semantic-robot-s",
  "title": "RoboEngine: Plug-and-Play Robot Data Augmentation with Semantic Robot Segmentation and Background Generation",
  "abstract": "Visual augmentation has become a crucial technique for enhancing the visual robustness of imitation learning. However, existing methods are often limited by prerequisites such as camera calibration or the need for controlled environments (e.g., green screen setups). In this work, we introduce RoboEngine, the first plug-and-play visual robot data augmentation toolkit. For the first time, users can effortlessly generate physics- and task-aware robot scenes with just a few lines of code. To achieve this, we present a novel robot scene segmentation dataset, a generalizable high-quality robot segmentation model, and a fine-tuned background generation model, which together form the core components of the out-of-the-box toolkit. Using RoboEngine, we demonstrate the ability to generalize robot manipulation tasks across six entirely new scenes, based solely on demonstrations collected from a single scene, achieving a more than 200% performance improvement compared to the no-augmentation baseline. All datasets, model weights, and the toolkit are released https://roboengine.github.io/",
  "published": "2025-03-24",
  "updated": "2025-07-13",
  "year": "2025",
  "authors": [
   "Chengbo Yuan",
   "Suraj Joshi",
   "Shaoting Zhu",
   "Hang Su",
   "Hang Zhao",
   "Yang Gao"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS",
  "venue_source": "semantic-scholar",
  "citations": 51,
  "influential_citations": 5,
  "tldr": "RoboEngine is introduced, the first plug-and-play visual robot data augmentation toolkit that can generalize robot manipulation tasks across six entirely new scenes, based solely on demonstrations collected from a single scene, achieving a more than 200% performance improvement compared to the no-augmentation baseline.",
  "doi": "10.1109/IROS60139.2025.11247097",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Chengbo Yuan",
    "id": "2280187985",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Suraj Joshi",
    "id": "2332883577",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Shaoting Zhu",
    "id": "2216251632",
    "h_index": 8,
    "papers": 16
   },
   {
    "name": "Hang Su",
    "id": "2302165612",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Hang Zhao",
    "id": "2312396784",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Yang Gao",
    "id": "2330756947",
    "h_index": 4,
    "papers": 8
   }
  ],
  "comment": "Project Page: https://roboengine.github.io/",
  "topics": [
   "imitation-diffusion",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2503.18738v2",
  "pdf_url": "https://arxiv.org/pdf/2503.18738v2",
  "html_url": "https://arxiv.org/html/2503.18738v2",
  "code_url": "https://roboengine.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.22
 },
 {
  "id": "2503.16396",
  "slug": "sv4d-2-0-enhancing-spatio-temporal-consistency-in-multi-view-video-dif",
  "title": "SV4D 2.0: Enhancing Spatio-Temporal Consistency in Multi-View Video Diffusion for High-Quality 4D Generation",
  "abstract": "We present Stable Video 4D 2.0 (SV4D 2.0), a multi-view video diffusion model for dynamic 3D asset generation. Compared to its predecessor SV4D, SV4D 2.0 is more robust to occlusions and large motion, generalizes better to real-world videos, and produces higher-quality outputs in terms of detail sharpness and spatio-temporal consistency. We achieve this by introducing key improvements in multiple aspects: 1) network architecture: eliminating the dependency of reference multi-views and designing blending mechanism for 3D and frame attention, 2) data: enhancing quality and quantity of training data, 3) training strategy: adopting progressive 3D-4D training for better generalization, and 4) 4D optimization: handling 3D inconsistency and large motion via 2-stage refinement and progressive frame sampling. Extensive experiments demonstrate significant performance gain by SV4D 2.0 both visually and quantitatively, achieving better detail (-14\\% LPIPS) and 4D consistency (-44\\% FV4D) in novel-view video synthesis and 4D optimization (-12\\% LPIPS and -24\\% FV4D) compared to SV4D. Project page: https://sv4d20.github.io.",
  "published": "2025-03-20",
  "updated": "2025-03-25",
  "year": "2025",
  "authors": [
   "Chun-Han Yao",
   "Yiming Xie",
   "Vikram Voleti",
   "Huaizu Jiang",
   "Varun Jampani"
  ],
  "author_count": 5,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "ICCV",
  "venue_source": "semantic-scholar",
  "citations": 55,
  "influential_citations": 12,
  "tldr": "SV4D 2.0 is more robust to occlusions and large motion, generalizes better to realworld videos, and produces higher-quality outputs in terms of detail sharpness and spatio-temporal consistency.",
  "doi": "10.1109/ICCV51701.2025.01231",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "C. Yao",
    "id": "2292210811",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Yiming Xie",
    "id": "46268911",
    "h_index": 13,
    "papers": 25
   },
   {
    "name": "Vikram S. Voleti",
    "id": "2961618",
    "h_index": 17,
    "papers": 45
   },
   {
    "name": "Huaizu Jiang",
    "id": "2254230515",
    "h_index": 11,
    "papers": 18
   },
   {
    "name": "Varun Jampani",
    "id": "2131639924",
    "h_index": 34,
    "papers": 88
   }
  ],
  "comment": "Project page: https://sv4d20.github.io/",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2503.16396v3",
  "pdf_url": "https://arxiv.org/pdf/2503.16396v3",
  "html_url": "https://arxiv.org/html/2503.16396v3",
  "code_url": "https://sv4d20.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.25
 },
 {
  "id": "2503.14734",
  "slug": "gr00t-n1-an-open-foundation-model-for-generalist-humanoid-robots",
  "title": "GR00T N1: An Open Foundation Model for Generalist Humanoid Robots",
  "abstract": "General-purpose robots need a versatile body and an intelligent mind. Recent advancements in humanoid robots have shown great promise as a hardware platform for building generalist autonomy in the human world. A robot foundation model, trained on massive and diverse data sources, is essential for enabling the robots to reason about novel situations, robustly handle real-world variability, and rapidly learn new tasks. To this end, we introduce GR00T N1, an open foundation model for humanoid robots. GR00T N1 is a Vision-Language-Action (VLA) model with a dual-system architecture. The vision-language module (System 2) interprets the environment through vision and language instructions. The subsequent diffusion transformer module (System 1) generates fluid motor actions in real time. Both modules are tightly coupled and jointly trained end-to-end. We train GR00T N1 with a heterogeneous mixture of real-robot trajectories, human videos, and synthetically generated datasets. We show that our generalist robot model GR00T N1 outperforms the state-of-the-art imitation learning baselines on standard simulation benchmarks across multiple robot embodiments. Furthermore, we deploy our model on the Fourier GR-1 humanoid robot for language-conditioned bimanual manipulation tasks, achieving strong performance with high data efficiency.",
  "published": "2025-03-18",
  "updated": "2025-03-27",
  "year": "2025",
  "authors": [
   " NVIDIA",
   " :",
   "Johan Bjorck",
   "Fernando Casta\u00f1eda",
   "Nikita Cherniadev",
   "Xingye Da",
   "Runyu Ding",
   "Linxi \"Jim\" Fan",
   "Yu Fang",
   "Dieter Fox",
   "Fengyuan Hu",
   "Spencer Huang",
   "Joel Jang",
   "Zhenyu Jiang",
   "Jan Kautz",
   "Kaushil Kundalia",
   "Lawrence Lao",
   "Zhiqi Li",
   "Zongyu Lin",
   "Kevin Lin",
   "Guilin Liu",
   "Edith Llontop",
   "Loic Magne",
   "Ajay Mandlekar",
   "Avnish Narayan",
   "Soroush Nasiriany",
   "Scott Reed",
   "You Liang Tan",
   "Guanzhi Wang",
   "Zu Wang",
   "Jing Wang",
   "Qi Wang",
   "Jiannan Xiang",
   "Yuqi Xie",
   "Yinzhen Xu",
   "Zhenjia Xu",
   "Seonghyeon Ye",
   "Zhiding Yu",
   "Ao Zhang",
   "Hao Zhang",
   "Yizhou Zhao",
   "Ruijie Zheng",
   "Yuke Zhu"
  ],
  "author_count": 43,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1220,
  "influential_citations": 186,
  "tldr": "This work introduces GR00T N1, an open foundation model for humanoid robots that outperforms the state-of-the-art imitation learning baselines on standard simulation benchmarks across multiple robot embodiments and deploys the model on the Fourier GR-1 humanoid robot for language-conditioned bimanual manipulation tasks.",
  "doi": "10.48550/arXiv.2503.14734",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Nvidia",
    "id": "2276243439",
    "h_index": 11,
    "papers": 39
   },
   {
    "name": "Johan Bjorck",
    "id": "46278353",
    "h_index": 17,
    "papers": 31
   },
   {
    "name": "Fernando Casta\u00f1eda",
    "id": "2350859158",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Nikita Cherniadev",
    "id": "2350861671",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Xingye Da",
    "id": "2350863929",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Runyu Ding",
    "id": "2350936924",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "LinxiJimFan",
    "id": "2350861618",
    "h_index": 10,
    "papers": 15
   },
   {
    "name": "Yu Fang",
    "id": "2351241177",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Dieter Fox",
    "id": "2258436157",
    "h_index": 12,
    "papers": 14
   },
   {
    "name": "Fengyuan Hu",
    "id": "2352992614",
    "h_index": 8,
    "papers": 11
   },
   {
    "name": "Spencer Huang",
    "id": "2350990085",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "J. Jang",
    "id": "2333419032",
    "h_index": 11,
    "papers": 13
   },
   {
    "name": "Zhenyuan Jiang",
    "id": "2319430243",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Jan Kautz",
    "id": "2364684748",
    "h_index": 21,
    "papers": 30
   },
   {
    "name": "Kaushil Kundalia",
    "id": "1471804481",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Lawrence Lao",
    "id": "2350862410",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Zhiqi Li",
    "id": "2348789944",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Zongyu Lin",
    "id": "2350991916",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Kevin Lin",
    "id": "2315897190",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Guilin Liu",
    "id": "2269510103",
    "h_index": 11,
    "papers": 21
   },
   {
    "name": "Edith Llontop",
    "id": "2192605890",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Loic Magne",
    "id": "2350862986",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "A. Mandlekar",
    "id": "49686756",
    "h_index": 36,
    "papers": 67
   },
   {
    "name": "Avnish Narayan",
    "id": "2014184266",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Soroush Nasiriany",
    "id": "3457048",
    "h_index": 18,
    "papers": 24
   },
   {
    "name": "Scott Reed",
    "id": "2344615904",
    "h_index": 8,
    "papers": 9
   },
   {
    "name": "Y. Tan",
    "id": "2350869215",
    "h_index": 7,
    "papers": 9
   },
   {
    "name": "Guanzhi Wang",
    "id": "96374437",
    "h_index": 16,
    "papers": 22
   },
   {
    "name": "Zu Wang",
    "id": "2350859563",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Jing Wang",
    "id": "2350827994",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Qi Wang",
    "id": "2326830174",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Jiannan Xiang",
    "id": "2057477005",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Yuqi Xie",
    "id": "2218866691",
    "h_index": 11,
    "papers": 17
   },
   {
    "name": "Yinzhen Xu",
    "id": "2351665438",
    "h_index": 7,
    "papers": 8
   },
   {
    "name": "Zhen-Teng Xu",
    "id": "2318097278",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Seonghyeon Ye",
    "id": "2152111477",
    "h_index": 23,
    "papers": 30
   },
   {
    "name": "Zhiding Yu",
    "id": "2269841405",
    "h_index": 21,
    "papers": 38
   },
   {
    "name": "Ao Zhang",
    "id": "2350752882",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Hao Zhang",
    "id": "2316357156",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Yizhou Zhao",
    "id": "2350998025",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Ruijie Zheng",
    "id": "2345931905",
    "h_index": 9,
    "papers": 15
   },
   {
    "name": "Yuke Zhu",
    "id": "2253507326",
    "h_index": 16,
    "papers": 20
   }
  ],
  "comment": "Authors are listed alphabetically. Project leads are Linxi \"Jim\" Fan and Yuke Zhu. For more information, see https://developer.nvidia.com/isaac/gr00t",
  "topics": [
   "vla",
   "humanoids",
   "egocentric-data",
   "sim2real",
   "imitation-diffusion",
   "foundation-pretraining"
  ],
  "orgs": [
   "NVIDIA"
  ],
  "abs_url": "https://arxiv.org/abs/2503.14734v2",
  "pdf_url": "https://arxiv.org/pdf/2503.14734v2",
  "html_url": "https://arxiv.org/html/2503.14734v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.5
 },
 {
  "id": "2503.13441",
  "slug": "humanoid-policy-human-policy",
  "title": "Humanoid Policy ~ Human Policy",
  "abstract": "Training manipulation policies for humanoid robots with diverse data enhances their robustness and generalization across tasks and platforms. However, learning solely from robot demonstrations is labor-intensive, requiring expensive tele-operated data collection which is difficult to scale. This paper investigates a more scalable data source, egocentric human demonstrations, to serve as cross-embodiment training data for robot learning. We mitigate the embodiment gap between humanoids and humans from both the data and modeling perspectives. We collect an egocentric task-oriented dataset (PH2D) that is directly aligned with humanoid manipulation demonstrations. We then train a human-humanoid behavior policy, which we term Human Action Transformer (HAT). The state-action space of HAT is unified for both humans and humanoid robots and can be differentiably retargeted to robot actions. Co-trained with smaller-scale robot data, HAT directly models humanoid robots and humans as different embodiments without additional supervision. We show that human data improves both generalization and robustness of HAT with significantly better data collection efficiency. Code and data: https://human-as-robot.github.io/",
  "published": "2025-03-17",
  "updated": "2025-10-05",
  "year": "2025",
  "authors": [
   "Ri-Zhao Qiu",
   "Shiqi Yang",
   "Xuxin Cheng",
   "Chaitanya Chawla",
   "Jialong Li",
   "Tairan He",
   "Ge Yan",
   "David J. Yoon",
   "Ryan Hoque",
   "Lars Paulsen",
   "Ge Yang",
   "Jian Zhang",
   "Sha Yi",
   "Guanya Shi",
   "Xiaolong Wang"
  ],
  "author_count": 15,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 95,
  "influential_citations": 4,
  "tldr": "This paper investigates a more scalable data source, egocentric human demonstrations, to serve as cross-embodiment training data for robot learning and mitigate the embodiment gap between humanoids and humans from both the data and modeling perspectives.",
  "doi": "10.48550/arXiv.2503.13441",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ri-Zhao Qiu",
    "id": "2290904526",
    "h_index": 11,
    "papers": 19
   },
   {
    "name": "Shiqi Yang",
    "id": "2309666838",
    "h_index": 8,
    "papers": 9
   },
   {
    "name": "Xuxin Cheng",
    "id": "2287822264",
    "h_index": 8,
    "papers": 9
   },
   {
    "name": "Chaitanya Chawla",
    "id": "2337108209",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Jialong Li",
    "id": "2309196968",
    "h_index": 7,
    "papers": 7
   },
   {
    "name": "Tairan He",
    "id": "2055132189",
    "h_index": 20,
    "papers": 29
   },
   {
    "name": "Ge Yan",
    "id": "2253536756",
    "h_index": 7,
    "papers": 9
   },
   {
    "name": "David J. Yoon",
    "id": "2351774576",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Ryan Hoque",
    "id": "2335570611",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Lars Paulsen",
    "id": "2350752105",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Ge Yang",
    "id": "2288147740",
    "h_index": 9,
    "papers": 10
   },
   {
    "name": "Jian Zhang",
    "id": "2335574774",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Sha Yi",
    "id": "2316591556",
    "h_index": 5,
    "papers": 15
   },
   {
    "name": "Guanya Shi",
    "id": "2249759531",
    "h_index": 20,
    "papers": 31
   },
   {
    "name": "Xiaolong Wang",
    "id": "2294782536",
    "h_index": 12,
    "papers": 17
   }
  ],
  "comment": "Code and data: https://human-as-robot.github.io/",
  "topics": [
   "humanoids",
   "egocentric-data",
   "foundation-pretraining",
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2503.13441v3",
  "pdf_url": "https://arxiv.org/pdf/2503.13441v3",
  "html_url": "https://arxiv.org/html/2503.13441v3",
  "code_url": "https://human-as-robot.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.98
 },
 {
  "id": "2503.11423",
  "slug": "taste-rob-advancing-video-generation-of-task-oriented-hand-object-inte",
  "title": "TASTE-Rob: Advancing Video Generation of Task-Oriented Hand-Object Interaction for Generalizable Robotic Manipulation",
  "abstract": "We address key limitations in existing datasets and models for task-oriented hand-object interaction video generation, a critical approach of generating video demonstrations for robotic imitation learning. Current datasets, such as Ego4D, often suffer from inconsistent view perspectives and misaligned interactions, leading to reduced video quality and limiting their applicability for precise imitation learning tasks. Towards this end, we introduce TASTE-Rob -- a pioneering large-scale dataset of 100,856 ego-centric hand-object interaction videos. Each video is meticulously aligned with language instructions and recorded from a consistent camera viewpoint to ensure interaction clarity. By fine-tuning a Video Diffusion Model (VDM) on TASTE-Rob, we achieve realistic object interactions, though we observed occasional inconsistencies in hand grasping postures. To enhance realism, we introduce a three-stage pose-refinement pipeline that improves hand posture accuracy in generated videos. Our curated dataset, coupled with the specialized pose-refinement framework, provides notable performance gains in generating high-quality, task-oriented hand-object interaction videos, resulting in achieving superior generalizable robotic manipulation. The TASTE-Rob dataset is publicly available to foster further advancements in the field, TASTE-Rob dataset and source code will be made publicly available on our website https://taste-rob.github.io.",
  "published": "2025-03-14",
  "updated": "2025-06-06",
  "year": "2025",
  "authors": [
   "Hongxiang Zhao",
   "Xingchen Liu",
   "Mutian Xu",
   "Yiming Hao",
   "Weikai Chen",
   "Xiaoguang Han"
  ],
  "author_count": 6,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 36,
  "influential_citations": 4,
  "tldr": "This work introduces TASTE-Rob \u2014 a pioneering large-scale dataset of 100,856 ego-centric hand-object interaction videos, coupled with a three-stage pose-refinement pipeline that improves hand posture accuracy in generated videos and introduces a three-stage pose-refinement framework to enhance realism.",
  "doi": "10.1109/CVPR52734.2025.02578",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hongxia Zhao",
    "id": "2146229270",
    "h_index": 5,
    "papers": 42
   },
   {
    "name": "Xingchen Liu",
    "id": "2343693492",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Mutian Xu",
    "id": "2145273236",
    "h_index": 8,
    "papers": 13
   },
   {
    "name": "Yiming Hao",
    "id": "2351420709",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Weikai Chen",
    "id": "2109595033",
    "h_index": 13,
    "papers": 48
   },
   {
    "name": "Xiaoguang Han",
    "id": "2332475041",
    "h_index": 1,
    "papers": 4
   }
  ],
  "comment": "CVPR 2025; Project Page: https://taste-rob.github.io",
  "topics": [
   "dexterous-manipulation",
   "imitation-diffusion",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2503.11423v2",
  "pdf_url": "https://arxiv.org/pdf/2503.11423v2",
  "html_url": "https://arxiv.org/html/2503.11423v2",
  "code_url": "https://taste-rob.github.io",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.07
 },
 {
  "id": "2503.09642",
  "slug": "open-sora-2-0-training-a-commercial-level-video-generation-model-in-20",
  "title": "Open-Sora 2.0: Training a Commercial-Level Video Generation Model in $200k",
  "abstract": "Video generation models have achieved remarkable progress in the past year. The quality of AI video continues to improve, but at the cost of larger model size, increased data quantity, and greater demand for training compute. In this report, we present Open-Sora 2.0, a commercial-level video generation model trained for only $200k. With this model, we demonstrate that the cost of training a top-performing video generation model is highly controllable. We detail all techniques that contribute to this efficiency breakthrough, including data curation, model architecture, training strategy, and system optimization. According to human evaluation results and VBench scores, Open-Sora 2.0 is comparable to global leading video generation models including the open-source HunyuanVideo and the closed-source Runway Gen-3 Alpha. By making Open-Sora 2.0 fully open-source, we aim to democratize access to advanced video generation technology, fostering broader innovation and creativity in content creation. All resources are publicly available at: https://github.com/hpcaitech/Open-Sora.",
  "published": "2025-03-12",
  "updated": "2026-03-02",
  "year": "2025",
  "authors": [
   "Zangwei Zheng",
   "Xiangyu Peng",
   "Yuxuan Lou",
   "Chenhui Shen",
   "Tom Young",
   "Xinying Guo",
   "Binluo Wang",
   "Hang Xu",
   "Hongxin Liu",
   "Mingyan Jiang",
   "Wenjun Li",
   "Yuhui Wang",
   "Anbang Ye",
   "Gang Ren",
   "Qianran Ma",
   "Wanying Liang",
   "Xiang Lian",
   "Xiwen Wu",
   "Yuting Zhong",
   "Zhuangyan Li",
   "Chaoyu Gong",
   "Guojun Lei",
   "Leijun Cheng",
   "Limin Zhang",
   "Minghao Li",
   "Ruijie Zhang",
   "Silan Hu",
   "Shijie Huang",
   "Xiaokang Wang",
   "Yuanheng Zhao",
   "Yuqi Wang",
   "Ziang Wei",
   "Yang You"
  ],
  "author_count": 33,
  "categories": [
   "cs.GR",
   "cs.AI"
  ],
  "primary_category": "cs.GR",
  "venue": "",
  "venue_source": "",
  "citations": 163,
  "influential_citations": 22,
  "tldr": "This report presents Open-Sora 2.0, a commercial-level video generation model trained for only $200k, and demonstrates that the cost of training a top-performing video generation model is highly controllable.",
  "doi": "10.48550/arXiv.2503.09642",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xiangyu Peng",
    "id": "2256656781",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Zangwei Zheng",
    "id": "2109654065",
    "h_index": 14,
    "papers": 24
   },
   {
    "name": "Chenhui Shen",
    "id": "2152867918",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Tom Young",
    "id": "2269987910",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Xinying Guo",
    "id": "2183512577",
    "h_index": 7,
    "papers": 14
   },
   {
    "name": "Binluo Wang",
    "id": "2349946820",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Hang Xu",
    "id": "2367747099",
    "h_index": 5,
    "papers": 17
   },
   {
    "name": "Hongxin Liu",
    "id": "2337817537",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "M. Jiang",
    "id": "2367595564",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Wenjun Li",
    "id": "2293667534",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Yuhui Wang",
    "id": "2309656723",
    "h_index": 6,
    "papers": 14
   },
   {
    "name": "Anbang Ye",
    "id": "2340393726",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "G. Ren",
    "id": "2405979829",
    "h_index": 9,
    "papers": 34
   },
   {
    "name": "Qianran Ma",
    "id": "2341926040",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Wanying Liang",
    "id": "2336027162",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Xiangru Lian",
    "id": "2922996",
    "h_index": 19,
    "papers": 31
   },
   {
    "name": "Xiwen Wu",
    "id": "2267767379",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Yu Zhong",
    "id": "2197899723",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Zhuangyan Li",
    "id": "2351800689",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Chaoyu Gong",
    "id": "1492027607",
    "h_index": 10,
    "papers": 27
   },
   {
    "name": "Guojun Lei",
    "id": "2331321172",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Lei Cheng",
    "id": "2213709625",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Liming Zhang",
    "id": "2204116452",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Minghao Li",
    "id": "2329132409",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Ruijie Zhang",
    "id": "2381063226",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Silan Hu",
    "id": "2271367050",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Shijie Huang",
    "id": "2396529508",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Xiaokang Wang",
    "id": "2390658052",
    "h_index": 6,
    "papers": 16
   },
   {
    "name": "Yuanheng Zhao",
    "id": "2350140710",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Yuqi Wang",
    "id": "2300166309",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Ziang Wei",
    "id": "2328955818",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Yang You",
    "id": "2340393673",
    "h_index": 2,
    "papers": 2
   }
  ],
  "comment": "",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2503.09642v3",
  "pdf_url": "https://arxiv.org/pdf/2503.09642v3",
  "html_url": "https://arxiv.org/html/2503.09642v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.21
 },
 {
  "id": "2503.08548",
  "slug": "tla-tactile-language-action-model-for-contact-rich-manipulation",
  "title": "TLA: Tactile-Language-Action Model for Contact-Rich Manipulation",
  "abstract": "Significant progress has been made in vision-language models. However, language-conditioned robotic manipulation for contact-rich tasks remains underexplored, particularly in terms of tactile sensing. To address this gap, we introduce the Tactile-Language-Action (TLA) model, which effectively processes sequential tactile feedback via cross-modal language grounding to enable robust policy generation in contact-intensive scenarios. In addition, we construct a comprehensive dataset that contains 24k pairs of tactile action instruction data, customized for fingertip peg-in-hole assembly, providing essential resources for TLA training and evaluation. Our results show that TLA significantly outperforms traditional imitation learning methods (e.g., diffusion policy) in terms of effective action generation and action accuracy, while demonstrating strong generalization capabilities by achieving over 85\\% success rate on previously unseen assembly clearances and peg shapes. We publicly release all data and code in the hope of advancing research in language-conditioned tactile manipulation skill learning. Project website: https://sites.google.com/view/tactile-language-action/",
  "published": "2025-03-11",
  "updated": "2025-03-11",
  "year": "2025",
  "authors": [
   "Peng Hao",
   "Chaofan Zhang",
   "Dingzhe Li",
   "Xiaoge Cao",
   "Xiaoshuai Hao",
   "Shaowei Cui",
   "Shuo Wang"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "Robot Learning",
  "venue_source": "semantic-scholar",
  "citations": 71,
  "influential_citations": 3,
  "tldr": "The Tactile-Language-Action model is introduced, which effectively processes sequential tactile feedback via cross-modal language grounding to enable robust policy generation in contact-intensive scenarios and significantly outperforms traditional imitation learning methods in terms of effective action generation and action accuracy.",
  "doi": "10.48550/arXiv.2503.08548",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Peng Hao",
    "id": "2298907411",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Chaofan Zhang",
    "id": "2256775583",
    "h_index": 7,
    "papers": 20
   },
   {
    "name": "Dingzhe Li",
    "id": "2229782696",
    "h_index": 9,
    "papers": 14
   },
   {
    "name": "Xiaoge Cao",
    "id": "66819025",
    "h_index": 6,
    "papers": 14
   },
   {
    "name": "Xiaoshuai Hao",
    "id": "2313750556",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Shaowei Cui",
    "id": "1853836031",
    "h_index": 17,
    "papers": 58
   },
   {
    "name": "Shuo Wang",
    "id": "2117011458",
    "h_index": 7,
    "papers": 25
   }
  ],
  "comment": "",
  "topics": [
   "tactile",
   "imitation-diffusion",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2503.08548v1",
  "pdf_url": "https://arxiv.org/pdf/2503.08548v1",
  "html_url": "https://arxiv.org/html/2503.08548v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.36
 },
 {
  "id": "2503.06800",
  "slug": "videophy-2-a-challenging-action-centric-physical-commonsense-evaluatio",
  "title": "VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation",
  "abstract": "Large-scale video generative models, capable of creating realistic videos of diverse visual concepts, are strong candidates for general-purpose physical world simulators. However, their adherence to physical commonsense across real-world actions remains unclear (e.g., playing tennis, backflip). Existing benchmarks suffer from limitations such as limited size, lack of human evaluation, sim-to-real gaps, and absence of fine-grained physical rule analysis. To address this, we introduce VideoPhy-2, an action-centric dataset for evaluating physical commonsense in generated videos. We curate 200 diverse actions and detailed prompts for video synthesis from modern generative models. We perform human evaluation that assesses semantic adherence, physical commonsense, and grounding of physical rules in the generated videos. Our findings reveal major shortcomings, with even the best model achieving only 22% joint performance (i.e., high semantic and physical commonsense adherence) on the hard subset of VideoPhy-2. We find that the models particularly struggle with conservation laws like mass and momentum. Finally, we also train VideoPhy-AutoEval, an automatic evaluator for fast, reliable assessment on our dataset. Overall, VideoPhy-2 serves as a rigorous benchmark, exposing critical gaps in video generative models and guiding future research in physically-grounded video generation. The data and code is available at https://videophy2.github.io/.",
  "published": "2025-03-09",
  "updated": "2025-03-09",
  "year": "2025",
  "authors": [
   "Hritik Bansal",
   "Clark Peng",
   "Yonatan Bitton",
   "Roman Goldenberg",
   "Aditya Grover",
   "Kai-Wei Chang"
  ],
  "author_count": 6,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 121,
  "influential_citations": 28,
  "tldr": "This work introduces VideoPhy-2, an action-centric dataset for evaluating physical commonsense in generated videos, and performs human evaluation that assesses semantic adherence, physical commonsense, and grounding of physical rules in the generated videos.",
  "doi": "10.48550/arXiv.2503.06800",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hritik Bansal",
    "id": "103404553",
    "h_index": 28,
    "papers": 57
   },
   {
    "name": "C. Peng",
    "id": "2349957353",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Yonatan Bitton",
    "id": "1938499056",
    "h_index": 21,
    "papers": 47
   },
   {
    "name": "R. Goldenberg",
    "id": "2349384207",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Aditya Grover",
    "id": "2263888488",
    "h_index": 10,
    "papers": 13
   },
   {
    "name": "Kai-Wei Chang",
    "id": "2256646491",
    "h_index": 12,
    "papers": 17
   }
  ],
  "comment": "41 pages, 33 Figures",
  "topics": [
   "sim2real",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2503.06800v1",
  "pdf_url": "https://arxiv.org/pdf/2503.06800v1",
  "html_url": "https://arxiv.org/html/2503.06800v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.09
 },
 {
  "id": "2503.06669",
  "slug": "agibot-world-colosseo-a-large-scale-manipulation-platform-for-scalable",
  "title": "AgiBot World Colosseo: A Large-scale Manipulation Platform for Scalable and Intelligent Embodied Systems",
  "abstract": "We explore how scalable robot data can address real-world challenges for generalized robotic manipulation. Introducing AgiBot World, a large-scale platform comprising over 1 million trajectories across 217 tasks in five deployment scenarios, we achieve an order-of-magnitude increase in data scale compared to existing datasets. Accelerated by a standardized collection pipeline with human-in-the-loop verification, AgiBot World guarantees high-quality and diverse data distribution. It is extensible from grippers to dexterous hands and visuo-tactile sensors for fine-grained skill acquisition. Building on top of data, we introduce Genie Operator-1 (GO-1), a novel generalist policy that leverages latent action representations to maximize data utilization, demonstrating predictable performance scaling with increased data volume. Policies pre-trained on our dataset achieve an average performance improvement of 30% over those trained on Open X-Embodiment, both in in-domain and out-of-distribution scenarios. GO-1 exhibits exceptional capability in real-world dexterous and long-horizon tasks, achieving over 60% success rate on complex tasks and outperforming prior RDT approach by 32%. By open-sourcing the dataset, tools, and models, we aim to democratize access to large-scale, high-quality robot data, advancing the pursuit of scalable and general-purpose intelligence.",
  "published": "2025-03-09",
  "updated": "2025-08-04",
  "year": "2025",
  "authors": [
   " AgiBot-World-Contributors",
   "Qingwen Bu",
   "Jisong Cai",
   "Li Chen",
   "Xiuqi Cui",
   "Yan Ding",
   "Siyuan Feng",
   "Shenyuan Gao",
   "Xindong He",
   "Xuan Hu",
   "Xu Huang",
   "Shu Jiang",
   "Yuxin Jiang",
   "Cheng Jing",
   "Hongyang Li",
   "Jialu Li",
   "Chiming Liu",
   "Yi Liu",
   "Yuxiang Lu",
   "Jianlan Luo",
   "Ping Luo",
   "Yao Mu",
   "Yuehan Niu",
   "Yixuan Pan",
   "Jiangmiao Pang",
   "Yu Qiao",
   "Guanghui Ren",
   "Cheng Ruan",
   "Jiaqi Shan",
   "Yongjian Shen",
   "Chengshi Shi",
   "Mingkang Shi",
   "Modi Shi",
   "Chonghao Sima",
   "Jianheng Song",
   "Huijie Wang",
   "Wenhao Wang",
   "Dafeng Wei",
   "Chengen Xie",
   "Guo Xu",
   "Junchi Yan",
   "Cunbiao Yang",
   "Lei Yang",
   "Shukai Yang",
   "Maoqing Yao",
   "Jia Zeng",
   "Chi Zhang",
   "Qinglin Zhang",
   "Bin Zhao",
   "Chengyue Zhao",
   "Jiaqi Zhao",
   "Jianchao Zhu"
  ],
  "author_count": 52,
  "categories": [
   "cs.RO",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS",
  "venue_source": "semantic-scholar",
  "citations": 438,
  "influential_citations": 55,
  "tldr": "",
  "doi": "10.1109/IROS60139.2025.11247088",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "AgiBot-World-Contributors",
    "id": "2349375911",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Qingwen Bu",
    "id": "2290184536",
    "h_index": 13,
    "papers": 19
   },
   {
    "name": "Jisong Cai",
    "id": "2327001484",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Li Chen",
    "id": "2254272547",
    "h_index": 8,
    "papers": 13
   },
   {
    "name": "X. Cui",
    "id": "2235689081",
    "h_index": 6,
    "papers": 28
   },
   {
    "name": "Yan Ding",
    "id": "2316892567",
    "h_index": 11,
    "papers": 25
   },
   {
    "name": "Siyuan Feng",
    "id": "2350330740",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Shenyuan Gao",
    "id": "2350220976",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Xindong He",
    "id": "2349803110",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Xu Huang",
    "id": "2349370163",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Shu Jiang",
    "id": "2351073319",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Yuxin Jiang",
    "id": "2260839757",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Cheng Jing",
    "id": "2284084844",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Hongyang Li",
    "id": "2276316087",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Jialun Li",
    "id": "2311458203",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Chiming Liu",
    "id": "2349426158",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Yi Liu",
    "id": "2305371903",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Yuxiang Lu",
    "id": "2350049168",
    "h_index": 8,
    "papers": 31
   },
   {
    "name": "Jianlan Luo",
    "id": "2349439364",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Ping Luo",
    "id": "2262515628",
    "h_index": 14,
    "papers": 21
   },
   {
    "name": "Yao Mu",
    "id": "2301922237",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Yue Niu",
    "id": "2260691804",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Yixuan Pan",
    "id": "2349474156",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Jiangmiao Pang",
    "id": "2277447920",
    "h_index": 24,
    "papers": 61
   },
   {
    "name": "Yu Qiao",
    "id": "2290187706",
    "h_index": 6,
    "papers": 6
   },
   {
    "name": "Guanghui Ren",
    "id": "2338694069",
    "h_index": 13,
    "papers": 30
   },
   {
    "name": "Cheng-Xing Ruan",
    "id": "2299764770",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Jiaqi Shan",
    "id": "2358806358",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Yongjian Shen",
    "id": "2349389999",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Cheng Shi",
    "id": "2279315091",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Mi Shi",
    "id": "2051764754",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Modi Shi",
    "id": "2349410682",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Chonghao Sima",
    "id": "2144553163",
    "h_index": 15,
    "papers": 23
   },
   {
    "name": "Jia-Yi Song",
    "id": "2306357090",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Huijie Wang",
    "id": "2344788975",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Wenhao Wang",
    "id": "2260728905",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Dafeng Wei",
    "id": "2349556277",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Chengen Xie",
    "id": "2275812370",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Guofeng Xu",
    "id": "2115724391",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Junchi Yan",
    "id": "2325920309",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Cunbiao Yang",
    "id": "2349392472",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Lei Yang",
    "id": "2276187087",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Shukai Yang",
    "id": "47569533",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Maoqing Yao",
    "id": "2338695140",
    "h_index": 12,
    "papers": 22
   },
   {
    "name": "Jiansheng Zeng",
    "id": "2296001522",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Chi Zhang",
    "id": "2305001236",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Qingli Zhang",
    "id": "2258535380",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Bin Zhao",
    "id": "2295782011",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Chengyu Zhao",
    "id": "2340472051",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Jiaqi Zhao",
    "id": "2326448775",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Jianchao Zhu",
    "id": "2349662815",
    "h_index": 1,
    "papers": 5
   }
  ],
  "comment": "Project website: https://agibot-world.com/. Github repo: https://github.com/OpenDriveLab/AgiBot-World. The author list is ordered alphabetically by surname, with detailed contributions provided in the appendix",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "foundation-pretraining",
   "hri",
   "safety-eval"
  ],
  "orgs": [
   "AgiBot"
  ],
  "abs_url": "https://arxiv.org/abs/2503.06669v4",
  "pdf_url": "https://arxiv.org/pdf/2503.06669v4",
  "html_url": "https://arxiv.org/html/2503.06669v4",
  "code_url": "https://github.com/OpenDriveLab/AgiBot-World.",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.64
 },
 {
  "id": "2503.05652",
  "slug": "behavior-robot-suite-streamlining-real-world-whole-body-manipulation-f",
  "title": "BEHAVIOR Robot Suite: Streamlining Real-World Whole-Body Manipulation for Everyday Household Activities",
  "abstract": "Real-world household tasks present significant challenges for mobile manipulation robots. An analysis of existing robotics benchmarks reveals that successful task performance hinges on three key whole-body control capabilities: bimanual coordination, stable and precise navigation, and extensive end-effector reachability. Achieving these capabilities requires careful hardware design, but the resulting system complexity further complicates visuomotor policy learning. To address these challenges, we introduce the BEHAVIOR Robot Suite (BRS), a comprehensive framework for whole-body manipulation in diverse household tasks. Built on a bimanual, wheeled robot with a 4-DoF torso, BRS integrates a cost-effective whole-body teleoperation interface for data collection and a novel algorithm for learning whole-body visuomotor policies. We evaluate BRS on five challenging household tasks that not only emphasize the three core capabilities but also introduce additional complexities, such as long-range navigation, interaction with articulated and deformable objects, and manipulation in confined spaces. We believe that BRS's integrated robotic embodiment, data collection interface, and learning framework mark a significant step toward enabling real-world whole-body manipulation for everyday household tasks. BRS is open-sourced at https://behavior-robot-suite.github.io/",
  "published": "2025-03-07",
  "updated": "2025-08-24",
  "year": "2025",
  "authors": [
   "Yunfan Jiang",
   "Ruohan Zhang",
   "Josiah Wong",
   "Chen Wang",
   "Yanjie Ze",
   "Hang Yin",
   "Cem Gokmen",
   "Shuran Song",
   "Jiajun Wu",
   "Li Fei-Fei"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL 2025",
  "venue_source": "arxiv-comment",
  "citations": 55,
  "influential_citations": 4,
  "tldr": "This work introduces the BEHAVIOR Robot Suite (BRS), a comprehensive framework for whole-body manipulation in diverse household tasks that integrates a cost-effective whole-body teleoperation interface for data collection and a novel algorithm for learning whole-body visuomotor policies.",
  "doi": "10.48550/arXiv.2503.05652",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yunfan Jiang",
    "id": "2171112793",
    "h_index": 11,
    "papers": 16
   },
   {
    "name": "Ruohan Zhang",
    "id": "2344793490",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Josiah Wong",
    "id": "33808086",
    "h_index": 11,
    "papers": 13
   },
   {
    "name": "Chen Wang",
    "id": "2285193288",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Yanjie Ze",
    "id": "2325901084",
    "h_index": 10,
    "papers": 13
   },
   {
    "name": "Hang Yin",
    "id": "2292126834",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Cem Gokmen",
    "id": "46217329",
    "h_index": 10,
    "papers": 12
   },
   {
    "name": "Shuran Song",
    "id": "2336463656",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Jiajun Wu",
    "id": "2326444500",
    "h_index": 35,
    "papers": 214
   },
   {
    "name": "Fei-Fei Li",
    "id": "2253964624",
    "h_index": 15,
    "papers": 30
   }
  ],
  "comment": "9th Conference on Robot Learning (CoRL 2025), Seoul, Korea. Project website: https://behavior-robot-suite.github.io/",
  "topics": [
   "humanoids",
   "navigation",
   "data-teleop",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2503.05652v2",
  "pdf_url": "https://arxiv.org/pdf/2503.05652v2",
  "html_url": "https://arxiv.org/html/2503.05652v2",
  "code_url": "https://behavior-robot-suite.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.25
 },
 {
  "id": "2503.02881",
  "slug": "reactive-diffusion-policy-slow-fast-visual-tactile-policy-learning-for",
  "title": "Reactive Diffusion Policy: Slow-Fast Visual-Tactile Policy Learning for Contact-Rich Manipulation",
  "abstract": "Humans can accomplish complex contact-rich tasks using vision and touch, with highly reactive capabilities such as fast response to external changes and adaptive control of contact forces; however, this remains challenging for robots. Existing visual imitation learning (IL) approaches rely on action chunking to model complex behaviors, which lacks the ability to respond instantly to real-time tactile feedback during the chunk execution. Furthermore, most teleoperation systems struggle to provide fine-grained tactile / force feedback, which limits the range of tasks that can be performed. To address these challenges, we introduce TactAR, a low-cost teleoperation system that provides real-time tactile feedback through Augmented Reality (AR), along with Reactive Diffusion Policy (RDP), a novel slow-fast visual-tactile imitation learning algorithm for learning contact-rich manipulation skills. RDP employs a two-level hierarchy: (1) a slow latent diffusion policy for predicting high-level action chunks in latent space at low frequency, (2) a fast asymmetric tokenizer for closed-loop tactile feedback control at high frequency. This design enables both complex trajectory modeling and quick reactive behavior within a unified framework. Through extensive evaluation across three challenging contact-rich tasks, RDP significantly improves performance compared to state-of-the-art visual IL baselines. Furthermore, experiments show that RDP is applicable across different tactile / force sensors. Code and videos are available on https://reactive-diffusion-policy.github.io.",
  "published": "2025-03-04",
  "updated": "2025-04-23",
  "year": "2025",
  "authors": [
   "Han Xue",
   "Jieji Ren",
   "Wendi Chen",
   "Gu Zhang",
   "Yuan Fang",
   "Guoying Gu",
   "Huazhe Xu",
   "Cewu Lu"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "Robotics",
  "venue_source": "semantic-scholar",
  "citations": 190,
  "influential_citations": 13,
  "tldr": "TactAR, a low-cost teleoperation system that provides real-time tactile feedback through Augmented Reality (AR), along with Reactive Diffusion Policy (RDP), a novel slow-fast visual-tactile imitation learning algorithm for learning contact-rich manipulation skills are introduced.",
  "doi": "10.48550/arXiv.2503.02881",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Han Xue",
    "id": "2351107813",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Jieji Ren",
    "id": "2265565097",
    "h_index": 6,
    "papers": 28
   },
   {
    "name": "Wendi Chen",
    "id": "2326063487",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Gu Zhang",
    "id": "2291085400",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Yuan Fang",
    "id": "2326081245",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Guoying Gu",
    "id": "2238185319",
    "h_index": 14,
    "papers": 47
   },
   {
    "name": "Huazhe Xu",
    "id": "2386620187",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Cewu Lu",
    "id": "2264977082",
    "h_index": 18,
    "papers": 53
   }
  ],
  "comment": "Accepted to RSS 2025. Project page: https://reactive-diffusion-policy.github.io",
  "topics": [
   "tactile",
   "imitation-diffusion",
   "data-teleop",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2503.02881v3",
  "pdf_url": "https://arxiv.org/pdf/2503.02881v3",
  "html_url": "https://arxiv.org/html/2503.02881v3",
  "code_url": "https://reactive-diffusion-policy.github.io",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.78
 },
 {
  "id": "2503.00779",
  "slug": "phantom-training-robots-without-robots-using-only-human-videos",
  "title": "Phantom: Training Robots Without Robots Using Only Human Videos",
  "abstract": "Training general-purpose robots requires learning from large and diverse data sources. Current approaches rely heavily on teleoperated demonstrations which are difficult to scale. We present a scalable framework for training manipulation policies directly from human video demonstrations, requiring no robot data. Our method converts human demonstrations into robot-compatible observation-action pairs using hand pose estimation and visual data editing. We inpaint the human arm and overlay a rendered robot to align the visual domains. This enables zero-shot deployment on real hardware without any fine-tuning. We demonstrate strong success rates-up to 92%-on a range of tasks including deformable object manipulation, multi-object sweeping, and insertion. Our approach generalizes to novel environments and supports closed-loop execution. By demonstrating that effective policies can be trained using only human videos, our method broadens the path to scalable robot learning.",
  "published": "2025-03-02",
  "updated": "2026-05-28",
  "year": "2025",
  "authors": [
   "Marion Lepert",
   "Jiaying Fang",
   "Jeannette Bohg"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 93,
  "influential_citations": 8,
  "tldr": "This work presents a scalable framework for training manipulation policies directly from human video demonstrations, requiring no robot data, and converts human demonstrations into robot-compatible observation-action pairs using hand pose estimation and visual data editing.",
  "doi": "10.48550/arXiv.2503.00779",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Marion Lepert",
    "id": "10710717",
    "h_index": 8,
    "papers": 11
   },
   {
    "name": "Jiaying Fang",
    "id": "2344586943",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Jeannette Bohg",
    "id": "1775407",
    "h_index": 50,
    "papers": 161
   }
  ],
  "comment": "Project website at https://phantom-human-videos.github.io",
  "topics": [
   "egocentric-data"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2503.00779v2",
  "pdf_url": "https://arxiv.org/pdf/2503.00779v2",
  "html_url": "https://arxiv.org/html/2503.00779v2",
  "code_url": "https://phantom-human-videos.github.io",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.97
 },
 {
  "id": "2503.00200",
  "slug": "unified-video-action-model",
  "title": "Unified Video Action Model",
  "abstract": "A unified video and action model holds significant promise for robotics, where videos provide rich scene information for action prediction, and actions provide dynamics information for video prediction. However, effectively combining video generation and action prediction remains challenging, and current video generation-based methods struggle to match the performance of direct policy learning in action accuracy and inference speed. To bridge this gap, we introduce the Unified Video Action model (UVA), which jointly optimizes video and action predictions to achieve both high accuracy and efficient action inference. The key lies in learning a joint video-action latent representation and decoupling video-action decoding. The joint latent representation bridges the visual and action domains, effectively modeling the relationship between video and action sequences. Meanwhile, the decoupled decoding, powered by two lightweight diffusion heads, enables high-speed action inference by bypassing video generation during inference. Such a unified framework further enables versatile functionality through masked input training. By selectively masking actions or videos, a single model can tackle diverse tasks beyond policy learning, such as forward and inverse dynamics modeling and video generation. Via an extensive set of experiments, we demonstrate that UVA can serve as a general-purpose solution for a wide range of robotics tasks, such as policy learning, forward/inverse dynamics and video observation prediction, without compromising performance compared to methods tailored for specific applications. Results are best viewed on https://unified-video-action-model.github.io/.",
  "published": "2025-02-28",
  "updated": "2025-04-24",
  "year": "2025",
  "authors": [
   "Shuang Li",
   "Yihuai Gao",
   "Dorsa Sadigh",
   "Shuran Song"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "Robotics",
  "venue_source": "semantic-scholar",
  "citations": 192,
  "influential_citations": 17,
  "tldr": "The Unified Video Action model is introduced, which jointly optimizes video and action predictions to achieve both high accuracy and efficient action inference and can serve as a general-purpose solution for a wide range of robotics tasks, such as policy learning, forward/inverse dynamics and video observation prediction, without compromising performance.",
  "doi": "10.48550/arXiv.2503.00200",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Shuang Li",
    "id": "2348581010",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Yihuai Gao",
    "id": "2311944190",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Dorsa Sadigh",
    "id": "1779671",
    "h_index": 69,
    "papers": 226
   },
   {
    "name": "Shuran Song",
    "id": "2348261633",
    "h_index": 3,
    "papers": 4
   }
  ],
  "comment": "Project website: https://unified-video-action-model.github.io/",
  "topics": [
   "world-models",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2503.00200v3",
  "pdf_url": "https://arxiv.org/pdf/2503.00200v3",
  "html_url": "https://arxiv.org/html/2503.00200v3",
  "code_url": "https://unified-video-action-model.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.79
 },
 {
  "id": "2502.20391",
  "slug": "point-policy-unifying-observations-and-actions-with-key-points-for-rob",
  "title": "Point Policy: Unifying Observations and Actions with Key Points for Robot Manipulation",
  "abstract": "Building robotic agents capable of operating across diverse environments and object types remains a significant challenge, often requiring extensive data collection. This is particularly restrictive in robotics, where each data point must be physically executed in the real world. Consequently, there is a critical need for alternative data sources for robotics and frameworks that enable learning from such data. In this work, we present Point Policy, a new method for learning robot policies exclusively from offline human demonstration videos and without any teleoperation data. Point Policy leverages state-of-the-art vision models and policy architectures to translate human hand poses into robot poses while capturing object states through semantically meaningful key points. This approach yields a morphology-agnostic representation that facilitates effective policy learning. Our experiments on 8 real-world tasks demonstrate an overall 75% absolute improvement over prior works when evaluated in identical settings as training. Further, Point Policy exhibits a 74% gain across tasks for novel object instances and is robust to significant background clutter. Videos of the robot are best viewed at https://point-policy.github.io/.",
  "published": "2025-02-27",
  "updated": "2025-02-27",
  "year": "2025",
  "authors": [
   "Siddhant Haldar",
   "Lerrel Pinto"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 60,
  "influential_citations": 12,
  "tldr": "This work presents Point Policy, a new method for learning robot policies exclusively from offline human demonstration videos and without any teleoperation data, which yields a morphology-agnostic representation that facilitates effective policy learning.",
  "doi": "10.48550/arXiv.2502.20391",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Siddhant Haldar",
    "id": "51445278",
    "h_index": 13,
    "papers": 21
   },
   {
    "name": "Lerrel Pinto",
    "id": "2253567347",
    "h_index": 15,
    "papers": 25
   }
  ],
  "comment": "",
  "topics": [
   "egocentric-data",
   "data-teleop",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2502.20391v1",
  "pdf_url": "https://arxiv.org/pdf/2502.20391v1",
  "html_url": "https://arxiv.org/html/2502.20391v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.79
 },
 {
  "id": "2502.19638",
  "slug": "sensor-invariant-tactile-representation",
  "title": "Sensor-Invariant Tactile Representation",
  "abstract": "High-resolution tactile sensors have become critical for embodied perception and robotic manipulation. However, a key challenge in the field is the lack of transferability between sensors due to design and manufacturing variations, which result in significant differences in tactile signals. This limitation hinders the ability to transfer models or knowledge learned from one sensor to another. To address this, we introduce a novel method for extracting Sensor-Invariant Tactile Representations (SITR), enabling zero-shot transfer across optical tactile sensors. Our approach utilizes a transformer-based architecture trained on a diverse dataset of simulated sensor designs, allowing it to generalize to new sensors in the real world with minimal calibration. Experimental results demonstrate the method's effectiveness across various tactile sensing applications, facilitating data and model transferability for future advancements in the field.",
  "published": "2025-02-27",
  "updated": "2025-03-13",
  "year": "2025",
  "authors": [
   "Harsh Gupta",
   "Yuchen Mo",
   "Shengmiao Jin",
   "Wenzhen Yuan"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 15,
  "influential_citations": 2,
  "tldr": "This work introduces a novel method for extracting Sensor-Invariant Tactile Representations (SITR), enabling zero-shot transfer across optical tactile sensors, and demonstrates the method's effectiveness across various tactile sensing applications.",
  "doi": "10.48550/arXiv.2502.19638",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Harsh Gupta",
    "id": "2279928512",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Yuchen Mo",
    "id": "2343834053",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Shengmiao Jin",
    "id": "2309901723",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Wenzhen Yuan",
    "id": "2344216714",
    "h_index": 3,
    "papers": 6
   }
  ],
  "comment": "Accepted to ICLR'25. Project webpage: https://hgupt3.github.io/sitr/",
  "topics": [
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2502.19638v2",
  "pdf_url": "https://arxiv.org/pdf/2502.19638v2",
  "html_url": "https://arxiv.org/html/2502.19638v2",
  "code_url": "https://hgupt3.github.io/sitr/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.7
 },
 {
  "id": "2502.17432",
  "slug": "factr-force-attending-curriculum-training-for-contact-rich-policy-lear",
  "title": "FACTR: Force-Attending Curriculum Training for Contact-Rich Policy Learning",
  "abstract": "Many contact-rich tasks humans perform, such as box pickup or rolling dough, rely on force feedback for reliable execution. However, this force information, which is readily available in most robot arms, is not commonly used in teleoperation and policy learning. Consequently, robot behavior is often limited to quasi-static kinematic tasks that do not require intricate force-feedback. In this paper, we first present a low-cost, intuitive, bilateral teleoperation setup that relays external forces of the follower arm back to the teacher arm, facilitating data collection for complex, contact-rich tasks. We then introduce FACTR, a policy learning method that employs a curriculum which corrupts the visual input with decreasing intensity throughout training. The curriculum prevents our transformer-based policy from over-fitting to the visual input and guides the policy to properly attend to the force modality. We demonstrate that by fully utilizing the force information, our method significantly improves generalization to unseen objects by 43\\% compared to baseline approaches without a curriculum. Video results, codebases, and instructions at https://jasonjzliu.com/factr/",
  "published": "2025-02-24",
  "updated": "2025-04-24",
  "year": "2025",
  "authors": [
   "Jason Jingzhou Liu",
   "Yulong Li",
   "Kenneth Shaw",
   "Tony Tao",
   "Ruslan Salakhutdinov",
   "Deepak Pathak"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "Robotics",
  "venue_source": "semantic-scholar",
  "citations": 55,
  "influential_citations": 2,
  "tldr": "A low-cost, intuitive, bilateral teleoperation setup that relays external forces of the follower arm back to the teacher arm, facilitating data collection for complex, contact-rich tasks and introduces FACTR, a policy learning method that employs a curriculum which corrupts the visual input with decreasing intensity throughout training.",
  "doi": "10.48550/arXiv.2502.17432",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "J. Liu",
    "id": "2346996723",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Yulong Li",
    "id": "2331686482",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Kenneth Shaw",
    "id": "2263541750",
    "h_index": 10,
    "papers": 18
   },
   {
    "name": "Tony Tao",
    "id": "2346975492",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Ruslan Salakhutdinov",
    "id": "2256996419",
    "h_index": 9,
    "papers": 24
   },
   {
    "name": "Deepak Pathak",
    "id": "2269734979",
    "h_index": 11,
    "papers": 17
   }
  ],
  "comment": "Video results, codebases, and instructions: https://jasonjzliu.com/factr/",
  "topics": [
   "tactile",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2502.17432v2",
  "pdf_url": "https://arxiv.org/pdf/2502.17432v2",
  "html_url": "https://arxiv.org/html/2502.17432v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.25
 },
 {
  "id": "2502.16932",
  "slug": "demogen-synthetic-demonstration-generation-for-data-efficient-visuomot",
  "title": "DemoGen: Synthetic Demonstration Generation for Data-Efficient Visuomotor Policy Learning",
  "abstract": "Visuomotor policies have shown great promise in robotic manipulation but often require substantial amounts of human-collected data for effective performance. A key reason underlying the data demands is their limited spatial generalization capability, which necessitates extensive data collection across different object configurations. In this work, we present DemoGen, a low-cost, fully synthetic approach for automatic demonstration generation. Using only one human-collected demonstration per task, DemoGen generates spatially augmented demonstrations by adapting the demonstrated action trajectory to novel object configurations. Visual observations are synthesized by leveraging 3D point clouds as the modality and rearranging the subjects in the scene via 3D editing. Empirically, DemoGen significantly enhances policy performance across a diverse range of real-world manipulation tasks, showing its applicability even in challenging scenarios involving deformable objects, dexterous hand end-effectors, and bimanual platforms. Furthermore, DemoGen can be extended to enable additional out-of-distribution capabilities, including disturbance resistance and obstacle avoidance.",
  "published": "2025-02-24",
  "updated": "2025-02-24",
  "year": "2025",
  "authors": [
   "Zhengrong Xue",
   "Shuying Deng",
   "Zhenyang Chen",
   "Yixuan Wang",
   "Zhecheng Yuan",
   "Huazhe Xu"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "Robotics",
  "venue_source": "semantic-scholar",
  "citations": 107,
  "influential_citations": 12,
  "tldr": "DemoGen significantly enhances policy performance across a diverse range of real-world manipulation tasks, showing its applicability even in challenging scenarios involving deformable objects, dexterous hand end-effectors, and bimanual platforms.",
  "doi": "10.48550/arXiv.2502.16932",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhengrong Xue",
    "id": "2239106154",
    "h_index": 7,
    "papers": 14
   },
   {
    "name": "Shuying Deng",
    "id": "2293829732",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Zhenyang Chen",
    "id": "2321516735",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Yixuan Wang",
    "id": "2253810575",
    "h_index": 8,
    "papers": 10
   },
   {
    "name": "Zhecheng Yuan",
    "id": "2156151359",
    "h_index": 15,
    "papers": 24
   },
   {
    "name": "Huazhe Xu",
    "id": "2239159959",
    "h_index": 8,
    "papers": 13
   }
  ],
  "comment": "Project website: https://demo-generation.github.io",
  "topics": [
   "dexterous-manipulation",
   "spatial-3d",
   "navigation",
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2502.16932v1",
  "pdf_url": "https://arxiv.org/pdf/2502.16932v1",
  "html_url": "https://arxiv.org/html/2502.16932v1",
  "code_url": "https://demo-generation.github.io",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.53
 },
 {
  "id": "2502.15672",
  "slug": "vavim-and-vavam-autonomous-driving-through-video-generative-modeling",
  "title": "VaViM and VaVAM: Autonomous Driving through Video Generative Modeling",
  "abstract": "We explore the potential of large-scale generative video models for autonomous driving, introducing an open-source auto-regressive video model (VaViM) and its companion video-action model (VaVAM) to investigate how video pre-training transfers to real-world driving. VaViM is a simple auto-regressive video model that predicts frames using spatio-temporal token sequences. We show that it captures the semantics and dynamics of driving scenes. VaVAM, the video-action model, leverages the learned representations of VaViM to generate driving trajectories through imitation learning. Together, the models form a complete perception-to-action pipeline. We evaluate our models in open- and closed-loop driving scenarios, revealing that video-based pre-training holds promise for autonomous driving. Key insights include the semantic richness of the learned representations, the benefits of scaling for video synthesis, and the complex relationship between model size, data, and safety metrics in closed-loop evaluations. We release code and model weights at https://github.com/valeoai/VideoActionModel",
  "published": "2025-02-21",
  "updated": "2025-02-21",
  "year": "2025",
  "authors": [
   "Florent Bartoccioni",
   "Elias Ramzi",
   "Victor Besnier",
   "Shashanka Venkataramanan",
   "Tuan-Hung Vu",
   "Yihong Xu",
   "Loick Chambon",
   "Spyros Gidaris",
   "Serkan Odabas",
   "David Hurych",
   "Renaud Marlet",
   "Alexandre Boulch",
   "Mickael Chen",
   "\u00c9loi Zablocki",
   "Andrei Bursuc",
   "Eduardo Valle",
   "Matthieu Cord"
  ],
  "author_count": 17,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 30,
  "influential_citations": 2,
  "tldr": "This work introduces an open-source auto-regressive video model (VaViM) and its companion video-action model (VaVAM) to investigate how video pre-training transfers to real-world driving, revealing that video-based pre-training holds promise for autonomous driving.",
  "doi": "10.48550/arXiv.2502.15672",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Florent Bartoccioni",
    "id": "2125912167",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Elias Ramzi",
    "id": "2321572133",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Victor Besnier",
    "id": "1400349847",
    "h_index": 8,
    "papers": 11
   },
   {
    "name": "S. Venkataramanan",
    "id": "39863130",
    "h_index": 11,
    "papers": 23
   },
   {
    "name": "Tuan-Hung Vu",
    "id": "2320155225",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Yihong Xu",
    "id": "2257442712",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Lo\u00efck Chambon",
    "id": "2241316931",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Spyros Gidaris",
    "id": "2475428",
    "h_index": 23,
    "papers": 44
   },
   {
    "name": "Serkan Odabas",
    "id": "2346834663",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "David Hurych",
    "id": "2687885",
    "h_index": 10,
    "papers": 30
   },
   {
    "name": "Renaud Marlet",
    "id": "3250857",
    "h_index": 37,
    "papers": 145
   },
   {
    "name": "Alexandre Boulch",
    "id": "2300845",
    "h_index": 29,
    "papers": 83
   },
   {
    "name": "Mickael Chen",
    "id": "2335122090",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "'Eloi Zablocki",
    "id": "2125901086",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Andrei Bursuc",
    "id": "3056236",
    "h_index": 21,
    "papers": 46
   },
   {
    "name": "Eduardo Valle",
    "id": "2322444250",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Matthieu Cord",
    "id": "2238525173",
    "h_index": 12,
    "papers": 34
   }
  ],
  "comment": "Code and model: https://github.com/valeoai/VideoActionModel, project page: https://valeoai.github.io/vavim-vavam/",
  "topics": [
   "imitation-diffusion",
   "navigation",
   "foundation-pretraining",
   "video-generation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2502.15672v1",
  "pdf_url": "https://arxiv.org/pdf/2502.15672v1",
  "html_url": "https://arxiv.org/html/2502.15672v1",
  "code_url": "https://github.com/valeoai/VideoActionModel",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.49
 },
 {
  "id": "2502.14819",
  "slug": "learning-from-reward-free-offline-data-a-case-for-planning-with-latent",
  "title": "Learning from Reward-Free Offline Data: A Case for Planning with Latent Dynamics Models",
  "abstract": "A long-standing goal in AI is to develop agents capable of solving diverse tasks across a range of environments, including those never seen during training. Two dominant paradigms address this challenge: (i) reinforcement learning (RL), which learns policies via trial and error, and (ii) optimal control, which plans actions using a known or learned dynamics model. However, their comparative strengths in the offline setting - where agents must learn from reward-free trajectories - remain underexplored. In this work, we systematically evaluate RL and control-based methods on a suite of navigation tasks, using offline datasets of varying quality. On the RL side, we consider goal-conditioned and zero-shot methods. On the control side, we train a latent dynamics model using the Joint Embedding Predictive Architecture (JEPA) and employ it for planning. We investigate how factors such as data diversity, trajectory quality, and environment variability influence the performance of these approaches. Our results show that model-free RL benefits most from large amounts of high-quality data, whereas model-based planning generalizes better to unseen layouts and is more data-efficient, while achieving trajectory stitching performance comparable to leading model-free methods. Notably, planning with a latent dynamics model proves to be a strong approach for handling suboptimal offline data and adapting to diverse environments.",
  "published": "2025-02-20",
  "updated": "2025-10-29",
  "year": "2025",
  "authors": [
   "Vlad Sobal",
   "Wancong Zhang",
   "Kyunghyun Cho",
   "Randall Balestriero",
   "Tim G. J. Rudner",
   "Yann LeCun"
  ],
  "author_count": 6,
  "categories": [
   "cs.LG"
  ],
  "primary_category": "cs.LG",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 64,
  "influential_citations": 13,
  "tldr": "This work systematically evaluates RL and control-based methods on a suite of navigation tasks, using offline datasets of varying quality and shows that model-free RL benefits most from large amounts of high-quality data, whereas model-based planning generalizes better to unseen layouts and is more data-efficient, while achieving trajectory stitching performance comparable to leading model-free methods.",
  "doi": "10.48550/arXiv.2502.14819",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Vlad Sobal",
    "id": "2162736903",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Wancong Zhang",
    "id": "2108277713",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Kynghyun Cho",
    "id": "2346533186",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Randall Balestriero",
    "id": "2277741253",
    "h_index": 22,
    "papers": 53
   },
   {
    "name": "Tim G. J. Rudner",
    "id": "2107677984",
    "h_index": 22,
    "papers": 75
   },
   {
    "name": "Yann LeCun",
    "id": "2270469816",
    "h_index": 12,
    "papers": 46
   }
  ],
  "comment": "Project web page: https://latent-planning.github.io/",
  "topics": [
   "world-models",
   "rl-control",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2502.14819v4",
  "pdf_url": "https://arxiv.org/pdf/2502.14819v4",
  "html_url": "https://arxiv.org/html/2502.14819v4",
  "code_url": "https://latent-planning.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.31
 },
 {
  "id": "2502.14795",
  "slug": "humanoid-vla-towards-universal-humanoid-control-with-visual-integratio",
  "title": "Humanoid-VLA: Towards Universal Humanoid Control with Visual Integration",
  "abstract": "This paper addresses the limitations of current humanoid robot control frameworks, which primarily rely on reactive mechanisms and lack autonomous interaction capabilities due to data scarcity. We propose Humanoid-VLA, a novel framework that integrates language understanding, egocentric scene perception, and motion control, enabling universal humanoid control. Humanoid-VLA begins with language-motion pre-alignment using non-egocentric human motion datasets paired with textual descriptions, allowing the model to learn universal motion patterns and action semantics. We then incorporate egocentric visual context through a parameter efficient video-conditioned fine-tuning, enabling context-aware motion generation. Furthermore, we introduce a self-supervised data augmentation strategy that automatically generates pseudoannotations directly derived from motion data. This process converts raw motion sequences into informative question-answer pairs, facilitating the effective use of large-scale unlabeled video data. Built upon whole-body control architectures, extensive experiments show that Humanoid-VLA achieves object interaction and environment exploration tasks with enhanced contextual awareness, demonstrating a more human-like capacity for adaptive and intelligent engagement.",
  "published": "2025-02-20",
  "updated": "2025-02-21",
  "year": "2025",
  "authors": [
   "Pengxiang Ding",
   "Jianfei Ma",
   "Xinyang Tong",
   "Binghong Zou",
   "Xinxin Luo",
   "Yiguo Fan",
   "Ting Wang",
   "Hongchao Lu",
   "Panzhong Mo",
   "Jinxin Liu",
   "Yuefan Wang",
   "Huaicheng Zhou",
   "Wenshuo Feng",
   "Jiacheng Liu",
   "Siteng Huang",
   "Donglin Wang"
  ],
  "author_count": 16,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 53,
  "influential_citations": 3,
  "tldr": "Humanoid-VLA is proposed, a novel framework that integrates language understanding, egocentric scene perception, and motion control, enabling universal humanoid control, and introduces a self-supervised data augmentation strategy that automatically generates pseudoannotations directly derived from motion data.",
  "doi": "10.48550/arXiv.2502.14795",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Pengxiang Ding",
    "id": "2275186266",
    "h_index": 23,
    "papers": 61
   },
   {
    "name": "Jianfei Ma",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Xinyang Tong",
    "id": "2338977132",
    "h_index": 8,
    "papers": 10
   },
   {
    "name": "B. Zou",
    "id": "2246922045",
    "h_index": 6,
    "papers": 24
   },
   {
    "name": "Xin Luo",
    "id": "2370961672",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Yiguo Fan",
    "id": "2336736864",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Ting Wang",
    "id": "2347771371",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Hongchao Lu",
    "id": "2346429624",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Panzhong Mo",
    "id": "2346324922",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Jinxi Liu",
    "id": "2375077787",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Yuefan Wang",
    "id": "2368720436",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Huaicheng Zhou",
    "id": "2346855145",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Wenshuo Feng",
    "id": "2348191726",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Jiacheng Liu",
    "id": "2346433253",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Siteng Huang",
    "id": "122132048",
    "h_index": 20,
    "papers": 37
   },
   {
    "name": "Donglin Wang",
    "id": "2237951974",
    "h_index": 7,
    "papers": 12
   }
  ],
  "comment": "",
  "topics": [
   "vla",
   "humanoids",
   "egocentric-data",
   "navigation",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2502.14795v2",
  "pdf_url": "https://arxiv.org/pdf/2502.14795v2",
  "html_url": "https://arxiv.org/html/2502.14795v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.73
 },
 {
  "id": "2502.13923",
  "slug": "qwen2-5-vl-technical-report",
  "title": "Qwen2.5-VL Technical Report",
  "abstract": "We introduce Qwen2.5-VL, the latest flagship model of Qwen vision-language series, which demonstrates significant advancements in both foundational capabilities and innovative functionalities. Qwen2.5-VL achieves a major leap forward in understanding and interacting with the world through enhanced visual recognition, precise object localization, robust document parsing, and long-video comprehension. A standout feature of Qwen2.5-VL is its ability to localize objects using bounding boxes or points accurately. It provides robust structured data extraction from invoices, forms, and tables, as well as detailed analysis of charts, diagrams, and layouts. To handle complex inputs, Qwen2.5-VL introduces dynamic resolution processing and absolute time encoding, enabling it to process images of varying sizes and videos of extended durations (up to hours) with second-level event localization. This allows the model to natively perceive spatial scales and temporal dynamics without relying on traditional normalization techniques. By training a native dynamic-resolution Vision Transformer (ViT) from scratch and incorporating Window Attention, we reduce computational overhead while maintaining native resolution. As a result, Qwen2.5-VL excels not only in static image and document understanding but also as an interactive visual agent capable of reasoning, tool usage, and task execution in real-world scenarios such as operating computers and mobile devices. Qwen2.5-VL is available in three sizes, addressing diverse use cases from edge AI to high-performance computing. The flagship Qwen2.5-VL-72B model matches state-of-the-art models like GPT-4o and Claude 3.5 Sonnet, particularly excelling in document and diagram understanding. Additionally, Qwen2.5-VL maintains robust linguistic performance, preserving the core language competencies of the Qwen2.5 LLM.",
  "published": "2025-02-19",
  "updated": "2025-02-19",
  "year": "2025",
  "authors": [
   "Shuai Bai",
   "Keqin Chen",
   "Xuejing Liu",
   "Jialin Wang",
   "Wenbin Ge",
   "Sibo Song",
   "Kai Dang",
   "Peng Wang",
   "Shijie Wang",
   "Jun Tang",
   "Humen Zhong",
   "Yuanzhi Zhu",
   "Mingkun Yang",
   "Zhaohai Li",
   "Jianqiang Wan",
   "Pengfei Wang",
   "Wei Ding",
   "Zheren Fu",
   "Yiheng Xu",
   "Jiabo Ye",
   "Xi Zhang",
   "Tianbao Xie",
   "Zesen Cheng",
   "Hang Zhang",
   "Zhibo Yang",
   "Haiyang Xu",
   "Junyang Lin"
  ],
  "author_count": 27,
  "categories": [
   "cs.CV",
   "cs.CL"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 5649,
  "influential_citations": 1321,
  "tldr": "This work introduces Qwen2.5-VL, the latest flagship model of Qwen vision-language series, which demonstrates significant advancements in both foundational capabilities and innovative functionalities, and maintains robust linguistic performance, preserving the core language competencies of the Qwen2.5 LLM.",
  "doi": "10.48550/arXiv.2502.13923",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Shuai Bai",
    "id": "2247821453",
    "h_index": 20,
    "papers": 46
   },
   {
    "name": "Ke-qin Chen",
    "id": "2344244387",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Xuejing Liu",
    "id": "2321923438",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Jialin Wang",
    "id": "2182966132",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Wenbin Ge",
    "id": "2311391178",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Sibo Song",
    "id": "2527741",
    "h_index": 10,
    "papers": 11
   },
   {
    "name": "K. Dang",
    "id": "2247877609",
    "h_index": 18,
    "papers": 52
   },
   {
    "name": "Peng Wang",
    "id": "2316259834",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "Shijie Wang",
    "id": "2311456728",
    "h_index": 5,
    "papers": 16
   },
   {
    "name": "Jun Tang",
    "id": "2112534494",
    "h_index": 7,
    "papers": 12
   },
   {
    "name": "Humen Zhong",
    "id": "2063801171",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Yuanzhi Zhu",
    "id": "2312265028",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Mingkun Yang",
    "id": "2333657808",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Zhaohai Li",
    "id": "2317078397",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Jianqiang Wan",
    "id": "1733915174",
    "h_index": 10,
    "papers": 12
   },
   {
    "name": "Pengfei Wang",
    "id": "2329116815",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Wei Ding",
    "id": "2346266830",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Zheren Fu",
    "id": "2106681735",
    "h_index": 6,
    "papers": 25
   },
   {
    "name": "Yiheng Xu",
    "id": "3032611",
    "h_index": 19,
    "papers": 21
   },
   {
    "name": "Jiabo Ye",
    "id": "2153258288",
    "h_index": 20,
    "papers": 31
   },
   {
    "name": "Xi Zhang",
    "id": "2304527137",
    "h_index": 10,
    "papers": 21
   },
   {
    "name": "Tianbao Xie",
    "id": "2346173497",
    "h_index": 8,
    "papers": 13
   },
   {
    "name": "Zesen Cheng",
    "id": "79673589",
    "h_index": 16,
    "papers": 36
   },
   {
    "name": "Hang Zhang",
    "id": "2282383887",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Zhibo Yang",
    "id": "2109432908",
    "h_index": 26,
    "papers": 76
   },
   {
    "name": "Haiyang Xu",
    "id": "2341400186",
    "h_index": 12,
    "papers": 26
   },
   {
    "name": "Junyang Lin",
    "id": "2333476820",
    "h_index": 10,
    "papers": 21
   }
  ],
  "comment": "",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2502.13923v1",
  "pdf_url": "https://arxiv.org/pdf/2502.13923v1",
  "html_url": "https://arxiv.org/html/2502.13923v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.0
 },
 {
  "id": "2502.13013",
  "slug": "homie-humanoid-loco-manipulation-with-isomorphic-exoskeleton-cockpit",
  "title": "HOMIE: Humanoid Loco-Manipulation with Isomorphic Exoskeleton Cockpit",
  "abstract": "Generalizable humanoid loco-manipulation poses significant challenges, requiring coordinated whole-body control and precise, contact-rich object manipulation. To address this, this paper introduces HOMIE, a semi-autonomous teleoperation system that combines a reinforcement learning policy for body control mapped to a pedal, an isomorphic exoskeleton arm for arm control, and motion-sensing gloves for hand control, forming a unified cockpit to freely operate humanoids and establish a data flywheel. The policy incorporates novel designs, including an upper-body pose curriculum, a height-tracking reward, and symmetry utilization. These features enable the system to perform walking and squatting to specific heights while seamlessly adapting to arbitrary upper-body poses. The exoskeleton, by eliminating the reliance on inverse dynamics, delivers faster and more precise arm control. The gloves utilize Hall sensors instead of servos, allowing even compact devices to achieve 15 or more degrees of freedom and freely adapt to any model of dexterous hands. Compared to previous teleoperation systems, HOMIE stands out for its exceptional efficiency, completing tasks in half the time; its expanded working range, allowing users to freely reach high and low areas as well as interact with any objects; and its affordability, with a price of just $500. The system is fully open-source, demos and code can be found in our https://homietele.github.io/.",
  "published": "2025-02-18",
  "updated": "2025-04-28",
  "year": "2025",
  "authors": [
   "Qingwei Ben",
   "Feiyu Jia",
   "Jia Zeng",
   "Junting Dong",
   "Dahua Lin",
   "Jiangmiao Pang"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.HC"
  ],
  "primary_category": "cs.RO",
  "venue": "Robotics",
  "venue_source": "semantic-scholar",
  "citations": 157,
  "influential_citations": 9,
  "tldr": "HOMIE is a semi-autonomous teleoperation system that combines a reinforcement learning policy for body control mapped to a pedal, an isomorphic exoskeleton arm for arm control, and motion-sensing gloves for hand control, forming a unified cockpit to freely operate humanoids and establish a data flywheel.",
  "doi": "10.48550/arXiv.2502.13013",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Qingwei Ben",
    "id": "2293395502",
    "h_index": 11,
    "papers": 16
   },
   {
    "name": "Feiyu Jia",
    "id": "2345925355",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Jia Zeng",
    "id": "2337356727",
    "h_index": 10,
    "papers": 24
   },
   {
    "name": "Junting Dong",
    "id": "2273884693",
    "h_index": 9,
    "papers": 16
   },
   {
    "name": "Dahua Lin",
    "id": "2237091231",
    "h_index": 16,
    "papers": 31
   },
   {
    "name": "Jiangmiao Pang",
    "id": "2277447920",
    "h_index": 24,
    "papers": 61
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "humanoids",
   "tactile",
   "rl-control",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2502.13013v2",
  "pdf_url": "https://arxiv.org/pdf/2502.13013v2",
  "html_url": "https://arxiv.org/html/2502.13013v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.7
 },
 {
  "id": "2502.11831",
  "slug": "intuitive-physics-understanding-emerges-from-self-supervised-pretraini",
  "title": "Intuitive physics understanding emerges from self-supervised pretraining on natural videos",
  "abstract": "We investigate the emergence of intuitive physics understanding in general-purpose deep neural network models trained to predict masked regions in natural videos. Leveraging the violation-of-expectation framework, we find that video prediction models trained to predict outcomes in a learned representation space demonstrate an understanding of various intuitive physics properties, such as object permanence and shape consistency. In contrast, video prediction in pixel space and multimodal large language models, which reason through text, achieve performance closer to chance. Our comparisons of these architectures reveal that jointly learning an abstract representation space while predicting missing parts of sensory input, akin to predictive coding, is sufficient to acquire an understanding of intuitive physics, and that even models trained on one week of unique video achieve above chance performance. This challenges the idea that core knowledge -- a set of innate systems to help understand the world -- needs to be hardwired to develop an understanding of intuitive physics.",
  "published": "2025-02-17",
  "updated": "2025-02-17",
  "year": "2025",
  "authors": [
   "Quentin Garrido",
   "Nicolas Ballas",
   "Mahmoud Assran",
   "Adrien Bardes",
   "Laurent Najman",
   "Michael Rabbat",
   "Emmanuel Dupoux",
   "Yann LeCun"
  ],
  "author_count": 8,
  "categories": [
   "cs.CV",
   "cs.AI"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 71,
  "influential_citations": 4,
  "tldr": "This work finds that video prediction models trained to predict outcomes in a learned representation space demonstrate an understanding of various intuitive physics properties, such as object permanence and shape consistency, in contrast to video prediction in pixel space and multimodal large language models, which reason through text.",
  "doi": "10.48550/arXiv.2502.11831",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Q. Garrido",
    "id": "2048163343",
    "h_index": 16,
    "papers": 22
   },
   {
    "name": "Nicolas Ballas",
    "id": "2289844757",
    "h_index": 13,
    "papers": 19
   },
   {
    "name": "Mahmoud Assran",
    "id": "38698856",
    "h_index": 15,
    "papers": 21
   },
   {
    "name": "Adrien Bardes",
    "id": "1453740540",
    "h_index": 16,
    "papers": 23
   },
   {
    "name": "Laurent Najman",
    "id": "2288435774",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Michael G. Rabbat",
    "id": "2066127975",
    "h_index": 25,
    "papers": 42
   },
   {
    "name": "Emmanuel Dupoux",
    "id": "2345856777",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Yann LeCun",
    "id": "2265899558",
    "h_index": 22,
    "papers": 47
   }
  ],
  "comment": "24 pages,14 figures, 5 tables",
  "topics": [
   "world-models",
   "foundation-pretraining",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2502.11831v1",
  "pdf_url": "https://arxiv.org/pdf/2502.11831v1",
  "html_url": "https://arxiv.org/html/2502.11831v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.86
 },
 {
  "id": "2502.12191",
  "slug": "anytouch-learning-unified-static-dynamic-representation-across-multipl",
  "title": "AnyTouch: Learning Unified Static-Dynamic Representation across Multiple Visuo-tactile Sensors",
  "abstract": "Visuo-tactile sensors aim to emulate human tactile perception, enabling robots to precisely understand and manipulate objects. Over time, numerous meticulously designed visuo-tactile sensors have been integrated into robotic systems, aiding in completing various tasks. However, the distinct data characteristics of these low-standardized visuo-tactile sensors hinder the establishment of a powerful tactile perception system. We consider that the key to addressing this issue lies in learning unified multi-sensor representations, thereby integrating the sensors and promoting tactile knowledge transfer between them. To achieve unified representation of this nature, we introduce TacQuad, an aligned multi-modal multi-sensor tactile dataset from four different visuo-tactile sensors, which enables the explicit integration of various sensors. Recognizing that humans perceive the physical environment by acquiring diverse tactile information such as texture and pressure changes, we further propose to learn unified multi-sensor representations from both static and dynamic perspectives. By integrating tactile images and videos, we present AnyTouch, a unified static-dynamic multi-sensor representation learning framework with a multi-level structure, aimed at both enhancing comprehensive perceptual abilities and enabling effective cross-sensor transfer. This multi-level architecture captures pixel-level details from tactile data via masked modeling and enhances perception and transferability by learning semantic-level sensor-agnostic features through multi-modal alignment and cross-sensor matching. We provide a comprehensive analysis of multi-sensor transferability, and validate our method on various datasets and in the real-world pouring task. Experimental results show that our method outperforms existing methods, exhibits outstanding static and dynamic perception capabilities across various sensors.",
  "published": "2025-02-15",
  "updated": "2025-04-01",
  "year": "2025",
  "authors": [
   "Ruoxuan Feng",
   "Jiangyu Hu",
   "Wenke Xia",
   "Tianci Gao",
   "Ao Shen",
   "Yuhao Sun",
   "Bin Fang",
   "Di Hu"
  ],
  "author_count": 8,
  "categories": [
   "cs.LG",
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.LG",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 68,
  "influential_citations": 8,
  "tldr": "AnyTouch is presented, a unified static-dynamic multi-sensor representation learning framework with a multi-level structure aimed at both enhancing comprehensive perceptual abilities and enabling effective cross-sensor transfer, and provides a comprehensive analysis of multi-sensor transferability.",
  "doi": "10.48550/arXiv.2502.12191",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ruoxuan Feng",
    "id": "2204645360",
    "h_index": 7,
    "papers": 12
   },
   {
    "name": "Jiangyu Hu",
    "id": "2346057168",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Wenke Xia",
    "id": "2201319923",
    "h_index": 8,
    "papers": 16
   },
   {
    "name": "Tianci Gao",
    "id": "2319604976",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Ao Shen",
    "id": "2282594075",
    "h_index": 7,
    "papers": 14
   },
   {
    "name": "Yuhao Sun",
    "id": "2189498375",
    "h_index": 10,
    "papers": 23
   },
   {
    "name": "Bin Fang",
    "id": "2345041551",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Di Hu",
    "id": "2291128131",
    "h_index": 5,
    "papers": 6
   }
  ],
  "comment": "Accepted by ICLR 2025",
  "topics": [
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2502.12191v3",
  "pdf_url": "https://arxiv.org/pdf/2502.12191v3",
  "html_url": "https://arxiv.org/html/2502.12191v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.34
 },
 {
  "id": "2502.09960",
  "slug": "global-local-interface-for-on-demand-teleoperation",
  "title": "Global-Local Interface for On-Demand Teleoperation",
  "abstract": "Teleoperation is a critical method for human-robot interface, holds significant potential for enabling robotic applications in industrial and unstructured environments. Existing teleoperation methods have distinct strengths and limitations in flexibility, range of workspace and precision. To fuse these advantages, we introduce the Global-Local (G-L) Teleoperation Interface. This interface decouples robotic teleoperation into global behavior, which ensures the robot motion range and intuitiveness, and local behavior, which enhances human operator's dexterity and capability for performing fine tasks. The G-L interface enables efficient teleoperation not only for conventional tasks like pick-and-place, but also for challenging fine manipulation and large-scale movements. Based on the G-L interface, we constructed a single-arm and a dual-arm teleoperation system with different remote control devices, then demonstrated tasks requiring large motion range, precise manipulation or dexterous end-effector control. Extensive experiments validated the user-friendliness, accuracy, and generalizability of the proposed interface.",
  "published": "2025-02-14",
  "updated": "2025-09-10",
  "year": "2025",
  "authors": [
   "Jianshu Zhou",
   "Boyuan Liang",
   "Junda Huang",
   "Ian Zhang",
   "Masayoshi Tomizuka"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 4,
  "influential_citations": 0,
  "tldr": "This work introduces the G-L interface, which decouples robotic teleoperation into global behavior, which ensures the robot motion range and intuitiveness, and local behavior, which enhances human operator's dexterity and capability for performing fine tasks.",
  "doi": "10.48550/arXiv.2502.09960",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jianshu Zhou",
    "id": "20476309",
    "h_index": 20,
    "papers": 53
   },
   {
    "name": "Boyuan Liang",
    "id": "2292198026",
    "h_index": 4,
    "papers": 14
   },
   {
    "name": "Junda Huang",
    "id": "2118227122",
    "h_index": 8,
    "papers": 26
   },
   {
    "name": "Ian Zhang",
    "id": "2345696512",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Pieter Abbeel",
    "id": "2265490900",
    "h_index": 8,
    "papers": 13
   },
   {
    "name": "Masayoshi Tomizuka",
    "id": "2261974717",
    "h_index": 7,
    "papers": 27
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2502.09960v3",
  "pdf_url": "https://arxiv.org/pdf/2502.09960v3",
  "html_url": "https://arxiv.org/html/2502.09960v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.7
 },
 {
  "id": "2502.09560",
  "slug": "embodiedbench-comprehensive-benchmarking-multi-modal-large-language-mo",
  "title": "EmbodiedBench: Comprehensive Benchmarking Multi-modal Large Language Models for Vision-Driven Embodied Agents",
  "abstract": "Leveraging Multi-modal Large Language Models (MLLMs) to create embodied agents offers a promising avenue for tackling real-world tasks. While language-centric embodied agents have garnered substantial attention, MLLM-based embodied agents remain underexplored due to the lack of comprehensive evaluation frameworks. To bridge this gap, we introduce EmbodiedBench, an extensive benchmark designed to evaluate vision-driven embodied agents. EmbodiedBench features: (1) a diverse set of 1,128 testing tasks across four environments, ranging from high-level semantic tasks (e.g., household) to low-level tasks involving atomic actions (e.g., navigation and manipulation); and (2) six meticulously curated subsets evaluating essential agent capabilities like commonsense reasoning, complex instruction understanding, spatial awareness, visual perception, and long-term planning. Through extensive experiments, we evaluated 24 leading proprietary and open-source MLLMs within EmbodiedBench. Our findings reveal that: MLLMs excel at high-level tasks but struggle with low-level manipulation, with the best model, GPT-4o, scoring only 28.9\\% on average. EmbodiedBench provides a multifaceted standardized evaluation platform that not only highlights existing challenges but also offers valuable insights to advance MLLM-based embodied agents. Our code and dataset are available at https://embodiedbench.github.io.",
  "published": "2025-02-13",
  "updated": "2025-06-05",
  "year": "2025",
  "authors": [
   "Rui Yang",
   "Hanyang Chen",
   "Junyu Zhang",
   "Mark Zhao",
   "Cheng Qian",
   "Kangrui Wang",
   "Qineng Wang",
   "Teja Venkat Koripella",
   "Marziyeh Movahedi",
   "Manling Li",
   "Heng Ji",
   "Huan Zhang",
   "Tong Zhang"
  ],
  "author_count": 13,
  "categories": [
   "cs.AI",
   "cs.CL",
   "cs.CV"
  ],
  "primary_category": "cs.AI",
  "venue": "ICML",
  "venue_source": "semantic-scholar",
  "citations": 213,
  "influential_citations": 20,
  "tldr": "EmbodiedBench is an extensive benchmark designed to evaluate vision-driven embodied agents and provides a multifaceted standardized evaluation platform that not only highlights existing challenges but also offers valuable insights to advance MLLM-based embodied agents.",
  "doi": "10.48550/arXiv.2502.09560",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Rui Yang",
    "id": "145094495",
    "h_index": 13,
    "papers": 28
   },
   {
    "name": "Hanyang Chen",
    "id": "2345238533",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Junyu Zhang",
    "id": "2329114461",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Mark Zhao",
    "id": "2345322659",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Cheng Qian",
    "id": "2082473972",
    "h_index": 18,
    "papers": 21
   },
   {
    "name": "Kangrui Wang",
    "id": "2325203435",
    "h_index": 10,
    "papers": 21
   },
   {
    "name": "Qineng Wang",
    "id": "2282995868",
    "h_index": 12,
    "papers": 19
   },
   {
    "name": "Teja Venkat Koripella",
    "id": "2345186506",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "M. Movahedi",
    "id": "32364837",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Manling Li",
    "id": "2118482058",
    "h_index": 16,
    "papers": 45
   },
   {
    "name": "Heng Ji",
    "id": "2331855378",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Huan Zhang",
    "id": "2346046247",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Tong Zhang",
    "id": "2345443315",
    "h_index": 3,
    "papers": 8
   }
  ],
  "comment": "Accepted to ICML 2025",
  "topics": [
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2502.09560v3",
  "pdf_url": "https://arxiv.org/pdf/2502.09560v3",
  "html_url": "https://arxiv.org/html/2502.09560v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.83
 },
 {
  "id": "2502.07730",
  "slug": "doglove-dexterous-manipulation-with-a-low-cost-open-source-haptic-forc",
  "title": "DOGlove: Dexterous Manipulation with a Low-Cost Open-Source Haptic Force Feedback Glove",
  "abstract": "Dexterous hand teleoperation plays a pivotal role in enabling robots to achieve human-level manipulation dexterity. However, current teleoperation systems often rely on expensive equipment and lack multi-modal sensory feedback, restricting human operators' ability to perceive object properties and perform complex manipulation tasks. To address these limitations, we present DOGlove, a low-cost, precise, and haptic force feedback glove system for teleoperation and manipulation. DoGlove can be assembled in hours at a cost under 600 USD. It features a customized joint structure for 21-DoF motion capture, a compact cable-driven torque transmission mechanism for 5-DoF multidirectional force feedback, and a linear resonate actuator for 5-DoF fingertip haptic feedback. Leveraging action and haptic force retargeting, DOGlove enables precise and immersive teleoperation of dexterous robotic hands, achieving high success rates in complex, contact-rich tasks. We further evaluate DOGlove in scenarios without visual feedback, demonstrating the critical role of haptic force feedback in task performance. In addition, we utilize the collected demonstrations to train imitation learning policies, highlighting the potential and effectiveness of DOGlove. DOGlove's hardware and software system will be fully open-sourced at https://do-glove.github.io/.",
  "published": "2025-02-11",
  "updated": "2025-02-11",
  "year": "2025",
  "authors": [
   "Han Zhang",
   "Songbo Hu",
   "Zhecheng Yuan",
   "Huazhe Xu"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "Robotics",
  "venue_source": "semantic-scholar",
  "citations": 61,
  "influential_citations": 1,
  "tldr": "DOGlove enables precise and immersive teleoperation of dexterous robotic hands, achieving high success rates in complex, contact-rich tasks, and is utilized to train imitation learning policies, highlighting the potential and effectiveness of DOGlove.",
  "doi": "10.48550/arXiv.2502.07730",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "H. Zhang",
    "id": "2183907239",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Songbo Hu",
    "id": "2344965720",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Zhecheng Yuan",
    "id": "2156151359",
    "h_index": 15,
    "papers": 24
   },
   {
    "name": "Huazhe Xu",
    "id": "2255379295",
    "h_index": 14,
    "papers": 21
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "egocentric-data",
   "tactile",
   "imitation-diffusion",
   "data-teleop",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2502.07730v1",
  "pdf_url": "https://arxiv.org/pdf/2502.07730v1",
  "html_url": "https://arxiv.org/html/2502.07730v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.29
 },
 {
  "id": "2502.05855",
  "slug": "dexvla-vision-language-model-with-plug-in-diffusion-expert-for-general",
  "title": "DexVLA: Vision-Language Model with Plug-In Diffusion Expert for General Robot Control",
  "abstract": "Enabling robots to perform diverse tasks across varied environments is a central challenge in robot learning. While vision-language-action (VLA) models have shown promise for generalizable robot skills, realizing their full potential requires addressing limitations in action representation and efficient training. Current VLA models often focus on scaling the vision-language model (VLM) component, while the action space representation remains a critical bottleneck. This paper introduces DexVLA, a novel framework designed to enhance the efficiency and generalization capabilities of VLAs for complex, long-horizon tasks across diverse robot embodiments. DexVLA features a novel diffusion-based action expert, scaled to one billion parameters, designed for cross-embodiment learning. A novel embodiment curriculum learning strategy facilitates efficient training: (1) pre-training the diffusion expert that is separable from the VLA on cross-embodiment data, (2) aligning the VLA model to specific embodiments, and (3) post-training for rapid adaptation to new tasks. We conduct comprehensive experiments across multiple embodiments, including single-arm, bimanual, and dexterous hand, demonstrating DexVLA's adaptability to challenging tasks without task-specific adaptation, its ability to learn dexterous skills on novel embodiments with limited data, and its capacity to complete complex, long-horizon tasks using only direct language prompting, such as laundry folding. In all settings, our method demonstrates superior performance compared to state-of-the-art models like Octo, OpenVLA, and Diffusion Policy.",
  "published": "2025-02-09",
  "updated": "2025-08-09",
  "year": "2025",
  "authors": [
   "Junjie Wen",
   "Yichen Zhu",
   "Jinming Li",
   "Zhibin Tang",
   "Chaomin Shen",
   "Feifei Feng"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL 2025",
  "venue_source": "arxiv-comment",
  "citations": 219,
  "influential_citations": 8,
  "tldr": "DexVLA is introduced, a novel framework designed to enhance the efficiency and generalization capabilities of VLAs for complex, long-horizon tasks across diverse robot embodiments, and demonstrates superior performance compared to state-of-the-art models like Octo, OpenVLA, and Diffusion Policy.",
  "doi": "10.48550/arXiv.2502.05855",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Junjie Wen",
    "id": "2278247833",
    "h_index": 17,
    "papers": 22
   },
   {
    "name": "Yichen Zhu",
    "id": "2275531481",
    "h_index": 21,
    "papers": 36
   },
   {
    "name": "Jinming Li",
    "id": "2278339388",
    "h_index": 14,
    "papers": 16
   },
   {
    "name": "Zhibin Tang",
    "id": "2333823781",
    "h_index": 7,
    "papers": 8
   },
   {
    "name": "Chaomin Shen",
    "id": "2335417053",
    "h_index": 9,
    "papers": 11
   },
   {
    "name": "Feifei Feng",
    "id": "2143808446",
    "h_index": 17,
    "papers": 28
   }
  ],
  "comment": "The webpage is at https://dex-vla.github.io/. DexVLA is accepted by CoRL 2025",
  "topics": [
   "vla",
   "dexterous-manipulation",
   "imitation-diffusion",
   "foundation-pretraining",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2502.05855v3",
  "pdf_url": "https://arxiv.org/pdf/2502.05855v3",
  "html_url": "https://arxiv.org/html/2502.05855v3",
  "code_url": "https://dex-vla.github.io/.",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.84
 },
 {
  "id": "2502.04307",
  "slug": "dexteritygen-foundation-controller-for-unprecedented-dexterity",
  "title": "DexterityGen: Foundation Controller for Unprecedented Dexterity",
  "abstract": "Teaching robots dexterous manipulation skills, such as tool use, presents a significant challenge. Current approaches can be broadly categorized into two strategies: human teleoperation (for imitation learning) and sim-to-real reinforcement learning. The first approach is difficult as it is hard for humans to produce safe and dexterous motions on a different embodiment without touch feedback. The second RL-based approach struggles with the domain gap and involves highly task-specific reward engineering on complex tasks. Our key insight is that RL is effective at learning low-level motion primitives, while humans excel at providing coarse motion commands for complex, long-horizon tasks. Therefore, the optimal solution might be a combination of both approaches. In this paper, we introduce DexterityGen (DexGen), which uses RL to pretrain large-scale dexterous motion primitives, such as in-hand rotation or translation. We then leverage this learned dataset to train a dexterous foundational controller. In the real world, we use human teleoperation as a prompt to the controller to produce highly dexterous behavior. We evaluate the effectiveness of DexGen in both simulation and real world, demonstrating that it is a general-purpose controller that can realize input dexterous manipulation commands and significantly improves stability by 10-100x measured as duration of holding objects across diverse tasks. Notably, with DexGen we demonstrate unprecedented dexterous skills including diverse object reorientation and dexterous tool use such as pen, syringe, and screwdriver for the first time.",
  "published": "2025-02-06",
  "updated": "2025-02-06",
  "year": "2025",
  "authors": [
   "Zhao-Heng Yin",
   "Changhao Wang",
   "Luis Pineda",
   "Francois Hogan",
   "Krishna Bodduluri",
   "Akash Sharma",
   "Patrick Lancaster",
   "Ishita Prasad",
   "Mrinal Kalakrishnan",
   "Jitendra Malik",
   "Mike Lambeta",
   "Tingfan Wu",
   "Pieter Abbeel",
   "Mustafa Mukadam"
  ],
  "author_count": 14,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "Robotics",
  "venue_source": "semantic-scholar",
  "citations": 55,
  "influential_citations": 1,
  "tldr": "",
  "doi": "10.48550/arXiv.2502.04307",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhao-Heng Yin",
    "id": "2290035529",
    "h_index": 7,
    "papers": 12
   },
   {
    "name": "Changhao Wang",
    "id": "2344075962",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Luis Pineda",
    "id": "2248222186",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Francois Hogan",
    "id": "2344089432",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Chaithanya Krishna Bodduluri",
    "id": "2328409907",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Akash Sharma",
    "id": "2109364933",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "Patrick Lancaster",
    "id": "2328413960",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Ishita Prasad",
    "id": "2344087152",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Mrinal Kalakrishnan",
    "id": "1729262",
    "h_index": 35,
    "papers": 58
   },
   {
    "name": "Jitendra Malik",
    "id": "2242761335",
    "h_index": 15,
    "papers": 32
   },
   {
    "name": "Mike Lambeta",
    "id": "3427691",
    "h_index": 13,
    "papers": 20
   },
   {
    "name": "Tingfan Wu",
    "id": "2254158966",
    "h_index": 12,
    "papers": 21
   },
   {
    "name": "Pieter Abbeel",
    "id": "2257003229",
    "h_index": 11,
    "papers": 17
   },
   {
    "name": "Mustafa Mukadam",
    "id": "2874057",
    "h_index": 30,
    "papers": 68
   }
  ],
  "comment": "Project: https://zhaohengyin.github.io/dexteritygen",
  "topics": [
   "dexterous-manipulation",
   "sim2real",
   "imitation-diffusion",
   "rl-control",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2502.04307v1",
  "pdf_url": "https://arxiv.org/pdf/2502.04307v1",
  "html_url": "https://arxiv.org/html/2502.04307v1",
  "code_url": "https://zhaohengyin.github.io/dexteritygen",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.25
 },
 {
  "id": "2502.04296",
  "slug": "learning-real-world-action-video-dynamics-with-heterogeneous-masked-au",
  "title": "Learning Real-World Action-Video Dynamics with Heterogeneous Masked Autoregression",
  "abstract": "We propose Heterogeneous Masked Autoregression (HMA) for modeling action-video dynamics to generate high-quality data and evaluation in scaling robot learning. Building interactive video world models and policies for robotics is difficult due to the challenge of handling diverse settings while maintaining computational efficiency to run in real time. HMA uses heterogeneous pre-training from observations and action sequences across different robotic embodiments, domains, and tasks. HMA uses masked autoregression to generate quantized or soft tokens for video predictions. \\ourshort achieves better visual fidelity and controllability than the previous robotic video generation models with 15 times faster speed in the real world. After post-training, this model can be used as a video simulator from low-level action inputs for evaluating policies and generating synthetic data. See this link https://liruiw.github.io/hma for more information.",
  "published": "2025-02-06",
  "updated": "2025-02-06",
  "year": "2025",
  "authors": [
   "Lirui Wang",
   "Kevin Zhao",
   "Chaoqi Liu",
   "Xinlei Chen"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 21,
  "influential_citations": 1,
  "tldr": "The proposed Heterogeneous Masked Autoregression (HMA) model achieves better visual fidelity and controllability than the previous robotic video generation models with 15 times faster speed in the real world.",
  "doi": "10.48550/arXiv.2502.04296",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Lirui Wang",
    "id": "2253973819",
    "h_index": 6,
    "papers": 6
   },
   {
    "name": "Kevin Zhao",
    "id": "2074109526",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Chaoqi Liu",
    "id": "2353314038",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Xinlei Chen",
    "id": "2281064954",
    "h_index": 8,
    "papers": 10
   }
  ],
  "comment": "Website: https://liruiw.github.io/hma/",
  "topics": [
   "world-models",
   "sim2real",
   "foundation-pretraining",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2502.04296v1",
  "pdf_url": "https://arxiv.org/pdf/2502.04296v1",
  "html_url": "https://arxiv.org/html/2502.04296v1",
  "code_url": "https://liruiw.github.io/hma/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.34
 },
 {
  "id": "2502.03444",
  "slug": "masked-autoencoders-are-effective-tokenizers-for-diffusion-models",
  "title": "Masked Autoencoders Are Effective Tokenizers for Diffusion Models",
  "abstract": "Recent advances in latent diffusion models have demonstrated their effectiveness for high-resolution image synthesis. However, the properties of the latent space from tokenizer for better learning and generation of diffusion models remain under-explored. Theoretically and empirically, we find that improved generation quality is closely tied to the latent distributions with better structure, such as the ones with fewer Gaussian Mixture modes and more discriminative features. Motivated by these insights, we propose MAETok, an autoencoder (AE) leveraging mask modeling to learn semantically rich latent space while maintaining reconstruction fidelity. Extensive experiments validate our analysis, demonstrating that the variational form of autoencoders is not necessary, and a discriminative latent space from AE alone enables state-of-the-art performance on ImageNet generation using only 128 tokens. MAETok achieves significant practical improvements, enabling a gFID of 1.69 with 76x faster training and 31x higher inference throughput for 512x512 generation. Our findings show that the structure of the latent space, rather than variational constraints, is crucial for effective diffusion models. Code and trained models are released.",
  "published": "2025-02-05",
  "updated": "2025-05-30",
  "year": "2025",
  "authors": [
   "Hao Chen",
   "Yujin Han",
   "Fangyi Chen",
   "Xiang Li",
   "Yidong Wang",
   "Jindong Wang",
   "Ze Wang",
   "Zicheng Liu",
   "Difan Zou",
   "Bhiksha Raj"
  ],
  "author_count": 10,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.CV",
  "venue": "ICML",
  "venue_source": "semantic-scholar",
  "citations": 89,
  "influential_citations": 16,
  "tldr": "MAETok is proposed, an autoencoder (AE) leveraging mask modeling to learn semantically rich latent space while maintaining reconstruction fidelity and shows that the structure of the latent space, rather than variational constraints, is crucial for effective diffusion models.",
  "doi": "10.48550/arXiv.2502.03444",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hao Chen",
    "id": "2307335951",
    "h_index": 9,
    "papers": 18
   },
   {
    "name": "Yujin Han",
    "id": "2297887026",
    "h_index": 6,
    "papers": 19
   },
   {
    "name": "Fangyi Chen",
    "id": "2335650319",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Xiang Li",
    "id": "2304463477",
    "h_index": 9,
    "papers": 14
   },
   {
    "name": "Yidong Wang",
    "id": "2343897801",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Jindong Wang",
    "id": "2333375516",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Ze Wang",
    "id": "2336047375",
    "h_index": 6,
    "papers": 21
   },
   {
    "name": "Zicheng Liu",
    "id": "2334872571",
    "h_index": 12,
    "papers": 38
   },
   {
    "name": "Difan Zou",
    "id": "2297772501",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Bhiksha Raj",
    "id": "2288787089",
    "h_index": 12,
    "papers": 67
   }
  ],
  "comment": "",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2502.03444v2",
  "pdf_url": "https://arxiv.org/pdf/2502.03444v2",
  "html_url": "https://arxiv.org/html/2502.03444v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.45
 },
 {
  "id": "2501.18982",
  "slug": "omniphysgs-3d-constitutive-gaussians-for-general-physics-based-dynamic",
  "title": "OmniPhysGS: 3D Constitutive Gaussians for General Physics-Based Dynamics Generation",
  "abstract": "Recently, significant advancements have been made in the reconstruction and generation of 3D assets, including static cases and those with physical interactions. To recover the physical properties of 3D assets, existing methods typically assume that all materials belong to a specific predefined category (e.g., elasticity). However, such assumptions ignore the complex composition of multiple heterogeneous objects in real scenarios and tend to render less physically plausible animation given a wider range of objects. We propose OmniPhysGS for synthesizing a physics-based 3D dynamic scene composed of more general objects. A key design of OmniPhysGS is treating each 3D asset as a collection of constitutive 3D Gaussians. For each Gaussian, its physical material is represented by an ensemble of 12 physical domain-expert sub-models (rubber, metal, honey, water, etc.), which greatly enhances the flexibility of the proposed model. In the implementation, we define a scene by user-specified prompts and supervise the estimation of material weighting factors via a pretrained video diffusion model. Comprehensive experiments demonstrate that OmniPhysGS achieves more general and realistic physical dynamics across a broader spectrum of materials, including elastic, viscoelastic, plastic, and fluid substances, as well as interactions between different materials. Our method surpasses existing methods by approximately 3% to 16% in metrics of visual quality and text alignment.",
  "published": "2025-01-31",
  "updated": "2025-01-31",
  "year": "2025",
  "authors": [
   "Yuchen Lin",
   "Chenguo Lin",
   "Jianjin Xu",
   "Yadong Mu"
  ],
  "author_count": 4,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 64,
  "influential_citations": 12,
  "tldr": "Comprehensive experiments demonstrate that OmniPhysGS achieves more general and realistic physical dynamics across a broader spectrum of materials, including elastic, viscoelastic, plastic, and fluid substances, as well as interactions between different materials.",
  "doi": "10.48550/arXiv.2501.18982",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yuchen Lin",
    "id": "2310656109",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Chenguo Lin",
    "id": "2283193527",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "Jianjin Xu",
    "id": "2352041965",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Yadong Mu",
    "id": "2283135775",
    "h_index": 7,
    "papers": 12
   }
  ],
  "comment": "Accepted to ICLR 2025; Project page: https://wgsxm.github.io/projects/omniphysgs/",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2501.18982v1",
  "pdf_url": "https://arxiv.org/pdf/2501.18982v1",
  "html_url": "https://arxiv.org/html/2501.18982v1",
  "code_url": "https://wgsxm.github.io/projects/omniphysgs/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.31
 },
 {
  "id": "2501.14400",
  "slug": "skil-semantic-keypoint-imitation-learning-for-generalizable-data-effic",
  "title": "SKIL: Semantic Keypoint Imitation Learning for Generalizable Data-efficient Manipulation",
  "abstract": "Real-world tasks such as garment manipulation and table rearrangement demand robots to perform generalizable, highly precise, and long-horizon actions. Although imitation learning has proven to be an effective approach for teaching robots new skills, large amounts of expert demonstration data are still indispensible for these complex tasks, resulting in high sample complexity and costly data collection. To address this, we propose Semantic Keypoint Imitation Learning (SKIL), a framework which automatically obtains semantic keypoints with the help of vision foundation models, and forms the descriptor of semantic keypoints that enables efficient imitation learning of complex robotic tasks with significantly lower sample complexity. In real-world experiments, SKIL doubles the performance of baseline methods in tasks such as picking a cup or mouse, while demonstrating exceptional robustness to variations in objects, environmental changes, and distractors. For long-horizon tasks like hanging a towel on a rack where previous methods fail completely, SKIL achieves a mean success rate of 70\\% with as few as 30 demonstrations. Furthermore, SKIL naturally supports cross-embodiment learning due to its semantic keypoints abstraction. Our experiments demonstrate that even human videos bring considerable improvement to the learning performance. All these results demonstrate the great success of SKIL in achieving data-efficient generalizable robotic learning. Visualizations and code are available at: https://skil-robotics.github.io/SKIL-robotics/.",
  "published": "2025-01-24",
  "updated": "2025-07-02",
  "year": "2025",
  "authors": [
   "Shengjie Wang",
   "Jiacheng You",
   "Yihang Hu",
   "Jiongye Li",
   "Yang Gao"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "Robotics",
  "venue_source": "semantic-scholar",
  "citations": 33,
  "influential_citations": 2,
  "tldr": "Semantic Keypoint Imitation Learning (SKIL), a framework which automatically obtains semantic keypoints with the help of vision foundation models, and forms the descriptor of semantic keypoints that enables efficient imitation learning of complex robotic tasks with significantly lower sample complexity, is proposed.",
  "doi": "10.48550/arXiv.2501.14400",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Shengjie Wang",
    "id": "2255485492",
    "h_index": 8,
    "papers": 17
   },
   {
    "name": "Jiacheng You",
    "id": "2289844086",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "Yihan Hu",
    "id": "2325677836",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Jiongye Li",
    "id": "2326816637",
    "h_index": 8,
    "papers": 16
   },
   {
    "name": "Yang Gao",
    "id": "2258569969",
    "h_index": 4,
    "papers": 10
   }
  ],
  "comment": "22 pages, 22 figures",
  "topics": [
   "egocentric-data",
   "imitation-diffusion",
   "foundation-pretraining",
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2501.14400v2",
  "pdf_url": "https://arxiv.org/pdf/2501.14400v2",
  "html_url": "https://arxiv.org/html/2501.14400v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.03
 },
 {
  "id": "2501.13918",
  "slug": "improving-video-generation-with-human-feedback",
  "title": "Improving Video Generation with Human Feedback",
  "abstract": "Video generation has achieved significant advances through rectified flow techniques, but issues like unsmooth motion and misalignment between videos and prompts persist. In this work, we develop a systematic pipeline that harnesses human feedback to mitigate these problems and refine the video generation model. Specifically, we begin by constructing a large-scale human preference dataset focused on modern video generation models, incorporating pairwise annotations across multi-dimensions. We then introduce VideoReward, a multi-dimensional video reward model, and examine how annotations and various design choices impact its rewarding efficacy. From a unified reinforcement learning perspective aimed at maximizing reward with KL regularization, we introduce three alignment algorithms for flow-based models. These include two training-time strategies: direct preference optimization for flow (Flow-DPO) and reward weighted regression for flow (Flow-RWR), and an inference-time technique, Flow-NRG, which applies reward guidance directly to noisy videos. Experimental results indicate that VideoReward significantly outperforms existing reward models, and Flow-DPO demonstrates superior performance compared to both Flow-RWR and supervised fine-tuning methods. Additionally, Flow-NRG lets users assign custom weights to multiple objectives during inference, meeting personalized video quality needs.",
  "published": "2025-01-23",
  "updated": "2025-10-27",
  "year": "2025",
  "authors": [
   "Jie Liu",
   "Gongye Liu",
   "Jiajun Liang",
   "Ziyang Yuan",
   "Xiaokun Liu",
   "Mingwu Zheng",
   "Xiele Wu",
   "Qiulin Wang",
   "Menghan Xia",
   "Xintao Wang",
   "Xiaohong Liu",
   "Fei Yang",
   "Pengfei Wan",
   "Di Zhang",
   "Kun Gai",
   "Yujiu Yang",
   "Wanli Ouyang"
  ],
  "author_count": 17,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.GR",
   "cs.LG"
  ],
  "primary_category": "cs.CV",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 248,
  "influential_citations": 46,
  "tldr": "This work builds a large-scale human preference dataset focused on modern video generation models, and introduces VideoReward, a multi-dimensional video reward model, and examines how annotations and various design choices impact its rewarding efficacy.",
  "doi": "10.48550/arXiv.2501.13918",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jie Liu",
    "id": "2285060791",
    "h_index": 9,
    "papers": 19
   },
   {
    "name": "Gongye Liu",
    "id": "2269171464",
    "h_index": 9,
    "papers": 18
   },
   {
    "name": "Jiajun Liang",
    "id": "2342646262",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Ziyang Yuan",
    "id": "2264188878",
    "h_index": 12,
    "papers": 20
   },
   {
    "name": "Xiaokun Liu",
    "id": "2341704358",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Mingwu Zheng",
    "id": "2146523688",
    "h_index": 11,
    "papers": 14
   },
   {
    "name": "Xiele Wu",
    "id": "2298414297",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Qiulin Wang",
    "id": "2296745026",
    "h_index": 8,
    "papers": 17
   },
   {
    "name": "Wenyu Qin",
    "id": "2341720942",
    "h_index": 5,
    "papers": 18
   },
   {
    "name": "Menghan Xia",
    "id": "2257035878",
    "h_index": 16,
    "papers": 33
   },
   {
    "name": "Xintao Wang",
    "id": "2305033532",
    "h_index": 21,
    "papers": 67
   },
   {
    "name": "Xiaohong Liu",
    "id": "2282544408",
    "h_index": 7,
    "papers": 18
   },
   {
    "name": "Fei Yang",
    "id": "2325842209",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Pengfei Wan",
    "id": "2276606835",
    "h_index": 27,
    "papers": 84
   },
   {
    "name": "Di Zhang",
    "id": "2332361648",
    "h_index": 18,
    "papers": 28
   },
   {
    "name": "Kun Gai",
    "id": "2238953242",
    "h_index": 17,
    "papers": 39
   },
   {
    "name": "Yujiu Yang",
    "id": "2283881403",
    "h_index": 28,
    "papers": 181
   },
   {
    "name": "Wanli Ouyang",
    "id": "2254269925",
    "h_index": 20,
    "papers": 28
   }
  ],
  "comment": "https://github.com/KwaiVGI/VideoAlign",
  "topics": [
   "rl-control",
   "data-teleop",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2501.13918v2",
  "pdf_url": "https://arxiv.org/pdf/2501.13918v2",
  "html_url": "https://arxiv.org/html/2501.13918v2",
  "code_url": "https://github.com/KwaiVGI/VideoAlign",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.9
 },
 {
  "id": "2501.10357",
  "slug": "zero-shot-monocular-scene-flow-estimation-in-the-wild",
  "title": "Zero-Shot Monocular Scene Flow Estimation in the Wild",
  "abstract": "Large models have shown generalization across datasets for many low-level vision tasks, like depth estimation, but no such general models exist for scene flow. Even though scene flow has wide potential use, it is not used in practice because current predictive models do not generalize well. We identify three key challenges and propose solutions for each. First, we create a method that jointly estimates geometry and motion for accurate prediction. Second, we alleviate scene flow data scarcity with a data recipe that affords us 1M annotated training samples across diverse synthetic scenes. Third, we evaluate different parameterizations for scene flow prediction and adopt a natural and effective parameterization. Our resulting model outperforms existing methods as well as baselines built on large-scale models in terms of 3D end-point error, and shows zero-shot generalization to the casually captured videos from DAVIS and the robotic manipulation scenes from RoboTAP. Overall, our approach makes scene flow prediction more practical in-the-wild.",
  "published": "2025-01-17",
  "updated": "2025-01-20",
  "year": "2025",
  "authors": [
   "Yiqing Liang",
   "Abhishek Badki",
   "Hang Su",
   "James Tompkin",
   "Orazio Gallo"
  ],
  "author_count": 5,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 22,
  "influential_citations": 3,
  "tldr": "This work creates a method that jointly estimates geometry and motion for accurate prediction and shows zero-shot generalization to the casually captured videos from DAVIS and the robotic manipulation scenes from RoboTAP, which makes scene flow prediction more practical in thewild.",
  "doi": "10.1109/CVPR52734.2025.01959",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yiqing Liang",
    "id": "2257367768",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Abhishek Badki",
    "id": "2679481",
    "h_index": 8,
    "papers": 14
   },
   {
    "name": "Hang Su",
    "id": "2281070310",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "James Tompkin",
    "id": "1854493",
    "h_index": 30,
    "papers": 143
   },
   {
    "name": "O. Gallo",
    "id": "39775678",
    "h_index": 27,
    "papers": 60
   }
  ],
  "comment": "Project Website: https://research.nvidia.com/labs/lpr/zero_msf//",
  "topics": [
   "spatial-3d"
  ],
  "orgs": [
   "NVIDIA"
  ],
  "abs_url": "https://arxiv.org/abs/2501.10357v2",
  "pdf_url": "https://arxiv.org/pdf/2501.10357v2",
  "html_url": "https://arxiv.org/html/2501.10357v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.36
 },
 {
  "id": "2501.09898",
  "slug": "foundationstereo-zero-shot-stereo-matching",
  "title": "FoundationStereo: Zero-Shot Stereo Matching",
  "abstract": "Tremendous progress has been made in deep stereo matching to excel on benchmark datasets through per-domain fine-tuning. However, achieving strong zero-shot generalization - a hallmark of foundation models in other computer vision tasks - remains challenging for stereo matching. We introduce FoundationStereo, a foundation model for stereo depth estimation designed to achieve strong zero-shot generalization. To this end, we first construct a large-scale (1M stereo pairs) synthetic training dataset featuring large diversity and high photorealism, followed by an automatic self-curation pipeline to remove ambiguous samples. We then design a number of network architecture components to enhance scalability, including a side-tuning feature backbone that adapts rich monocular priors from vision foundation models to mitigate the sim-to-real gap, and long-range context reasoning for effective cost volume filtering. Together, these components lead to strong robustness and accuracy across domains, establishing a new standard in zero-shot stereo depth estimation. Project page: https://nvlabs.github.io/FoundationStereo/",
  "published": "2025-01-17",
  "updated": "2025-04-04",
  "year": "2025",
  "authors": [
   "Bowen Wen",
   "Matthew Trepte",
   "Joseph Aribido",
   "Jan Kautz",
   "Orazio Gallo",
   "Stan Birchfield"
  ],
  "author_count": 6,
  "categories": [
   "cs.CV",
   "cs.LG",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 261,
  "influential_citations": 40,
  "tldr": "This work introduces FoundationStereo, a foundation model for stereo depth estimation designed to achieve strong zero-shot generalization and designs a number of network architecture components to enhance scalability, including a side-tuning feature backbone that adapts rich monocular priors from vision foundation models to mitigate the sim-to-real gap.",
  "doi": "10.1109/CVPR52734.2025.00495",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Bowen Wen",
    "id": "2340686160",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Matthew Trepte",
    "id": "2340686469",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "J. Aribido",
    "id": "2174393100",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Jan Kautz",
    "id": "2281034686",
    "h_index": 18,
    "papers": 29
   },
   {
    "name": "O. Gallo",
    "id": "39775678",
    "h_index": 27,
    "papers": 60
   },
   {
    "name": "S. T. Birchfield",
    "id": "2285517121",
    "h_index": 5,
    "papers": 6
   }
  ],
  "comment": "CVPR 2025",
  "topics": [
   "sim2real",
   "spatial-3d",
   "foundation-pretraining",
   "safety-eval"
  ],
  "orgs": [
   "NVIDIA"
  ],
  "abs_url": "https://arxiv.org/abs/2501.09898v4",
  "pdf_url": "https://arxiv.org/pdf/2501.09898v4",
  "html_url": "https://arxiv.org/html/2501.09898v4",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.42
 },
 {
  "id": "2501.09747",
  "slug": "fast-efficient-action-tokenization-for-vision-language-action-models",
  "title": "FAST: Efficient Action Tokenization for Vision-Language-Action Models",
  "abstract": "Autoregressive sequence models, such as Transformer-based vision-language action (VLA) policies, can be tremendously effective for capturing complex and generalizable robotic behaviors. However, such models require us to choose a tokenization of our continuous action signals, which determines how the discrete symbols predicted by the model map to continuous robot actions. We find that current approaches for robot action tokenization, based on simple per-dimension, per-timestep binning schemes, typically perform poorly when learning dexterous skills from high-frequency robot data. To address this challenge, we propose a new compression-based tokenization scheme for robot actions, based on the discrete cosine transform. Our tokenization approach, Frequency-space Action Sequence Tokenization (FAST), enables us to train autoregressive VLAs for highly dexterous and high-frequency tasks where standard discretization methods fail completely. Based on FAST, we release FAST+, a universal robot action tokenizer, trained on 1M real robot action trajectories. It can be used as a black-box tokenizer for a wide range of robot action sequences, with diverse action spaces and control frequencies. Finally, we show that, when combined with the pi0 VLA, our method can scale to training on 10k hours of robot data and match the performance of diffusion VLAs, while reducing training time by up to 5x.",
  "published": "2025-01-16",
  "updated": "2025-01-16",
  "year": "2025",
  "authors": [
   "Karl Pertsch",
   "Kyle Stachowicz",
   "Brian Ichter",
   "Danny Driess",
   "Suraj Nair",
   "Quan Vuong",
   "Oier Mees",
   "Chelsea Finn",
   "Sergey Levine"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "Robotics",
  "venue_source": "semantic-scholar",
  "citations": 607,
  "influential_citations": 91,
  "tldr": "This work proposes a new compression-based tokenization scheme for robot actions, based on the discrete cosine transform, and releases FAST+, a universal robot action tokenizer, trained on 1M real robot action trajectories, and shows that it can scale to training on 10k hours of robot data and match the performance of diffusion VLAs, while reducing training time by up to 5x.",
  "doi": "10.48550/arXiv.2501.09747",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Karl Pertsch",
    "id": "31719101",
    "h_index": 32,
    "papers": 56
   },
   {
    "name": "Kyle Stachowicz",
    "id": "2106415427",
    "h_index": 12,
    "papers": 21
   },
   {
    "name": "Brian Ichter",
    "id": "2704814",
    "h_index": 37,
    "papers": 60
   },
   {
    "name": "Danny Driess",
    "id": "2283848260",
    "h_index": 27,
    "papers": 35
   },
   {
    "name": "Suraj Nair",
    "id": "2286638954",
    "h_index": 12,
    "papers": 19
   },
   {
    "name": "Quan Vuong",
    "id": "2288210223",
    "h_index": 19,
    "papers": 29
   },
   {
    "name": "Oier Mees",
    "id": "7264115",
    "h_index": 28,
    "papers": 45
   },
   {
    "name": "Chelsea Finn",
    "id": "2257346440",
    "h_index": 23,
    "papers": 32
   },
   {
    "name": "Sergey Levine",
    "id": "2249615151",
    "h_index": 32,
    "papers": 53
   }
  ],
  "comment": "Website: https://www.pi.website/research/fast",
  "topics": [
   "vla",
   "dexterous-manipulation",
   "video-generation"
  ],
  "orgs": [
   "Physical Intelligence"
  ],
  "abs_url": "https://arxiv.org/abs/2501.09747v1",
  "pdf_url": "https://arxiv.org/pdf/2501.09747v1",
  "html_url": "https://arxiv.org/html/2501.09747v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.78
 },
 {
  "id": "2501.09038",
  "slug": "do-generative-video-models-understand-physical-principles",
  "title": "Do generative video models understand physical principles?",
  "abstract": "AI video generation is undergoing a revolution, with quality and realism advancing rapidly. These advances have led to a passionate scientific debate: Do video models learn \"world models\" that discover laws of physics -- or, alternatively, are they merely sophisticated pixel predictors that achieve visual realism without understanding the physical principles of reality? We address this question by developing Physics-IQ, a comprehensive benchmark dataset that can only be solved by acquiring a deep understanding of various physical principles, like fluid dynamics, optics, solid mechanics, magnetism and thermodynamics. We find that across a range of current models (Sora, Runway, Pika, Lumiere, Stable Video Diffusion, and VideoPoet), physical understanding is severely limited, and unrelated to visual realism. At the same time, some test cases can already be successfully solved. This indicates that acquiring certain physical principles from observation alone may be possible, but significant challenges remain. While we expect rapid advances ahead, our work demonstrates that visual realism does not imply physical understanding. Our project page is at https://physics-iq.github.io; code at https://github.com/google-deepmind/physics-IQ-benchmark.",
  "published": "2025-01-14",
  "updated": "2025-02-27",
  "year": "2025",
  "authors": [
   "Saman Motamed",
   "Laura Culp",
   "Kevin Swersky",
   "Priyank Jaini",
   "Robert Geirhos"
  ],
  "author_count": 5,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.GR",
   "cs.LG"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 139,
  "influential_citations": 18,
  "tldr": "Physic-IQ is developed, a comprehensive benchmark dataset that can only be solved by acquiring a deep understanding of various physical principles, like fluid dynamics, optics, solid mechanics, magnetism and thermodynamics, and demonstrates that visual realism does not imply physical understanding.",
  "doi": "10.1109/WACV61042.2026.00099",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Saman Motamed",
    "id": "1387976878",
    "h_index": 9,
    "papers": 19
   },
   {
    "name": "Laura Culp",
    "id": "2219763699",
    "h_index": 7,
    "papers": 12
   },
   {
    "name": "Kevin Swersky",
    "id": "1754860",
    "h_index": 24,
    "papers": 51
   },
   {
    "name": "P. Jaini",
    "id": "144818264",
    "h_index": 17,
    "papers": 46
   },
   {
    "name": "Robert Geirhos",
    "id": "1949747",
    "h_index": 20,
    "papers": 38
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "data-teleop",
   "video-generation"
  ],
  "orgs": [
   "Google DeepMind"
  ],
  "abs_url": "https://arxiv.org/abs/2501.09038v3",
  "pdf_url": "https://arxiv.org/pdf/2501.09038v3",
  "html_url": "https://arxiv.org/html/2501.09038v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.65
 },
 {
  "id": "2501.06994",
  "slug": "motion-tracks-a-unified-representation-for-human-robot-transfer-in-few",
  "title": "Motion Tracks: A Unified Representation for Human-Robot Transfer in Few-Shot Imitation Learning",
  "abstract": "Teaching robots to autonomously complete everyday tasks remains a challenge. Imitation Learning (IL) is a powerful approach that imbues robots with skills via demonstrations, but is limited by the labor-intensive process of collecting teleoperated robot data. Human videos offer a scalable alternative, but it remains difficult to directly train IL policies from them due to the lack of robot action labels. To address this, we propose to represent actions as short-horizon 2D trajectories on an image. These actions, or motion tracks, capture the predicted direction of motion for either human hands or robot end-effectors. We instantiate an IL policy called Motion Track Policy (MT-pi) which receives image observations and outputs motion tracks as actions. By leveraging this unified, cross-embodiment action space, MT-pi completes tasks with high success given just minutes of human video and limited additional robot demonstrations. At test time, we predict motion tracks from two camera views, recovering 6DoF trajectories via multi-view synthesis. MT-pi achieves an average success rate of 86.5% across 4 real-world tasks, outperforming state-of-the-art IL baselines which do not leverage human data or our action space by 40%, and generalizes to scenarios seen only in human videos. Code and videos are available on our website https://portal-cornell.github.io/motion_track_policy/.",
  "published": "2025-01-13",
  "updated": "2025-10-10",
  "year": "2025",
  "authors": [
   "Juntao Ren",
   "Priya Sundaresan",
   "Dorsa Sadigh",
   "Sanjiban Choudhury",
   "Jeannette Bohg"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA",
  "venue_source": "semantic-scholar",
  "citations": 79,
  "influential_citations": 5,
  "tldr": "This work instantiates an IL policy called Motion Track Policy (MT- $\\pi$ ) which receives image observations and outputs motion tracks as actions, and leveraging this unified, cross-embodiment action space completes tasks with high success given just minutes of human video and limited additional robot demonstrations.",
  "doi": "10.1109/ICRA55743.2025.11128834",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Juntao Ren",
    "id": "2284191057",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Priya Sundaresan",
    "id": "123235030",
    "h_index": 19,
    "papers": 31
   },
   {
    "name": "Dorsa Sadigh",
    "id": "1779671",
    "h_index": 69,
    "papers": 226
   },
   {
    "name": "Sanjiban Choudhury",
    "id": "2266752840",
    "h_index": 9,
    "papers": 14
   },
   {
    "name": "Jeannette Bohg",
    "id": "2323565347",
    "h_index": 6,
    "papers": 12
   }
  ],
  "comment": "",
  "topics": [
   "egocentric-data",
   "imitation-diffusion",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2501.06994v2",
  "pdf_url": "https://arxiv.org/pdf/2501.06994v2",
  "html_url": "https://arxiv.org/html/2501.06994v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.4
 },
 {
  "id": "2501.05420",
  "slug": "robopanoptes-the-all-seeing-robot-with-whole-body-dexterity",
  "title": "RoboPanoptes: The All-seeing Robot with Whole-body Dexterity",
  "abstract": "We present RoboPanoptes, a capable yet practical robot system that achieves whole-body dexterity through whole-body vision. Its whole-body dexterity allows the robot to utilize its entire body surface for manipulation, such as leveraging multiple contact points or navigating constrained spaces. Meanwhile, whole-body vision uses a camera system distributed over the robot's surface to provide comprehensive, multi-perspective visual feedback of its own and the environment's state. At its core, RoboPanoptes uses a whole-body visuomotor policy that learns complex manipulation skills directly from human demonstrations, efficiently aggregating information from the distributed cameras while maintaining resilience to sensor failures. Together, these design aspects unlock new capabilities and tasks, allowing RoboPanoptes to unbox in narrow spaces, sweep multiple or oversized objects, and succeed in multi-step stowing in cluttered environments, outperforming baselines in adaptability and efficiency. Results are best viewed on https://robopanoptes.github.io.",
  "published": "2025-01-09",
  "updated": "2026-01-11",
  "year": "2025",
  "authors": [
   "Xiaomeng Xu",
   "Dominik Bauer",
   "Shuran Song"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "Robotics",
  "venue_source": "semantic-scholar",
  "citations": 21,
  "influential_citations": 1,
  "tldr": "The design aspects unlock new capabilities and tasks, allowing RoboPanoptes to unbox in narrow spaces, sweep multiple or oversized objects, and succeed in multi-step stowing in cluttered environments, outperforming baselines in adaptability and efficiency.",
  "doi": "10.48550/arXiv.2501.05420",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xiaomeng Xu",
    "id": "2286521452",
    "h_index": 7,
    "papers": 9
   },
   {
    "name": "Dominik Bauer",
    "id": "2297668171",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Shuran Song",
    "id": "2297819190",
    "h_index": 4,
    "papers": 5
   }
  ],
  "comment": "Project website: https://robopanoptes.github.io",
  "topics": [
   "dexterous-manipulation",
   "humanoids",
   "egocentric-data"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2501.05420v3",
  "pdf_url": "https://arxiv.org/pdf/2501.05420v3",
  "html_url": "https://arxiv.org/html/2501.05420v3",
  "code_url": "https://robopanoptes.github.io",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.84
 },
 {
  "id": "2501.04169",
  "slug": "learning-to-transfer-human-hand-skills-for-robot-manipulations",
  "title": "Learning to Transfer Human Hand Skills for Robot Manipulations",
  "abstract": "We present a method for teaching dexterous manipulation tasks to robots from human hand motion demonstrations. Unlike existing approaches that solely rely on kinematics information without taking into account the plausibility of robot and object interaction, our method directly infers plausible robot manipulation actions from human motion demonstrations. To address the embodiment gap between the human hand and the robot system, our approach learns a joint motion manifold that maps human hand movements, robot hand actions, and object movements in 3D, enabling us to infer one motion component from others. Our key idea is the generation of pseudo-supervision triplets, which pair human, object, and robot motion trajectories synthetically. Through real-world experiments with robot hand manipulation, we demonstrate that our data-driven retargeting method significantly outperforms conventional retargeting techniques, effectively bridging the embodiment gap between human and robotic hands. Website at https://rureadyo.github.io/MocapRobot/.",
  "published": "2025-01-07",
  "updated": "2025-01-07",
  "year": "2025",
  "authors": [
   "Sungjae Park",
   "Seungho Lee",
   "Mingi Choi",
   "Jiye Lee",
   "Jeonghwan Kim",
   "Jisoo Kim",
   "Hanbyul Joo"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 16,
  "influential_citations": 0,
  "tldr": "The key idea is the generation of pseudo-supervision triplets, which pair human, object, and robot motion trajectories synthetically and significantly outperforms conventional retargeting techniques, effectively bridging the embodiment gap between human and robotic hands.",
  "doi": "10.48550/arXiv.2501.04169",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Sungjae Park",
    "id": "2371412504",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Seungho Lee",
    "id": "2343468565",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Mingi Choi",
    "id": "2262455307",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Jiye Lee",
    "id": "2277508449",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Jeonghwan Kim",
    "id": "2279829723",
    "h_index": 2,
    "papers": 11
   },
   {
    "name": "Jisoo Kim",
    "id": "2279809276",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Hanbyul Joo",
    "id": "2277246764",
    "h_index": 8,
    "papers": 27
   }
  ],
  "comment": "Preprint. Under Review",
  "topics": [
   "dexterous-manipulation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2501.04169v1",
  "pdf_url": "https://arxiv.org/pdf/2501.04169v1",
  "html_url": "https://arxiv.org/html/2501.04169v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.23
 },
 {
  "id": "2501.03575",
  "slug": "cosmos-world-foundation-model-platform-for-physical-ai",
  "title": "Cosmos World Foundation Model Platform for Physical AI",
  "abstract": "Physical AI needs to be trained digitally first. It needs a digital twin of itself, the policy model, and a digital twin of the world, the world model. In this paper, we present the Cosmos World Foundation Model Platform to help developers build customized world models for their Physical AI setups. We position a world foundation model as a general-purpose world model that can be fine-tuned into customized world models for downstream applications. Our platform covers a video curation pipeline, pre-trained world foundation models, examples of post-training of pre-trained world foundation models, and video tokenizers. To help Physical AI builders solve the most critical problems of our society, we make Cosmos open-source and our models open-weight with permissive licenses available via https://github.com/nvidia-cosmos/cosmos-predict1.",
  "published": "2025-01-07",
  "updated": "2025-07-09",
  "year": "2025",
  "authors": [
   " NVIDIA",
   " :",
   "Niket Agarwal",
   "Arslan Ali",
   "Maciej Bala",
   "Yogesh Balaji",
   "Erik Barker",
   "Tiffany Cai",
   "Prithvijit Chattopadhyay",
   "Yongxin Chen",
   "Yin Cui",
   "Yifan Ding",
   "Daniel Dworakowski",
   "Jiaojiao Fan",
   "Michele Fenzi",
   "Francesco Ferroni",
   "Sanja Fidler",
   "Dieter Fox",
   "Songwei Ge",
   "Yunhao Ge",
   "Jinwei Gu",
   "Siddharth Gururani",
   "Ethan He",
   "Jiahui Huang",
   "Jacob Huffman",
   "Pooya Jannaty",
   "Jingyi Jin",
   "Seung Wook Kim",
   "Gergely Kl\u00e1r",
   "Grace Lam",
   "Shiyi Lan",
   "Laura Leal-Taixe",
   "Anqi Li",
   "Zhaoshuo Li",
   "Chen-Hsuan Lin",
   "Tsung-Yi Lin",
   "Huan Ling",
   "Ming-Yu Liu",
   "Xian Liu",
   "Alice Luo",
   "Qianli Ma",
   "Hanzi Mao",
   "Kaichun Mo",
   "Arsalan Mousavian",
   "Seungjun Nah",
   "Sriharsha Niverty",
   "David Page",
   "Despoina Paschalidou",
   "Zeeshan Patel",
   "Lindsey Pavao",
   "Morteza Ramezanali",
   "Fitsum Reda",
   "Xiaowei Ren",
   "Vasanth Rao Naik Sabavat",
   "Ed Schmerling",
   "Stella Shi",
   "Bartosz Stefaniak",
   "Shitao Tang",
   "Lyne Tchapmi",
   "Przemek Tredak",
   "Wei-Cheng Tseng",
   "Jibin Varghese",
   "Hao Wang",
   "Haoxiang Wang",
   "Heng Wang",
   "Ting-Chun Wang",
   "Fangyin Wei",
   "Xinyue Wei",
   "Jay Zhangjie Wu",
   "Jiashu Xu",
   "Wei Yang",
   "Lin Yen-Chen",
   "Xiaohui Zeng",
   "Yu Zeng",
   "Jing Zhang",
   "Qinsheng Zhang",
   "Yuxuan Zhang",
   "Qingqing Zhao",
   "Artur Zolkowski"
  ],
  "author_count": 79,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.LG",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 810,
  "influential_citations": 113,
  "tldr": "The Cosmos World Foundation Model Platform is presented to help developers build customized world models for their Physical AI setups and position a world foundation model as a general-purpose world model that can be fine-tuned into customized world models for downstream applications.",
  "doi": "10.48550/arXiv.2501.03575",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "N. Agarwal",
    "id": "2338889954",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Arslan Ali",
    "id": "1405851901",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "M. Bala",
    "id": "2330191272",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Yogesh Balaji",
    "id": "2310337201",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Erik Barker",
    "id": "2338890412",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Tiffany Cai",
    "id": "2330193931",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Prithvijit Chattopadhyay",
    "id": "40424000",
    "h_index": 15,
    "papers": 30
   },
   {
    "name": "Yongxin Chen",
    "id": "2221031822",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Yin Cui",
    "id": "2299115514",
    "h_index": 10,
    "papers": 17
   },
   {
    "name": "Yifan Ding",
    "id": "2263775555",
    "h_index": 6,
    "papers": 18
   },
   {
    "name": "Daniel Dworakowski",
    "id": "2338890333",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Jiaojiao Fan",
    "id": "2328007579",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Michele Fenzi",
    "id": "1779415",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Francesco Ferroni",
    "id": "2260336030",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "Sanja Fidler",
    "id": "2261282058",
    "h_index": 20,
    "papers": 41
   },
   {
    "name": "Dieter Fox",
    "id": "2257169746",
    "h_index": 9,
    "papers": 13
   },
   {
    "name": "Songwei Ge",
    "id": "2338889811",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Yunhao Ge",
    "id": "2299104362",
    "h_index": 11,
    "papers": 20
   },
   {
    "name": "Jinwei Gu",
    "id": "2338980231",
    "h_index": 9,
    "papers": 13
   },
   {
    "name": "Siddharth Gururani",
    "id": "3454904",
    "h_index": 16,
    "papers": 35
   },
   {
    "name": "Ethan He",
    "id": "2338890507",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Jiahui Huang",
    "id": "2244137145",
    "h_index": 26,
    "papers": 68
   },
   {
    "name": "J. Huffman",
    "id": "2298965422",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Pooya Jannaty",
    "id": "2313203519",
    "h_index": 8,
    "papers": 9
   },
   {
    "name": "Jingyi Jin",
    "id": "2313473772",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "S. Kim",
    "id": "2262213592",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "G. Kl\u00e1r",
    "id": "3086857",
    "h_index": 7,
    "papers": 20
   },
   {
    "name": "Grace Lam",
    "id": "2330190570",
    "h_index": 8,
    "papers": 9
   },
   {
    "name": "Shiyi Lan",
    "id": "2338890616",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "L. Leal-Taix\u00e9",
    "id": "1388407684",
    "h_index": 46,
    "papers": 113
   },
   {
    "name": "Anqi Li",
    "id": "2364873725",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Zhaoshuo Li",
    "id": "2313218585",
    "h_index": 12,
    "papers": 18
   },
   {
    "name": "Chen-Hsuan Lin",
    "id": "2313497801",
    "h_index": 8,
    "papers": 11
   },
   {
    "name": "Tsung-Yi Lin",
    "id": "2300141490",
    "h_index": 15,
    "papers": 22
   },
   {
    "name": "Huan Ling",
    "id": "18900686",
    "h_index": 28,
    "papers": 44
   },
   {
    "name": "Ming-Yu Liu",
    "id": "2299018949",
    "h_index": 15,
    "papers": 23
   },
   {
    "name": "Xian Liu",
    "id": "2323079661",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Alice Luo",
    "id": "2330185830",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Qianli Ma",
    "id": "2313310048",
    "h_index": 10,
    "papers": 12
   },
   {
    "name": "Hanzi Mao",
    "id": "2313167739",
    "h_index": 8,
    "papers": 8
   },
   {
    "name": "Kaichun Mo",
    "id": "2261042152",
    "h_index": 8,
    "papers": 14
   },
   {
    "name": "A. Mousavian",
    "id": "3040583",
    "h_index": 36,
    "papers": 57
   },
   {
    "name": "Seungjun Nah",
    "id": "40648435",
    "h_index": 21,
    "papers": 30
   },
   {
    "name": "Sriharsha Niverty",
    "id": "2338889490",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "David Page",
    "id": "2338890044",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Despoina Paschalidou",
    "id": "2328977814",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Zeeshan Patel",
    "id": "2338890341",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Lindsey Pavao",
    "id": "2338890844",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Morteza Ramezanali",
    "id": "2338890557",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "F. Reda",
    "id": "3291967",
    "h_index": 17,
    "papers": 37
   },
   {
    "name": "Xiao-Shuai Ren",
    "id": "2201343711",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Vasanth Rao Naik Sabavat",
    "id": "2306951791",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Ed Schmerling",
    "id": "2286298438",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Stella Shi",
    "id": "2330236588",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Bartosz Stefaniak",
    "id": "2338889798",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Shitao Tang",
    "id": "2338977747",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Lyne P. Tchapmi",
    "id": "26917145",
    "h_index": 9,
    "papers": 16
   },
   {
    "name": "Przemek Tredak",
    "id": "2338890568",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Wei-Cheng Tseng",
    "id": "2321873035",
    "h_index": 8,
    "papers": 17
   },
   {
    "name": "J. Varghese",
    "id": "145853825",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Hao Wang",
    "id": "2339536289",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Haoxiang Wang",
    "id": "2338962955",
    "h_index": 11,
    "papers": 31
   },
   {
    "name": "Hengyi Wang",
    "id": "2298494612",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Tingwei Wang",
    "id": "2322488670",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Fangyin Wei",
    "id": "2330307301",
    "h_index": 8,
    "papers": 9
   },
   {
    "name": "Xinyue Wei",
    "id": "2336255140",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Jay Zhangjie Wu",
    "id": "2167473942",
    "h_index": 16,
    "papers": 22
   },
   {
    "name": "Jiashu Xu",
    "id": "2301402981",
    "h_index": 11,
    "papers": 25
   },
   {
    "name": "Wei Yang",
    "id": "2336525939",
    "h_index": 1,
    "papers": 6
   },
   {
    "name": "Lin Yen-Chen",
    "id": "1485124622",
    "h_index": 6,
    "papers": 6
   },
   {
    "name": "Xiaohui Zeng",
    "id": "2152293554",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Yuan Zeng",
    "id": "2313482950",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Jing Zhang",
    "id": "2339986899",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Qinsheng Zhang",
    "id": "2288856236",
    "h_index": 14,
    "papers": 17
   },
   {
    "name": "Yuxuan Zhang",
    "id": "2339352559",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Qingqing Zhao",
    "id": "2258609525",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Artur Zolkowski",
    "id": "2338890689",
    "h_index": 3,
    "papers": 4
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "foundation-pretraining",
   "video-generation"
  ],
  "orgs": [
   "NVIDIA"
  ],
  "abs_url": "https://arxiv.org/abs/2501.03575v3",
  "pdf_url": "https://arxiv.org/pdf/2501.03575v3",
  "html_url": "https://arxiv.org/html/2501.03575v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.41
 },
 {
  "id": "2501.02973",
  "slug": "hawor-world-space-hand-motion-reconstruction-from-egocentric-videos",
  "title": "HaWoR: World-Space Hand Motion Reconstruction from Egocentric Videos",
  "abstract": "Despite the advent in 3D hand pose estimation, current methods predominantly focus on single-image 3D hand reconstruction in the camera frame, overlooking the world-space motion of the hands. Such limitation prohibits their direct use in egocentric video settings, where hands and camera are continuously in motion. In this work, we propose HaWoR, a high-fidelity method for hand motion reconstruction in world coordinates from egocentric videos. We propose to decouple the task by reconstructing the hand motion in the camera space and estimating the camera trajectory in the world coordinate system. To achieve precise camera trajectory estimation, we propose an adaptive egocentric SLAM framework that addresses the shortcomings of traditional SLAM methods, providing robust performance under challenging camera dynamics. To ensure robust hand motion trajectories, even when the hands move out of view frustum, we devise a novel motion infiller network that effectively completes the missing frames of the sequence. Through extensive quantitative and qualitative evaluations, we demonstrate that HaWoR achieves state-of-the-art performance on both hand motion reconstruction and world-frame camera trajectory estimation under different egocentric benchmark datasets. Code and models are available on https://hawor-project.github.io/ .",
  "published": "2025-01-06",
  "updated": "2025-01-06",
  "year": "2025",
  "authors": [
   "Jinglei Zhang",
   "Jiankang Deng",
   "Chao Ma",
   "Rolandos Alexandros Potamias"
  ],
  "author_count": 4,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 60,
  "influential_citations": 11,
  "tldr": "This work proposes HaWoR, a high-fidelity method for hand motion reconstruction in world coordinates from egocentric videos, and proposes an adaptive egocentric SLAM framework that addresses the shortcomings of traditional SLAM methods, providing robust performance under challenging camera dynamics.",
  "doi": "10.1109/CVPR52734.2025.00175",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jinglei Zhang",
    "id": "2321886378",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Jiankang Deng",
    "id": "2332095520",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Chao Ma",
    "id": "2372843698",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Rolandos Alexandros Potamias",
    "id": "121927450",
    "h_index": 13,
    "papers": 41
   }
  ],
  "comment": "",
  "topics": [
   "egocentric-data",
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2501.02973v1",
  "pdf_url": "https://arxiv.org/pdf/2501.02973v1",
  "html_url": "https://arxiv.org/html/2501.02973v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.29
 },
 {
  "id": "2501.02116",
  "slug": "humanoid-locomotion-and-manipulation-current-progress-and-challenges-i",
  "title": "Humanoid Locomotion and Manipulation: Current Progress and Challenges in Control, Planning, and Learning",
  "abstract": "Humanoid robots hold great potential to perform various human-level skills, involving unified locomotion and manipulation in real-world settings. Driven by advances in machine learning and the strength of existing model-based approaches, these capabilities have progressed rapidly, but often separately. This survey offers a comprehensive overview of the state-of-the-art in humanoid locomotion and manipulation (HLM), with a focus on control, planning, and learning methods. We first review the model-based methods that have been the backbone of humanoid robotics for the past three decades. We discuss contact planning, motion planning, and whole-body control, highlighting the trade-offs between model fidelity and computational efficiency. Then the focus is shifted to examine emerging learning-based methods, with an emphasis on reinforcement and imitation learning that enhance the robustness and versatility of loco-manipulation skills. Furthermore, we assess the potential of integrating foundation models with humanoid embodiments to enable the development of generalist humanoid agents. This survey also highlights the emerging role of tactile sensing, particularly whole-body tactile feedback, as a crucial modality for handling contact-rich interactions. Finally, we compare the strengths and limitations of model-based and learning-based paradigms from multiple perspectives, such as robustness, computational efficiency, versatility, and generalizability, and suggest potential solutions to existing challenges.",
  "published": "2025-01-03",
  "updated": "2025-04-19",
  "year": "2025",
  "authors": [
   "Zhaoyuan Gu",
   "Junheng Li",
   "Wenlan Shen",
   "Wenhao Yu",
   "Zhaoming Xie",
   "Stephen McCrory",
   "Xianyi Cheng",
   "Abdulaziz Shamsah",
   "Robert Griffin",
   "C. Karen Liu",
   "Abderrahmane Kheddar",
   "Xue Bin Peng",
   "Yuke Zhu",
   "Guanya Shi",
   "Quan Nguyen",
   "Gordon Cheng",
   "Huijun Gao",
   "Ye Zhao"
  ],
  "author_count": 18,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 140,
  "influential_citations": 1,
  "tldr": "This survey offers a comprehensive overview of the state-of-the-art in humanoid locomotion and manipulation, with a focus on control, planning, and learning methods, and highlights the emerging role of tactile sensing as a crucial modality for handling contact-rich interactions.",
  "doi": "10.1109/TMECH.2025.3579247",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhaoyuan Gu",
    "id": "2151700177",
    "h_index": 7,
    "papers": 16
   },
   {
    "name": "Junheng Li",
    "id": "2108988997",
    "h_index": 7,
    "papers": 25
   },
   {
    "name": "Wenlan Shen",
    "id": "2338877528",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Wenhao Yu",
    "id": "2304565105",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Zhaoming Xie",
    "id": "2333841902",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Stephen McCrory",
    "id": "4114011",
    "h_index": 11,
    "papers": 23
   },
   {
    "name": "Xianyi Cheng",
    "id": "2297376678",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "Abdulaziz Shamsah",
    "id": "66555007",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Robert J. Griffin",
    "id": "2278284852",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "C. K. Liu",
    "id": "2266010832",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "A. Kheddar",
    "id": "1725002",
    "h_index": 52,
    "papers": 384
   },
   {
    "name": "Xue Bin Peng",
    "id": "2326248112",
    "h_index": 6,
    "papers": 6
   },
   {
    "name": "Yuke Zhu",
    "id": "2338857250",
    "h_index": 7,
    "papers": 14
   },
   {
    "name": "Guanya Shi",
    "id": "2249759531",
    "h_index": 20,
    "papers": 31
   },
   {
    "name": "Quan Nguyen",
    "id": "2286899309",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Gordon Cheng",
    "id": "2338895049",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Huijun Gao",
    "id": "2271781485",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Ye Zhao",
    "id": "2287167298",
    "h_index": 1,
    "papers": 4
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "tactile",
   "imitation-diffusion",
   "foundation-pretraining",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2501.02116v2",
  "pdf_url": "https://arxiv.org/pdf/2501.02116v2",
  "html_url": "https://arxiv.org/html/2501.02116v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.15
 },
 {
  "id": "2501.00103",
  "slug": "ltx-video-realtime-video-latent-diffusion",
  "title": "LTX-Video: Realtime Video Latent Diffusion",
  "abstract": "We introduce LTX-Video, a transformer-based latent diffusion model that adopts a holistic approach to video generation by seamlessly integrating the responsibilities of the Video-VAE and the denoising transformer. Unlike existing methods, which treat these components as independent, LTX-Video aims to optimize their interaction for improved efficiency and quality. At its core is a carefully designed Video-VAE that achieves a high compression ratio of 1:192, with spatiotemporal downscaling of 32 x 32 x 8 pixels per token, enabled by relocating the patchifying operation from the transformer's input to the VAE's input. Operating in this highly compressed latent space enables the transformer to efficiently perform full spatiotemporal self-attention, which is essential for generating high-resolution videos with temporal consistency. However, the high compression inherently limits the representation of fine details. To address this, our VAE decoder is tasked with both latent-to-pixel conversion and the final denoising step, producing the clean result directly in pixel space. This approach preserves the ability to generate fine details without incurring the runtime cost of a separate upsampling module. Our model supports diverse use cases, including text-to-video and image-to-video generation, with both capabilities trained simultaneously. It achieves faster-than-real-time generation, producing 5 seconds of 24 fps video at 768x512 resolution in just 2 seconds on an Nvidia H100 GPU, outperforming all existing models of similar scale. The source code and pre-trained models are publicly available, setting a new benchmark for accessible and scalable video generation.",
  "published": "2024-12-30",
  "updated": "2024-12-30",
  "year": "2024",
  "authors": [
   "Yoav HaCohen",
   "Nisan Chiprut",
   "Benny Brazowski",
   "Daniel Shalem",
   "Dudu Moshe",
   "Eitan Richardson",
   "Eran Levin",
   "Guy Shiran",
   "Nir Zabari",
   "Ori Gordon",
   "Poriya Panet",
   "Sapir Weissbuch",
   "Victor Kulikov",
   "Yaki Bitterman",
   "Zeev Melumian",
   "Ofir Bibi"
  ],
  "author_count": 16,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 538,
  "influential_citations": 84,
  "tldr": "LTX-Video is introduced, a transformer-based latent diffusion model that adopts a holistic approach to video generation by seamlessly integrating the responsibilities of the Video-VAE and the denoising transformer, and achieves faster-than-real-time generation.",
  "doi": "10.48550/arXiv.2501.00103",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yoav HaCohen",
    "id": "32169553",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Nisan Chiprut",
    "id": "2338265934",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Benny Brazowski",
    "id": "2338266151",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Daniel Shalem",
    "id": "2338266426",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "David-Pur Moshe",
    "id": "116793653",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Eitan Richardson",
    "id": "2338266358",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "E. Levin",
    "id": "2056085993",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Guy Shiran",
    "id": "2338266050",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Nir Zabari",
    "id": "1630345313",
    "h_index": 9,
    "papers": 14
   },
   {
    "name": "Ori Gordon",
    "id": "2338266099",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Poriya Panet",
    "id": "2338266047",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Sapir Weissbuch",
    "id": "2284222326",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "V. Kulikov",
    "id": "2073929312",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Yaki Bitterman",
    "id": "2338266086",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Zeev Melumian",
    "id": "2338265914",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Ofir Bibi",
    "id": "2338266423",
    "h_index": 3,
    "papers": 5
   }
  ],
  "comment": "",
  "topics": [
   "video-generation"
  ],
  "orgs": [
   "NVIDIA"
  ],
  "abs_url": "https://arxiv.org/abs/2501.00103v1",
  "pdf_url": "https://arxiv.org/pdf/2501.00103v1",
  "html_url": "https://arxiv.org/html/2501.00103v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.23
 },
 {
  "id": "2412.21059",
  "slug": "visionreward-fine-grained-multi-dimensional-human-preference-learning",
  "title": "VisionReward: Fine-Grained Multi-Dimensional Human Preference Learning for Image and Video Generation",
  "abstract": "Visual generative models have achieved remarkable progress in synthesizing photorealistic images and videos, yet aligning their outputs with human preferences across critical dimensions remains a persistent challenge. Though reinforcement learning from human feedback offers promise for preference alignment, existing reward models for visual generation face limitations, including black-box scoring without interpretability and potentially resultant unexpected biases. We present VisionReward, a general framework for learning human visual preferences in both image and video generation. Specifically, we employ a hierarchical visual assessment framework to capture fine-grained human preferences, and leverages linear weighting to enable interpretable preference learning. Furthermore, we propose a multi-dimensional consistent strategy when using VisionReward as a reward model during preference optimization for visual generation. Experiments show that VisionReward can significantly outperform existing image and video reward models on both machine metrics and human evaluation. Notably, VisionReward surpasses VideoScore by 17.2% in preference prediction accuracy, and text-to-video models with VisionReward achieve a 31.6% higher pairwise win rate compared to the same models using VideoScore. All code and datasets are provided at https://github.com/THUDM/VisionReward.",
  "published": "2024-12-30",
  "updated": "2026-01-05",
  "year": "2024",
  "authors": [
   "Jiazheng Xu",
   "Yu Huang",
   "Jiale Cheng",
   "Yuanming Yang",
   "Jiajun Xu",
   "Yuan Wang",
   "Wenbo Duan",
   "Shen Yang",
   "Qunlin Jin",
   "Shurun Li",
   "Jiayan Teng",
   "Zhuoyi Yang",
   "Wendi Zheng",
   "Xiao Liu",
   "Dan Zhang",
   "Ming Ding",
   "Xiaohan Zhang",
   "Xiaotao Gu",
   "Shiyu Huang",
   "Minlie Huang",
   "Jie Tang",
   "Yuxiao Dong"
  ],
  "author_count": 22,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 153,
  "influential_citations": 13,
  "tldr": "This work presents VisionReward, a general framework for learning human visual preferences in both image and video generation that employs a hierarchical visual assessment framework to capture fine-grained human preferences, and leverages linear weighting to enable interpretable preference learning.",
  "doi": "10.48550/arXiv.2412.21059",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jiazheng Xu",
    "id": "2214082934",
    "h_index": 10,
    "papers": 14
   },
   {
    "name": "Yu Huang",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Jiale Cheng",
    "id": "2308160059",
    "h_index": 8,
    "papers": 13
   },
   {
    "name": "Yuanming Yang",
    "id": "2315948290",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Jiajun Xu",
    "id": "2408440401",
    "h_index": 7,
    "papers": 20
   },
   {
    "name": "Yuan Wang",
    "id": "2337869444",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Wenbo Duan",
    "id": "2342413994",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Shengchao Yang",
    "id": "2339233970",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Qunlin Jin",
    "id": "2337797400",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Shurun Li",
    "id": "40977083",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Jiayan Teng",
    "id": "2238205354",
    "h_index": 8,
    "papers": 11
   },
   {
    "name": "Zhuoyi Yang",
    "id": "2109506541",
    "h_index": 12,
    "papers": 16
   },
   {
    "name": "Wendi Zheng",
    "id": "2163967642",
    "h_index": 9,
    "papers": 13
   },
   {
    "name": "Xiao Liu",
    "id": "2308072332",
    "h_index": 10,
    "papers": 13
   },
   {
    "name": "Ming Ding",
    "id": "2055623340",
    "h_index": 23,
    "papers": 30
   },
   {
    "name": "Xiaohan Zhang",
    "id": "2268628279",
    "h_index": 12,
    "papers": 15
   },
   {
    "name": "Xiaotao Gu",
    "id": "2290625851",
    "h_index": 15,
    "papers": 31
   },
   {
    "name": "Shiyu Huang",
    "id": "2305795673",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "Minlie Huang",
    "id": "2254009342",
    "h_index": 20,
    "papers": 61
   },
   {
    "name": "Jie Tang",
    "id": "2238207092",
    "h_index": 16,
    "papers": 21
   },
   {
    "name": "Yuxiao Dong",
    "id": "2243402027",
    "h_index": 41,
    "papers": 94
   }
  ],
  "comment": "27 pages",
  "topics": [
   "rl-control",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2412.21059v4",
  "pdf_url": "https://arxiv.org/pdf/2412.21059v4",
  "html_url": "https://arxiv.org/html/2412.21059v4",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.19
 },
 {
  "id": "2412.20404",
  "slug": "open-sora-democratizing-efficient-video-production-for-all",
  "title": "Open-Sora: Democratizing Efficient Video Production for All",
  "abstract": "Vision and language are the two foundational senses for humans, and they build up our cognitive ability and intelligence. While significant breakthroughs have been made in AI language ability, artificial visual intelligence, especially the ability to generate and simulate the world we see, is far lagging behind. To facilitate the development and accessibility of artificial visual intelligence, we created Open-Sora, an open-source video generation model designed to produce high-fidelity video content. Open-Sora supports a wide spectrum of visual generation tasks, including text-to-image generation, text-to-video generation, and image-to-video generation. The model leverages advanced deep learning architectures and training/inference techniques to enable flexible video synthesis, which could generate video content of up to 15 seconds, up to 720p resolution, and arbitrary aspect ratios. Specifically, we introduce Spatial-Temporal Diffusion Transformer (STDiT), an efficient diffusion framework for videos that decouples spatial and temporal attention. We also introduce a highly compressive 3D autoencoder to make representations compact and further accelerate training with an ad hoc training strategy. Through this initiative, we aim to foster innovation, creativity, and inclusivity within the community of AI content creation. By embracing the open-source principle, Open-Sora democratizes full access to all the training/inference/data preparation codes as well as model weights. All resources are publicly available at: https://github.com/hpcaitech/Open-Sora.",
  "published": "2024-12-29",
  "updated": "2024-12-29",
  "year": "2024",
  "authors": [
   "Zangwei Zheng",
   "Xiangyu Peng",
   "Tianji Yang",
   "Chenhui Shen",
   "Shenggui Li",
   "Hongxin Liu",
   "Yukun Zhou",
   "Tianyi Li",
   "Yang You"
  ],
  "author_count": 9,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 783,
  "influential_citations": 74,
  "tldr": "Open-Sora, an open-source video generation model designed to produce high-fidelity video content, is created and Spatial-Temporal Diffusion Transformer (STDiT) is introduced, an efficient diffusion framework for videos that decouples spatial and temporal attention.",
  "doi": "10.48550/arXiv.2412.20404",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zangwei Zheng",
    "id": "2109654065",
    "h_index": 14,
    "papers": 24
   },
   {
    "name": "Xiangyu Peng",
    "id": "2256656781",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Tianji Yang",
    "id": "2387162460",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Chenhui Shen",
    "id": "2152867918",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Shenggui Li",
    "id": "2153703322",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Hongxin Liu",
    "id": "2337817537",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Yukun Zhou",
    "id": "2300824614",
    "h_index": 14,
    "papers": 86
   },
   {
    "name": "Tianyi Li",
    "id": "2376486877",
    "h_index": 5,
    "papers": 19
   },
   {
    "name": "Yang You",
    "id": "2303393275",
    "h_index": 4,
    "papers": 4
   }
  ],
  "comment": "",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2412.20404v1",
  "pdf_url": "https://arxiv.org/pdf/2412.20404v1",
  "html_url": "https://arxiv.org/html/2412.20404v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.89
 },
 {
  "id": "2412.18600",
  "slug": "zerohsi-zero-shot-4d-human-scene-interaction-by-video-generation",
  "title": "ZeroHSI: Zero-Shot 4D Human-Scene Interaction by Video Generation",
  "abstract": "Human-scene interaction (HSI) generation is crucial for applications in embodied AI, virtual reality, and robotics. Yet, existing methods cannot synthesize interactions in unseen environments such as in-the-wild scenes or reconstructed scenes, as they rely on paired 3D scenes and captured human motion data for training, which are unavailable for unseen environments. We present ZeroHSI, a novel approach that enables zero-shot 4D human-scene interaction synthesis, eliminating the need for training on any MoCap data. Our key insight is to distill human-scene interactions from state-of-the-art video generation models, which have been trained on vast amounts of natural human movements and interactions, and use differentiable rendering to reconstruct human-scene interactions. ZeroHSI can synthesize realistic human motions in both static scenes and environments with dynamic objects, without requiring any ground-truth motion data. We evaluate ZeroHSI on a curated dataset of different types of various indoor and outdoor scenes with different interaction prompts, demonstrating its ability to generate diverse and contextually appropriate human-scene interactions.",
  "published": "2024-12-24",
  "updated": "2025-03-21",
  "year": "2024",
  "authors": [
   "Hongjie Li",
   "Hong-Xing Yu",
   "Jiaman Li",
   "Jiajun Wu"
  ],
  "author_count": 4,
  "categories": [
   "cs.CV",
   "cs.GR"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 27,
  "influential_citations": 2,
  "tldr": "ZeroHSI is presented, a novel approach that enables zero-shot 4D human-scene interaction synthesis, eliminating the need for training on any MoCap data, and can synthesize realistic human motions in both static scenes and environments with dynamic objects, without requiring any ground-truth motion data.",
  "doi": "10.1109/3DV69130.2026.00080",
  "oa_pdf": "https://arxiv.org/pdf/2412.18600",
  "s2_authors": [
   {
    "name": "Hongjie Li",
    "id": "2291072184",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Hong-Xing Yu",
    "id": "2239448099",
    "h_index": 18,
    "papers": 32
   },
   {
    "name": "Jiaman Li",
    "id": "22133106",
    "h_index": 16,
    "papers": 22
   },
   {
    "name": "Jiajun Wu",
    "id": "2326444500",
    "h_index": 35,
    "papers": 214
   }
  ],
  "comment": "Project website: https://awfuact.github.io/zerohsi/ The first two authors contribute equally",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2412.18600v2",
  "pdf_url": "https://arxiv.org/pdf/2412.18600v2",
  "html_url": "https://arxiv.org/html/2412.18600v2",
  "code_url": "https://awfuact.github.io/zerohsi/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.45
 },
 {
  "id": "2412.14803",
  "slug": "video-prediction-policy-a-generalist-robot-policy-with-predictive-visu",
  "title": "Video Prediction Policy: A Generalist Robot Policy with Predictive Visual Representations",
  "abstract": "Visual representations play a crucial role in developing generalist robotic policies. Previous vision encoders, typically pre-trained with single-image reconstruction or two-image contrastive learning, tend to capture static information, often neglecting the dynamic aspects vital for embodied tasks. Recently, video diffusion models (VDMs) demonstrate the ability to predict future frames and showcase a strong understanding of physical world. We hypothesize that VDMs inherently produce visual representations that encompass both current static information and predicted future dynamics, thereby providing valuable guidance for robot action learning. Based on this hypothesis, we propose the Video Prediction Policy (VPP), which learns implicit inverse dynamics model conditioned on predicted future representations inside VDMs. To predict more precise future, we fine-tune pre-trained video foundation model on robot datasets along with internet human manipulation data. In experiments, VPP achieves a 18.6\\% relative improvement on the Calvin ABC-D generalization benchmark compared to the previous state-of-the-art, and demonstrates a 31.6\\% increase in success rates for complex real-world dexterous manipulation tasks. Project page at https://video-prediction-policy.github.io",
  "published": "2024-12-19",
  "updated": "2025-05-04",
  "year": "2024",
  "authors": [
   "Yucheng Hu",
   "Yanjiang Guo",
   "Pengchao Wang",
   "Xiaoyu Chen",
   "Yen-Jen Wang",
   "Jianke Zhang",
   "Koushil Sreenath",
   "Chaochao Lu",
   "Jianyu Chen"
  ],
  "author_count": 9,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "ICML",
  "venue_source": "semantic-scholar",
  "citations": 292,
  "influential_citations": 34,
  "tldr": "The Video Prediction Policy (VPP), which learns implicit inverse dynamics model conditioned on predicted future representations inside VDMs, and achieves a 18.6% relative improvement on the Calvin ABC-D generalization benchmark compared to the previous state-of-the-art.",
  "doi": "10.48550/arXiv.2412.14803",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yucheng Hu",
    "id": "2325107577",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Yanjiang Guo",
    "id": "2181339548",
    "h_index": 12,
    "papers": 26
   },
   {
    "name": "Pengchao Wang",
    "id": "2266168880",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Xiaoyu Chen",
    "id": "2325107465",
    "h_index": 6,
    "papers": 20
   },
   {
    "name": "Yen-Jen Wang",
    "id": "2115740911",
    "h_index": 11,
    "papers": 18
   },
   {
    "name": "Jianke Zhang",
    "id": "2325003014",
    "h_index": 8,
    "papers": 17
   },
   {
    "name": "K. Sreenath",
    "id": "144116765",
    "h_index": 55,
    "papers": 231
   },
   {
    "name": "Chaochao Lu",
    "id": "2332705161",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Jianyu Chen",
    "id": "2280257464",
    "h_index": 10,
    "papers": 20
   }
  ],
  "comment": "ICML 2025 Spotlight Paper. The first two authors contribute equally",
  "topics": [
   "world-models",
   "dexterous-manipulation",
   "foundation-pretraining",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2412.14803v2",
  "pdf_url": "https://arxiv.org/pdf/2412.14803v2",
  "html_url": "https://arxiv.org/html/2412.14803v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.97
 },
 {
  "id": "2412.12861",
  "slug": "dyn-hamr-recovering-4d-interacting-hand-motion-from-a-dynamic-camera",
  "title": "Dyn-HaMR: Recovering 4D Interacting Hand Motion from a Dynamic Camera",
  "abstract": "We propose Dyn-HaMR, to the best of our knowledge, the first approach to reconstruct 4D global hand motion from monocular videos recorded by dynamic cameras in the wild. Reconstructing accurate 3D hand meshes from monocular videos is a crucial task for understanding human behaviour, with significant applications in augmented and virtual reality (AR/VR). However, existing methods for monocular hand reconstruction typically rely on a weak perspective camera model, which simulates hand motion within a limited camera frustum. As a result, these approaches struggle to recover the full 3D global trajectory and often produce noisy or incorrect depth estimations, particularly when the video is captured by dynamic or moving cameras, which is common in egocentric scenarios. Our Dyn-HaMR consists of a multi-stage, multi-objective optimization pipeline, that factors in (i) simultaneous localization and mapping (SLAM) to robustly estimate relative camera motion, (ii) an interacting-hand prior for generative infilling and to refine the interaction dynamics, ensuring plausible recovery under (self-)occlusions, and (iii) hierarchical initialization through a combination of state-of-the-art hand tracking methods. Through extensive evaluations on both in-the-wild and indoor datasets, we show that our approach significantly outperforms state-of-the-art methods in terms of 4D global mesh recovery. This establishes a new benchmark for hand motion reconstruction from monocular video with moving cameras. Our project page is at https://dyn-hamr.github.io/.",
  "published": "2024-12-17",
  "updated": "2025-05-31",
  "year": "2024",
  "authors": [
   "Zhengdi Yu",
   "Stefanos Zafeiriou",
   "Tolga Birdal"
  ],
  "author_count": 3,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 46,
  "influential_citations": 7,
  "tldr": "Dyn-HaMR is proposed, to the best of the authors' knowledge, the first approach to reconstruct 4D global hand motion from monocular videos recorded by dynamic cameras in the wild, and significantly outperforms state-of-the-art methods in terms of 4D global mesh recovery.",
  "doi": "10.1109/CVPR52734.2025.02581",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhengdi Yu",
    "id": "2264493507",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "S. Zafeiriou",
    "id": "1776444",
    "h_index": 83,
    "papers": 448
   },
   {
    "name": "Tolga Birdal",
    "id": "2355828",
    "h_index": 29,
    "papers": 107
   }
  ],
  "comment": "Project page is available at https://dyn-hamr.github.io/",
  "topics": [
   "egocentric-data",
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2412.12861v3",
  "pdf_url": "https://arxiv.org/pdf/2412.12861v3",
  "html_url": "https://arxiv.org/html/2412.12861v3",
  "code_url": "https://dyn-hamr.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.17
 },
 {
  "id": "2412.11673",
  "slug": "dino-foresight-looking-into-the-future-with-dino",
  "title": "DINO-Foresight: Looking into the Future with DINO",
  "abstract": "Predicting future dynamics is crucial for applications like autonomous driving and robotics, where understanding the environment is key. Existing pixel-level methods are computationally expensive and often focus on irrelevant details. To address these challenges, we introduce DINO-Foresight, a novel framework that operates in the semantic feature space of pretrained Vision Foundation Models (VFMs). Our approach trains a masked feature transformer in a self-supervised manner to predict the evolution of VFM features over time. By forecasting these features, we can apply off-the-shelf, task-specific heads for various scene understanding tasks. In this framework, VFM features are treated as a latent space, to which different heads attach to perform specific tasks for future-frame analysis. Extensive experiments show the very strong performance, robustness and scalability of our framework. Project page and code at https://dino-foresight.github.io/ .",
  "published": "2024-12-16",
  "updated": "2025-11-28",
  "year": "2024",
  "authors": [
   "Efstathios Karypidis",
   "Ioannis Kakogeorgiou",
   "Spyros Gidaris",
   "Nikos Komodakis"
  ],
  "author_count": 4,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 44,
  "influential_citations": 8,
  "tldr": "DINO-Foresight is introduced, a novel framework that operates in the semantic feature space of pretrained Vision Foundation Models (VFMs) that trains a masked feature transformer in a self-supervised manner to predict the evolution of VFM features over time.",
  "doi": "10.48550/arXiv.2412.11673",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Efstathios Karypidis",
    "id": "2162188151",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Ioannis Kakogeorgiou",
    "id": "2064596258",
    "h_index": 9,
    "papers": 15
   },
   {
    "name": "Spyros Gidaris",
    "id": "2475428",
    "h_index": 23,
    "papers": 44
   },
   {
    "name": "Nikos Komodakis",
    "id": "2304769241",
    "h_index": 6,
    "papers": 9
   }
  ],
  "comment": "NeurIPS 2025",
  "topics": [
   "navigation",
   "foundation-pretraining",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2412.11673v2",
  "pdf_url": "https://arxiv.org/pdf/2412.11673v2",
  "html_url": "https://arxiv.org/html/2412.11673v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.15
 },
 {
  "id": "2412.10631",
  "slug": "armada-augmented-reality-for-robot-manipulation-and-robot-free-data-ac",
  "title": "ARMADA: Augmented Reality for Robot Manipulation and Robot-Free Data Acquisition",
  "abstract": "Teleoperation for robot imitation learning is bottlenecked by hardware availability. Can high-quality robot data be collected without a physical robot? We present a system for augmenting Apple Vision Pro with real-time virtual robot feedback. By providing users with an intuitive understanding of how their actions translate to robot motions, we enable the collection of natural barehanded human data that is compatible with the limitations of physical robot hardware. We conducted a user study with 15 participants demonstrating 3 different tasks each under 3 different feedback conditions and directly replayed the collected trajectories on physical robot hardware. Results suggest live robot feedback dramatically improves the quality of the collected data, suggesting a new avenue for scalable human data collection without access to robot hardware. Videos and more are available at https://nataliya.dev/armada.",
  "published": "2024-12-14",
  "updated": "2024-12-14",
  "year": "2024",
  "authors": [
   "Nataliya Nechyporenko",
   "Ryan Hoque",
   "Christopher Webb",
   "Mouli Sivapurapu",
   "Jian Zhang"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 7,
  "influential_citations": 1,
  "tldr": "This work presents a system for augmenting Apple Vision Pro with real-time virtual robot feedback that enables the collection of natural barehanded human data that is compatible with the limitations of physical robot hardware.",
  "doi": "10.48550/arXiv.2412.10631",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Nataliya Nechyporenko",
    "id": "30478223",
    "h_index": 5,
    "papers": 15
   },
   {
    "name": "Ryan Hoque",
    "id": "2335570611",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "C. Webb",
    "id": "2311887058",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Mouli Sivapurapu",
    "id": "3317431",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Jian Zhang",
    "id": "2335574774",
    "h_index": 5,
    "papers": 7
   }
  ],
  "comment": "",
  "topics": [
   "imitation-diffusion",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2412.10631v1",
  "pdf_url": "https://arxiv.org/pdf/2412.10631v1",
  "html_url": "https://arxiv.org/html/2412.10631v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.9
 },
 {
  "id": "2412.10345",
  "slug": "tracevla-visual-trace-prompting-enhances-spatial-temporal-awareness-fo",
  "title": "TraceVLA: Visual Trace Prompting Enhances Spatial-Temporal Awareness for Generalist Robotic Policies",
  "abstract": "Although large vision-language-action (VLA) models pretrained on extensive robot datasets offer promising generalist policies for robotic learning, they still struggle with spatial-temporal dynamics in interactive robotics, making them less effective in handling complex tasks, such as manipulation. In this work, we introduce visual trace prompting, a simple yet effective approach to facilitate VLA models' spatial-temporal awareness for action prediction by encoding state-action trajectories visually. We develop a new TraceVLA model by finetuning OpenVLA on our own collected dataset of 150K robot manipulation trajectories using visual trace prompting. Evaluations of TraceVLA across 137 configurations in SimplerEnv and 4 tasks on a physical WidowX robot demonstrate state-of-the-art performance, outperforming OpenVLA by 10% on SimplerEnv and 3.5x on real-robot tasks and exhibiting robust generalization across diverse embodiments and scenarios. To further validate the effectiveness and generality of our method, we present a compact VLA model based on 4B Phi-3-Vision, pretrained on the Open-X-Embodiment and finetuned on our dataset, rivals the 7B OpenVLA baseline while significantly improving inference efficiency.",
  "published": "2024-12-13",
  "updated": "2025-06-05",
  "year": "2024",
  "authors": [
   "Ruijie Zheng",
   "Yongyuan Liang",
   "Shuaiyi Huang",
   "Jianfeng Gao",
   "Hal Daum\u00e9",
   "Andrey Kolobov",
   "Furong Huang",
   "Jianwei Yang"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 300,
  "influential_citations": 30,
  "tldr": "A compact VLA model based on 4B Phi-3-Vision, pretrained on the Open-X-Embodiment and finetuned on the authors' dataset, rivals the 7B OpenVLA baseline while significantly improving inference efficiency.",
  "doi": "10.48550/arXiv.2412.10345",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ruijie Zheng",
    "id": "2108815599",
    "h_index": 13,
    "papers": 28
   },
   {
    "name": "Yongyuan Liang",
    "id": "83158497",
    "h_index": 12,
    "papers": 24
   },
   {
    "name": "Shuaiyi Huang",
    "id": "2294631774",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Jianfeng Gao",
    "id": "2295522725",
    "h_index": 12,
    "papers": 18
   },
   {
    "name": "Hal Daum'e",
    "id": "2200167546",
    "h_index": 12,
    "papers": 26
   },
   {
    "name": "A. Kolobov",
    "id": "2335445895",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Furong Huang",
    "id": "2238405926",
    "h_index": 10,
    "papers": 19
   },
   {
    "name": "Jianwei Yang",
    "id": "2279705714",
    "h_index": 6,
    "papers": 6
   }
  ],
  "comment": "",
  "topics": [
   "vla"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2412.10345v3",
  "pdf_url": "https://arxiv.org/pdf/2412.10345v3",
  "html_url": "https://arxiv.org/html/2412.10345v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.98
 },
 {
  "id": "2412.09621",
  "slug": "stereo4d-learning-how-things-move-in-3d-from-internet-stereo-videos",
  "title": "Stereo4D: Learning How Things Move in 3D from Internet Stereo Videos",
  "abstract": "Learning to understand dynamic 3D scenes from imagery is crucial for applications ranging from robotics to scene reconstruction. Yet, unlike other problems where large-scale supervised training has enabled rapid progress, directly supervising methods for recovering 3D motion remains challenging due to the fundamental difficulty of obtaining ground truth annotations. We present a system for mining high-quality 4D reconstructions from internet stereoscopic, wide-angle videos. Our system fuses and filters the outputs of camera pose estimation, stereo depth estimation, and temporal tracking methods into high-quality dynamic 3D reconstructions. We use this method to generate large-scale data in the form of world-consistent, pseudo-metric 3D point clouds with long-term motion trajectories. We demonstrate the utility of this data by training a variant of DUSt3R to predict structure and 3D motion from real-world image pairs, showing that training on our reconstructed data enables generalization to diverse real-world scenes. Project page and data at: https://stereo4d.github.io",
  "published": "2024-12-12",
  "updated": "2025-04-30",
  "year": "2024",
  "authors": [
   "Linyi Jin",
   "Richard Tucker",
   "Zhengqi Li",
   "David Fouhey",
   "Noah Snavely",
   "Aleksander Holynski"
  ],
  "author_count": 6,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 93,
  "influential_citations": 6,
  "tldr": "A system for mining high-quality 4D reconstructions from internet stereoscopic, wide-angle videos and demonstrates the utility of this data by training a variant of DUSt3R to predict structure and 3D motion from real-world image pairs, showing that training on the reconstructed data enables generalization to diverse real-world scenes.",
  "doi": "10.1109/CVPR52734.2025.00982",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Linyi Jin",
    "id": "151091452",
    "h_index": 10,
    "papers": 15
   },
   {
    "name": "Richard Tucker",
    "id": "2061556403",
    "h_index": 22,
    "papers": 33
   },
   {
    "name": "Zhengqi Li",
    "id": "2145369560",
    "h_index": 23,
    "papers": 30
   },
   {
    "name": "David F. Fouhey",
    "id": "1786435",
    "h_index": 30,
    "papers": 67
   },
   {
    "name": "Noah Snavely",
    "id": "1830653",
    "h_index": 75,
    "papers": 184
   },
   {
    "name": "Aleksander Holynski",
    "id": "2333897738",
    "h_index": 8,
    "papers": 10
   }
  ],
  "comment": "CVPR 2025 Camera Ready; Data released",
  "topics": [
   "spatial-3d",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2412.09621v2",
  "pdf_url": "https://arxiv.org/pdf/2412.09621v2",
  "html_url": "https://arxiv.org/html/2412.09621v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.47
 },
 {
  "id": "2412.08442",
  "slug": "from-multimodal-llms-to-generalist-embodied-agents-methods-and-lessons",
  "title": "From Multimodal LLMs to Generalist Embodied Agents: Methods and Lessons",
  "abstract": "We examine the capability of Multimodal Large Language Models (MLLMs) to tackle diverse domains that extend beyond the traditional language and vision tasks these models are typically trained on. Specifically, our focus lies in areas such as Embodied AI, Games, UI Control, and Planning. To this end, we introduce a process of adapting an MLLM to a Generalist Embodied Agent (GEA). GEA is a single unified model capable of grounding itself across these varied domains through a multi-embodiment action tokenizer. GEA is trained with supervised learning on a large dataset of embodied experiences and with online RL in interactive simulators. We explore the data and algorithmic choices necessary to develop such a model. Our findings reveal the importance of training with cross-domain data and online RL for building generalist agents. The final GEA model achieves strong generalization performance to unseen tasks across diverse benchmarks compared to other generalist models and benchmark-specific approaches.",
  "published": "2024-12-11",
  "updated": "2024-12-11",
  "year": "2024",
  "authors": [
   "Andrew Szot",
   "Bogdan Mazoure",
   "Omar Attia",
   "Aleksei Timofeev",
   "Harsh Agrawal",
   "Devon Hjelm",
   "Zhe Gan",
   "Zsolt Kira",
   "Alexander Toshev"
  ],
  "author_count": 9,
  "categories": [
   "cs.LG"
  ],
  "primary_category": "cs.LG",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 46,
  "influential_citations": 2,
  "tldr": "GEA is a single unified model capable of grounding itself across these varied domains through a multi-embodiment action tokenizer and achieves strong generalization performance to unseen tasks across diverse benchmarks compared to other generalist models and benchmark-specific approaches.",
  "doi": "10.1109/CVPR52734.2025.00995",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Andrew Szot",
    "id": "1580188581",
    "h_index": 14,
    "papers": 22
   },
   {
    "name": "Bogdan Mazoure",
    "id": "2262216401",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Omar Attia",
    "id": "2334737469",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Aleksei Timofeev",
    "id": "2257035938",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Harsh Agrawal",
    "id": "37825612",
    "h_index": 17,
    "papers": 26
   },
   {
    "name": "Devon Hjelm",
    "id": "88844399",
    "h_index": 8,
    "papers": 16
   },
   {
    "name": "Zhe Gan",
    "id": "2268495128",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "Z. Kira",
    "id": "145276578",
    "h_index": 48,
    "papers": 174
   },
   {
    "name": "Alexander Toshev",
    "id": "1726415",
    "h_index": 42,
    "papers": 82
   }
  ],
  "comment": "",
  "topics": [
   "sim2real",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2412.08442v1",
  "pdf_url": "https://arxiv.org/pdf/2412.08442v1",
  "html_url": "https://arxiv.org/html/2412.08442v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.17
 },
 {
  "id": "2412.08261",
  "slug": "flip-flow-centric-generative-planning-as-general-purpose-manipulation",
  "title": "FLIP: Flow-Centric Generative Planning as General-Purpose Manipulation World Model",
  "abstract": "We aim to develop a model-based planning framework for world models that can be scaled with increasing model and data budgets for general-purpose manipulation tasks with only language and vision inputs. To this end, we present FLow-centric generative Planning (FLIP), a model-based planning algorithm on visual space that features three key modules: 1. a multi-modal flow generation model as the general-purpose action proposal module; 2. a flow-conditioned video generation model as the dynamics module; and 3. a vision-language representation learning model as the value module. Given an initial image and language instruction as the goal, FLIP can progressively search for long-horizon flow and video plans that maximize the discounted return to accomplish the task. FLIP is able to synthesize long-horizon plans across objects, robots, and tasks with image flows as the general action representation, and the dense flow information also provides rich guidance for long-horizon video generation. In addition, the synthesized flow and video plans can guide the training of low-level control policies for robot execution. Experiments on diverse benchmarks demonstrate that FLIP can improve both the success rates and quality of long-horizon video plan synthesis and has the interactive world model property, opening up wider applications for future works.Video demos are on our website: https://nus-lins-lab.github.io/flipweb/.",
  "published": "2024-12-11",
  "updated": "2025-02-16",
  "year": "2024",
  "authors": [
   "Chongkai Gao",
   "Haozhuo Zhang",
   "Zhixuan Xu",
   "Zhehao Cai",
   "Lin Shao"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 41,
  "influential_citations": 4,
  "tldr": "Experiments on diverse benchmarks demonstrate that FLIP can improve both the success rates and quality of long-horizon video plan synthesis and has the interactive world model property, opening up wider applications for future works.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Chongkai Gao",
    "id": "2294721059",
    "h_index": 8,
    "papers": 17
   },
   {
    "name": "Haozhuo Zhang",
    "id": "2334828527",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Zhixuan Xu",
    "id": "2284845139",
    "h_index": 8,
    "papers": 24
   },
   {
    "name": "Zhehao Cai",
    "id": "2326255106",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Lin Shao",
    "id": "2334740929",
    "h_index": 4,
    "papers": 15
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2412.08261v2",
  "pdf_url": "https://arxiv.org/pdf/2412.08261v2",
  "html_url": "https://arxiv.org/html/2412.08261v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.62
 },
 {
  "id": "2412.07773",
  "slug": "mobile-television-predictive-motion-priors-for-humanoid-whole-body-con",
  "title": "Mobile-TeleVision: Predictive Motion Priors for Humanoid Whole-Body Control",
  "abstract": "Humanoid robots require both robust lower-body locomotion and precise upper-body manipulation. While recent Reinforcement Learning (RL) approaches provide whole-body loco-manipulation policies, they lack precise manipulation with high DoF arms. In this paper, we propose decoupling upper-body control from locomotion, using inverse kinematics (IK) and motion retargeting for precise manipulation, while RL focuses on robust lower-body locomotion. We introduce PMP (Predictive Motion Priors), trained with Conditional Variational Autoencoder (CVAE) to effectively represent upper-body motions. The locomotion policy is trained conditioned on this upper-body motion representation, ensuring that the system remains robust with both manipulation and locomotion. We show that CVAE features are crucial for stability and robustness, and significantly outperforms RL-based whole-body control in precise manipulation. With precise upper-body motion and robust lower-body locomotion control, operators can remotely control the humanoid to walk around and explore different environments, while performing diverse manipulation tasks.",
  "published": "2024-12-10",
  "updated": "2025-03-09",
  "year": "2024",
  "authors": [
   "Chenhao Lu",
   "Xuxin Cheng",
   "Jialong Li",
   "Shiqi Yang",
   "Mazeyu Ji",
   "Chengjing Yuan",
   "Ge Yang",
   "Sha Yi",
   "Xiaolong Wang"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA",
  "venue_source": "semantic-scholar",
  "citations": 98,
  "influential_citations": 4,
  "tldr": "PMP (Predictive Motion Priors) is introduced, trained with Conditional Variational Autoencoder (CVAE) to effectively represent upper-body motions, and significantly outperforms RL-based whole-body control in precise manipulation.",
  "doi": "10.1109/ICRA55743.2025.11128652",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Chenhao Lu",
    "id": "2265619607",
    "h_index": 8,
    "papers": 20
   },
   {
    "name": "Xuxin Cheng",
    "id": "90080090",
    "h_index": 13,
    "papers": 13
   },
   {
    "name": "Jialong Li",
    "id": "2309196968",
    "h_index": 7,
    "papers": 7
   },
   {
    "name": "Shiqi Yang",
    "id": "2309666838",
    "h_index": 8,
    "papers": 9
   },
   {
    "name": "Mazeyu Ji",
    "id": "2319410473",
    "h_index": 7,
    "papers": 8
   },
   {
    "name": "Chengjing Yuan",
    "id": "2325972752",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Ge Yang",
    "id": "2288147740",
    "h_index": 9,
    "papers": 10
   },
   {
    "name": "Sha Yi",
    "id": "2316591556",
    "h_index": 5,
    "papers": 15
   },
   {
    "name": "Xiaolong Wang",
    "id": "2294782536",
    "h_index": 12,
    "papers": 17
   }
  ],
  "comment": "Accepted for ICRA 2025",
  "topics": [
   "humanoids",
   "rl-control",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2412.07773v2",
  "pdf_url": "https://arxiv.org/pdf/2412.07773v2",
  "html_url": "https://arxiv.org/html/2412.07773v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.5
 },
 {
  "id": "2412.06784",
  "slug": "p3-po-prescriptive-point-priors-for-visuo-spatial-generalization-of-ro",
  "title": "P3-PO: Prescriptive Point Priors for Visuo-Spatial Generalization of Robot Policies",
  "abstract": "Developing generalizable robot policies that can robustly handle varied environmental conditions and object instances remains a fundamental challenge in robot learning. While considerable efforts have focused on collecting large robot datasets and developing policy architectures to learn from such data, naively learning from visual inputs often results in brittle policies that fail to transfer beyond the training data. This work presents Prescriptive Point Priors for Policies or P3-PO, a novel framework that constructs a unique state representation of the environment leveraging recent advances in computer vision and robot learning to achieve improved out-of-distribution generalization for robot manipulation. This representation is obtained through two steps. First, a human annotator prescribes a set of semantically meaningful points on a single demonstration frame. These points are then propagated through the dataset using off-the-shelf vision models. The derived points serve as an input to state-of-the-art policy architectures for policy learning. Our experiments across four real-world tasks demonstrate an overall 43% absolute improvement over prior methods when evaluated in identical settings as training. Further, P3-PO exhibits 58% and 80% gains across tasks for new object instances and more cluttered environments respectively. Videos illustrating the robot's performance are best viewed at point-priors.github.io.",
  "published": "2024-12-09",
  "updated": "2024-12-09",
  "year": "2024",
  "authors": [
   "Mara Levy",
   "Siddhant Haldar",
   "Lerrel Pinto",
   "Abhinav Shirivastava"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA",
  "venue_source": "semantic-scholar",
  "citations": 25,
  "influential_citations": 5,
  "tldr": "Prescriptive Point Priors for Policies or P3-PO is presented, a novel framework that constructs a unique state representation of the environment leveraging recent advances in computer vision and robot learning to achieve improved out-of-distribution generalization for robot manipulation.",
  "doi": "10.1109/ICRA55743.2025.11128755",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Mara Levy",
    "id": "2307465829",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Siddhant Haldar",
    "id": "51445278",
    "h_index": 13,
    "papers": 21
   },
   {
    "name": "Lerrel Pinto",
    "id": "2253567347",
    "h_index": 15,
    "papers": 25
   },
   {
    "name": "Abhinav Shirivastava",
    "id": "2334567031",
    "h_index": 1,
    "papers": 1
   }
  ],
  "comment": "",
  "topics": [
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2412.06784v1",
  "pdf_url": "https://arxiv.org/pdf/2412.06784v1",
  "html_url": "https://arxiv.org/html/2412.06784v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.91
 },
 {
  "id": "2412.05278",
  "slug": "birth-and-death-of-a-rose",
  "title": "Birth and Death of a Rose",
  "abstract": "We study the problem of generating temporal object intrinsics -- temporally evolving sequences of object geometry, reflectance, and texture, such as a blooming rose -- from pre-trained 2D foundation models. Unlike conventional 3D modeling and animation techniques that require extensive manual effort and expertise, we introduce a method that generates such assets with signals distilled from pre-trained 2D diffusion models. To ensure the temporal consistency of object intrinsics, we propose Neural Templates for temporal-state-guided distillation, derived automatically from image features from self-supervised learning. Our method can generate high-quality temporal object intrinsics for several natural phenomena and enable the sampling and controllable rendering of these dynamic objects from any viewpoint, under any environmental lighting conditions, at any time of their lifespan. Project website: https://chen-geng.com/rose4d",
  "published": "2024-12-06",
  "updated": "2025-06-05",
  "year": "2024",
  "authors": [
   "Chen Geng",
   "Yunzhi Zhang",
   "Shangzhe Wu",
   "Jiajun Wu"
  ],
  "author_count": 4,
  "categories": [
   "cs.CV",
   "cs.GR"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 6,
  "influential_citations": 0,
  "tldr": "This work introduces a method that generates high-quality temporal object intrinsics for several natural phenomena with signals distilled from pre-trained 2D diffusion models, and proposes Neural Templates for temporal-state-guided distillation, derived automatically from image features from self-supervised learning.",
  "doi": "10.1109/CVPR52734.2025.02431",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Chen Geng",
    "id": "2334353317",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Yunzhi Zhang",
    "id": "2261420360",
    "h_index": 14,
    "papers": 30
   },
   {
    "name": "Shangzhe Wu",
    "id": "2112538311",
    "h_index": 11,
    "papers": 21
   },
   {
    "name": "Jiajun Wu",
    "id": "2326444500",
    "h_index": 35,
    "papers": 214
   }
  ],
  "comment": "CVPR 2025 Oral. Project website: https://chen-geng.com/rose4d",
  "topics": [
   "foundation-pretraining",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2412.05278v2",
  "pdf_url": "https://arxiv.org/pdf/2412.05278v2",
  "html_url": "https://arxiv.org/html/2412.05278v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.35
 },
 {
  "id": "2412.04987",
  "slug": "flowpolicy-enabling-fast-and-robust-3d-flow-based-policy-via-consisten",
  "title": "FlowPolicy: Enabling Fast and Robust 3D Flow-based Policy via Consistency Flow Matching for Robot Manipulation",
  "abstract": "Robots can acquire complex manipulation skills by learning policies from expert demonstrations, which is often known as vision-based imitation learning. Generating policies based on diffusion and flow matching models has been shown to be effective, particularly in robotic manipulation tasks. However, recursion-based approaches are inference inefficient in working from noise distributions to policy distributions, posing a challenging trade-off between efficiency and quality. This motivates us to propose FlowPolicy, a novel framework for fast policy generation based on consistency flow matching and 3D vision. Our approach refines the flow dynamics by normalizing the self-consistency of the velocity field, enabling the model to derive task execution policies in a single inference step. Specifically, FlowPolicy conditions on the observed 3D point cloud, where consistency flow matching directly defines straight-line flows from different time states to the same action space, while simultaneously constraining their velocity values, that is, we approximate the trajectories from noise to robot actions by normalizing the self-consistency of the velocity field within the action space, thus improving the inference efficiency. We validate the effectiveness of FlowPolicy in Adroit and Metaworld, demonstrating a 7$\\times$ increase in inference speed while maintaining competitive average success rates compared to state-of-the-art methods. Code is available at https://github.com/zql-kk/FlowPolicy.",
  "published": "2024-12-06",
  "updated": "2024-12-15",
  "year": "2024",
  "authors": [
   "Qinglun Zhang",
   "Zhen Liu",
   "Haoqiang Fan",
   "Guanghui Liu",
   "Bing Zeng",
   "Shuaicheng Liu"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 134,
  "influential_citations": 14,
  "tldr": "The proposed FlowPolicy refines the flow dynamics by normalizing the self-consistency of the velocity field, enabling the model to derive task execution policies in a single inference step, thus improving the inference efficiency.",
  "doi": "10.48550/arXiv.2412.04987",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Qinglun Zhang",
    "id": "2334438982",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Zhen Liu",
    "id": "2329853953",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "Haoqiang Fan",
    "id": "2249400286",
    "h_index": 8,
    "papers": 13
   },
   {
    "name": "Guanghui Liu",
    "id": "2268757026",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Bing Zeng",
    "id": "2267261912",
    "h_index": 7,
    "papers": 26
   },
   {
    "name": "Shuaicheng Liu",
    "id": "2268797502",
    "h_index": 12,
    "papers": 28
   }
  ],
  "comment": "",
  "topics": [
   "imitation-diffusion",
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2412.04987v2",
  "pdf_url": "https://arxiv.org/pdf/2412.04987v2",
  "html_url": "https://arxiv.org/html/2412.04987v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.13
 },
 {
  "id": "2412.04814",
  "slug": "lift-leveraging-human-feedback-for-text-to-video-model-alignment",
  "title": "LiFT: Leveraging Human Feedback for Text-to-Video Model Alignment",
  "abstract": "Recent advances in text-to-video (T2V) generative models have shown impressive capabilities. However, these models are still inadequate in aligning synthesized videos with human preferences (e.g., accurately reflecting text descriptions), which is particularly difficult to address, as human preferences are subjective and challenging to formalize as objective functions. Existing studies train video quality assessment models that rely on human-annotated ratings for video evaluation but overlook the reasoning behind evaluations, limiting their ability to capture nuanced human criteria. Moreover, aligning T2V model using video-based human feedback remains unexplored. Therefore, this paper proposes LiFT, the first method designed to leverage human feedback for T2V model alignment. Specifically, we first construct a Human Rating Annotation dataset, LiFT-HRA, consisting of approximately 10k human annotations, each including a score and its corresponding rationale. Based on this, we train a reward model LiFT-Critic to learn reward function effectively, which serves as a proxy for human judgment, measuring the alignment between given videos and human expectations. Lastly, we leverage the learned reward function to align the T2V model by maximizing the reward-weighted likelihood. As a case study, we apply our pipeline to CogVideoX-2B, showing that the fine-tuned model outperforms the CogVideoX-5B across all 16 metrics, highlighting the potential of human feedback in improving the alignment and quality of synthesized videos.",
  "published": "2024-12-06",
  "updated": "2025-03-05",
  "year": "2024",
  "authors": [
   "Yibin Wang",
   "Zhiyu Tan",
   "Junyan Wang",
   "Xiaomeng Yang",
   "Cheng Jin",
   "Hao Li"
  ],
  "author_count": 6,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 60,
  "influential_citations": 5,
  "tldr": "This paper proposes LiFT, the first method designed to leverage human feedback for T2V model alignment, and trains a reward model LiFT-Critic to learn reward function effectively, which serves as a proxy for human judgment, measuring the alignment between given videos and human expectations.",
  "doi": "10.48550/arXiv.2412.04814",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yibin Wang",
    "id": "2267308173",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "Zhiyu Tan",
    "id": "2093185926",
    "h_index": 14,
    "papers": 42
   },
   {
    "name": "Junyan Wang",
    "id": "2290516780",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Xiaomeng Yang",
    "id": "2308043884",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Cheng Jin",
    "id": "2334445793",
    "h_index": 8,
    "papers": 9
   },
   {
    "name": "Hao Li",
    "id": "2290962436",
    "h_index": 10,
    "papers": 27
   }
  ],
  "comment": "Project page: https://codegoat24.github.io/LiFT",
  "topics": [
   "rl-control",
   "data-teleop",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2412.04814v3",
  "pdf_url": "https://arxiv.org/pdf/2412.04814v3",
  "html_url": "https://arxiv.org/html/2412.04814v3",
  "code_url": "https://codegoat24.github.io/LiFT",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.79
 },
 {
  "id": "2412.03603",
  "slug": "hunyuanvideo-a-systematic-framework-for-large-video-generative-models",
  "title": "HunyuanVideo: A Systematic Framework For Large Video Generative Models",
  "abstract": "Recent advancements in video generation have significantly impacted daily life for both individuals and industries. However, the leading video generation models remain closed-source, resulting in a notable performance gap between industry capabilities and those available to the public. In this report, we introduce HunyuanVideo, an innovative open-source video foundation model that demonstrates performance in video generation comparable to, or even surpassing, that of leading closed-source models. HunyuanVideo encompasses a comprehensive framework that integrates several key elements, including data curation, advanced architectural design, progressive model scaling and training, and an efficient infrastructure tailored for large-scale model training and inference. As a result, we successfully trained a video generative model with over 13 billion parameters, making it the largest among all open-source models. We conducted extensive experiments and implemented a series of targeted designs to ensure high visual quality, motion dynamics, text-video alignment, and advanced filming techniques. According to evaluations by professionals, HunyuanVideo outperforms previous state-of-the-art models, including Runway Gen-3, Luma 1.6, and three top-performing Chinese video generative models. By releasing the code for the foundation model and its applications, we aim to bridge the gap between closed-source and open-source communities. This initiative will empower individuals within the community to experiment with their ideas, fostering a more dynamic and vibrant video generation ecosystem. The code is publicly available at https://github.com/Tencent/HunyuanVideo.",
  "published": "2024-12-03",
  "updated": "2025-03-11",
  "year": "2024",
  "authors": [
   "Weijie Kong",
   "Qi Tian",
   "Zijian Zhang",
   "Rox Min",
   "Zuozhuo Dai",
   "Jin Zhou",
   "Jiangfeng Xiong",
   "Xin Li",
   "Bo Wu",
   "Jianwei Zhang",
   "Kathrina Wu",
   "Qin Lin",
   "Junkun Yuan",
   "Yanxin Long",
   "Aladdin Wang",
   "Andong Wang",
   "Changlin Li",
   "Duojun Huang",
   "Fang Yang",
   "Hao Tan",
   "Hongmei Wang",
   "Jacob Song",
   "Jiawang Bai",
   "Jianbing Wu",
   "Jinbao Xue",
   "Joey Wang",
   "Kai Wang",
   "Mengyang Liu",
   "Pengyu Li",
   "Shuai Li",
   "Weiyan Wang",
   "Wenqing Yu",
   "Xinchi Deng",
   "Yang Li",
   "Yi Chen",
   "Yutao Cui",
   "Yuanbo Peng",
   "Zhentao Yu",
   "Zhiyu He",
   "Zhiyong Xu",
   "Zixiang Zhou",
   "Zunnan Xu",
   "Yangyu Tao",
   "Qinglin Lu",
   "Songtao Liu",
   "Dax Zhou",
   "Hongfa Wang",
   "Yong Yang",
   "Di Wang",
   "Yuhong Liu",
   "Jie Jiang",
   "Caesar Zhong"
  ],
  "author_count": 52,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 1604,
  "influential_citations": 263,
  "tldr": "HunyuanVideo is introduced, an innovative open-source video foundation model that demonstrates performance in video generation comparable to, or even surpassing, that of leading closed-source models.",
  "doi": "10.48550/arXiv.2412.03603",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Weijie Kong",
    "id": "2333896208",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Qi Tian",
    "id": "2304750503",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Zijian Zhang",
    "id": "2292883051",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Rox Min",
    "id": "2333907917",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Zuozhuo Dai",
    "id": "2334031240",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Jin Zhou",
    "id": "2314553207",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Jia-Liang Xiong",
    "id": "2262109774",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Xin Li",
    "id": "2334331255",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Bo Wu",
    "id": "2333959995",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Jianwei Zhang",
    "id": "2301232669",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Kathrina Wu",
    "id": "2333906659",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Qin Lin",
    "id": "2301456725",
    "h_index": 11,
    "papers": 13
   },
   {
    "name": "Junkun Yuan",
    "id": "2304610230",
    "h_index": 18,
    "papers": 33
   },
   {
    "name": "Yanxin Long",
    "id": "2301272172",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Aladdin Wang",
    "id": "2334442817",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Andong Wang",
    "id": "2301515438",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Changlin Li",
    "id": "2333990892",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Duojun Huang",
    "id": "2333971771",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Fan Yang",
    "id": "2329438495",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Hao Tan",
    "id": "2262871912",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Hongmei Wang",
    "id": "2319390362",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Jacob Song",
    "id": "2329134064",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Jiawang Bai",
    "id": "2335519457",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Jianbing Wu",
    "id": "2259644290",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Jinbao Xue",
    "id": "2302814808",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Joey Wang",
    "id": "2333879277",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Kai Wang",
    "id": "2324210490",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Mengyang Liu",
    "id": "2268185479",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Pengyuan Li",
    "id": "2349807061",
    "h_index": 4,
    "papers": 28
   },
   {
    "name": "Shuai Li",
    "id": "2320225721",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Weiyan Wang",
    "id": "2292424837",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Wenqing Yu",
    "id": "2334328325",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Xi Deng",
    "id": "2269123220",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Yang Li",
    "id": "2334456131",
    "h_index": 8,
    "papers": 19
   },
   {
    "name": "Yi Chen",
    "id": "2333179401",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "Yutao Cui",
    "id": "2334226695",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Yuanbo Peng",
    "id": "2311693690",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Zhen Yu",
    "id": "2315128969",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Zhiyu He",
    "id": "2334227440",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Zhiyong Xu",
    "id": "2145166331",
    "h_index": 13,
    "papers": 30
   },
   {
    "name": "Zixiang Zhou",
    "id": "2244243543",
    "h_index": 8,
    "papers": 20
   },
   {
    "name": "Zunnan Xu",
    "id": "2224549852",
    "h_index": 15,
    "papers": 35
   },
   {
    "name": "Yang-Dan Tao",
    "id": "2267016579",
    "h_index": 8,
    "papers": 16
   },
   {
    "name": "Qinglin Lu",
    "id": "2333353148",
    "h_index": 13,
    "papers": 31
   },
   {
    "name": "Songtao Liu",
    "id": "2302788106",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Daquan Zhou",
    "id": "2333966740",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Hongfa Wang",
    "id": "2253816773",
    "h_index": 9,
    "papers": 11
   },
   {
    "name": "Yong Yang",
    "id": "2284866832",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Di Wang",
    "id": "2295556567",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Yuhong Liu",
    "id": "2309874739",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Jie Jiang",
    "id": "2308066874",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Caesar Zhong",
    "id": "2333978977",
    "h_index": 2,
    "papers": 2
   }
  ],
  "comment": "",
  "topics": [
   "foundation-pretraining",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2412.03603v6",
  "pdf_url": "https://arxiv.org/pdf/2412.03603v6",
  "html_url": "https://arxiv.org/html/2412.03603v6",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.0
 },
 {
  "id": "2412.02617",
  "slug": "improving-dynamic-object-interactions-in-text-to-video-generation-with",
  "title": "Improving Dynamic Object Interactions in Text-to-Video Generation with AI Feedback",
  "abstract": "Large text-to-video models hold immense potential for a wide range of downstream applications. However, they struggle to accurately depict dynamic object interactions, often resulting in unrealistic movements and frequent violations of real-world physics. One solution inspired by large language models is to align generated outputs with desired outcomes using external feedback. In this work, we investigate the use of feedback to enhance the quality of object dynamics in text-to-video models. We aim to answer a critical question: what types of feedback, paired with which specific self-improvement algorithms, can most effectively overcome movement misalignment and realistic object interactions? We first point out that offline RL-finetuning algorithms for text-to-video models can be equivalent as derived from a unified probabilistic objective. This perspective highlights that there is no algorithmically dominant method in principle; rather, we should care about the property of reward and data. While human feedback is less scalable, vision-language models could notice the video scenes as humans do. We then propose leveraging vision-language models to provide perceptual feedback specifically tailored to object dynamics in videos. Compared to popular video quality metrics measuring alignment or dynamics, the experiments demonstrate that our approach with binary AI feedback drives the most significant improvements in the quality of interaction scenes in video, as confirmed by AI, human, and quality metric evaluations. Notably, we observe substantial gains when using signals from vision language models, particularly in scenarios involving complex interactions between multiple objects and realistic depictions of objects falling.",
  "published": "2024-12-03",
  "updated": "2026-04-17",
  "year": "2024",
  "authors": [
   "Hiroki Furuta",
   "Heiga Zen",
   "Dale Schuurmans",
   "Aleksandra Faust",
   "Yutaka Matsuo",
   "Percy Liang",
   "Sherry Yang"
  ],
  "author_count": 7,
  "categories": [
   "cs.LG",
   "cs.AI",
   "cs.CV"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 46,
  "influential_citations": 1,
  "tldr": "This work investigates the use of feedback to enhance the quality of object dynamics in text-to-video models and proposes leveraging vision-language models to provide perceptual feedback specifically tailored to object dynamics in videos.",
  "doi": "10.48550/arXiv.2412.02617",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hiroki Furuta",
    "id": "2052903664",
    "h_index": 16,
    "papers": 31
   },
   {
    "name": "H. Zen",
    "id": "1691713",
    "h_index": 52,
    "papers": 153
   },
   {
    "name": "Dale Schuurmans",
    "id": "2265994815",
    "h_index": 11,
    "papers": 32
   },
   {
    "name": "Aleksandra Faust",
    "id": "2268757423",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Yutaka Matsuo",
    "id": "2320464508",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Percy Liang",
    "id": "2325905225",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Sherry Yang",
    "id": "2287810415",
    "h_index": 7,
    "papers": 8
   }
  ],
  "comment": "Website: https://sites.google.com/view/aif-dynamic-t2v/",
  "topics": [
   "rl-control",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2412.02617v2",
  "pdf_url": "https://arxiv.org/pdf/2412.02617v2",
  "html_url": "https://arxiv.org/html/2412.02617v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.67
 },
 {
  "id": "2411.19167",
  "slug": "hot3d-hand-and-object-tracking-in-3d-from-egocentric-multi-view-videos",
  "title": "HOT3D: Hand and Object Tracking in 3D from Egocentric Multi-View Videos",
  "abstract": "We introduce HOT3D, a publicly available dataset for egocentric hand and object tracking in 3D. The dataset offers over 833 minutes (3.7M+ images) of recordings that feature 19 subjects interacting with 33 diverse rigid objects. In addition to simple pick-up, observe, and put-down actions, the subjects perform actions typical for a kitchen, office, and living room environment. The recordings include multiple synchronized data streams containing egocentric multi-view RGB/monochrome images, eye gaze signal, scene point clouds, and 3D poses of cameras, hands, and objects. The dataset is recorded with two headsets from Meta: Project Aria, which is a research prototype of AI glasses, and Quest 3, a virtual-reality headset that has shipped millions of units. Ground-truth poses were obtained by a motion-capture system using small optical markers attached to hands and objects. Hand annotations are provided in the UmeTrack and MANO formats, and objects are represented by 3D meshes with PBR materials obtained by an in-house scanner. In our experiments, we demonstrate the effectiveness of multi-view egocentric data for three popular tasks: 3D hand tracking, model-based 6DoF object pose estimation, and 3D lifting of unknown in-hand objects. The evaluated multi-view methods, whose benchmarking is uniquely enabled by HOT3D, significantly outperform their single-view counterparts.",
  "published": "2024-11-28",
  "updated": "2025-04-30",
  "year": "2024",
  "authors": [
   "Prithviraj Banerjee",
   "Sindi Shkodrani",
   "Pierre Moulon",
   "Shreyas Hampali",
   "Shangchen Han",
   "Fan Zhang",
   "Linguang Zhang",
   "Jade Fountain",
   "Edward Miller",
   "Selen Basol",
   "Richard Newcombe",
   "Robert Wang",
   "Jakob Julian Engel",
   "Tomas Hodan"
  ],
  "author_count": 14,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 141,
  "influential_citations": 16,
  "tldr": "In these experiments, the effectiveness of multi-view egocentric data for three popular tasks: 3D hand tracking, model-based 6DoF object pose estimation, and 3D lifting of unknown in-hand objects is demonstrated.",
  "doi": "10.1109/CVPR52734.2025.00662",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Prithviraj Banerjee",
    "id": "2306783561",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Sindi Shkodrani",
    "id": "51208845",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Pierre Moulon",
    "id": "2284863436",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Shreyas Hampali",
    "id": "150296901",
    "h_index": 11,
    "papers": 20
   },
   {
    "name": "Shangchen Han",
    "id": "10461747",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "Fan Zhang",
    "id": "2307188098",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Linguang Zhang",
    "id": "2240167296",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Jade Fountain",
    "id": "2306783804",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Edward Miller",
    "id": "2234024715",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Selen Basol",
    "id": "2904166",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Richard A. Newcombe",
    "id": "2292257340",
    "h_index": 9,
    "papers": 20
   },
   {
    "name": "Robert Wang",
    "id": "2307017739",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "J. Engel",
    "id": "2241357086",
    "h_index": 9,
    "papers": 13
   },
   {
    "name": "Tomas Hodan",
    "id": "2396902",
    "h_index": 21,
    "papers": 33
   }
  ],
  "comment": "CVPR 2025",
  "topics": [
   "dexterous-manipulation",
   "egocentric-data",
   "spatial-3d",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2411.19167v2",
  "pdf_url": "https://arxiv.org/pdf/2411.19167v2",
  "html_url": "https://arxiv.org/html/2411.19167v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.65
 },
 {
  "id": "2412.01791",
  "slug": "dextrah-rgb-visuomotor-policies-to-grasp-anything-with-dexterous-hands",
  "title": "DextrAH-RGB: Visuomotor Policies to Grasp Anything with Dexterous Hands",
  "abstract": "One of the most important, yet challenging, skills for a dexterous robot is grasping a diverse range of objects. Much of the prior work has been limited by speed, generality, or reliance on depth maps and object poses. In this paper, we introduce DextrAH-RGB, a system that can perform dexterous arm-hand grasping end-to-end from RGB image input. We train a privileged fabric-guided policy (FGP) in simulation through reinforcement learning that acts on a geometric fabric controller to dexterously grasp a wide variety of objects. We then distill this privileged FGP into a RGB-based FGP strictly in simulation using photorealistic tiled rendering. To our knowledge, this is the first work that is able to demonstrate robust sim2real transfer of an end2end RGB-based policy for complex, dynamic, contact-rich tasks such as dexterous grasping. DextrAH-RGB is competitive with depth-based dexterous grasping policies, and generalizes to novel objects with unseen geometry, texture, and lighting conditions in the real world. Videos of our system grasping a diverse range of unseen objects are available at \\url{https://dextrah-rgb.github.io/}.",
  "published": "2024-11-27",
  "updated": "2025-02-01",
  "year": "2024",
  "authors": [
   "Ritvik Singh",
   "Arthur Allshire",
   "Ankur Handa",
   "Nathan Ratliff",
   "Karl Van Wyk"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 67,
  "influential_citations": 2,
  "tldr": "This paper introduces DextrAH-RGB, a system that can perform dexterous arm-hand grasping end-to-end from RGB image input, and trains a privileged fabric-guided policy in simulation through reinforcement learning that acts on a geometric fabric controller to dexterously grasp a wide variety of objects.",
  "doi": "10.48550/arXiv.2412.01791",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ritvik Singh",
    "id": "2328020615",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Arthur Allshire",
    "id": "2061149217",
    "h_index": 9,
    "papers": 12
   },
   {
    "name": "Ankur Handa",
    "id": "2328010136",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Nathan D. Ratliff",
    "id": "2240527931",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "Karl Van Wyk",
    "id": "2423933",
    "h_index": 20,
    "papers": 54
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "sim2real",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2412.01791v2",
  "pdf_url": "https://arxiv.org/pdf/2412.01791v2",
  "html_url": "https://arxiv.org/html/2412.01791v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.83
 },
 {
  "id": "2411.18613",
  "slug": "cat4d-create-anything-in-4d-with-multi-view-video-diffusion-models",
  "title": "CAT4D: Create Anything in 4D with Multi-View Video Diffusion Models",
  "abstract": "We present CAT4D, a method for creating 4D (dynamic 3D) scenes from monocular video. CAT4D leverages a multi-view video diffusion model trained on a diverse combination of datasets to enable novel view synthesis at any specified camera poses and timestamps. Combined with a novel sampling approach, this model can transform a single monocular video into a multi-view video, enabling robust 4D reconstruction via optimization of a deformable 3D Gaussian representation. We demonstrate competitive performance on novel view synthesis and dynamic scene reconstruction benchmarks, and highlight the creative capabilities for 4D scene generation from real or generated videos. See our project page for results and interactive demos: https://cat-4d.github.io/.",
  "published": "2024-11-27",
  "updated": "2024-12-18",
  "year": "2024",
  "authors": [
   "Rundi Wu",
   "Ruiqi Gao",
   "Ben Poole",
   "Alex Trevithick",
   "Changxi Zheng",
   "Jonathan T. Barron",
   "Aleksander Holynski"
  ],
  "author_count": 7,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 172,
  "influential_citations": 9,
  "tldr": "CAT4D leverages a multi-view video diffusion model trained on a diverse combination of datasets to enable novel view synthesis at any specified camera poses and timestamps, enabling robust 4D reconstruction via optimization of a deformable 3D Gaussian representation.",
  "doi": "10.1109/CVPR52734.2025.02427",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Rundi Wu",
    "id": "1406236938",
    "h_index": 13,
    "papers": 21
   },
   {
    "name": "Ruiqi Gao",
    "id": "2269735504",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Ben Poole",
    "id": "2269733468",
    "h_index": 7,
    "papers": 8
   },
   {
    "name": "Alex Trevithick",
    "id": "1993540993",
    "h_index": 8,
    "papers": 14
   },
   {
    "name": "Changxi Zheng",
    "id": "2297884549",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Jonathan T. Barron",
    "id": "2279837686",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Aleksander Holynski",
    "id": "2248172435",
    "h_index": 12,
    "papers": 17
   }
  ],
  "comment": "Project page: https://cat-4d.github.io/",
  "topics": [
   "spatial-3d",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2411.18613v2",
  "pdf_url": "https://arxiv.org/pdf/2411.18613v2",
  "html_url": "https://arxiv.org/html/2411.18613v2",
  "code_url": "https://cat-4d.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.74
 },
 {
  "id": "2411.18369",
  "slug": "g3flow-generative-3d-semantic-flow-for-pose-aware-and-generalizable-ob",
  "title": "G3Flow: Generative 3D Semantic Flow for Pose-aware and Generalizable Object Manipulation",
  "abstract": "Recent advances in imitation learning for 3D robotic manipulation have shown promising results with diffusion-based policies. However, achieving human-level dexterity requires seamless integration of geometric precision and semantic understanding. We present G3Flow, a novel framework that constructs real-time semantic flow, a dynamic, object-centric 3D semantic representation by leveraging foundation models. Our approach uniquely combines 3D generative models for digital twin creation, vision foundation models for semantic feature extraction, and robust pose tracking for continuous semantic flow updates. This integration enables complete semantic understanding even under occlusions while eliminating manual annotation requirements. By incorporating semantic flow into diffusion policies, we demonstrate significant improvements in both terminal-constrained manipulation and cross-object generalization. Extensive experiments across five simulation tasks show that G3Flow consistently outperforms existing approaches, achieving up to 68.3% and 50.1% average success rates on terminal-constrained manipulation and cross-object generalization tasks respectively. Our results demonstrate the effectiveness of G3Flow in enhancing real-time dynamic semantic feature understanding for robotic manipulation policies.",
  "published": "2024-11-27",
  "updated": "2025-06-22",
  "year": "2024",
  "authors": [
   "Tianxing Chen",
   "Yao Mu",
   "Zhixuan Liang",
   "Zanxin Chen",
   "Shijia Peng",
   "Qiangyu Chen",
   "Mingkun Xu",
   "Ruizhen Hu",
   "Hongyuan Zhang",
   "Xuelong Li",
   "Ping Luo"
  ],
  "author_count": 11,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 49,
  "influential_citations": 3,
  "tldr": "G3Flow is a novel framework that constructs real-time semantic flow, a dynamic, object-centric 3D semantic representation by leveraging foundation models that uniquely combines 3D generative models for digital twin creation, vision foundation models for semantic feature extraction, and robust pose tracking for continuous semantic flow updates.",
  "doi": "10.1109/CVPR52734.2025.00169",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Tianxing Chen",
    "id": "2316455829",
    "h_index": 10,
    "papers": 31
   },
   {
    "name": "Yao Mu",
    "id": "2248348669",
    "h_index": 12,
    "papers": 16
   },
   {
    "name": "Zhixuan Liang",
    "id": "2257485304",
    "h_index": 10,
    "papers": 35
   },
   {
    "name": "Zanxin Chen",
    "id": "2319612137",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Shijia Peng",
    "id": "2319812486",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Qiangyu Chen",
    "id": "2332909061",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Min Xu",
    "id": "2273795815",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Ruizhen Hu",
    "id": "2284945736",
    "h_index": 8,
    "papers": 25
   },
   {
    "name": "Hongyuan Zhang",
    "id": "2108879612",
    "h_index": 18,
    "papers": 48
   },
   {
    "name": "Xuelong Li",
    "id": "2332601293",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Ping Luo",
    "id": "2257349788",
    "h_index": 8,
    "papers": 23
   }
  ],
  "comment": "Webpage: https://tianxingchen.github.io/G3Flow/, accepted to CVPR 2025",
  "topics": [
   "dexterous-manipulation",
   "imitation-diffusion",
   "foundation-pretraining",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2411.18369v3",
  "pdf_url": "https://arxiv.org/pdf/2411.18369v3",
  "html_url": "https://arxiv.org/html/2411.18369v3",
  "code_url": "https://tianxingchen.github.io/G3Flow/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.2
 },
 {
  "id": "2411.17636",
  "slug": "malmm-multi-agent-large-language-models-for-zero-shot-robotics-manipul",
  "title": "MALMM: Multi-Agent Large Language Models for Zero-Shot Robotics Manipulation",
  "abstract": "Large Language Models (LLMs) have demonstrated remarkable planning abilities across various domains, including robotics manipulation and navigation. While recent efforts in robotics have leveraged LLMs both for high-level and low-level planning, these approaches often face significant challenges, such as hallucinations in long-horizon tasks and limited adaptability due to the generation of plans in a single pass without real-time feedback. To address these limitations, we propose a novel multi-agent LLM framework, Multi-Agent Large Language Model for Manipulation (MALMM) that distributes high-level planning and low-level control code generation across specialized LLM agents, supervised by an additional agent that dynamically manages transitions. By incorporating observations from the environment after each step, our framework effectively handles intermediate failures and enables adaptive re-planning. Unlike existing methods, our approach does not rely on pre-trained skill policies or in-context learning examples and generalizes to a variety of new tasks. We evaluate our approach on nine RLBench tasks, including long-horizon tasks, and demonstrate its ability to solve robotics manipulation in a zero-shot setting, thereby overcoming key limitations of existing LLM-based manipulation methods.",
  "published": "2024-11-26",
  "updated": "2025-08-25",
  "year": "2024",
  "authors": [
   "Harsh Singh",
   "Rocktim Jyoti Das",
   "Mingfei Han",
   "Preslav Nakov",
   "Ivan Laptev"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS",
  "venue_source": "semantic-scholar",
  "citations": 27,
  "influential_citations": 0,
  "tldr": "A novel multi-agent LLM framework, Multi-Agent Large Language Model for Manipulation (MALMM), which demonstrates excellent performance in solving previously unseen long-horizon manipulation tasks, and outperforms existing zero-shot LLM-based methods in RLBench by a large margin.",
  "doi": "10.1109/IROS60139.2025.11247340",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Harsh Singh",
    "id": "2332305613",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Rocktim Jyoti Das",
    "id": "2211732585",
    "h_index": 9,
    "papers": 19
   },
   {
    "name": "Mingfei Han",
    "id": "2334521542",
    "h_index": 5,
    "papers": 15
   },
   {
    "name": "Preslav Nakov",
    "id": "2026545715",
    "h_index": 49,
    "papers": 334
   },
   {
    "name": "Ivan Laptev",
    "id": "2311508734",
    "h_index": 10,
    "papers": 21
   }
  ],
  "comment": "48 pages",
  "topics": [
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2411.17636v2",
  "pdf_url": "https://arxiv.org/pdf/2411.17636v2",
  "html_url": "https://arxiv.org/html/2411.17636v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.95
 },
 {
  "id": "2411.13677",
  "slug": "bimanual-dexterity-for-complex-tasks",
  "title": "Bimanual Dexterity for Complex Tasks",
  "abstract": "To train generalist robot policies, machine learning methods often require a substantial amount of expert human teleoperation data. An ideal robot for humans collecting data is one that closely mimics them: bimanual arms and dexterous hands. However, creating such a bimanual teleoperation system with over 50 DoF is a significant challenge. To address this, we introduce Bidex, an extremely dexterous, low-cost, low-latency and portable bimanual dexterous teleoperation system which relies on motion capture gloves and teacher arms. We compare Bidex to a Vision Pro teleoperation system and a SteamVR system and find Bidex to produce better quality data for more complex tasks at a faster rate. Additionally, we show Bidex operating a mobile bimanual robot for in the wild tasks. The robot hands (5k USD) and teleoperation system (7k USD) is readily reproducible and can be used on many robot arms including two xArms (16k USD). Website at https://bidex-teleop.github.io/",
  "published": "2024-11-20",
  "updated": "2024-11-20",
  "year": "2024",
  "authors": [
   "Kenneth Shaw",
   "Yulong Li",
   "Jiahui Yang",
   "Mohan Kumar Srirama",
   "Ray Liu",
   "Haoyu Xiong",
   "Russell Mendonca",
   "Deepak Pathak"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 56,
  "influential_citations": 5,
  "tldr": "This work introduces Bidex, an extremely dexterous, low-cost, low-latency and portable bimanual dexterous teleoperation system which relies on motion capture gloves and teacher arms and finds Bidex to produce better quality data for more complex tasks at a faster rate.",
  "doi": "10.48550/arXiv.2411.13677",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Kenneth Shaw",
    "id": "2263541750",
    "h_index": 10,
    "papers": 18
   },
   {
    "name": "Yulong Li",
    "id": "2331686482",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Jiahui Yang",
    "id": "2320289466",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "M. K. Srirama",
    "id": "2193493900",
    "h_index": 12,
    "papers": 14
   },
   {
    "name": "R. Liu",
    "id": "2331677309",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Haoyu Xiong",
    "id": "2281036863",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "R. Mendonca",
    "id": "35509365",
    "h_index": 15,
    "papers": 24
   },
   {
    "name": "Deepak Pathak",
    "id": "2269734979",
    "h_index": 11,
    "papers": 17
   }
  ],
  "comment": "In CoRL 2024. Website at https://bidex-teleop.github.io/",
  "topics": [
   "dexterous-manipulation",
   "egocentric-data",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2411.13677v1",
  "pdf_url": "https://arxiv.org/pdf/2411.13677v1",
  "html_url": "https://arxiv.org/html/2411.13677v1",
  "code_url": "https://bidex-teleop.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.26
 },
 {
  "id": "2411.11839",
  "slug": "robogsim-a-real2sim2real-robotic-gaussian-splatting-simulator",
  "title": "RoboGSim: A Real2Sim2Real Robotic Gaussian Splatting Simulator",
  "abstract": "Efficient acquisition of real-world embodied data has been increasingly critical. However, large-scale demonstrations captured by remote operation tend to take extremely high costs and fail to scale up the data size in an efficient manner. Sampling the episodes under a simulated environment is a promising way for large-scale collection while existing simulators fail to high-fidelity modeling on texture and physics. To address these limitations, we introduce the RoboGSim, a real2sim2real robotic simulator, powered by 3D Gaussian Splatting and the physics engine. RoboGSim mainly includes four parts: Gaussian Reconstructor, Digital Twins Builder, Scene Composer, and Interactive Engine. It can synthesize the simulated data with novel views, objects, trajectories, and scenes. RoboGSim also provides an online, reproducible, and safe evaluation for different manipulation policies. The real2sim and sim2real transfer experiments show a high consistency in the texture and physics. We compared the test results of RoboGSim data and real robot data on both RoboGSim and real robot platforms. The experimental results show that the RoboGSim data model can achieve zero-shot performance on the real robot, with results comparable to real robot data. Additionally, in experiments with novel perspectives and novel scenes, the RoboGSim data model performed even better on the real robot than the real robot data model. This not only helps reduce the sim2real gap but also addresses the limitations of real robot data collection, such as its single-source and high cost. We hope RoboGSim serves as a closed-loop simulator for fair comparison on policy learning. More information can be found on our project page https://robogsim.github.io/.",
  "published": "2024-11-18",
  "updated": "2025-08-03",
  "year": "2024",
  "authors": [
   "Xinhai Li",
   "Jialin Li",
   "Ziheng Zhang",
   "Rui Zhang",
   "Fan Jia",
   "Tiancai Wang",
   "Haoqiang Fan",
   "Kuo-Kun Tseng",
   "Ruiping Wang"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 77,
  "influential_citations": 3,
  "tldr": "The experimental results show that the RoboGSim data model can achieve zero-shot performance on the real robot, with results comparable to real robot data, and the real2sim and sim2real transfer experiments show a high consistency in the texture and physics.",
  "doi": "10.48550/arXiv.2411.11839",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xinhai Li",
    "id": "2267385564",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Jialin Li",
    "id": "2303433806",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Ziheng Zhang",
    "id": "2303885890",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Rui Zhang",
    "id": "2325810526",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Fan Jia",
    "id": "2325726966",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Tiancai Wang",
    "id": "2325923837",
    "h_index": 10,
    "papers": 30
   },
   {
    "name": "Haoqiang Fan",
    "id": "2326357387",
    "h_index": 8,
    "papers": 14
   },
   {
    "name": "Kuo-Kun Tseng",
    "id": "2267332483",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Ruiping Wang",
    "id": "2331377109",
    "h_index": 3,
    "papers": 4
   }
  ],
  "comment": "",
  "topics": [
   "sim2real",
   "spatial-3d",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2411.11839v2",
  "pdf_url": "https://arxiv.org/pdf/2411.11839v2",
  "html_url": "https://arxiv.org/html/2411.11839v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.89
 },
 {
  "id": "2411.07223",
  "slug": "grounding-video-models-to-actions-through-goal-conditioned-exploration",
  "title": "Grounding Video Models to Actions through Goal Conditioned Exploration",
  "abstract": "Large video models, pretrained on massive amounts of Internet video, provide a rich source of physical knowledge about the dynamics and motions of objects and tasks. However, video models are not grounded in the embodiment of an agent, and do not describe how to actuate the world to reach the visual states depicted in a video. To tackle this problem, current methods use a separate vision-based inverse dynamic model trained on embodiment-specific data to map image states to actions. Gathering data to train such a model is often expensive and challenging, and this model is limited to visual settings similar to the ones in which data are available. In this paper, we investigate how to directly ground video models to continuous actions through self-exploration in the embodied environment -- using generated video states as visual goals for exploration. We propose a framework that uses trajectory level action generation in combination with video guidance to enable an agent to solve complex tasks without any external supervision, e.g., rewards, action labels, or segmentation masks. We validate the proposed approach on 8 tasks in Libero, 6 tasks in MetaWorld, 4 tasks in Calvin, and 12 tasks in iThor Visual Navigation. We show how our approach is on par with or even surpasses multiple behavior cloning baselines trained on expert demonstrations while without requiring any action annotations.",
  "published": "2024-11-11",
  "updated": "2025-03-12",
  "year": "2024",
  "authors": [
   "Yunhao Luo",
   "Yilun Du"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 38,
  "influential_citations": 4,
  "tldr": "This paper proposes a framework that uses trajectory level action generation in combination with video guidance to enable an agent to solve complex tasks without any external supervision, e.g., rewards, action labels, or segmentation masks.",
  "doi": "10.48550/arXiv.2411.07223",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yunhao Luo",
    "id": "2308053404",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Yilun Du",
    "id": "2330228153",
    "h_index": 2,
    "papers": 6
   }
  ],
  "comment": "ICLR 2025 (Spotlight). Project page: https://video-to-action.github.io/",
  "topics": [
   "imitation-diffusion",
   "navigation",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2411.07223v2",
  "pdf_url": "https://arxiv.org/pdf/2411.07223v2",
  "html_url": "https://arxiv.org/html/2411.07223v2",
  "code_url": "https://video-to-action.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.09
 },
 {
  "id": "2411.04983",
  "slug": "dino-wm-world-models-on-pre-trained-visual-features-enable-zero-shot-p",
  "title": "DINO-WM: World Models on Pre-trained Visual Features enable Zero-shot Planning",
  "abstract": "The ability to predict future outcomes given control actions is fundamental for physical reasoning. However, such predictive models, often called world models, remains challenging to learn and are typically developed for task-specific solutions with online policy learning. To unlock world models' true potential, we argue that they should 1) be trainable on offline, pre-collected trajectories, 2) support test-time behavior optimization, and 3) facilitate task-agnostic reasoning. To this end, we present DINO World Model (DINO-WM), a new method to model visual dynamics without reconstructing the visual world. DINO-WM leverages spatial patch features pre-trained with DINOv2, enabling it to learn from offline behavioral trajectories by predicting future patch features. This allows DINO-WM to achieve observational goals through action sequence optimization, facilitating task-agnostic planning by treating goal features as prediction targets. We demonstrate that DINO-WM achieves zero-shot behavioral solutions at test time on six environments without expert demonstrations, reward modeling, or pre-learned inverse models, outperforming prior state-of-the-art work across diverse task families such as arbitrarily configured mazes, push manipulation with varied object shapes, and multi-particle scenarios.",
  "published": "2024-11-07",
  "updated": "2025-02-01",
  "year": "2024",
  "authors": [
   "Gaoyue Zhou",
   "Hengkai Pan",
   "Yann LeCun",
   "Lerrel Pinto"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "ICML",
  "venue_source": "semantic-scholar",
  "citations": 321,
  "influential_citations": 59,
  "tldr": "DINO World Model (DINO-WM), a new method to model visual dynamics without reconstructing the visual world, is presented, which achieves zero-shot behavioral solutions at test time on six environments without expert demonstrations, reward modeling, or pre-learned inverse models.",
  "doi": "10.48550/arXiv.2411.04983",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Gaoyue Zhou",
    "id": "2257386929",
    "h_index": 9,
    "papers": 12
   },
   {
    "name": "Hengkai Pan",
    "id": "2323521814",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Yann LeCun",
    "id": "2270469816",
    "h_index": 12,
    "papers": 46
   },
   {
    "name": "Lerrel Pinto",
    "id": "2253567347",
    "h_index": 15,
    "papers": 25
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2411.04983v2",
  "pdf_url": "https://arxiv.org/pdf/2411.04983v2",
  "html_url": "https://arxiv.org/html/2411.04983v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.01
 },
 {
  "id": "2411.03682",
  "slug": "legato-cross-embodiment-imitation-using-a-grasping-tool",
  "title": "LEGATO: Cross-Embodiment Imitation Using a Grasping Tool",
  "abstract": "Cross-embodiment imitation learning enables policies trained on specific embodiments to transfer across different robots, unlocking the potential for large-scale imitation learning that is both cost-effective and highly reusable. This paper presents LEGATO, a cross-embodiment imitation learning framework for visuomotor skill transfer across varied kinematic morphologies. We introduce a handheld gripper that unifies action and observation spaces, allowing tasks to be defined consistently across robots. We train visuomotor policies on task demonstrations using this gripper through imitation learning, applying transformation to a motion-invariant space for computing the training loss. Gripper motions generated by the policies are retargeted into high-degree-of-freedom whole-body motions using inverse kinematics for deployment across diverse embodiments. Our evaluations in simulation and real-robot experiments highlight the framework's effectiveness in learning and transferring visuomotor skills across various robots. More information can be found on the project page: https://ut-hcrl.github.io/LEGATO.",
  "published": "2024-11-06",
  "updated": "2025-02-19",
  "year": "2024",
  "authors": [
   "Mingyo Seo",
   "H. Andy Park",
   "Shenli Yuan",
   "Yuke Zhu",
   "Luis Sentis"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "RA-L",
  "venue_source": "semantic-scholar",
  "citations": 35,
  "influential_citations": 2,
  "tldr": "",
  "doi": "10.1109/LRA.2025.3535182",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Mingyo Seo",
    "id": "23190833",
    "h_index": 7,
    "papers": 12
   },
   {
    "name": "H. A. Park",
    "id": "2229122169",
    "h_index": 6,
    "papers": 6
   },
   {
    "name": "Shenli Yuan",
    "id": "2315661821",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Yuke Zhu",
    "id": "2322736301",
    "h_index": 13,
    "papers": 21
   },
   {
    "name": "Luis Sentis",
    "id": "2237810419",
    "h_index": 6,
    "papers": 18
   }
  ],
  "comment": "Published in RA-L",
  "topics": [
   "dexterous-manipulation",
   "humanoids",
   "imitation-diffusion",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2411.03682v3",
  "pdf_url": "https://arxiv.org/pdf/2411.03682v3",
  "html_url": "https://arxiv.org/html/2411.03682v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.06
 },
 {
  "id": "2411.02385",
  "slug": "how-far-is-video-generation-from-world-model-a-physical-law-perspectiv",
  "title": "How Far is Video Generation from World Model: A Physical Law Perspective",
  "abstract": "OpenAI's Sora highlights the potential of video generation for developing world models that adhere to fundamental physical laws. However, the ability of video generation models to discover such laws purely from visual data without human priors can be questioned. A world model learning the true law should give predictions robust to nuances and correctly extrapolate on unseen scenarios. In this work, we evaluate across three key scenarios: in-distribution, out-of-distribution, and combinatorial generalization. We developed a 2D simulation testbed for object movement and collisions to generate videos deterministically governed by one or more classical mechanics laws. This provides an unlimited supply of data for large-scale experimentation and enables quantitative evaluation of whether the generated videos adhere to physical laws. We trained diffusion-based video generation models to predict object movements based on initial frames. Our scaling experiments show perfect generalization within the distribution, measurable scaling behavior for combinatorial generalization, but failure in out-of-distribution scenarios. Further experiments reveal two key insights about the generalization mechanisms of these models: (1) the models fail to abstract general physical rules and instead exhibit \"case-based\" generalization behavior, i.e., mimicking the closest training example; (2) when generalizing to new cases, models are observed to prioritize different factors when referencing training data: color > size > velocity > shape. Our study suggests that scaling alone is insufficient for video generation models to uncover fundamental physical laws, despite its role in Sora's broader success. See our project page at https://phyworld.github.io",
  "published": "2024-11-04",
  "updated": "2025-06-22",
  "year": "2024",
  "authors": [
   "Bingyi Kang",
   "Yang Yue",
   "Rui Lu",
   "Zhijie Lin",
   "Yang Zhao",
   "Kaixin Wang",
   "Gao Huang",
   "Jiashi Feng"
  ],
  "author_count": 8,
  "categories": [
   "cs.CV",
   "cs.AI"
  ],
  "primary_category": "cs.CV",
  "venue": "ICML",
  "venue_source": "semantic-scholar",
  "citations": 214,
  "influential_citations": 16,
  "tldr": "This study suggests that scaling alone is insufficient for video generation models to uncover fundamental physical laws, despite its role in Sora's broader success.",
  "doi": "10.48550/arXiv.2411.02385",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Bingyi Kang",
    "id": "2256992544",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Yang Yue",
    "id": "2256993684",
    "h_index": 13,
    "papers": 28
   },
   {
    "name": "Rui Lu",
    "id": "2055872341",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Zhijie Lin",
    "id": "2266462787",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Yang Zhao",
    "id": "2324632295",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Kaixin Wang",
    "id": "2283624439",
    "h_index": 8,
    "papers": 25
   },
   {
    "name": "Gao Huang",
    "id": "2329326400",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Jiashi Feng",
    "id": "2276580610",
    "h_index": 16,
    "papers": 24
   }
  ],
  "comment": "ICML 2025",
  "topics": [
   "world-models",
   "video-generation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2411.02385v2",
  "pdf_url": "https://arxiv.org/pdf/2411.02385v2",
  "html_url": "https://arxiv.org/html/2411.02385v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.83
 },
 {
  "id": "2411.02214",
  "slug": "dexhub-and-dart-towards-internet-scale-robot-data-collection",
  "title": "DexHub and DART: Towards Internet Scale Robot Data Collection",
  "abstract": "The quest to build a generalist robotic system is impeded by the scarcity of diverse and high-quality data. While real-world data collection effort exist, requirements for robot hardware, physical environment setups, and frequent resets significantly impede the scalability needed for modern learning frameworks. We introduce DART, a teleoperation platform designed for crowdsourcing that reimagines robotic data collection by leveraging cloud-based simulation and augmented reality (AR) to address many limitations of prior data collection efforts. Our user studies highlight that DART enables higher data collection throughput and lower physical fatigue compared to real-world teleoperation. We also demonstrate that policies trained using DART-collected datasets successfully transfer to reality and are robust to unseen visual disturbances. All data collected through DART is automatically stored in our cloud-hosted database, DexHub, which will be made publicly available upon curation, paving the path for DexHub to become an ever-growing data hub for robot learning. Videos are available at: https://dexhub.ai/project",
  "published": "2024-11-04",
  "updated": "2024-11-04",
  "year": "2024",
  "authors": [
   "Younghyo Park",
   "Jagdeep Singh Bhatia",
   "Lars Ankile",
   "Pulkit Agrawal"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 27,
  "influential_citations": 1,
  "tldr": "DART is introduced, a teleoperation platform designed for crowdsourcing that reimagines robotic data collection by leveraging cloud-based simulation and augmented reality (AR) to address many limitations of prior data collection efforts.",
  "doi": "10.48550/arXiv.2411.02214",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Younghyo Park",
    "id": "2313155531",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Jagdeep Singh Bhatia",
    "id": "2276637979",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Lars Ankile",
    "id": "2378852823",
    "h_index": 7,
    "papers": 9
   },
   {
    "name": "Pulkit Agrawal",
    "id": "2257003971",
    "h_index": 18,
    "papers": 41
   }
  ],
  "comment": "Visit https://dexhub.ai/project for more details",
  "topics": [
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2411.02214v1",
  "pdf_url": "https://arxiv.org/pdf/2411.02214v1",
  "html_url": "https://arxiv.org/html/2411.02214v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.45
 },
 {
  "id": "2411.00965",
  "slug": "spot-se-3-pose-trajectory-diffusion-for-object-centric-manipulation",
  "title": "SPOT: SE(3) Pose Trajectory Diffusion for Object-Centric Manipulation",
  "abstract": "We introduce SPOT, an object-centric imitation learning framework. The key idea is to capture each task by an object-centric representation, specifically the SE(3) object pose trajectory relative to the target. This approach decouples embodiment actions from sensory inputs, facilitating learning from various demonstration types, including both action-based and action-less human hand demonstrations, as well as cross-embodiment generalization. Additionally, object pose trajectories inherently capture planning constraints from demonstrations without the need for manually-crafted rules. To guide the robot in executing the task, the object trajectory is used to condition a diffusion policy. We systematically evaluate our method on simulation and real-world tasks. In real-world evaluation, using only eight demonstrations shot on an iPhone, our approach completed all tasks while fully complying with task constraints. Project page: https://nvlabs.github.io/object_centric_diffusion",
  "published": "2024-11-01",
  "updated": "2025-05-13",
  "year": "2024",
  "authors": [
   "Cheng-Chun Hsu",
   "Bowen Wen",
   "Jie Xu",
   "Yashraj Narang",
   "Xiaolong Wang",
   "Yuke Zhu",
   "Joydeep Biswas",
   "Stan Birchfield"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA",
  "venue_source": "semantic-scholar",
  "citations": 47,
  "influential_citations": 2,
  "tldr": "SPOT, an object-centric imitation learning framework that decouples embodiment actions from sensory inputs, facilitating learning from various demonstration types, including both action-based and action-less human hand demonstrations, as well as crossembodiment generalization.",
  "doi": "10.1109/ICRA55743.2025.11127562",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Cheng-Chun Hsu",
    "id": "2110514281",
    "h_index": 7,
    "papers": 9
   },
   {
    "name": "Bowen Wen",
    "id": "2261740421",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Jie Xu",
    "id": "2273556820",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Yashraj S. Narang",
    "id": "2387216730",
    "h_index": 12,
    "papers": 16
   },
   {
    "name": "Xiaolong Wang",
    "id": "2239141122",
    "h_index": 12,
    "papers": 13
   },
   {
    "name": "Yuke Zhu",
    "id": "2322736301",
    "h_index": 13,
    "papers": 21
   },
   {
    "name": "Joydeep Biswas",
    "id": "2293427870",
    "h_index": 6,
    "papers": 22
   },
   {
    "name": "S. T. Birchfield",
    "id": "2285517121",
    "h_index": 5,
    "papers": 6
   }
  ],
  "comment": "",
  "topics": [
   "imitation-diffusion",
   "foundation-pretraining"
  ],
  "orgs": [
   "NVIDIA"
  ],
  "abs_url": "https://arxiv.org/abs/2411.00965v2",
  "pdf_url": "https://arxiv.org/pdf/2411.00965v2",
  "html_url": "https://arxiv.org/html/2411.00965v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.68
 },
 {
  "id": "2411.00769",
  "slug": "gamegen-x-interactive-open-world-game-video-generation",
  "title": "GameGen-X: Interactive Open-world Game Video Generation",
  "abstract": "We introduce GameGen-X, the first diffusion transformer model specifically designed for both generating and interactively controlling open-world game videos. This model facilitates high-quality, open-domain generation by simulating an extensive array of game engine features, such as innovative characters, dynamic environments, complex actions, and diverse events. Additionally, it provides interactive controllability, predicting and altering future content based on the current clip, thus allowing for gameplay simulation. To realize this vision, we first collected and built an Open-World Video Game Dataset from scratch. It is the first and largest dataset for open-world game video generation and control, which comprises over a million diverse gameplay video clips sampling from over 150 games with informative captions from GPT-4o. GameGen-X undergoes a two-stage training process, consisting of foundation model pre-training and instruction tuning. Firstly, the model was pre-trained via text-to-video generation and video continuation, endowing it with the capability for long-sequence, high-quality open-domain game video generation. Further, to achieve interactive controllability, we designed InstructNet to incorporate game-related multi-modal control signal experts. This allows the model to adjust latent representations based on user inputs, unifying character interaction and scene content control for the first time in video generation. During instruction tuning, only the InstructNet is updated while the pre-trained foundation model is frozen, enabling the integration of interactive controllability without loss of diversity and quality of generated video content.",
  "published": "2024-11-01",
  "updated": "2024-12-06",
  "year": "2024",
  "authors": [
   "Haoxuan Che",
   "Xuanhua He",
   "Quande Liu",
   "Cheng Jin",
   "Hao Chen"
  ],
  "author_count": 5,
  "categories": [
   "cs.CV",
   "cs.AI"
  ],
  "primary_category": "cs.CV",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 124,
  "influential_citations": 11,
  "tldr": "This work introduces GameGen-X, the first diffusion transformer model specifically designed for both generating and interactively controlling open-world game videos, and designed InstructNet to incorporate game-related multi-modal control signal experts, allowing the model to adjust latent representations based on user inputs for the first time in video generation.",
  "doi": "10.48550/arXiv.2411.00769",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Haoxuan Che",
    "id": "2136739844",
    "h_index": 12,
    "papers": 21
   },
   {
    "name": "Xuanhua He",
    "id": "2329309860",
    "h_index": 5,
    "papers": 19
   },
   {
    "name": "Quande Liu",
    "id": "2298019138",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Cheng Jin",
    "id": "2329715096",
    "h_index": 7,
    "papers": 19
   },
   {
    "name": "Hao Chen",
    "id": "2256936593",
    "h_index": 5,
    "papers": 6
   }
  ],
  "comment": "Homepage: https://gamegen-x.github.io/ Github: https://github.com/GameGen-X/GameGen-X",
  "topics": [
   "foundation-pretraining",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2411.00769v3",
  "pdf_url": "https://arxiv.org/pdf/2411.00769v3",
  "html_url": "https://arxiv.org/html/2411.00769v3",
  "code_url": "https://gamegen-x.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.6
 },
 {
  "id": "2410.24221",
  "slug": "egomimic-scaling-imitation-learning-via-egocentric-video",
  "title": "EgoMimic: Scaling Imitation Learning via Egocentric Video",
  "abstract": "The scale and diversity of demonstration data required for imitation learning is a significant challenge. We present EgoMimic, a full-stack framework which scales manipulation via human embodiment data, specifically egocentric human videos paired with 3D hand tracking. EgoMimic achieves this through: (1) a system to capture human embodiment data using the ergonomic Project Aria glasses, (2) a low-cost bimanual manipulator that minimizes the kinematic gap to human data, (3) cross-domain data alignment techniques, and (4) an imitation learning architecture that co-trains on human and robot data. Compared to prior works that only extract high-level intent from human videos, our approach treats human and robot data equally as embodied demonstration data and learns a unified policy from both data sources. EgoMimic achieves significant improvement on a diverse set of long-horizon, single-arm and bimanual manipulation tasks over state-of-the-art imitation learning methods and enables generalization to entirely new scenes. Finally, we show a favorable scaling trend for EgoMimic, where adding 1 hour of additional hand data is significantly more valuable than 1 hour of additional robot data. Videos and additional information can be found at https://egomimic.github.io/",
  "published": "2024-10-31",
  "updated": "2024-10-31",
  "year": "2024",
  "authors": [
   "Simar Kareer",
   "Dhruv Patel",
   "Ryan Punamiya",
   "Pranay Mathur",
   "Shuo Cheng",
   "Chen Wang",
   "Judy Hoffman",
   "Danfei Xu"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA",
  "venue_source": "semantic-scholar",
  "citations": 219,
  "influential_citations": 14,
  "tldr": "EgoMimic achieves significant improvement on a diverse set of long-horizon, single-arm and bimanual manipulation tasks over state-of-the-art imitation learning methods and enables generalization to entirely new scenes.",
  "doi": "10.1109/ICRA55743.2025.11127989",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Simar Kareer",
    "id": "2188833033",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Dhruv Patel",
    "id": "2328566408",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Ryan Punamiya",
    "id": "2328411560",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Pranay Mathur",
    "id": "2305685371",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Shuo Cheng",
    "id": "2232588215",
    "h_index": 6,
    "papers": 16
   },
   {
    "name": "Chen Wang",
    "id": "2240836173",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Judy Hoffman",
    "id": "2328413304",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Danfei Xu",
    "id": "2315451166",
    "h_index": 8,
    "papers": 10
   }
  ],
  "comment": "",
  "topics": [
   "egocentric-data",
   "imitation-diffusion"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2410.24221v1",
  "pdf_url": "https://arxiv.org/pdf/2410.24221v1",
  "html_url": "https://arxiv.org/html/2410.24221v1",
  "code_url": null,
  "presented": [
   {
    "club": "saturday-robotics",
    "club_name": "Saturday Robotics",
    "session_number": 8,
    "session_title": "Robotics & World Models Reading Club 08: Embodied Human Data as the \u201cInternet of Motion and Behavior\u201d \u2014 San Francisco 0516",
    "date_text": "Saturday, May 16, 2026 | 2:00 PM \u2013 5:00 PM",
    "city": "San Francisco, CA",
    "url": "https://lu.ma/qoxioge7",
    "listed_as": "Learning Dexterous Manipulation from Egocentric Human Videos"
   },
   {
    "club": "saturday-robotics",
    "club_name": "Saturday Robotics",
    "session_number": 13,
    "session_title": "Robotics & World Models Reading Club 13: HumanEgo: Train Robot Policy from 30 min Egocentric Videos \u2014 SF 0620",
    "date_text": "Saturday, June 20, 2026 | 2:00 PM \u2013 5:00 PM",
    "city": "San Francisco, CA",
    "url": "https://lu.ma/6vkhxnum",
    "listed_as": "X: https://x.com/TX\\Leo\\Wang/status/2059320921228546220 (Highly recommend)"
   }
  ],
  "club_note": "Learns robot policies from egocentric human video via latent action inference + temporal alignment.",
  "featured": true,
  "signal": 6.84
 },
 {
  "id": "2410.24164",
  "slug": "0-a-vision-language-action-flow-model-for-general-robot-control",
  "title": "$\u03c0_0$: A Vision-Language-Action Flow Model for General Robot Control",
  "abstract": "Robot learning holds tremendous promise to unlock the full potential of flexible, general, and dexterous robot systems, as well as to address some of the deepest questions in artificial intelligence. However, bringing robot learning to the level of generality required for effective real-world systems faces major obstacles in terms of data, generalization, and robustness. In this paper, we discuss how generalist robot policies (i.e., robot foundation models) can address these challenges, and how we can design effective generalist robot policies for complex and highly dexterous tasks. We propose a novel flow matching architecture built on top of a pre-trained vision-language model (VLM) to inherit Internet-scale semantic knowledge. We then discuss how this model can be trained on a large and diverse dataset from multiple dexterous robot platforms, including single-arm robots, dual-arm robots, and mobile manipulators. We evaluate our model in terms of its ability to perform tasks in zero shot after pre-training, follow language instructions from people and from a high-level VLM policy, and its ability to acquire new skills via fine-tuning. Our results cover a wide variety of tasks, such as laundry folding, table cleaning, and assembling boxes.",
  "published": "2024-10-31",
  "updated": "2026-01-08",
  "year": "2024",
  "authors": [
   "Kevin Black",
   "Noah Brown",
   "Danny Driess",
   "Adnan Esmail",
   "Michael Equi",
   "Chelsea Finn",
   "Niccolo Fusai",
   "Lachy Groom",
   "Karol Hausman",
   "Brian Ichter",
   "Szymon Jakubczak",
   "Tim Jones",
   "Liyiming Ke",
   "Sergey Levine",
   "Adrian Li-Bell",
   "Mohith Mothukuri",
   "Suraj Nair",
   "Karl Pertsch",
   "Lucy Xiaoyang Shi",
   "James Tanner",
   "Quan Vuong",
   "Anna Walling",
   "Haohuan Wang",
   "Ury Zhilinsky"
  ],
  "author_count": 24,
  "categories": [
   "cs.LG",
   "cs.RO"
  ],
  "primary_category": "cs.LG",
  "venue": "RSS 2025",
  "venue_source": "arxiv-comment",
  "citations": 2461,
  "influential_citations": 455,
  "tldr": "A novel flow matching architecture built on top of a pre-trained vision-language model (VLM) to inherit Internet-scale semantic knowledge is proposed and evaluated in terms of its ability to perform tasks in zero shot after pre-training, follow language instructions from people and from a high-level VLM policy, and its ability to acquire new skills via fine-tuning.",
  "doi": "10.48550/arXiv.2410.24164",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Kevin Black",
    "id": "2258959388",
    "h_index": 13,
    "papers": 15
   },
   {
    "name": "Noah Brown",
    "id": "2161343011",
    "h_index": 10,
    "papers": 16
   },
   {
    "name": "Danny Driess",
    "id": "2283848260",
    "h_index": 27,
    "papers": 35
   },
   {
    "name": "A. Esmail",
    "id": "2332926590",
    "h_index": 7,
    "papers": 12
   },
   {
    "name": "Michael Equi",
    "id": "2298901898",
    "h_index": 10,
    "papers": 15
   },
   {
    "name": "Chelsea Finn",
    "id": "2257346440",
    "h_index": 23,
    "papers": 32
   },
   {
    "name": "Niccolo Fusai",
    "id": "2332926006",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Lachy Groom",
    "id": "2332926792",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Karol Hausman",
    "id": "1944801",
    "h_index": 47,
    "papers": 122
   },
   {
    "name": "Brian Ichter",
    "id": "2704814",
    "h_index": 37,
    "papers": 60
   },
   {
    "name": "S. Jakubczak",
    "id": "2332926523",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Tim Jones",
    "id": "2333409218",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Liyiming Ke",
    "id": "2332976956",
    "h_index": 7,
    "papers": 8
   },
   {
    "name": "Sergey Levine",
    "id": "2249615151",
    "h_index": 32,
    "papers": 53
   },
   {
    "name": "Adrian Li-Bell",
    "id": "2332927424",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Mohith Mothukuri",
    "id": "2332926533",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Suraj Nair",
    "id": "2286638954",
    "h_index": 12,
    "papers": 19
   },
   {
    "name": "Karl Pertsch",
    "id": "31719101",
    "h_index": 32,
    "papers": 56
   },
   {
    "name": "L. Shi",
    "id": "2292341452",
    "h_index": 12,
    "papers": 17
   },
   {
    "name": "James Tanner",
    "id": "2332926892",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Quan Vuong",
    "id": "2288210223",
    "h_index": 19,
    "papers": 29
   },
   {
    "name": "Anna Walling",
    "id": "2333982746",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Haohuan Wang",
    "id": "2332952253",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Ury Zhilinsky",
    "id": "3187915",
    "h_index": 5,
    "papers": 7
   }
  ],
  "comment": "See project website for videos: https://physicalintelligence.company/blog/pi0 Published in RSS 2025",
  "topics": [
   "vla",
   "dexterous-manipulation",
   "foundation-pretraining",
   "video-generation",
   "safety-eval"
  ],
  "orgs": [
   "Physical Intelligence"
  ],
  "abs_url": "https://arxiv.org/abs/2410.24164v4",
  "pdf_url": "https://arxiv.org/pdf/2410.24164v4",
  "html_url": "https://arxiv.org/html/2410.24164v4",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.0
 },
 {
  "id": "2410.24091",
  "slug": "3d-vitac-learning-fine-grained-manipulation-with-visuo-tactile-sensing",
  "title": "3D-ViTac: Learning Fine-Grained Manipulation with Visuo-Tactile Sensing",
  "abstract": "Tactile and visual perception are both crucial for humans to perform fine-grained interactions with their environment. Developing similar multi-modal sensing capabilities for robots can significantly enhance and expand their manipulation skills. This paper introduces \\textbf{3D-ViTac}, a multi-modal sensing and learning system designed for dexterous bimanual manipulation. Our system features tactile sensors equipped with dense sensing units, each covering an area of 3$mm^2$. These sensors are low-cost and flexible, providing detailed and extensive coverage of physical contacts, effectively complementing visual information. To integrate tactile and visual data, we fuse them into a unified 3D representation space that preserves their 3D structures and spatial relationships. The multi-modal representation can then be coupled with diffusion policies for imitation learning. Through concrete hardware experiments, we demonstrate that even low-cost robots can perform precise manipulations and significantly outperform vision-only policies, particularly in safe interactions with fragile items and executing long-horizon tasks involving in-hand manipulation. Our project page is available at \\url{https://binghao-huang.github.io/3D-ViTac/}.",
  "published": "2024-10-31",
  "updated": "2025-01-06",
  "year": "2024",
  "authors": [
   "Binghao Huang",
   "Yixuan Wang",
   "Xinyi Yang",
   "Yiyue Luo",
   "Yunzhu Li"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 154,
  "influential_citations": 9,
  "tldr": "It is demonstrated that even low-cost robots can perform precise manipulations and significantly outperform vision-only policies, particularly in safe interactions with fragile items and executing long-horizon tasks involving in-hand manipulation.",
  "doi": "10.48550/arXiv.2410.24091",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Binghao Huang",
    "id": "2287019710",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Yixuan Wang",
    "id": "2253810575",
    "h_index": 8,
    "papers": 10
   },
   {
    "name": "Xinyi Yang",
    "id": "2329040389",
    "h_index": 7,
    "papers": 26
   },
   {
    "name": "Yiyue Luo",
    "id": "2295090159",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Yunzhu Li",
    "id": "2374457025",
    "h_index": 4,
    "papers": 9
   }
  ],
  "comment": "Accepted at Conference on Robot Learning (CoRL) 2024",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "imitation-diffusion"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2410.24091v2",
  "pdf_url": "https://arxiv.org/pdf/2410.24091v2",
  "html_url": "https://arxiv.org/html/2410.24091v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.69
 },
 {
  "id": "2410.24090",
  "slug": "sparsh-self-supervised-touch-representations-for-vision-based-tactile",
  "title": "Sparsh: Self-supervised touch representations for vision-based tactile sensing",
  "abstract": "In this work, we introduce general purpose touch representations for the increasingly accessible class of vision-based tactile sensors. Such sensors have led to many recent advances in robot manipulation as they markedly complement vision, yet solutions today often rely on task and sensor specific handcrafted perception models. Collecting real data at scale with task centric ground truth labels, like contact forces and slip, is a challenge further compounded by sensors of various form factor differing in aspects like lighting and gel markings. To tackle this we turn to self-supervised learning (SSL) that has demonstrated remarkable performance in computer vision. We present Sparsh, a family of SSL models that can support various vision-based tactile sensors, alleviating the need for custom labels through pre-training on 460k+ tactile images with masking and self-distillation in pixel and latent spaces. We also build TacBench, to facilitate standardized benchmarking across sensors and models, comprising of six tasks ranging from comprehending tactile properties to enabling physical perception and manipulation planning. In evaluations, we find that SSL pre-training for touch representation outperforms task and sensor-specific end-to-end training by 95.1% on average over TacBench, and Sparsh (DINO) and Sparsh (IJEPA) are the most competitive, indicating the merits of learning in latent space for tactile images. Project page: https://sparsh-ssl.github.io/",
  "published": "2024-10-31",
  "updated": "2024-10-31",
  "year": "2024",
  "authors": [
   "Carolina Higuera",
   "Akash Sharma",
   "Chaithanya Krishna Bodduluri",
   "Taosha Fan",
   "Patrick Lancaster",
   "Mrinal Kalakrishnan",
   "Michael Kaess",
   "Byron Boots",
   "Mike Lambeta",
   "Tingfan Wu",
   "Mustafa Mukadam"
  ],
  "author_count": 11,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 101,
  "influential_citations": 5,
  "tldr": "In evaluations, it is found that SSL pre-training for touch representation outperforms task and sensor-specific end-to-end training by 95.1% on average over TacBench, and Sparsh (DINO) and Sparsh (IJEPA) are the most competitive, indicating the merits of learning in latent space for tactile images.",
  "doi": "10.48550/arXiv.2410.24090",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Carolina Higuera",
    "id": "2248215923",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Akash Sharma",
    "id": "2109364933",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "Chaithanya Krishna Bodduluri",
    "id": "2328409907",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Taosha Fan",
    "id": "2275595472",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Patrick Lancaster",
    "id": "2328413960",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Mrinal Kalakrishnan",
    "id": "1729262",
    "h_index": 35,
    "papers": 58
   },
   {
    "name": "Michael Kaess",
    "id": "2279716032",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Byron Boots",
    "id": "2276429486",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Mike Lambeta",
    "id": "3427691",
    "h_index": 13,
    "papers": 20
   },
   {
    "name": "Tingfan Wu",
    "id": "2254158966",
    "h_index": 12,
    "papers": 21
   },
   {
    "name": "Mustafa Mukadam",
    "id": "2874057",
    "h_index": 30,
    "papers": 68
   }
  ],
  "comment": "Conference on Robot Learning (CoRL), 2024",
  "topics": [
   "world-models",
   "tactile",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2410.24090v1",
  "pdf_url": "https://arxiv.org/pdf/2410.24090v1",
  "html_url": "https://arxiv.org/html/2410.24090v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.51
 },
 {
  "id": "2410.23701",
  "slug": "get-a-grip-multi-finger-grasp-evaluation-at-scale-enables-robust-sim-t",
  "title": "Get a Grip: Multi-Finger Grasp Evaluation at Scale Enables Robust Sim-to-Real Transfer",
  "abstract": "This work explores conditions under which multi-finger grasping algorithms can attain robust sim-to-real transfer. While numerous large datasets facilitate learning generative models for multi-finger grasping at scale, reliable real-world dexterous grasping remains challenging, with most methods degrading when deployed on hardware. An alternate strategy is to use discriminative grasp evaluation models for grasp selection and refinement, conditioned on real-world sensor measurements. This paradigm has produced state-of-the-art results for vision-based parallel-jaw grasping, but remains unproven in the multi-finger setting. In this work, we find that existing datasets and methods have been insufficient for training discriminitive models for multi-finger grasping. To train grasp evaluators at scale, datasets must provide on the order of millions of grasps, including both positive and negative examples, with corresponding visual data resembling measurements at inference time. To that end, we release a new, open-source dataset of 3.5M grasps on 4.3K objects annotated with RGB images, point clouds, and trained NeRFs. Leveraging this dataset, we train vision-based grasp evaluators that outperform both analytic and generative modeling-based baselines on extensive simulated and real-world trials across a diverse range of objects. We show via numerous ablations that the key factor for performance is indeed the evaluator, and that its quality degrades as the dataset shrinks, demonstrating the importance of our new dataset. Project website at: https://sites.google.com/view/get-a-grip-dataset.",
  "published": "2024-10-31",
  "updated": "2024-10-31",
  "year": "2024",
  "authors": [
   "Tyler Ga Wei Lum",
   "Albert H. Li",
   "Preston Culbertson",
   "Krishnan Srinivasan",
   "Aaron D. Ames",
   "Mac Schwager",
   "Jeannette Bohg"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 17,
  "influential_citations": 6,
  "tldr": "It is found that existing datasets and methods have been insufficient for training discriminitive models for multi-finger grasping, and it is shown that the key factor for performance is indeed the evaluator, and that its quality degrades as the dataset shrinks, demonstrating the importance of the new dataset.",
  "doi": "10.48550/arXiv.2410.23701",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Tyler Ga Wei Lum",
    "id": "2309245667",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Albert H. Li",
    "id": "2249724042",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Preston Culbertson",
    "id": "49535307",
    "h_index": 14,
    "papers": 25
   },
   {
    "name": "K. Srinivasan",
    "id": "2093939303",
    "h_index": 16,
    "papers": 29
   },
   {
    "name": "Aaron D. Ames",
    "id": "2265495408",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Mac Schwager",
    "id": "2360172520",
    "h_index": 6,
    "papers": 14
   },
   {
    "name": "Jeannette Bohg",
    "id": "1775407",
    "h_index": 50,
    "papers": 161
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "sim2real",
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2410.23701v1",
  "pdf_url": "https://arxiv.org/pdf/2410.23701v1",
  "html_url": "https://arxiv.org/html/2410.23701v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.76
 },
 {
  "id": "2410.23289",
  "slug": "bridging-the-human-to-robot-dexterity-gap-through-object-oriented-rewa",
  "title": "Bridging the Human to Robot Dexterity Gap through Object-Oriented Rewards",
  "abstract": "Training robots directly from human videos is an emerging area in robotics and computer vision. While there has been notable progress with two-fingered grippers, learning autonomous tasks for multi-fingered robot hands in this way remains challenging. A key reason for this difficulty is that a policy trained on human hands may not directly transfer to a robot hand due to morphology differences. In this work, we present HuDOR, a technique that enables online fine-tuning of policies by directly computing rewards from human videos. Importantly, this reward function is built using object-oriented trajectories derived from off-the-shelf point trackers, providing meaningful learning signals despite the morphology gap and visual differences between human and robot hands. Given a single video of a human solving a task, such as gently opening a music box, HuDOR enables our four-fingered Allegro hand to learn the task with just an hour of online interaction. Our experiments across four tasks show that HuDOR achieves a 4x improvement over baselines. Code and videos are available on our website, https://object-rewards.github.io.",
  "published": "2024-10-30",
  "updated": "2024-10-30",
  "year": "2024",
  "authors": [
   "Irmak Guzey",
   "Yinlong Dai",
   "Georgy Savva",
   "Raunaq Bhirangi",
   "Lerrel Pinto"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA",
  "venue_source": "semantic-scholar",
  "citations": 37,
  "influential_citations": 1,
  "tldr": "HUDOR, a technique that enables online fine-tuning of the policy by constructing a reward function from the human video, is presented, which allows for meaningful learning signals even when the robot hand is in the visual observation, while the human hand is used to construct the reward.",
  "doi": "10.1109/ICRA55743.2025.11128690",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Irmak G\u00fczey",
    "id": "2212471096",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Yinlong Dai",
    "id": "2243500130",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "George Savva",
    "id": "2265852851",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Raunaq M. Bhirangi",
    "id": "152466940",
    "h_index": 11,
    "papers": 17
   },
   {
    "name": "Lerrel Pinto",
    "id": "2320806817",
    "h_index": 10,
    "papers": 12
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "egocentric-data",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2410.23289v1",
  "pdf_url": "https://arxiv.org/pdf/2410.23289v1",
  "html_url": "https://arxiv.org/html/2410.23289v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.08
 },
 {
  "id": "2410.18647",
  "slug": "data-scaling-laws-in-imitation-learning-for-robotic-manipulation",
  "title": "Data Scaling Laws in Imitation Learning for Robotic Manipulation",
  "abstract": "Data scaling has revolutionized fields like natural language processing and computer vision, providing models with remarkable generalization capabilities. In this paper, we investigate whether similar data scaling laws exist in robotics, particularly in robotic manipulation, and whether appropriate data scaling can yield single-task robot policies that can be deployed zero-shot for any object within the same category in any environment. To this end, we conduct a comprehensive empirical study on data scaling in imitation learning. By collecting data across numerous environments and objects, we study how a policy's generalization performance changes with the number of training environments, objects, and demonstrations. Throughout our research, we collect over 40,000 demonstrations and execute more than 15,000 real-world robot rollouts under a rigorous evaluation protocol. Our findings reveal several intriguing results: the generalization performance of the policy follows a roughly power-law relationship with the number of environments and objects. The diversity of environments and objects is far more important than the absolute number of demonstrations; once the number of demonstrations per environment or object reaches a certain threshold, additional demonstrations have minimal effect. Based on these insights, we propose an efficient data collection strategy. With four data collectors working for one afternoon, we collect sufficient data to enable the policies for two tasks to achieve approximately 90% success rates in novel environments with unseen objects.",
  "published": "2024-10-24",
  "updated": "2026-06-26",
  "year": "2024",
  "authors": [
   "Fanqi Lin",
   "Yingdong Hu",
   "Pingyue Sheng",
   "Chuan Wen",
   "Jiacheng You",
   "Yang Gao"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 199,
  "influential_citations": 9,
  "tldr": "This paper investigates whether similar data scaling laws exist in robotics, particularly in robotic manipulation, and whether appropriate data scaling can yield single-task robot policies that can be deployed zero-shot for any object within the same category in any environment.",
  "doi": "10.48550/arXiv.2410.18647",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Fanqi Lin",
    "id": "2270740581",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "Yingdong Hu",
    "id": "2149297811",
    "h_index": 13,
    "papers": 31
   },
   {
    "name": "Pingyue Sheng",
    "id": "2327340052",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Chuan Wen",
    "id": "2068033698",
    "h_index": 14,
    "papers": 25
   },
   {
    "name": "Jiacheng You",
    "id": "2289844086",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "Yang Gao",
    "id": "2257027030",
    "h_index": 7,
    "papers": 10
   }
  ],
  "comment": "",
  "topics": [
   "sim2real",
   "imitation-diffusion",
   "foundation-pretraining",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2410.18647v4",
  "pdf_url": "https://arxiv.org/pdf/2410.18647v4",
  "html_url": "https://arxiv.org/html/2410.18647v4",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.8
 },
 {
  "id": "2410.18072",
  "slug": "worldsimbench-towards-video-generation-models-as-world-simulators",
  "title": "WorldSimBench: Towards Video Generation Models as World Simulators",
  "abstract": "Recent advancements in predictive models have demonstrated exceptional capabilities in predicting the future state of objects and scenes. However, the lack of categorization based on inherent characteristics continues to hinder the progress of predictive model development. Additionally, existing benchmarks are unable to effectively evaluate higher-capability, highly embodied predictive models from an embodied perspective. In this work, we classify the functionalities of predictive models into a hierarchy and take the first step in evaluating World Simulators by proposing a dual evaluation framework called WorldSimBench. WorldSimBench includes Explicit Perceptual Evaluation and Implicit Manipulative Evaluation, encompassing human preference assessments from the visual perspective and action-level evaluations in embodied tasks, covering three representative embodied scenarios: Open-Ended Embodied Environment, Autonomous, Driving, and Robot Manipulation. In the Explicit Perceptual Evaluation, we introduce the HF-Embodied Dataset, a video assessment dataset based on fine-grained human feedback, which we use to train a Human Preference Evaluator that aligns with human perception and explicitly assesses the visual fidelity of World Simulators. In the Implicit Manipulative Evaluation, we assess the video-action consistency of World Simulators by evaluating whether the generated situation-aware video can be accurately translated into the correct control signals in dynamic environments. Our comprehensive evaluation offers key insights that can drive further innovation in video generation models, positioning World Simulators as a pivotal advancement toward embodied artificial intelligence.",
  "published": "2024-10-23",
  "updated": "2024-10-23",
  "year": "2024",
  "authors": [
   "Yiran Qin",
   "Zhelun Shi",
   "Jiwen Yu",
   "Xijun Wang",
   "Enshen Zhou",
   "Lijun Li",
   "Zhenfei Yin",
   "Xihui Liu",
   "Lu Sheng",
   "Jing Shao",
   "Lei Bai",
   "Wanli Ouyang",
   "Ruimao Zhang"
  ],
  "author_count": 13,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 1254,
  "influential_citations": 80,
  "tldr": "This work classify the functionalities of predictive models into a hierarchy and takes the first step in evaluating World Simulators by proposing a dual evaluation framework called WorldSimBench, encompassing human preference assessments from the visual perspective and action-level evaluations in embodied tasks.",
  "doi": "10.48550/arXiv.2410.18072",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yiran Qin",
    "id": "2240266091",
    "h_index": 7,
    "papers": 20
   },
   {
    "name": "Zhelun Shi",
    "id": "2144146362",
    "h_index": 9,
    "papers": 16
   },
   {
    "name": "Jiwen Yu",
    "id": "2116420946",
    "h_index": 16,
    "papers": 28
   },
   {
    "name": "Xijun Wang",
    "id": "2327287661",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Enshen Zhou",
    "id": "2273688063",
    "h_index": 9,
    "papers": 20
   },
   {
    "name": "Lijun Li",
    "id": "2280261471",
    "h_index": 10,
    "papers": 29
   },
   {
    "name": "Zhen-fei Yin",
    "id": "13050405",
    "h_index": 19,
    "papers": 37
   },
   {
    "name": "Xihui Liu",
    "id": "2284733067",
    "h_index": 9,
    "papers": 13
   },
   {
    "name": "Lu Sheng",
    "id": "2290983876",
    "h_index": 8,
    "papers": 16
   },
   {
    "name": "Jing Shao",
    "id": "2328076278",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Lei Bai",
    "id": "2253472588",
    "h_index": 16,
    "papers": 30
   },
   {
    "name": "Wanli Ouyang",
    "id": "2253464521",
    "h_index": 20,
    "papers": 81
   },
   {
    "name": "Ruimao Zhang",
    "id": "2274031380",
    "h_index": 7,
    "papers": 15
   }
  ],
  "comment": "",
  "topics": [
   "sim2real",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2410.18072v1",
  "pdf_url": "https://arxiv.org/pdf/2410.18072v1",
  "html_url": "https://arxiv.org/html/2410.18072v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.0
 },
 {
  "id": "2410.18013",
  "slug": "scalable-ranked-preference-optimization-for-text-to-image-generation",
  "title": "Scalable Ranked Preference Optimization for Text-to-Image Generation",
  "abstract": "Direct Preference Optimization (DPO) has emerged as a powerful approach to align text-to-image (T2I) models with human feedback. Unfortunately, successful application of DPO to T2I models requires a huge amount of resources to collect and label large-scale datasets, e.g., millions of generated paired images annotated with human preferences. In addition, these human preference datasets can get outdated quickly as the rapid improvements of T2I models lead to higher quality images. In this work, we investigate a scalable approach for collecting large-scale and fully synthetic datasets for DPO training. Specifically, the preferences for paired images are generated using a pre-trained reward function, eliminating the need for involving humans in the annotation process, greatly improving the dataset collection efficiency. Moreover, we demonstrate that such datasets allow averaging predictions across multiple models and collecting ranked preferences as opposed to pairwise preferences. Furthermore, we introduce RankDPO to enhance DPO-based methods using the ranking feedback. Applying RankDPO on SDXL and SD3-Medium models with our synthetically generated preference dataset \"Syn-Pic\" improves both prompt-following (on benchmarks like T2I-Compbench, GenEval, and DPG-Bench) and visual quality (through user studies). This pipeline presents a practical and scalable solution to develop better preference datasets to enhance the performance of text-to-image models.",
  "published": "2024-10-23",
  "updated": "2024-10-30",
  "year": "2024",
  "authors": [
   "Shyamgopal Karthik",
   "Huseyin Coskun",
   "Zeynep Akata",
   "Sergey Tulyakov",
   "Jian Ren",
   "Anil Kag"
  ],
  "author_count": 6,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "ICCV",
  "venue_source": "semantic-scholar",
  "citations": 36,
  "influential_citations": 0,
  "tldr": "This work investigates a scalable approach for collecting large-scale and fully synthetic datasets for DPO training and introduces RankDPO to enhance DiffusionDPO based methods using the ranking feedback.",
  "doi": "10.1109/ICCV51701.2025.01710",
  "oa_pdf": "https://arxiv.org/pdf/2410.18013",
  "s2_authors": [
   {
    "name": "Shyamgopal Karthik",
    "id": "1387873091",
    "h_index": 16,
    "papers": 30
   },
   {
    "name": "Huseyin Coskun",
    "id": "2327216532",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Zeynep Akata",
    "id": "1854487018",
    "h_index": 23,
    "papers": 100
   },
   {
    "name": "S. Tulyakov",
    "id": "145582202",
    "h_index": 48,
    "papers": 171
   },
   {
    "name": "Jian Ren",
    "id": "2286038173",
    "h_index": 8,
    "papers": 11
   },
   {
    "name": "Anil Kag",
    "id": "2284982329",
    "h_index": 9,
    "papers": 23
   }
  ],
  "comment": "Project Page: https://snap-research.github.io/RankDPO/",
  "topics": [
   "data-teleop",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2410.18013v2",
  "pdf_url": "https://arxiv.org/pdf/2410.18013v2",
  "html_url": "https://arxiv.org/html/2410.18013v2",
  "code_url": "https://snap-research.github.io/RankDPO/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.07
 },
 {
  "id": "2410.13851",
  "slug": "differentiable-robot-rendering",
  "title": "Differentiable Robot Rendering",
  "abstract": "Vision foundation models trained on massive amounts of visual data have shown unprecedented reasoning and planning skills in open-world settings. A key challenge in applying them to robotic tasks is the modality gap between visual data and action data. We introduce differentiable robot rendering, a method allowing the visual appearance of a robot body to be directly differentiable with respect to its control parameters. Our model integrates a kinematics-aware deformable model and Gaussians Splatting and is compatible with any robot form factors and degrees of freedom. We demonstrate its capability and usage in applications including reconstruction of robot poses from images and controlling robots through vision language models. Quantitative and qualitative results show that our differentiable rendering model provides effective gradients for robotic control directly from pixels, setting the foundation for the future applications of vision foundation models in robotics.",
  "published": "2024-10-17",
  "updated": "2024-10-17",
  "year": "2024",
  "authors": [
   "Ruoshi Liu",
   "Alper Canberk",
   "Shuran Song",
   "Carl Vondrick"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.CV",
   "cs.GR"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 28,
  "influential_citations": 4,
  "tldr": "Quantitative and qualitative results show that the differentiable rendering model provides effective gradients for robotic control directly from pixels, setting the foundation for the future applications of vision foundation models in robotics.",
  "doi": "10.48550/arXiv.2410.13851",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ruoshi Liu",
    "id": "2143183492",
    "h_index": 14,
    "papers": 24
   },
   {
    "name": "Alper Canberk",
    "id": "2061747814",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Shuran Song",
    "id": "2289085682",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Carl Vondrick",
    "id": "2306336178",
    "h_index": 5,
    "papers": 9
   }
  ],
  "comment": "Project Page: https://drrobot.cs.columbia.edu/",
  "topics": [
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2410.13851v1",
  "pdf_url": "https://arxiv.org/pdf/2410.13851v1",
  "html_url": "https://arxiv.org/html/2410.13851v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.96
 },
 {
  "id": "2410.13720",
  "slug": "movie-gen-a-cast-of-media-foundation-models",
  "title": "Movie Gen: A Cast of Media Foundation Models",
  "abstract": "We present Movie Gen, a cast of foundation models that generates high-quality, 1080p HD videos with different aspect ratios and synchronized audio. We also show additional capabilities such as precise instruction-based video editing and generation of personalized videos based on a user's image. Our models set a new state-of-the-art on multiple tasks: text-to-video synthesis, video personalization, video editing, video-to-audio generation, and text-to-audio generation. Our largest video generation model is a 30B parameter transformer trained with a maximum context length of 73K video tokens, corresponding to a generated video of 16 seconds at 16 frames-per-second. We show multiple technical innovations and simplifications on the architecture, latent spaces, training objectives and recipes, data curation, evaluation protocols, parallelization techniques, and inference optimizations that allow us to reap the benefits of scaling pre-training data, model size, and training compute for training large scale media generation models. We hope this paper helps the research community to accelerate progress and innovation in media generation models. All videos from this paper are available at https://go.fb.me/MovieGenResearchVideos.",
  "published": "2024-10-17",
  "updated": "2025-02-26",
  "year": "2024",
  "authors": [
   "Adam Polyak",
   "Amit Zohar",
   "Andrew Brown",
   "Andros Tjandra",
   "Animesh Sinha",
   "Ann Lee",
   "Apoorv Vyas",
   "Bowen Shi",
   "Chih-Yao Ma",
   "Ching-Yao Chuang",
   "David Yan",
   "Dhruv Choudhary",
   "Dingkang Wang",
   "Geet Sethi",
   "Guan Pang",
   "Haoyu Ma",
   "Ishan Misra",
   "Ji Hou",
   "Jialiang Wang",
   "Kiran Jagadeesh",
   "Kunpeng Li",
   "Luxin Zhang",
   "Mannat Singh",
   "Mary Williamson",
   "Matt Le",
   "Matthew Yu",
   "Mitesh Kumar Singh",
   "Peizhao Zhang",
   "Peter Vajda",
   "Quentin Duval",
   "Rohit Girdhar",
   "Roshan Sumbaly",
   "Sai Saketh Rambhatla",
   "Sam Tsai",
   "Samaneh Azadi",
   "Samyak Datta",
   "Sanyuan Chen",
   "Sean Bell",
   "Sharadh Ramaswamy",
   "Shelly Sheynin",
   "Siddharth Bhattacharya",
   "Simran Motwani",
   "Tao Xu",
   "Tianhe Li",
   "Tingbo Hou",
   "Wei-Ning Hsu",
   "Xi Yin",
   "Xiaoliang Dai",
   "Yaniv Taigman",
   "Yaqiao Luo",
   "Yen-Cheng Liu",
   "Yi-Chiao Wu",
   "Yue Zhao",
   "Yuval Kirstain",
   "Zecheng He",
   "Zijian He",
   "Albert Pumarola",
   "Ali Thabet",
   "Artsiom Sanakoyeu",
   "Arun Mallya",
   "Baishan Guo",
   "Boris Araya",
   "Breena Kerr",
   "Carleigh Wood",
   "Ce Liu",
   "Cen Peng",
   "Dimitry Vengertsev",
   "Edgar Schonfeld",
   "Elliot Blanchard",
   "Felix Juefei-Xu",
   "Fraylie Nord",
   "Jeff Liang",
   "John Hoffman",
   "Jonas Kohler",
   "Kaolin Fire",
   "Karthik Sivakumar",
   "Lawrence Chen",
   "Licheng Yu",
   "Luya Gao",
   "Markos Georgopoulos",
   "Rashel Moritz",
   "Sara K. Sampson",
   "Shikai Li",
   "Simone Parmeggiani",
   "Steve Fine",
   "Tara Fowler",
   "Vladan Petrovic",
   "Yuming Du"
  ],
  "author_count": 88,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.LG",
   "eess.IV"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 627,
  "influential_citations": 50,
  "tldr": "Movie Gen is presented, a cast of foundation models that generates high-quality, 1080p HD videos with different aspect ratios and synchronized audio and set a new state-of-the-art on multiple tasks: text-to-video synthesis, video personalization, video editing, video-to-audio generation, and text-to-audio generation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Adam Polyak",
    "id": "33964593",
    "h_index": 28,
    "papers": 38
   },
   {
    "name": "Amit Zohar",
    "id": "2266842423",
    "h_index": 7,
    "papers": 8
   },
   {
    "name": "Andrew Brown",
    "id": "2267305491",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Andros Tjandra",
    "id": "2894428",
    "h_index": 26,
    "papers": 73
   },
   {
    "name": "Animesh Sinha",
    "id": "2147354973",
    "h_index": 10,
    "papers": 19
   },
   {
    "name": "Ann Lee",
    "id": "2324572663",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Apoorv Vyas",
    "id": "2992087",
    "h_index": 16,
    "papers": 24
   },
   {
    "name": "Bowen Shi",
    "id": "2261676061",
    "h_index": 12,
    "papers": 31
   },
   {
    "name": "Chih-Yao Ma",
    "id": "2250489747",
    "h_index": 6,
    "papers": 6
   },
   {
    "name": "Ching-Yao Chuang",
    "id": "2275606869",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "David Yan",
    "id": "2267338180",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Dhruv Choudhary",
    "id": "2303390957",
    "h_index": 11,
    "papers": 46
   },
   {
    "name": "Dingkang Wang",
    "id": "2283843884",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Geet Sethi",
    "id": "3382094",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Guan Pang",
    "id": "2273066537",
    "h_index": 11,
    "papers": 45
   },
   {
    "name": "Haoyu Ma",
    "id": "2126795",
    "h_index": 19,
    "papers": 36
   },
   {
    "name": "Ishan Misra",
    "id": "2267241285",
    "h_index": 13,
    "papers": 60
   },
   {
    "name": "Ji Hou",
    "id": "2249723114",
    "h_index": 11,
    "papers": 22
   },
   {
    "name": "Jialiang Wang",
    "id": "2247901055",
    "h_index": 10,
    "papers": 22
   },
   {
    "name": "K. Jagadeesh",
    "id": "2326296433",
    "h_index": 4,
    "papers": 26
   },
   {
    "name": "Kunpeng Li",
    "id": "2256701243",
    "h_index": 10,
    "papers": 17
   },
   {
    "name": "Luxin Zhang",
    "id": "2273376477",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Mannat Singh",
    "id": "152964870",
    "h_index": 14,
    "papers": 51
   },
   {
    "name": "Mary Williamson",
    "id": "2272909484",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Matt Le",
    "id": "2261636821",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "Matthew Yu",
    "id": "2110144262",
    "h_index": 8,
    "papers": 10
   },
   {
    "name": "Mitesh Kumar Singh",
    "id": "2247874378",
    "h_index": 6,
    "papers": 6
   },
   {
    "name": "Peizhao Zhang",
    "id": "2323251815",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Peter Vajda",
    "id": "2283763097",
    "h_index": 8,
    "papers": 14
   },
   {
    "name": "Quentin Duval",
    "id": "2101830371",
    "h_index": 10,
    "papers": 11
   },
   {
    "name": "Rohit Girdhar",
    "id": "3102850",
    "h_index": 31,
    "papers": 95
   },
   {
    "name": "Roshan Sumbaly",
    "id": "1722889",
    "h_index": 11,
    "papers": 46
   },
   {
    "name": "Sai Saketh Rambhatla",
    "id": "3403576",
    "h_index": 12,
    "papers": 27
   },
   {
    "name": "Sam S. Tsai",
    "id": "2225238191",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "S. Azadi",
    "id": "2300366",
    "h_index": 18,
    "papers": 29
   },
   {
    "name": "Samyak Datta",
    "id": "19200118",
    "h_index": 11,
    "papers": 15
   },
   {
    "name": "Sanyuan Chen",
    "id": "2326358059",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Sean Bell",
    "id": "2277511475",
    "h_index": 7,
    "papers": 27
   },
   {
    "name": "Sharadh Ramaswamy",
    "id": "48347720",
    "h_index": 9,
    "papers": 24
   },
   {
    "name": "Shelly Sheynin",
    "id": "2086827528",
    "h_index": 12,
    "papers": 14
   },
   {
    "name": "Siddharth Bhattacharya",
    "id": "2326301307",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Simran Motwani",
    "id": "121255235",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Tao Xu",
    "id": "2318244226",
    "h_index": 6,
    "papers": 17
   },
   {
    "name": "Tianhe Li",
    "id": "2314332791",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "Tingbo Hou",
    "id": "2326294705",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Wei-Ning Hsu",
    "id": "2265493884",
    "h_index": 10,
    "papers": 22
   },
   {
    "name": "Xi Yin",
    "id": "2267386757",
    "h_index": 6,
    "papers": 19
   },
   {
    "name": "Xiaoliang Dai",
    "id": "2271159599",
    "h_index": 8,
    "papers": 14
   },
   {
    "name": "Yaniv Taigman",
    "id": "2188620",
    "h_index": 30,
    "papers": 41
   },
   {
    "name": "Yaqiao Luo",
    "id": "2271325625",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Yen-Cheng Liu",
    "id": "2326336755",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Yi-Chiao Wu",
    "id": "2276483591",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Yue Zhao",
    "id": "2270809396",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Yuval Kirstain",
    "id": "2044194129",
    "h_index": 9,
    "papers": 10
   },
   {
    "name": "Zecheng He",
    "id": "2313688915",
    "h_index": 6,
    "papers": 16
   },
   {
    "name": "Zijian He",
    "id": "2558787",
    "h_index": 18,
    "papers": 25
   },
   {
    "name": "Albert Pumarola",
    "id": "2264453654",
    "h_index": 8,
    "papers": 14
   },
   {
    "name": "Ali K. Thabet",
    "id": "2276426062",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Artsiom Sanakoyeu",
    "id": "3451249",
    "h_index": 19,
    "papers": 30
   },
   {
    "name": "Arun Mallya",
    "id": "2313205743",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Baishan Guo",
    "id": "2276436526",
    "h_index": 6,
    "papers": 6
   },
   {
    "name": "Boris Araya",
    "id": "2326294516",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Breena Kerr",
    "id": "2326294505",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Carleigh Wood",
    "id": "2218040866",
    "h_index": 10,
    "papers": 17
   },
   {
    "name": "Ce Liu",
    "id": "2353993404",
    "h_index": 4,
    "papers": 24
   },
   {
    "name": "Cen Peng",
    "id": "2326996565",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Dimitry Vengertsev",
    "id": "2326294816",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "E. Schonfeld",
    "id": "2326296203",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Elliot Blanchard",
    "id": "2267338968",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Felix Juefei-Xu",
    "id": "2313677278",
    "h_index": 12,
    "papers": 37
   },
   {
    "name": "F. Nord",
    "id": "2326298048",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Jeff Liang",
    "id": "2363834562",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "John Hoffman",
    "id": "2326296420",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "J. Kohler",
    "id": "2275352009",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Kaolin Fire",
    "id": "2096072080",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Karthik Sivakumar",
    "id": "2326294686",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Lawrence Chen",
    "id": "2271703027",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Licheng Yu",
    "id": "2269696579",
    "h_index": 7,
    "papers": 29
   },
   {
    "name": "Luya Gao",
    "id": "2326454230",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Markos Georgopoulos",
    "id": "34291068",
    "h_index": 15,
    "papers": 27
   },
   {
    "name": "Rashel Moritz",
    "id": "2219692579",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "S. Sampson",
    "id": "39738459",
    "h_index": 13,
    "papers": 21
   },
   {
    "name": "Shikai Li",
    "id": "2326974085",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Simone Parmeggiani",
    "id": "2326296601",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "S. Fine",
    "id": "145376527",
    "h_index": 12,
    "papers": 107
   },
   {
    "name": "T. Fowler",
    "id": "2313918585",
    "h_index": 9,
    "papers": 36
   },
   {
    "name": "Vladan Petrovic",
    "id": "2162195471",
    "h_index": 7,
    "papers": 19
   },
   {
    "name": "Yuming Du",
    "id": "2326480732",
    "h_index": 4,
    "papers": 8
   }
  ],
  "comment": "",
  "topics": [
   "sim2real",
   "foundation-pretraining",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2410.13720v2",
  "pdf_url": "https://arxiv.org/pdf/2410.13720v2",
  "html_url": "https://arxiv.org/html/2410.13720v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.8
 },
 {
  "id": "2410.13126",
  "slug": "aloha-unleashed-a-simple-recipe-for-robot-dexterity",
  "title": "ALOHA Unleashed: A Simple Recipe for Robot Dexterity",
  "abstract": "Recent work has shown promising results for learning end-to-end robot policies using imitation learning. In this work we address the question of how far can we push imitation learning for challenging dexterous manipulation tasks. We show that a simple recipe of large scale data collection on the ALOHA 2 platform, combined with expressive models such as Diffusion Policies, can be effective in learning challenging bimanual manipulation tasks involving deformable objects and complex contact rich dynamics. We demonstrate our recipe on 5 challenging real-world and 3 simulated tasks and demonstrate improved performance over state-of-the-art baselines. The project website and videos can be found at aloha-unleashed.github.io.",
  "published": "2024-10-17",
  "updated": "2024-10-17",
  "year": "2024",
  "authors": [
   "Tony Z. Zhao",
   "Jonathan Tompson",
   "Danny Driess",
   "Pete Florence",
   "Kamyar Ghasemipour",
   "Chelsea Finn",
   "Ayzaan Wahid"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 213,
  "influential_citations": 12,
  "tldr": "This work shows that a simple recipe of large scale data collection on the ALOHA 2 platform, combined with expressive models such as Diffusion Policies, can be effective in learning challenging bimanual manipulation tasks involving deformable objects and complex contact rich dynamics.",
  "doi": "10.48550/arXiv.2410.13126",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Tony Z. Zhao",
    "id": "2327844315",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Jonathan Tompson",
    "id": "2704494",
    "h_index": 43,
    "papers": 71
   },
   {
    "name": "Danny Driess",
    "id": "2283848260",
    "h_index": 27,
    "papers": 35
   },
   {
    "name": "Pete Florence",
    "id": "2264974363",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Kamyar Ghasemipour",
    "id": "2256999030",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Chelsea Finn",
    "id": "2257346440",
    "h_index": 23,
    "papers": 32
   },
   {
    "name": "Ayzaan Wahid",
    "id": "88728227",
    "h_index": 21,
    "papers": 27
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "imitation-diffusion",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2410.13126v1",
  "pdf_url": "https://arxiv.org/pdf/2410.13126v1",
  "html_url": "https://arxiv.org/html/2410.13126v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.83
 },
 {
  "id": "2410.12557",
  "slug": "one-step-diffusion-via-shortcut-models",
  "title": "One Step Diffusion via Shortcut Models",
  "abstract": "Diffusion models and flow-matching models have enabled generating diverse and realistic images by learning to transfer noise to data. However, sampling from these models involves iterative denoising over many neural network passes, making generation slow and expensive. Previous approaches for speeding up sampling require complex training regimes, such as multiple training phases, multiple networks, or fragile scheduling. We introduce shortcut models, a family of generative models that use a single network and training phase to produce high-quality samples in a single or multiple sampling steps. Shortcut models condition the network not only on the current noise level but also on the desired step size, allowing the model to skip ahead in the generation process. Across a wide range of sampling step budgets, shortcut models consistently produce higher quality samples than previous approaches, such as consistency models and reflow. Compared to distillation, shortcut models reduce complexity to a single network and training phase and additionally allow varying step budgets at inference time.",
  "published": "2024-10-16",
  "updated": "2025-06-23",
  "year": "2024",
  "authors": [
   "Kevin Frans",
   "Danijar Hafner",
   "Sergey Levine",
   "Pieter Abbeel"
  ],
  "author_count": 4,
  "categories": [
   "cs.LG",
   "cs.CV"
  ],
  "primary_category": "cs.LG",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 366,
  "influential_citations": 54,
  "tldr": "Shortcut models are introduced, a family of generative models that use a single network and training phase to produce high-quality samples in a single or multiple sampling steps and additionally allow varying step budgets at inference time.",
  "doi": "10.48550/arXiv.2410.12557",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Kevin Frans",
    "id": "2287848095",
    "h_index": 12,
    "papers": 22
   },
   {
    "name": "Danijar Hafner",
    "id": "2285299430",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Sergey Levine",
    "id": "2279022150",
    "h_index": 9,
    "papers": 22
   },
   {
    "name": "Pieter Abbeel",
    "id": "2279021699",
    "h_index": 13,
    "papers": 32
   }
  ],
  "comment": "",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2410.12557v3",
  "pdf_url": "https://arxiv.org/pdf/2410.12557v3",
  "html_url": "https://arxiv.org/html/2410.12557v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.06
 },
 {
  "id": "2410.11758",
  "slug": "latent-action-pretraining-from-videos",
  "title": "Latent Action Pretraining from Videos",
  "abstract": "We introduce Latent Action Pretraining for general Action models (LAPA), an unsupervised method for pretraining Vision-Language-Action (VLA) models without ground-truth robot action labels. Existing Vision-Language-Action models require action labels typically collected by human teleoperators during pretraining, which significantly limits possible data sources and scale. In this work, we propose a method to learn from internet-scale videos that do not have robot action labels. We first train an action quantization model leveraging VQ-VAE-based objective to learn discrete latent actions between image frames, then pretrain a latent VLA model to predict these latent actions from observations and task descriptions, and finally finetune the VLA on small-scale robot manipulation data to map from latent to robot actions. Experimental results demonstrate that our method significantly outperforms existing techniques that train robot manipulation policies from large-scale videos. Furthermore, it outperforms the state-of-the-art VLA model trained with robotic action labels on real-world manipulation tasks that require language conditioning, generalization to unseen objects, and semantic generalization to unseen instructions. Training only on human manipulation videos also shows positive transfer, opening up the potential for leveraging web-scale data for robotics foundation model.",
  "published": "2024-10-15",
  "updated": "2025-05-15",
  "year": "2024",
  "authors": [
   "Seonghyeon Ye",
   "Joel Jang",
   "Byeongguk Jeon",
   "Sejune Joo",
   "Jianwei Yang",
   "Baolin Peng",
   "Ajay Mandlekar",
   "Reuben Tan",
   "Yu-Wei Chao",
   "Bill Yuchen Lin",
   "Lars Liden",
   "Kimin Lee",
   "Jianfeng Gao",
   "Luke Zettlemoyer",
   "Dieter Fox",
   "Minjoon Seo"
  ],
  "author_count": 16,
  "categories": [
   "cs.RO",
   "cs.CL",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 315,
  "influential_citations": 45,
  "tldr": "Experimental results demonstrate that the LAPA method significantly outperforms existing techniques that train robot manipulation policies from large-scale videos and outperforms the state-of-the-art VLA model trained with robotic action labels on real-world manipulation tasks that require language conditioning, generalization to unseen objects, and semantic generalization to unseen instructions.",
  "doi": "10.48550/arXiv.2410.11758",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Seonghyeon Ye",
    "id": "2152111477",
    "h_index": 23,
    "papers": 30
   },
   {
    "name": "Joel Jang",
    "id": "2000091730",
    "h_index": 17,
    "papers": 26
   },
   {
    "name": "Byeongguk Jeon",
    "id": "2325955855",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Se June Joo",
    "id": "102540266",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Jianwei Yang",
    "id": "2279705714",
    "h_index": 6,
    "papers": 6
   },
   {
    "name": "Baolin Peng",
    "id": "1780690",
    "h_index": 44,
    "papers": 100
   },
   {
    "name": "A. Mandlekar",
    "id": "49686756",
    "h_index": 36,
    "papers": 67
   },
   {
    "name": "Reuben Tan",
    "id": "73441526",
    "h_index": 15,
    "papers": 36
   },
   {
    "name": "Yu-Wei Chao",
    "id": "2306062337",
    "h_index": 7,
    "papers": 12
   },
   {
    "name": "Bill Yuchen Lin",
    "id": "51583409",
    "h_index": 39,
    "papers": 64
   },
   {
    "name": "Lars Lid\u00e9n",
    "id": "145417002",
    "h_index": 16,
    "papers": 36
   },
   {
    "name": "Kimin Lee",
    "id": "2359689822",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Jianfeng Gao",
    "id": "2295522725",
    "h_index": 12,
    "papers": 18
   },
   {
    "name": "Luke S. Zettlemoyer",
    "id": "2325955099",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Dieter Fox",
    "id": "2258436157",
    "h_index": 12,
    "papers": 14
   },
   {
    "name": "Minjoon Seo",
    "id": "2266468609",
    "h_index": 9,
    "papers": 19
   }
  ],
  "comment": "ICLR 2025 Website: https://latentactionpretraining.github.io",
  "topics": [
   "vla",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2410.11758v2",
  "pdf_url": "https://arxiv.org/pdf/2410.11758v2",
  "html_url": "https://arxiv.org/html/2410.11758v2",
  "code_url": "https://latentactionpretraining.github.io",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.0
 },
 {
  "id": "2410.11081",
  "slug": "simplifying-stabilizing-and-scaling-continuous-time-consistency-models",
  "title": "Simplifying, Stabilizing and Scaling Continuous-Time Consistency Models",
  "abstract": "Consistency models (CMs) are a powerful class of diffusion-based generative models optimized for fast sampling. Most existing CMs are trained using discretized timesteps, which introduce additional hyperparameters and are prone to discretization errors. While continuous-time formulations can mitigate these issues, their success has been limited by training instability. To address this, we propose a simplified theoretical framework that unifies previous parameterizations of diffusion models and CMs, identifying the root causes of instability. Based on this analysis, we introduce key improvements in diffusion process parameterization, network architecture, and training objectives. These changes enable us to train continuous-time CMs at an unprecedented scale, reaching 1.5B parameters on ImageNet 512x512. Our proposed training algorithm, using only two sampling steps, achieves FID scores of 2.06 on CIFAR-10, 1.48 on ImageNet 64x64, and 1.88 on ImageNet 512x512, narrowing the gap in FID scores with the best existing diffusion models to within 10%.",
  "published": "2024-10-14",
  "updated": "2025-03-01",
  "year": "2024",
  "authors": [
   "Cheng Lu",
   "Yang Song"
  ],
  "author_count": 2,
  "categories": [
   "cs.LG",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "ICLR 2025",
  "venue_source": "arxiv-comment",
  "citations": 269,
  "influential_citations": 47,
  "tldr": "A simplified theoretical framework is proposed that unifies previous parameterizations of diffusion models and CMs, identifying the root causes of instability and introducing key improvements in diffusion process parameterization, network architecture, and training objectives that enable to train continuous-time CMs at an unprecedented scale.",
  "doi": "10.48550/arXiv.2410.11081",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Cheng Lu",
    "id": "2326216127",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Yang Song",
    "id": "2326075777",
    "h_index": 1,
    "papers": 2
   }
  ],
  "comment": "ICLR 2025 Oral",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2410.11081v2",
  "pdf_url": "https://arxiv.org/pdf/2410.11081v2",
  "html_url": "https://arxiv.org/html/2410.11081v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.93
 },
 {
  "id": "2410.09309",
  "slug": "adaptive-compliance-policy-learning-approximate-compliance-for-diffusi",
  "title": "Adaptive Compliance Policy: Learning Approximate Compliance for Diffusion Guided Control",
  "abstract": "Compliance plays a crucial role in manipulation, as it balances between the concurrent control of position and force under uncertainties. Yet compliance is often overlooked by today's visuomotor policies that solely focus on position control. This paper introduces Adaptive Compliance Policy (ACP), a novel framework that learns to dynamically adjust system compliance both spatially and temporally for given manipulation tasks from human demonstrations, improving upon previous approaches that rely on pre-selected compliance parameters or assume uniform constant stiffness. However, computing full compliance parameters from human demonstrations is an ill-defined problem. Instead, we estimate an approximate compliance profile with two useful properties: avoiding large contact forces and encouraging accurate tracking. Our approach enables robots to handle complex contact-rich manipulation tasks and achieves over 50\\% performance improvement compared to state-of-the-art visuomotor policy methods. For result videos, see https://adaptive-compliance.github.io/",
  "published": "2024-10-12",
  "updated": "2025-03-07",
  "year": "2024",
  "authors": [
   "Yifan Hou",
   "Zeyi Liu",
   "Cheng Chi",
   "Eric Cousineau",
   "Naveen Kuppuswamy",
   "Siyuan Feng",
   "Benjamin Burchfiel",
   "Shuran Song"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA",
  "venue_source": "semantic-scholar",
  "citations": 86,
  "influential_citations": 6,
  "tldr": "Adaptive Compliance Policy is introduced, a novel framework that learns to dynamically adjust system com-pliance both spatially and temporally for given manipulation tasks from human demonstrations, improving upon previous approaches that rely on pre-selected compliance parameters or assume uniform constant stiffness.",
  "doi": "10.1109/ICRA55743.2025.11128452",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yifan Hou",
    "id": "2327832445",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Zeyi Liu",
    "id": "2176845464",
    "h_index": 9,
    "papers": 11
   },
   {
    "name": "Cheng Chi",
    "id": "2253746565",
    "h_index": 8,
    "papers": 11
   },
   {
    "name": "Eric Cousineau",
    "id": "2090529",
    "h_index": 12,
    "papers": 16
   },
   {
    "name": "Naveen Kuppuswamy",
    "id": "2275353318",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Siyuan Feng",
    "id": "2284620540",
    "h_index": 8,
    "papers": 10
   },
   {
    "name": "B. Burchfiel",
    "id": "2302757",
    "h_index": 18,
    "papers": 30
   },
   {
    "name": "Shuran Song",
    "id": "2254874914",
    "h_index": 12,
    "papers": 12
   }
  ],
  "comment": "",
  "topics": [
   "egocentric-data",
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2410.09309v2",
  "pdf_url": "https://arxiv.org/pdf/2410.09309v2",
  "html_url": "https://arxiv.org/html/2410.09309v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.44
 },
 {
  "id": "2410.08464",
  "slug": "arcap-collecting-high-quality-human-demonstrations-for-robot-learning",
  "title": "ARCap: Collecting High-quality Human Demonstrations for Robot Learning with Augmented Reality Feedback",
  "abstract": "Recent progress in imitation learning from human demonstrations has shown promising results in teaching robots manipulation skills. To further scale up training datasets, recent works start to use portable data collection devices without the need for physical robot hardware. However, due to the absence of on-robot feedback during data collection, the data quality depends heavily on user expertise, and many devices are limited to specific robot embodiments. We propose ARCap, a portable data collection system that provides visual feedback through augmented reality (AR) and haptic warnings to guide users in collecting high-quality demonstrations. Through extensive user studies, we show that ARCap enables novice users to collect robot-executable data that matches robot kinematics and avoids collisions with the scenes. With data collected from ARCap, robots can perform challenging tasks, such as manipulation in cluttered environments and long-horizon cross-embodiment manipulation. ARCap is fully open-source and easy to calibrate; all components are built from off-the-shelf products. More details and results can be found on our website: https://stanford-tml.github.io/ARCap",
  "published": "2024-10-11",
  "updated": "2024-10-11",
  "year": "2024",
  "authors": [
   "Sirui Chen",
   "Chen Wang",
   "Kaden Nguyen",
   "Li Fei-Fei",
   "C. Karen Liu"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA",
  "venue_source": "semantic-scholar",
  "citations": 64,
  "influential_citations": 5,
  "tldr": "It is shown that ARCap enables novice users to collect robot-executable data that matches robot kinematics and avoids collisions with the scenes, and with data collected from ARCap, robots can perform challenging tasks, such as manipulation in cluttered environments and long-horizon cross-embodiment manipulation.",
  "doi": "10.1109/ICRA55743.2025.11128717",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Sirui Chen",
    "id": "2209905328",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Chen Wang",
    "id": "2261162246",
    "h_index": 7,
    "papers": 8
   },
   {
    "name": "Kaden Nguyen",
    "id": "2328346839",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Fei-Fei Li",
    "id": "2238030496",
    "h_index": 7,
    "papers": 8
   },
   {
    "name": "C. K. Liu",
    "id": "2278583770",
    "h_index": 5,
    "papers": 7
   }
  ],
  "comment": "8 pages, 8 Figures, submitted to ICRA 2025",
  "topics": [
   "egocentric-data",
   "imitation-diffusion",
   "foundation-pretraining",
   "data-teleop"
  ],
  "orgs": [
   "Stanford"
  ],
  "abs_url": "https://arxiv.org/abs/2410.08464v1",
  "pdf_url": "https://arxiv.org/pdf/2410.08464v1",
  "html_url": "https://arxiv.org/html/2410.08464v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.81
 },
 {
  "id": "2410.07864",
  "slug": "rdt-1b-a-diffusion-foundation-model-for-bimanual-manipulation",
  "title": "RDT-1B: a Diffusion Foundation Model for Bimanual Manipulation",
  "abstract": "Bimanual manipulation is essential in robotics, yet developing foundation models is extremely challenging due to the inherent complexity of coordinating two robot arms (leading to multi-modal action distributions) and the scarcity of training data. In this paper, we present the Robotics Diffusion Transformer (RDT), a pioneering diffusion foundation model for bimanual manipulation. RDT builds on diffusion models to effectively represent multi-modality, with innovative designs of a scalable Transformer to deal with the heterogeneity of multi-modal inputs and to capture the nonlinearity and high frequency of robotic data. To address data scarcity, we further introduce a Physically Interpretable Unified Action Space, which can unify the action representations of various robots while preserving the physical meanings of original actions, facilitating learning transferrable physical knowledge. With these designs, we managed to pre-train RDT on the largest collection of multi-robot datasets to date and scaled it up to 1.2B parameters, which is the largest diffusion-based foundation model for robotic manipulation. We finally fine-tuned RDT on a self-created multi-task bimanual dataset with over 6K+ episodes to refine its manipulation capabilities. Experiments on real robots demonstrate that RDT significantly outperforms existing methods. It exhibits zero-shot generalization to unseen objects and scenes, understands and follows language instructions, learns new skills with just 1~5 demonstrations, and effectively handles complex, dexterous tasks. We refer to https://rdt-robotics.github.io/rdt-robotics/ for the code and videos.",
  "published": "2024-10-10",
  "updated": "2025-03-01",
  "year": "2024",
  "authors": [
   "Songming Liu",
   "Lingxuan Wu",
   "Bangguo Li",
   "Hengkai Tan",
   "Huayu Chen",
   "Zhengyi Wang",
   "Ke Xu",
   "Hang Su",
   "Jun Zhu"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 832,
  "influential_citations": 76,
  "tldr": "A Physically Interpretable Unified Action Space is introduced, which can unify the action representations of various robots while preserving the physical meanings of original actions, facilitating learning transferrable physical knowledge.",
  "doi": "10.48550/arXiv.2410.07864",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Songming Liu",
    "id": "104037450",
    "h_index": 11,
    "papers": 17
   },
   {
    "name": "Lingxuan Wu",
    "id": "2294465744",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Bangguo Li",
    "id": "2342558395",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Hengkai Tan",
    "id": "2303970369",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "Huayu Chen",
    "id": "2257328842",
    "h_index": 11,
    "papers": 18
   },
   {
    "name": "Zhengyi Wang",
    "id": "2241653903",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Ke Xu",
    "id": "2361735991",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Hang Su",
    "id": "2245088111",
    "h_index": 11,
    "papers": 25
   },
   {
    "name": "Jun Zhu",
    "id": "2265957067",
    "h_index": 7,
    "papers": 9
   }
  ],
  "comment": "10 pages, conference",
  "topics": [
   "dexterous-manipulation",
   "foundation-pretraining",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2410.07864v2",
  "pdf_url": "https://arxiv.org/pdf/2410.07864v2",
  "html_url": "https://arxiv.org/html/2410.07864v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.42
 },
 {
  "id": "2410.06756",
  "slug": "dreammesh4d-video-to-4d-generation-with-sparse-controlled-gaussian-mes",
  "title": "DreamMesh4D: Video-to-4D Generation with Sparse-Controlled Gaussian-Mesh Hybrid Representation",
  "abstract": "Recent advancements in 2D/3D generative techniques have facilitated the generation of dynamic 3D objects from monocular videos. Previous methods mainly rely on the implicit neural radiance fields (NeRF) or explicit Gaussian Splatting as the underlying representation, and struggle to achieve satisfactory spatial-temporal consistency and surface appearance. Drawing inspiration from modern 3D animation pipelines, we introduce DreamMesh4D, a novel framework combining mesh representation with geometric skinning technique to generate high-quality 4D object from a monocular video. Instead of utilizing classical texture map for appearance, we bind Gaussian splats to triangle face of mesh for differentiable optimization of both the texture and mesh vertices. In particular, DreamMesh4D begins with a coarse mesh obtained through an image-to-3D generation procedure. Sparse points are then uniformly sampled across the mesh surface, and are used to build a deformation graph to drive the motion of the 3D object for the sake of computational efficiency and providing additional constraint. For each step, transformations of sparse control points are predicted using a deformation network, and the mesh vertices as well as the surface Gaussians are deformed via a novel geometric skinning algorithm, which is a hybrid approach combining LBS (linear blending skinning) and DQS (dual-quaternion skinning), mitigating drawbacks associated with both approaches. The static surface Gaussians and mesh vertices as well as the deformation network are learned via reference view photometric loss, score distillation loss as well as other regularizers in a two-stage manner. Extensive experiments demonstrate superior performance of our method. Furthermore, our method is compatible with modern graphic pipelines, showcasing its potential in the 3D gaming and film industry.",
  "published": "2024-10-09",
  "updated": "2024-10-09",
  "year": "2024",
  "authors": [
   "Zhiqi Li",
   "Yiming Chen",
   "Peidong Liu"
  ],
  "author_count": 3,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 68,
  "influential_citations": 8,
  "tldr": "This work introduces DreamMesh4D, a novel framework combining mesh representation with geometric skinning technique to generate high-quality 4D object from a monocular video, and binds Gaussian splats to triangle face of mesh for differentiable optimization of both the texture and mesh vertices.",
  "doi": "10.48550/arXiv.2410.06756",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhiqi Li",
    "id": "2268372460",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Yiming Chen",
    "id": "2268336838",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Peidong Liu",
    "id": "2256750170",
    "h_index": 8,
    "papers": 18
   }
  ],
  "comment": "NeurIPS 2024",
  "topics": [
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2410.06756v1",
  "pdf_url": "https://arxiv.org/pdf/2410.06756v1",
  "html_url": "https://arxiv.org/html/2410.06756v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.34
 },
 {
  "id": "2410.05677",
  "slug": "t2v-turbo-v2-enhancing-video-generation-model-post-training-through-da",
  "title": "T2V-Turbo-v2: Enhancing Video Generation Model Post-Training through Data, Reward, and Conditional Guidance Design",
  "abstract": "In this paper, we focus on enhancing a diffusion-based text-to-video (T2V) model during the post-training phase by distilling a highly capable consistency model from a pretrained T2V model. Our proposed method, T2V-Turbo-v2, introduces a significant advancement by integrating various supervision signals, including high-quality training data, reward model feedback, and conditional guidance, into the consistency distillation process. Through comprehensive ablation studies, we highlight the crucial importance of tailoring datasets to specific learning objectives and the effectiveness of learning from diverse reward models for enhancing both the visual quality and text-video alignment. Additionally, we highlight the vast design space of conditional guidance strategies, which centers on designing an effective energy function to augment the teacher ODE solver. We demonstrate the potential of this approach by extracting motion guidance from the training datasets and incorporating it into the ODE solver, showcasing its effectiveness in improving the motion quality of the generated videos with the improved motion-related metrics from VBench and T2V-CompBench. Empirically, our T2V-Turbo-v2 establishes a new state-of-the-art result on VBench, with a Total score of 85.13, surpassing proprietary systems such as Gen-3 and Kling.",
  "published": "2024-10-08",
  "updated": "2025-09-16",
  "year": "2024",
  "authors": [
   "Jiachen Li",
   "Qian Long",
   "Jian Zheng",
   "Xiaofeng Gao",
   "Robinson Piramuthu",
   "Wenhu Chen",
   "William Yang Wang"
  ],
  "author_count": 7,
  "categories": [
   "cs.CV",
   "cs.AI"
  ],
  "primary_category": "cs.CV",
  "venue": "ICLR 2025",
  "venue_source": "arxiv-comment",
  "citations": 57,
  "influential_citations": 5,
  "tldr": "This paper introduces a significant advancement by integrating various supervision signals, including high-quality training data, reward model feedback, and conditional guidance, into the consistency distillation process, and establishes a new state-of-the-art result on VBench, surpassing proprietary systems such as Gen-3 and Kling.",
  "doi": "10.48550/arXiv.2410.05677",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jiachen Li",
    "id": "2258750843",
    "h_index": 11,
    "papers": 19
   },
   {
    "name": "Qian Long",
    "id": "2324981827",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Jian Zheng",
    "id": "2300296962",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Xiaofeng Gao",
    "id": "2325188590",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Robinson Piramuthu",
    "id": "2268494451",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Wenhu Chen",
    "id": "2109664620",
    "h_index": 15,
    "papers": 24
   },
   {
    "name": "William Yang Wang",
    "id": "2258793748",
    "h_index": 8,
    "papers": 11
   }
  ],
  "comment": "Accepted by ICLR 2025. Project Page: https://t2v-turbo-v2.github.io/",
  "topics": [
   "rl-control",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2410.05677v3",
  "pdf_url": "https://arxiv.org/pdf/2410.05677v3",
  "html_url": "https://arxiv.org/html/2410.05677v3",
  "code_url": "https://t2v-turbo-v2.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.26
 },
 {
  "id": "2410.03665",
  "slug": "estimating-body-and-hand-motion-in-an-ego-sensed-world",
  "title": "Estimating Body and Hand Motion in an Ego-sensed World",
  "abstract": "We present EgoAllo, a system for human motion estimation from a head-mounted device. Using only egocentric SLAM poses and images, EgoAllo guides sampling from a conditional diffusion model to estimate 3D body pose, height, and hand parameters that capture a device wearer's actions in the allocentric coordinate frame of the scene. To achieve this, our key insight is in representation: we propose spatial and temporal invariance criteria for improving model performance, from which we derive a head motion conditioning parameterization that improves estimation by up to 18%. We also show how the bodies estimated by our system can improve hand estimation: the resulting kinematic and temporal constraints can reduce world-frame errors in single-frame estimates by 40%. Project page: https://egoallo.github.io/",
  "published": "2024-10-04",
  "updated": "2024-12-17",
  "year": "2024",
  "authors": [
   "Brent Yi",
   "Vickie Ye",
   "Maya Zheng",
   "Yunqi Li",
   "Lea M\u00fcller",
   "Georgios Pavlakos",
   "Yi Ma",
   "Jitendra Malik",
   "Angjoo Kanazawa"
  ],
  "author_count": 9,
  "categories": [
   "cs.CV",
   "cs.AI"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 54,
  "influential_citations": 13,
  "tldr": "This work proposes spatial and temporal invariance criteria for improving model performance, from which it derives a head motion conditioning parameterization that improves estimation by up to 18% and shows how the bodies estimated by the system can improve hand estimation.",
  "doi": "10.1109/CVPR52734.2025.00663",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Brent Yi",
    "id": "2242880086",
    "h_index": 18,
    "papers": 27
   },
   {
    "name": "Vickie Ye",
    "id": "31541718",
    "h_index": 12,
    "papers": 15
   },
   {
    "name": "M. Zheng",
    "id": "2374200032",
    "h_index": 4,
    "papers": 15
   },
   {
    "name": "Lea M\u00fcller",
    "id": "2330407100",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "G. Pavlakos",
    "id": "2829330",
    "h_index": 28,
    "papers": 65
   },
   {
    "name": "Yi Ma",
    "id": "2324935696",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Jitendra Malik",
    "id": "2295731211",
    "h_index": 7,
    "papers": 9
   },
   {
    "name": "Angjoo Kanazawa",
    "id": "20615377",
    "h_index": 60,
    "papers": 126
   }
  ],
  "comment": "Project page: https://egoallo.github.io/",
  "topics": [
   "egocentric-data",
   "spatial-3d",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2410.03665v3",
  "pdf_url": "https://arxiv.org/pdf/2410.03665v3",
  "html_url": "https://arxiv.org/html/2410.03665v3",
  "code_url": "https://egoallo.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.24
 },
 {
  "id": "2410.02073",
  "slug": "depth-pro-sharp-monocular-metric-depth-in-less-than-a-second",
  "title": "Depth Pro: Sharp Monocular Metric Depth in Less Than a Second",
  "abstract": "We present a foundation model for zero-shot metric monocular depth estimation. Our model, Depth Pro, synthesizes high-resolution depth maps with unparalleled sharpness and high-frequency details. The predictions are metric, with absolute scale, without relying on the availability of metadata such as camera intrinsics. And the model is fast, producing a 2.25-megapixel depth map in 0.3 seconds on a standard GPU. These characteristics are enabled by a number of technical contributions, including an efficient multi-scale vision transformer for dense prediction, a training protocol that combines real and synthetic datasets to achieve high metric accuracy alongside fine boundary tracing, dedicated evaluation metrics for boundary accuracy in estimated depth maps, and state-of-the-art focal length estimation from a single image. Extensive experiments analyze specific design choices and demonstrate that Depth Pro outperforms prior work along multiple dimensions. We release code and weights at https://github.com/apple/ml-depth-pro",
  "published": "2024-10-02",
  "updated": "2025-04-21",
  "year": "2024",
  "authors": [
   "Aleksei Bochkovskii",
   "Ama\u00ebl Delaunoy",
   "Hugo Germain",
   "Marcel Santos",
   "Yichao Zhou",
   "Stephan R. Richter",
   "Vladlen Koltun"
  ],
  "author_count": 7,
  "categories": [
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.CV",
  "venue": "ICLR 2025",
  "venue_source": "arxiv-comment",
  "citations": 541,
  "influential_citations": 69,
  "tldr": "This work presents a foundation model for zero-shot metric monocular depth estimation, which synthesizes high-resolution depth maps with unparalleled sharpness and high-frequency details and outperforms prior work along multiple dimensions.",
  "doi": "10.48550/arXiv.2410.02073",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Alexey Bochkovskiy",
    "id": "1651204675",
    "h_index": 7,
    "papers": 7
   },
   {
    "name": "Ama\u00ebl Delaunoy",
    "id": "3241610",
    "h_index": 12,
    "papers": 23
   },
   {
    "name": "Hugo Germain",
    "id": "2324055213",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Marcel Santos",
    "id": "2324213392",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Yichao Zhou",
    "id": "2324067238",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Stephan R. Richter",
    "id": "2724721",
    "h_index": 13,
    "papers": 21
   },
   {
    "name": "V. Koltun",
    "id": "145231047",
    "h_index": 114,
    "papers": 239
   }
  ],
  "comment": "Published at ICLR 2025. Code and weights available at https://github.com/apple/ml-depth-pro",
  "topics": [
   "spatial-3d",
   "foundation-pretraining",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2410.02073v2",
  "pdf_url": "https://arxiv.org/pdf/2410.02073v2",
  "html_url": "https://arxiv.org/html/2410.02073v2",
  "code_url": "https://github.com/apple/ml-depth-pro",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.23
 },
 {
  "id": "2410.00425",
  "slug": "maniskill3-gpu-parallelized-robotics-simulation-and-rendering-for-gene",
  "title": "ManiSkill3: GPU Parallelized Robotics Simulation and Rendering for Generalizable Embodied AI",
  "abstract": "Simulation has enabled unprecedented compute-scalable approaches to robot learning. However, many existing simulation frameworks typically support a narrow range of scenes/tasks and lack features critical for scaling generalizable robotics and sim2real. We introduce and open source ManiSkill3, the fastest state-visual GPU parallelized robotics simulator with contact-rich physics targeting generalizable manipulation. ManiSkill3 supports GPU parallelization of many aspects including simulation+rendering, heterogeneous simulation, pointclouds/voxels visual input, and more. Simulation with rendering on ManiSkill3 can run 10-1000x faster with 2-3x less GPU memory usage than other platforms, achieving up to 30,000+ FPS in benchmarked environments due to minimal python/pytorch overhead in the system, simulation on the GPU, and the use of the SAPIEN parallel rendering system. Tasks that used to take hours to train can now take minutes. We further provide the most comprehensive range of GPU parallelized environments/tasks spanning 12 distinct domains including but not limited to mobile manipulation for tasks such as drawing, humanoids, and dextrous manipulation in realistic scenes designed by artists or real-world digital twins. In addition, millions of demonstration frames are provided from motion planning, RL, and teleoperation. ManiSkill3 also provides a comprehensive set of baselines that span popular RL and learning-from-demonstrations algorithms.",
  "published": "2024-10-01",
  "updated": "2025-05-30",
  "year": "2024",
  "authors": [
   "Stone Tao",
   "Fanbo Xiang",
   "Arth Shukla",
   "Yuzhe Qin",
   "Xander Hinrichsen",
   "Xiaodi Yuan",
   "Chen Bao",
   "Xinsong Lin",
   "Yulin Liu",
   "Tse-kai Chan",
   "Yuan Gao",
   "Xuanlin Li",
   "Tongzhou Mu",
   "Nan Xiao",
   "Arnav Gurha",
   "Viswesh Nagaswamy Rajesh",
   "Yong Woo Choi",
   "Yen-Ru Chen",
   "Zhiao Huang",
   "Roberto Calandra",
   "Rui Chen",
   "Shan Luo",
   "Hao Su"
  ],
  "author_count": 23,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 282,
  "influential_citations": 35,
  "tldr": "This work introduces and open source ManiSkill3, the fastest state-visual GPU parallelized robotics simulator with contact-rich physics targeting generalizable manipulation, and provides the most comprehensive range of GPU parallelized environments/tasks spanning 12 distinct domains.",
  "doi": "10.48550/arXiv.2410.00425",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Stone Tao",
    "id": "2147350502",
    "h_index": 9,
    "papers": 18
   },
   {
    "name": "Fanbo Xiang",
    "id": "1572155914",
    "h_index": 13,
    "papers": 22
   },
   {
    "name": "Arth Shukla",
    "id": "2300095744",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Yuzhe Qin",
    "id": "2287945088",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Xander Hinrichsen",
    "id": "2323747890",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Xiao Yuan",
    "id": "2115846638",
    "h_index": 7,
    "papers": 9
   },
   {
    "name": "Chen Bao",
    "id": "2323741946",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Xinsong Lin",
    "id": "2323734850",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Yulin Liu",
    "id": "2292207780",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Tse-kai Chan",
    "id": "2300096196",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Yuan Gao",
    "id": "2323736334",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Xuanlin Li",
    "id": "2108263986",
    "h_index": 17,
    "papers": 21
   },
   {
    "name": "Tongzhou Mu",
    "id": "3431352",
    "h_index": 11,
    "papers": 28
   },
   {
    "name": "Nan Xiao",
    "id": "2323756662",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Arnav Gurha",
    "id": "2323741575",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Zhiao Huang",
    "id": "18036051",
    "h_index": 20,
    "papers": 32
   },
   {
    "name": "Roberto Calandra",
    "id": "2261268365",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Rui Chen",
    "id": "2285620389",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Shan Luo",
    "id": "2324087909",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Hao Su",
    "id": "2273885279",
    "h_index": 4,
    "papers": 7
   }
  ],
  "comment": "Project website: http://maniskill.ai/",
  "topics": [
   "humanoids",
   "tactile",
   "sim2real",
   "navigation",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2410.00425v2",
  "pdf_url": "https://arxiv.org/pdf/2410.00425v2",
  "html_url": "https://arxiv.org/html/2410.00425v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.45
 },
 {
  "id": "2409.20537",
  "slug": "scaling-proprioceptive-visual-learning-with-heterogeneous-pre-trained",
  "title": "Scaling Proprioceptive-Visual Learning with Heterogeneous Pre-trained Transformers",
  "abstract": "One of the roadblocks for training generalist robotic models today is heterogeneity. Previous robot learning methods often collect data to train with one specific embodiment for one task, which is expensive and prone to overfitting. This work studies the problem of learning policy representations through heterogeneous pre-training on robot data across different embodiments and tasks at scale. We propose Heterogeneous Pre-trained Transformers (HPT), which pre-train a large, shareable trunk of a policy neural network to learn a task and embodiment agnostic shared representation. This general architecture aligns the specific proprioception and vision inputs from distinct embodiments to a short sequence of tokens and then processes such tokens to map to control robots for different tasks. Leveraging the recent large-scale multi-embodiment real-world robotic datasets as well as simulation, deployed robots, and human video datasets, we investigate pre-training policies across heterogeneity. We conduct experiments to investigate the scaling behaviors of training objectives, to the extent of 52 datasets. HPTs outperform several baselines and enhance the fine-tuned policy performance by over 20% on unseen tasks in multiple simulator benchmarks and real-world settings. See the project website (https://liruiw.github.io/hpt/) for code and videos.",
  "published": "2024-09-30",
  "updated": "2024-09-30",
  "year": "2024",
  "authors": [
   "Lirui Wang",
   "Xinlei Chen",
   "Jialiang Zhao",
   "Kaiming He"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 186,
  "influential_citations": 11,
  "tldr": "This work proposes Heterogeneous Pre-trained Transformers (HPT), which pre-train a large, shareable trunk of a policy neural network to learn a task and embodiment agnostic shared representation through heterogeneous pre-training on robot data across different embodiments and tasks at scale.",
  "doi": "10.48550/arXiv.2409.20537",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Lirui Wang",
    "id": "2253973819",
    "h_index": 6,
    "papers": 6
   },
   {
    "name": "Xinlei Chen",
    "id": "2281064954",
    "h_index": 8,
    "papers": 10
   },
   {
    "name": "Jialiang Zhao",
    "id": "2243705199",
    "h_index": 12,
    "papers": 18
   },
   {
    "name": "Kaiming He",
    "id": "2058350112",
    "h_index": 9,
    "papers": 11
   }
  ],
  "comment": "See the project website (https://liruiw.github.io/hpt/) for code and videos",
  "topics": [
   "egocentric-data",
   "sim2real",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2409.20537v1",
  "pdf_url": "https://arxiv.org/pdf/2409.20537v1",
  "html_url": "https://arxiv.org/html/2409.20537v1",
  "code_url": "https://liruiw.github.io/hpt/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.77
 },
 {
  "id": "2409.20514",
  "slug": "opt2skill-imitating-dynamically-feasible-whole-body-trajectories-for-v",
  "title": "Opt2Skill: Imitating Dynamically-feasible Whole-Body Trajectories for Versatile Humanoid Loco-Manipulation",
  "abstract": "Humanoid robots are designed to perform diverse loco-manipulation tasks. However, they face challenges due to their high-dimensional and unstable dynamics, as well as the complex contact-rich nature of the tasks. Model-based optimal control methods offer flexibility to define precise motion but are limited by high computational complexity and accurate contact sensing. On the other hand, reinforcement learning (RL) handles high-dimensional spaces with strong robustness but suffers from inefficient learning, unnatural motion, and sim-to-real gaps. To address these challenges, we introduce Opt2Skill, an end-to-end pipeline that combines model-based trajectory optimization with RL to achieve robust whole-body loco-manipulation. Opt2Skill generates dynamic feasible and contact-consistent reference motions for the Digit humanoid robot using differential dynamic programming (DDP) and trains RL policies to track these optimal trajectories. Our results demonstrate that Opt2Skill outperforms baselines that rely on human demonstrations and inverse kinematics-based references, both in motion tracking and task success rates. Furthermore, we show that incorporating trajectories with torque information improves contact force tracking in contact-involved tasks, such as wiping a table. We have successfully transferred our approach to real-world applications.",
  "published": "2024-09-30",
  "updated": "2025-10-01",
  "year": "2024",
  "authors": [
   "Fukang Liu",
   "Zhaoyuan Gu",
   "Yilin Cai",
   "Ziyi Zhou",
   "Hyunyoung Jung",
   "Jaehwi Jang",
   "Shijie Zhao",
   "Sehoon Ha",
   "Yue Chen",
   "Danfei Xu",
   "Ye Zhao"
  ],
  "author_count": 11,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "RA-L",
  "venue_source": "semantic-scholar",
  "citations": 70,
  "influential_citations": 1,
  "tldr": "Opt2Skill, an end-to-end pipeline that combines model-based trajectory optimization with RL to achieve robust whole-body loco-manipulation, is introduced and it is shown that incorporating trajectories with torque information improves contact force tracking in contact-involved tasks, such as wiping a table.",
  "doi": "10.1109/LRA.2025.3620620",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Fukang Liu",
    "id": "2243317194",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Zhaoyuan Gu",
    "id": "2151700177",
    "h_index": 7,
    "papers": 16
   },
   {
    "name": "Yilin Cai",
    "id": "2295803354",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Ziyi Zhou",
    "id": "2121298138",
    "h_index": 10,
    "papers": 30
   },
   {
    "name": "Hyunyoung Jung",
    "id": "2249710352",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Jaehwi Jang",
    "id": "2354615174",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Shijie Zhao",
    "id": "2359962974",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Sehoon Ha",
    "id": "2249214925",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Yue Chen",
    "id": "2242992002",
    "h_index": 3,
    "papers": 15
   },
   {
    "name": "Danfei Xu",
    "id": "2264393671",
    "h_index": 4,
    "papers": 15
   },
   {
    "name": "Ye Zhao",
    "id": "2247917350",
    "h_index": 4,
    "papers": 8
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "egocentric-data",
   "tactile",
   "sim2real",
   "rl-control",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2409.20514v6",
  "pdf_url": "https://arxiv.org/pdf/2409.20514v6",
  "html_url": "https://arxiv.org/html/2409.20514v6",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.35
 },
 {
  "id": "2409.20291",
  "slug": "rl-gsbridge-3d-gaussian-splatting-based-real2sim2real-method-for-robot",
  "title": "RL-GSBridge: 3D Gaussian Splatting Based Real2Sim2Real Method for Robotic Manipulation Learning",
  "abstract": "Sim-to-Real refers to the process of transferring policies learned in simulation to the real world, which is crucial for achieving practical robotics applications. However, recent Sim2real methods either rely on a large amount of augmented data or large learning models, which is inefficient for specific tasks. In recent years, with the emergence of radiance field reconstruction methods, especially 3D Gaussian splatting, it has become possible to construct realistic real-world scenes. To this end, we propose RL-GSBridge, a novel real-to-sim-to-real framework which incorporates 3D Gaussian Splatting into the conventional RL simulation pipeline, enabling zero-shot sim-to-real transfer for vision-based deep reinforcement learning. We introduce a mesh-based 3D GS method with soft binding constraints, enhancing the rendering quality of mesh models. Then utilizing a GS editing approach to synchronize the rendering with the physics simulator, RL-GSBridge could reflect the visual interactions of the physical robot accurately. Through a series of sim-to-real experiments, including grasping and pick-and-place tasks, we demonstrate that RL-GSBridge maintains a satisfactory success rate in real-world task completion during sim-to-real transfer. Furthermore, a series of rendering metrics and visualization results indicate that our proposed mesh-based 3D GS reduces artifacts in unstructured objects, demonstrating more realistic rendering performance.",
  "published": "2024-09-30",
  "updated": "2025-02-22",
  "year": "2024",
  "authors": [
   "Yuxuan Wu",
   "Lei Pan",
   "Wenhua Wu",
   "Guangming Wang",
   "Yanzi Miao",
   "Fan Xu",
   "Hesheng Wang"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 35,
  "influential_citations": 0,
  "tldr": "RL-GSBridge is proposed, a novel real-to-sim-to-real framework which incorporates 3D Gaussian Splatting into the conventional RL simulation pipeline, enabling zero-shot sim-to-real transfer for vision-based deep reinforcement learning and introduces a mesh-based 3D GS method with soft binding constraints, enhancing the rendering quality of mesh models.",
  "doi": "10.48550/arXiv.2409.20291",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yuxuan Wu",
    "id": "2323523127",
    "h_index": 4,
    "papers": 16
   },
   {
    "name": "Lei Pan",
    "id": "2299780397",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Wenhua Wu",
    "id": "2148650892",
    "h_index": 8,
    "papers": 20
   },
   {
    "name": "Guangming Wang",
    "id": "2152583098",
    "h_index": 17,
    "papers": 42
   },
   {
    "name": "Yanzi Miao",
    "id": "2153966121",
    "h_index": 7,
    "papers": 29
   },
   {
    "name": "Hesheng Wang",
    "id": "2302856190",
    "h_index": 4,
    "papers": 7
   }
  ],
  "comment": "7 pages, 5 figures, 4 tables. Accepted by ICRA2025",
  "topics": [
   "dexterous-manipulation",
   "sim2real",
   "rl-control",
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2409.20291v2",
  "pdf_url": "https://arxiv.org/pdf/2409.20291v2",
  "html_url": "https://arxiv.org/html/2409.20291v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.56
 },
 {
  "id": "2409.18121",
  "slug": "robot-see-robot-do-imitating-articulated-object-manipulation-with-mono",
  "title": "Robot See Robot Do: Imitating Articulated Object Manipulation with Monocular 4D Reconstruction",
  "abstract": "Humans can learn to manipulate new objects by simply watching others; providing robots with the ability to learn from such demonstrations would enable a natural interface specifying new behaviors. This work develops Robot See Robot Do (RSRD), a method for imitating articulated object manipulation from a single monocular RGB human demonstration given a single static multi-view object scan. We first propose 4D Differentiable Part Models (4D-DPM), a method for recovering 3D part motion from a monocular video with differentiable rendering. This analysis-by-synthesis approach uses part-centric feature fields in an iterative optimization which enables the use of geometric regularizers to recover 3D motions from only a single video. Given this 4D reconstruction, the robot replicates object trajectories by planning bimanual arm motions that induce the demonstrated object part motion. By representing demonstrations as part-centric trajectories, RSRD focuses on replicating the demonstration's intended behavior while considering the robot's own morphological limits, rather than attempting to reproduce the hand's motion. We evaluate 4D-DPM's 3D tracking accuracy on ground truth annotated 3D part trajectories and RSRD's physical execution performance on 9 objects across 10 trials each on a bimanual YuMi robot. Each phase of RSRD achieves an average of 87% success rate, for a total end-to-end success rate of 60% across 90 trials. Notably, this is accomplished using only feature fields distilled from large pretrained vision models -- without any task-specific training, fine-tuning, dataset collection, or annotation. Project page: https://robot-see-robot-do.github.io",
  "published": "2024-09-26",
  "updated": "2024-09-26",
  "year": "2024",
  "authors": [
   "Justin Kerr",
   "Chung Min Kim",
   "Mingxuan Wu",
   "Brent Yi",
   "Qianqian Wang",
   "Ken Goldberg",
   "Angjoo Kanazawa"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 71,
  "influential_citations": 10,
  "tldr": "This work develops Robot See Robot Do (RSRD), a method for imitating articulated object manipulation from a single monocular RGB human demonstration given a single static multi-view object scan.",
  "doi": "10.48550/arXiv.2409.18121",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Justin Kerr",
    "id": "2311706302",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "C. Kim",
    "id": "2152613913",
    "h_index": 10,
    "papers": 18
   },
   {
    "name": "Mingxuan Wu",
    "id": "2279851824",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Brent Yi",
    "id": "2242880086",
    "h_index": 18,
    "papers": 27
   },
   {
    "name": "Qianqian Wang",
    "id": "2311992344",
    "h_index": 8,
    "papers": 10
   },
   {
    "name": "Ken Goldberg",
    "id": "2279739053",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Angjoo Kanazawa",
    "id": "20615377",
    "h_index": 60,
    "papers": 126
   }
  ],
  "comment": "CoRL 2024, Project page: https://robot-see-robot-do.github.io",
  "topics": [
   "egocentric-data",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2409.18121v1",
  "pdf_url": "https://arxiv.org/pdf/2409.18121v1",
  "html_url": "https://arxiv.org/html/2409.18121v1",
  "code_url": "https://robot-see-robot-do.github.io",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.36
 },
 {
  "id": "2409.16283",
  "slug": "gen2act-human-video-generation-in-novel-scenarios-enables-generalizabl",
  "title": "Gen2Act: Human Video Generation in Novel Scenarios enables Generalizable Robot Manipulation",
  "abstract": "How can robot manipulation policies generalize to novel tasks involving unseen object types and new motions? In this paper, we provide a solution in terms of predicting motion information from web data through human video generation and conditioning a robot policy on the generated video. Instead of attempting to scale robot data collection which is expensive, we show how we can leverage video generation models trained on easily available web data, for enabling generalization. Our approach Gen2Act casts language-conditioned manipulation as zero-shot human video generation followed by execution with a single policy conditioned on the generated video. To train the policy, we use an order of magnitude less robot interaction data compared to what the video prediction model was trained on. Gen2Act doesn't require fine-tuning the video model at all and we directly use a pre-trained model for generating human videos. Our results on diverse real-world scenarios show how Gen2Act enables manipulating unseen object types and performing novel motions for tasks not present in the robot data. Videos are at https://homangab.github.io/gen2act/",
  "published": "2024-09-24",
  "updated": "2024-09-24",
  "year": "2024",
  "authors": [
   "Homanga Bharadhwaj",
   "Debidatta Dwibedi",
   "Abhinav Gupta",
   "Shubham Tulsiani",
   "Carl Doersch",
   "Ted Xiao",
   "Dhruv Shah",
   "Fei Xia",
   "Dorsa Sadigh",
   "Sean Kirmani"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO",
   "cs.CV",
   "cs.LG",
   "eess.IV"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 177,
  "influential_citations": 7,
  "tldr": "The approach Gen2Act casts language-conditioned manipulation as zero-shot human video generation followed by execution with a single policy conditioned on the generated video, which doesn't require fine-tuning the video model at all and directly uses a pre-trained model for generating human videos.",
  "doi": "10.48550/arXiv.2409.16283",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Homanga Bharadhwaj",
    "id": "51113848",
    "h_index": 23,
    "papers": 59
   },
   {
    "name": "Debidatta Dwibedi",
    "id": "2420123",
    "h_index": 19,
    "papers": 35
   },
   {
    "name": "Abhinav Gupta",
    "id": "2240431852",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Shubham Tulsiani",
    "id": "2757335",
    "h_index": 45,
    "papers": 98
   },
   {
    "name": "Carl Doersch",
    "id": "2786693",
    "h_index": 33,
    "papers": 57
   },
   {
    "name": "Ted Xiao",
    "id": "2322507474",
    "h_index": 9,
    "papers": 12
   },
   {
    "name": "Dhruv Shah",
    "id": "2322628540",
    "h_index": 29,
    "papers": 63
   },
   {
    "name": "Fei Xia",
    "id": "2267320085",
    "h_index": 10,
    "papers": 12
   },
   {
    "name": "Dorsa Sadigh",
    "id": "1779671",
    "h_index": 69,
    "papers": 226
   },
   {
    "name": "Sean Kirmani",
    "id": "51881277",
    "h_index": 21,
    "papers": 29
   }
  ],
  "comment": "Preprint. Under Review",
  "topics": [
   "world-models",
   "vla",
   "egocentric-data",
   "data-teleop",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2409.16283v1",
  "pdf_url": "https://arxiv.org/pdf/2409.16283v1",
  "html_url": "https://arxiv.org/html/2409.16283v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.25
 },
 {
  "id": "2409.16211",
  "slug": "maskbit-embedding-free-image-generation-via-bit-tokens",
  "title": "MaskBit: Embedding-free Image Generation via Bit Tokens",
  "abstract": "Masked transformer models for class-conditional image generation have become a compelling alternative to diffusion models. Typically comprising two stages - an initial VQGAN model for transitioning between latent space and image space, and a subsequent Transformer model for image generation within latent space - these frameworks offer promising avenues for image synthesis. In this study, we present two primary contributions: Firstly, an empirical and systematic examination of VQGANs, leading to a modernized VQGAN. Secondly, a novel embedding-free generation network operating directly on bit tokens - a binary quantized representation of tokens with rich semantics. The first contribution furnishes a transparent, reproducible, and high-performing VQGAN model, enhancing accessibility and matching the performance of current state-of-the-art methods while revealing previously undisclosed details. The second contribution demonstrates that embedding-free image generation using bit tokens achieves a new state-of-the-art FID of 1.52 on the ImageNet 256x256 benchmark, with a compact generator model of mere 305M parameters. The code for this project is available on https://github.com/markweberdev/maskbit.",
  "published": "2024-09-24",
  "updated": "2024-12-08",
  "year": "2024",
  "authors": [
   "Mark Weber",
   "Lijun Yu",
   "Qihang Yu",
   "Xueqing Deng",
   "Xiaohui Shen",
   "Daniel Cremers",
   "Liang-Chieh Chen"
  ],
  "author_count": 7,
  "categories": [
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.CV",
  "venue": "Trans. Mach. Learn. Res.",
  "venue_source": "semantic-scholar",
  "citations": 98,
  "influential_citations": 10,
  "tldr": "A transparent, reproducible, and high-performing VQGAN model is furnishes, enhancing accessibility and matching the performance of current state-of-the-art methods while revealing previously undisclosed details, and a novel embedding-free generation network operating directly on bit tokens is presented.",
  "doi": "10.48550/arXiv.2409.16211",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Mark Weber",
    "id": "2110605521",
    "h_index": 8,
    "papers": 10
   },
   {
    "name": "Lijun Yu",
    "id": "2334519247",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Qihang Yu",
    "id": "2304633845",
    "h_index": 10,
    "papers": 19
   },
   {
    "name": "Xueqing Deng",
    "id": "2269123207",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "Xiaohui Shen",
    "id": "2266472250",
    "h_index": 11,
    "papers": 18
   },
   {
    "name": "Daniel Cremers",
    "id": "2311880050",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Liang-Chieh Chen",
    "id": "2266697544",
    "h_index": 13,
    "papers": 26
   }
  ],
  "comment": "Accepted to TMLR w. featured and reproducibility certification. Project page: https://weber-mark.github.io/projects/maskbit.html",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2409.16211v2",
  "pdf_url": "https://arxiv.org/pdf/2409.16211v2",
  "html_url": "https://arxiv.org/html/2409.16211v2",
  "code_url": "https://weber-mark.github.io/projects/maskbit.html",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "2409.12514",
  "slug": "tinyvla-towards-fast-data-efficient-vision-language-action-models-for",
  "title": "TinyVLA: Towards Fast, Data-Efficient Vision-Language-Action Models for Robotic Manipulation",
  "abstract": "Vision-Language-Action (VLA) models have shown remarkable potential in visuomotor control and instruction comprehension through end-to-end learning processes. However, current VLA models face significant challenges: they are slow during inference and require extensive pre-training on large amounts of robotic data, making real-world deployment difficult. In this paper, we introduce a new family of compact vision-language-action models, called TinyVLA, which offers two key advantages over existing VLA models: (1) faster inference speeds, and (2) improved data efficiency, eliminating the need for pre-training stage. Our framework incorporates two essential components to build TinyVLA: (1) initializing the policy backbone with robust, high-speed multimodal models, and (2) integrating a diffusion policy decoder during fine-tuning to enable precise robot actions. We conducted extensive evaluations of TinyVLA in both simulation and on real robots, demonstrating that our approach significantly outperforms the state-of-the-art VLA model, OpenVLA, in terms of speed and data efficiency, while delivering comparable or superior performance. Additionally, TinyVLA exhibits strong generalization capabilities across various dimensions, including language instructions, novel objects, unseen positions, changes in object appearance, background variations, and environmental shifts, often matching or exceeding the performance of OpenVLA. We believe that \\methodname offers an interesting perspective on utilizing pre-trained multimodal models for policy learning. Our project is at https://tiny-vla.github.io.",
  "published": "2024-09-19",
  "updated": "2025-05-13",
  "year": "2024",
  "authors": [
   "Junjie Wen",
   "Yichen Zhu",
   "Jinming Li",
   "Minjie Zhu",
   "Kun Wu",
   "Zhiyuan Xu",
   "Ning Liu",
   "Ran Cheng",
   "Chaomin Shen",
   "Yaxin Peng",
   "Feifei Feng",
   "Jian Tang"
  ],
  "author_count": 12,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "RA-L",
  "venue_source": "semantic-scholar",
  "citations": 415,
  "influential_citations": 14,
  "tldr": "This letter introduces a new family of compact vision-language-action models, called TinyVLA, which offers two key advantages over existing VLA models: faster inference speeds, and improved data efficiency, eliminating the need for pre-training stage.",
  "doi": "10.1109/LRA.2025.3544909",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Junjie Wen",
    "id": "2278247833",
    "h_index": 17,
    "papers": 22
   },
   {
    "name": "Yichen Zhu",
    "id": "2275531481",
    "h_index": 21,
    "papers": 36
   },
   {
    "name": "Jinming Li",
    "id": "2278339388",
    "h_index": 14,
    "papers": 16
   },
   {
    "name": "Minjie Zhu",
    "id": "2277906045",
    "h_index": 15,
    "papers": 18
   },
   {
    "name": "Zhibin Tang",
    "id": "2333823781",
    "h_index": 7,
    "papers": 8
   },
   {
    "name": "Kun Wu",
    "id": "2112562410",
    "h_index": 12,
    "papers": 36
   },
   {
    "name": "Zhiyuan Xu",
    "id": "48559420",
    "h_index": 27,
    "papers": 69
   },
   {
    "name": "Ning Liu",
    "id": "2279764376",
    "h_index": 10,
    "papers": 13
   },
   {
    "name": "Ran Cheng",
    "id": "2321805411",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Chaomin Shen",
    "id": "1684866",
    "h_index": 19,
    "papers": 74
   },
   {
    "name": "Yaxin Peng",
    "id": "2264150574",
    "h_index": 15,
    "papers": 21
   },
   {
    "name": "Feifei Feng",
    "id": "2143808446",
    "h_index": 17,
    "papers": 28
   },
   {
    "name": "Jian Tang",
    "id": "2277743747",
    "h_index": 8,
    "papers": 11
   }
  ],
  "comment": "add more citations",
  "topics": [
   "vla",
   "imitation-diffusion",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2409.12514v5",
  "pdf_url": "https://arxiv.org/pdf/2409.12514v5",
  "html_url": "https://arxiv.org/html/2409.12514v5",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.12
 },
 {
  "id": "2409.12259",
  "slug": "wilor-end-to-end-3d-hand-localization-and-reconstruction-in-the-wild",
  "title": "WiLoR: End-to-end 3D Hand Localization and Reconstruction in-the-wild",
  "abstract": "In recent years, 3D hand pose estimation methods have garnered significant attention due to their extensive applications in human-computer interaction, virtual reality, and robotics. In contrast, there has been a notable gap in hand detection pipelines, posing significant challenges in constructing effective real-world multi-hand reconstruction systems. In this work, we present a data-driven pipeline for efficient multi-hand reconstruction in the wild. The proposed pipeline is composed of two components: a real-time fully convolutional hand localization and a high-fidelity transformer-based 3D hand reconstruction model. To tackle the limitations of previous methods and build a robust and stable detection network, we introduce a large-scale dataset with over than 2M in-the-wild hand images with diverse lighting, illumination, and occlusion conditions. Our approach outperforms previous methods in both efficiency and accuracy on popular 2D and 3D benchmarks. Finally, we showcase the effectiveness of our pipeline to achieve smooth 3D hand tracking from monocular videos, without utilizing any temporal components. Code, models, and dataset are available https://rolpotamias.github.io/WiLoR.",
  "published": "2024-09-18",
  "updated": "2025-03-26",
  "year": "2024",
  "authors": [
   "Rolandos Alexandros Potamias",
   "Jinglei Zhang",
   "Jiankang Deng",
   "Stefanos Zafeiriou"
  ],
  "author_count": 4,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 155,
  "influential_citations": 25,
  "tldr": "This work presents a data-driven pipeline for efficient multi-hand reconstruction in the wild, composed of a real-time fully convolutional hand localization and a high-fidelity transformer-based 3D hand reconstruction model.",
  "doi": "10.1109/CVPR52734.2025.01143",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Rolandos Alexandros Potamias",
    "id": "121927450",
    "h_index": 13,
    "papers": 41
   },
   {
    "name": "Jinglei Zhang",
    "id": "2321886378",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Jiankang Deng",
    "id": "3234063",
    "h_index": 42,
    "papers": 87
   },
   {
    "name": "S. Zafeiriou",
    "id": "1776444",
    "h_index": 83,
    "papers": 448
   }
  ],
  "comment": "CVPR 2025, Project Page https://rolpotamias.github.io/WiLoR",
  "topics": [
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2409.12259v2",
  "pdf_url": "https://arxiv.org/pdf/2409.12259v2",
  "html_url": "https://arxiv.org/html/2409.12259v2",
  "code_url": "https://rolpotamias.github.io/WiLoR",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.69
 },
 {
  "id": "2409.10161",
  "slug": "splatsim-zero-shot-sim2real-transfer-of-rgb-manipulation-policies-usin",
  "title": "SplatSim: Zero-Shot Sim2Real Transfer of RGB Manipulation Policies Using Gaussian Splatting",
  "abstract": "Sim2Real transfer, particularly for manipulation policies relying on RGB images, remains a critical challenge in robotics due to the significant domain shift between synthetic and real-world visual data. In this paper, we propose SplatSim, a novel framework that leverages Gaussian Splatting as the primary rendering primitive to reduce the Sim2Real gap for RGB-based manipulation policies. By replacing traditional mesh representations with Gaussian Splats in simulators, SplatSim produces highly photorealistic synthetic data while maintaining the scalability and cost-efficiency of simulation. We demonstrate the effectiveness of our framework by training manipulation policies within SplatSim and deploying them in the real world in a zero-shot manner, achieving an average success rate of 86.25%, compared to 97.5% for policies trained on real-world data. Videos can be found on our project page: https://splatsim.github.io",
  "published": "2024-09-16",
  "updated": "2024-10-07",
  "year": "2024",
  "authors": [
   "Mohammad Nomaan Qureshi",
   "Sparsh Garg",
   "Francisco Yandun",
   "David Held",
   "George Kantor",
   "Abhisesh Silwal"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA",
  "venue_source": "semantic-scholar",
  "citations": 84,
  "influential_citations": 2,
  "tldr": "SplatSim is proposed, a novel framework that leverages Gaussian Splatting as the primary rendering primitive to reduce the Sim2Real gap for RGB-based manipulation policies by replacing traditional mesh representations with Gaussian Splats in simulators.",
  "doi": "10.1109/ICRA55743.2025.11128339",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "M. N. Qureshi",
    "id": "2277740672",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Sparsh Garg",
    "id": "2321422859",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Francisco Yand\u00fan",
    "id": "40958954",
    "h_index": 9,
    "papers": 21
   },
   {
    "name": "David Held",
    "id": "2249759751",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "George Kantor",
    "id": "2293392037",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Abhisesh Silwal",
    "id": "7536650",
    "h_index": 14,
    "papers": 30
   }
  ],
  "comment": "",
  "topics": [
   "sim2real",
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2409.10161v3",
  "pdf_url": "https://arxiv.org/pdf/2409.10161v3",
  "html_url": "https://arxiv.org/html/2409.10161v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.43
 },
 {
  "id": "2409.08276",
  "slug": "anyskin-plug-and-play-skin-sensing-for-robotic-touch",
  "title": "AnySkin: Plug-and-play Skin Sensing for Robotic Touch",
  "abstract": "While tactile sensing is widely accepted as an important and useful sensing modality, its use pales in comparison to other sensory modalities like vision and proprioception. AnySkin addresses the critical challenges that impede the use of tactile sensing -- versatility, replaceability, and data reusability. Building on the simplistic design of ReSkin, and decoupling the sensing electronics from the sensing interface, AnySkin simplifies integration making it as straightforward as putting on a phone case and connecting a charger. Furthermore, AnySkin is the first uncalibrated tactile-sensor with cross-instance generalizability of learned manipulation policies. To summarize, this work makes three key contributions: first, we introduce a streamlined fabrication process and a design tool for creating an adhesive-free, durable and easily replaceable magnetic tactile sensor; second, we characterize slip detection and policy learning with the AnySkin sensor; and third, we demonstrate zero-shot generalization of models trained on one instance of AnySkin to new instances, and compare it with popular existing tactile solutions like DIGIT and ReSkin. Videos of experiments, fabrication details and design files can be found on https://any-skin.github.io/",
  "published": "2024-09-12",
  "updated": "2024-09-27",
  "year": "2024",
  "authors": [
   "Raunaq Bhirangi",
   "Venkatesh Pattabiraman",
   "Enes Erciyes",
   "Yifeng Cao",
   "Tess Hellebrekers",
   "Lerrel Pinto"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA",
  "venue_source": "semantic-scholar",
  "citations": 56,
  "influential_citations": 2,
  "tldr": "AnySkin is the first uncalibrated tactile-sensor to report crossinstance generalizability of learned manipulation policies, and is the first uncalibrated tactile-sensor to report crossinstance generalizability of learned manipulation policies.",
  "doi": "10.1109/ICRA55743.2025.11128638",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Raunaq M. Bhirangi",
    "id": "152466940",
    "h_index": 11,
    "papers": 17
   },
   {
    "name": "Venkatesh Pattabiraman",
    "id": "2284217920",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Enes Erciyes",
    "id": "2320800733",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Yifeng Cao",
    "id": "2320891592",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "T. Hellebrekers",
    "id": "2576308",
    "h_index": 18,
    "papers": 31
   },
   {
    "name": "Lerrel Pinto",
    "id": "2320806817",
    "h_index": 10,
    "papers": 12
   }
  ],
  "comment": "",
  "topics": [
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2409.08276v3",
  "pdf_url": "https://arxiv.org/pdf/2409.08276v3",
  "html_url": "https://arxiv.org/html/2409.08276v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.26
 },
 {
  "id": "2409.08273",
  "slug": "hand-object-interaction-pretraining-from-videos",
  "title": "Hand-Object Interaction Pretraining from Videos",
  "abstract": "We present an approach to learn general robot manipulation priors from 3D hand-object interaction trajectories. We build a framework to use in-the-wild videos to generate sensorimotor robot trajectories. We do so by lifting both the human hand and the manipulated object in a shared 3D space and retargeting human motions to robot actions. Generative modeling on this data gives us a task-agnostic base policy. This policy captures a general yet flexible manipulation prior. We empirically demonstrate that finetuning this policy, with both reinforcement learning (RL) and behavior cloning (BC), enables sample-efficient adaptation to downstream tasks and simultaneously improves robustness and generalizability compared to prior approaches. Qualitative experiments are available at: \\url{https://hgaurav2k.github.io/hop/}.",
  "published": "2024-09-12",
  "updated": "2024-09-12",
  "year": "2024",
  "authors": [
   "Himanshu Gaurav Singh",
   "Antonio Loquercio",
   "Carmelo Sferrazza",
   "Jane Wu",
   "Haozhi Qi",
   "Pieter Abbeel",
   "Jitendra Malik"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA",
  "venue_source": "semantic-scholar",
  "citations": 56,
  "influential_citations": 2,
  "tldr": "An approach to learn general robot manipulation priors from 3D hand-object interaction trajectories from in-the-wild videos and empirically demonstrates that finetuning this policy enables sample-efficient adaptation to downstream tasks and simultaneously improves robustness and generalizability compared to prior approaches.",
  "doi": "10.1109/ICRA55743.2025.11127811",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Himanshu Gaurav Singh",
    "id": "2320952944",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Antonio Loquercio",
    "id": "20580939",
    "h_index": 26,
    "papers": 64
   },
   {
    "name": "Carmelo Sferrazza",
    "id": "47218071",
    "h_index": 21,
    "papers": 44
   },
   {
    "name": "Jane Wu",
    "id": "2295780115",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Haozhi Qi",
    "id": "2247951244",
    "h_index": 11,
    "papers": 18
   },
   {
    "name": "Pieter Abbeel",
    "id": "2257003229",
    "h_index": 11,
    "papers": 17
   },
   {
    "name": "Jitendra Malik",
    "id": "2242761335",
    "h_index": 15,
    "papers": 32
   }
  ],
  "comment": "",
  "topics": [
   "imitation-diffusion",
   "rl-control",
   "foundation-pretraining",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2409.08273v1",
  "pdf_url": "https://arxiv.org/pdf/2409.08273v1",
  "html_url": "https://arxiv.org/html/2409.08273v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.26
 },
 {
  "id": "2409.06765",
  "slug": "gsplat-an-open-source-library-for-gaussian-splatting",
  "title": "gsplat: An Open-Source Library for Gaussian Splatting",
  "abstract": "gsplat is an open-source library designed for training and developing Gaussian Splatting methods. It features a front-end with Python bindings compatible with the PyTorch library and a back-end with highly optimized CUDA kernels. gsplat offers numerous features that enhance the optimization of Gaussian Splatting models, which include optimization improvements for speed, memory, and convergence times. Experimental results demonstrate that gsplat achieves up to 10% less training time and 4x less memory than the original implementation. Utilized in several research projects, gsplat is actively maintained on GitHub. Source code is available at https://github.com/nerfstudio-project/gsplat under Apache License 2.0. We welcome contributions from the open-source community.",
  "published": "2024-09-10",
  "updated": "2024-09-10",
  "year": "2024",
  "authors": [
   "Vickie Ye",
   "Ruilong Li",
   "Justin Kerr",
   "Matias Turkulainen",
   "Brent Yi",
   "Zhuoyang Pan",
   "Otto Seiskari",
   "Jianbo Ye",
   "Jeffrey Hu",
   "Matthew Tancik",
   "Angjoo Kanazawa"
  ],
  "author_count": 11,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 358,
  "influential_citations": 20,
  "tldr": "Gsplat offers numerous features that enhance the optimization of Gaussian Splatting models, which include optimization improvements for speed, memory, and convergence times.",
  "doi": "10.48550/arXiv.2409.06765",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Vickie Ye",
    "id": "31541718",
    "h_index": 12,
    "papers": 15
   },
   {
    "name": "Ruilong Li",
    "id": "51104559",
    "h_index": 11,
    "papers": 13
   },
   {
    "name": "Justin Kerr",
    "id": "2311706302",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Matias Turkulainen",
    "id": "2292258917",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Brent Yi",
    "id": "2242880086",
    "h_index": 18,
    "papers": 27
   },
   {
    "name": "Zhuoyang Pan",
    "id": "2320984245",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Otto Seiskari",
    "id": "2449081",
    "h_index": 7,
    "papers": 14
   },
   {
    "name": "Jianbo Ye",
    "id": "145581826",
    "h_index": 18,
    "papers": 39
   },
   {
    "name": "Jeffrey Hu",
    "id": "2276188701",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Matthew Tancik",
    "id": "7638730",
    "h_index": 27,
    "papers": 47
   },
   {
    "name": "Angjoo Kanazawa",
    "id": "20615377",
    "h_index": 60,
    "papers": 126
   }
  ],
  "comment": "17 pages, 2 figures, JMLR MLOSS",
  "topics": [
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2409.06765v1",
  "pdf_url": "https://arxiv.org/pdf/2409.06765v1",
  "html_url": "https://arxiv.org/html/2409.06765v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.56
 },
 {
  "id": "2409.05865",
  "slug": "robot-utility-models-general-policies-for-zero-shot-deployment-in-new",
  "title": "Robot Utility Models: General Policies for Zero-Shot Deployment in New Environments",
  "abstract": "Robot models, particularly those trained with large amounts of data, have recently shown a plethora of real-world manipulation and navigation capabilities. Several independent efforts have shown that given sufficient training data in an environment, robot policies can generalize to demonstrated variations in that environment. However, needing to finetune robot models to every new environment stands in stark contrast to models in language or vision that can be deployed zero-shot for open-world problems. In this work, we present Robot Utility Models (RUMs), a framework for training and deploying zero-shot robot policies that can directly generalize to new environments without any finetuning. To create RUMs efficiently, we develop new tools to quickly collect data for mobile manipulation tasks, integrate such data into a policy with multi-modal imitation learning, and deploy policies on-device on Hello Robot Stretch, a cheap commodity robot, with an external mLLM verifier for retrying. We train five such utility models for opening cabinet doors, opening drawers, picking up napkins, picking up paper bags, and reorienting fallen objects. Our system, on average, achieves 90% success rate in unseen, novel environments interacting with unseen objects. Moreover, the utility models can also succeed in different robot and camera set-ups with no further data, training, or fine-tuning. Primary among our lessons are the importance of training data over training algorithm and policy class, guidance about data scaling, necessity for diverse yet high-quality demonstrations, and a recipe for robot introspection and retrying to improve performance on individual environments. Our code, data, models, hardware designs, as well as our experiment and deployment videos are open sourced and can be found on our project website: https://robotutilitymodels.com",
  "published": "2024-09-09",
  "updated": "2024-09-09",
  "year": "2024",
  "authors": [
   "Haritheja Etukuru",
   "Norihito Naka",
   "Zijin Hu",
   "Seungjae Lee",
   "Julian Mehu",
   "Aaron Edsinger",
   "Chris Paxton",
   "Soumith Chintala",
   "Lerrel Pinto",
   "Nur Muhammad Mahi Shafiullah"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA",
  "venue_source": "semantic-scholar",
  "citations": 55,
  "influential_citations": 2,
  "tldr": "This work presents Robot Utility Models (RUMs), a framework for training and deploying zero-shot robot policies that can directly generalize to new environments without any finetuning.",
  "doi": "10.1109/ICRA55743.2025.11127857",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Haritheja Etukuru",
    "id": "2268398574",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "N. Naka",
    "id": "153329078",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Zijing Hu",
    "id": "2350526300",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Seungjae Lee",
    "id": "2361410538",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Julian Mehu",
    "id": "2291557640",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "A. Edsinger",
    "id": "2326394",
    "h_index": 20,
    "papers": 36
   },
   {
    "name": "Chris Paxton",
    "id": "2280135281",
    "h_index": 8,
    "papers": 10
   },
   {
    "name": "Soumith Chintala",
    "id": "2127604",
    "h_index": 32,
    "papers": 49
   },
   {
    "name": "Lerrel Pinto",
    "id": "2253567347",
    "h_index": 15,
    "papers": 25
   },
   {
    "name": "Nur Muhammad (Mahi) Shafiullah",
    "id": "84146411",
    "h_index": 9,
    "papers": 11
   }
  ],
  "comment": "Project website https://robotutilitymodels.com",
  "topics": [
   "imitation-diffusion",
   "navigation",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2409.05865v1",
  "pdf_url": "https://arxiv.org/pdf/2409.05865v1",
  "html_url": "https://arxiv.org/html/2409.05865v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.25
 },
 {
  "id": "2409.04410",
  "slug": "open-magvit2-an-open-source-project-toward-democratizing-auto-regressi",
  "title": "Open-MAGVIT2: An Open-Source Project Toward Democratizing Auto-regressive Visual Generation",
  "abstract": "The Open-MAGVIT2 project produces an open-source replication of Google's MAGVIT-v2 tokenizer, a tokenizer with a super-large codebook (i.e., $2^{18}$ codes), and achieves the state-of-the-art reconstruction performance on ImageNet and UCF benchmarks. We also provide a tokenizer pre-trained on large-scale data, significantly outperforming Cosmos on zero-shot benchmarks (1.93 vs. 0.78 rFID on ImageNet original resolution). Furthermore, we explore its application in plain auto-regressive models to validate scalability properties, producing a family of auto-regressive image generation models ranging from 300M to 1.5B. To assist auto-regressive models in predicting with a super-large vocabulary, we factorize it into two sub-vocabulary of different sizes by asymmetric token factorization, and further introduce ``next sub-token prediction'' to enhance sub-token interaction for better generation quality. We release all models and codes to foster innovation and creativity in the field of auto-regressive visual generation.",
  "published": "2024-09-06",
  "updated": "2025-02-09",
  "year": "2024",
  "authors": [
   "Zhuoyan Luo",
   "Fengyuan Shi",
   "Yixiao Ge",
   "Yujiu Yang",
   "Limin Wang",
   "Ying Shan"
  ],
  "author_count": 6,
  "categories": [
   "cs.CV",
   "cs.AI"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 156,
  "influential_citations": 15,
  "tldr": "",
  "doi": "10.48550/arXiv.2409.04410",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhuoyan Luo",
    "id": "2219070252",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Fengyuan Shi",
    "id": "2261736425",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Yixiao Ge",
    "id": "152988335",
    "h_index": 45,
    "papers": 97
   },
   {
    "name": "Yujiu Yang",
    "id": "2283881403",
    "h_index": 28,
    "papers": 181
   },
   {
    "name": "Limin Wang",
    "id": "2261792357",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Ying Shan",
    "id": "2265579883",
    "h_index": 27,
    "papers": 50
   }
  ],
  "comment": "",
  "topics": [
   "video-generation"
  ],
  "orgs": [
   "NVIDIA"
  ],
  "abs_url": "https://arxiv.org/abs/2409.04410v3",
  "pdf_url": "https://arxiv.org/pdf/2409.04410v3",
  "html_url": "https://arxiv.org/html/2409.04410v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.7
 },
 {
  "id": "2409.03403",
  "slug": "rovi-aug-robot-and-viewpoint-augmentation-for-cross-embodiment-robot-l",
  "title": "RoVi-Aug: Robot and Viewpoint Augmentation for Cross-Embodiment Robot Learning",
  "abstract": "Scaling up robot learning requires large and diverse datasets, and how to efficiently reuse collected data and transfer policies to new embodiments remains an open question. Emerging research such as the Open-X Embodiment (OXE) project has shown promise in leveraging skills by combining datasets including different robots. However, imbalances in the distribution of robot types and camera angles in many datasets make policies prone to overfit. To mitigate this issue, we propose RoVi-Aug, which leverages state-of-the-art image-to-image generative models to augment robot data by synthesizing demonstrations with different robots and camera views. Through extensive physical experiments, we show that, by training on robot- and viewpoint-augmented data, RoVi-Aug can zero-shot deploy on an unseen robot with significantly different camera angles. Compared to test-time adaptation algorithms such as Mirage, RoVi-Aug requires no extra processing at test time, does not assume known camera angles, and allows policy fine-tuning. Moreover, by co-training on both the original and augmented robot datasets, RoVi-Aug can learn multi-robot and multi-task policies, enabling more efficient transfer between robots and skills and improving success rates by up to 30%. Project website: https://rovi-aug.github.io.",
  "published": "2024-09-05",
  "updated": "2024-09-09",
  "year": "2024",
  "authors": [
   "Lawrence Yunliang Chen",
   "Chenfeng Xu",
   "Karthik Dharmarajan",
   "Muhammad Zubair Irshad",
   "Richard Cheng",
   "Kurt Keutzer",
   "Masayoshi Tomizuka",
   "Quan Vuong",
   "Ken Goldberg"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 76,
  "influential_citations": 6,
  "tldr": "RoVi-Aug is proposed, which leverages state-of-the-art image-to-image generative models to augment robot data by synthesizing demonstrations with different robots and camera views, enabling more efficient transfer between robots and skills and improving success rates by up to 30%.",
  "doi": "10.48550/arXiv.2409.03403",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "L. Chen",
    "id": "2143804724",
    "h_index": 14,
    "papers": 24
   },
   {
    "name": "Chenfeng Xu",
    "id": "1490695028",
    "h_index": 25,
    "papers": 46
   },
   {
    "name": "K. Dharmarajan",
    "id": "1436202543",
    "h_index": 10,
    "papers": 19
   },
   {
    "name": "Muhammad Zubair Irshad",
    "id": "147495445",
    "h_index": 17,
    "papers": 37
   },
   {
    "name": "Richard Cheng",
    "id": "2321978405",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Kurt Keutzer",
    "id": "2242659602",
    "h_index": 31,
    "papers": 115
   },
   {
    "name": "Masayoshi Tomizuka",
    "id": "2293316662",
    "h_index": 8,
    "papers": 20
   },
   {
    "name": "Quan Vuong",
    "id": "2288210223",
    "h_index": 19,
    "papers": 29
   },
   {
    "name": "Ken Goldberg",
    "id": "2253727638",
    "h_index": 8,
    "papers": 13
   }
  ],
  "comment": "CoRL 2024 (Oral). Project website: https://rovi-aug.github.io",
  "topics": [
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2409.03403v2",
  "pdf_url": "https://arxiv.org/pdf/2409.03403v2",
  "html_url": "https://arxiv.org/html/2409.03403v2",
  "code_url": "https://rovi-aug.github.io",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.39
 },
 {
  "id": "2409.01652",
  "slug": "rekep-spatio-temporal-reasoning-of-relational-keypoint-constraints-for",
  "title": "ReKep: Spatio-Temporal Reasoning of Relational Keypoint Constraints for Robotic Manipulation",
  "abstract": "Representing robotic manipulation tasks as constraints that associate the robot and the environment is a promising way to encode desired robot behaviors. However, it remains unclear how to formulate the constraints such that they are 1) versatile to diverse tasks, 2) free of manual labeling, and 3) optimizable by off-the-shelf solvers to produce robot actions in real-time. In this work, we introduce Relational Keypoint Constraints (ReKep), a visually-grounded representation for constraints in robotic manipulation. Specifically, ReKep is expressed as Python functions mapping a set of 3D keypoints in the environment to a numerical cost. We demonstrate that by representing a manipulation task as a sequence of Relational Keypoint Constraints, we can employ a hierarchical optimization procedure to solve for robot actions (represented by a sequence of end-effector poses in SE(3)) with a perception-action loop at a real-time frequency. Furthermore, in order to circumvent the need for manual specification of ReKep for each new task, we devise an automated procedure that leverages large vision models and vision-language models to produce ReKep from free-form language instructions and RGB-D observations. We present system implementations on a wheeled single-arm platform and a stationary dual-arm platform that can perform a large variety of manipulation tasks, featuring multi-stage, in-the-wild, bimanual, and reactive behaviors, all without task-specific data or environment models. Website at https://rekep-robot.github.io/.",
  "published": "2024-09-03",
  "updated": "2024-11-12",
  "year": "2024",
  "authors": [
   "Wenlong Huang",
   "Chen Wang",
   "Yunzhu Li",
   "Ruohan Zhang",
   "Li Fei-Fei"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 398,
  "influential_citations": 40,
  "tldr": "This work introduces Relational Keypoint Constraints (ReKep), a visually-grounded representation for constraints in robotic manipulation that can employ a hierarchical optimization procedure to solve for robot actions with a perception-action loop at a real-time frequency.",
  "doi": "10.48550/arXiv.2409.01652",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Wenlong Huang",
    "id": "2319777572",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Chen Wang",
    "id": "2261162246",
    "h_index": 7,
    "papers": 8
   },
   {
    "name": "Yunzhu Li",
    "id": "2319504878",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Ruohan Zhang",
    "id": "2285397162",
    "h_index": 8,
    "papers": 17
   },
   {
    "name": "Fei-Fei Li",
    "id": "2238030496",
    "h_index": 7,
    "papers": 8
   }
  ],
  "comment": "",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2409.01652v2",
  "pdf_url": "https://arxiv.org/pdf/2409.01652v2",
  "html_url": "https://arxiv.org/html/2409.01652v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.1
 },
 {
  "id": "2409.01083",
  "slug": "affordance-based-robot-manipulation-with-flow-matching",
  "title": "Affordance-based Robot Manipulation with Flow Matching",
  "abstract": "We present a framework for assistive robot manipulation, which focuses on two fundamental challenges: first, efficiently adapting large-scale models to downstream scene affordance understanding tasks, especially in daily living scenarios where gathering multi-task data involving humans requires strenuous effort; second, effectively learning robot action trajectories by grounding the visual affordance model. We tackle the first challenge by employing a parameter-efficient prompt tuning method that prepends learnable text prompts to the frozen vision model to predict manipulation affordances in multi-task scenarios. Then we propose to learn robot action trajectories guided by affordances in a supervised flow matching method. Flow matching represents a robot visuomotor policy as a conditional process of flowing random waypoints to desired robot action trajectories. Finally, we introduce a real-world dataset with 10 tasks across Activities of Daily Living to test our framework. Our extensive evaluation highlights that the proposed prompt tuning method for learning manipulation affordance achieves competitive performance and even outperforms some other finetuning protocols across data scales, while satisfying parameter efficiency. Learning multi-task robot action trajectories with flow matching leads to consistently favorable results in several robot manipulation benchmarks than some alternative behavior cloning methods. This includes more stable training and evaluation, and noticeably faster inference, while maintaining comparable generalization performance to diffusion policy, where flow matching performs marginally better in most cases. Our framework seamlessly unifies affordance learning and action generation with flow matching for robot manipulation.",
  "published": "2024-09-02",
  "updated": "2025-11-07",
  "year": "2024",
  "authors": [
   "Fan Zhang",
   "Michael Gienger"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 78,
  "influential_citations": 4,
  "tldr": "This framework seamlessly unifies affordance learning and action generation with flow matching for robot manipulation, which leads to consistently favorable results in several robot manipulation benchmarks than some alternative behavior cloning methods.",
  "doi": "10.48550/arXiv.2409.01083",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Fan Zhang",
    "id": "2319382266",
    "h_index": 4,
    "papers": 16
   },
   {
    "name": "Michael Gienger",
    "id": "2300094695",
    "h_index": 6,
    "papers": 21
   }
  ],
  "comment": "",
  "topics": [
   "imitation-diffusion",
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2409.01083v5",
  "pdf_url": "https://arxiv.org/pdf/2409.01083v5",
  "html_url": "https://arxiv.org/html/2409.01083v5",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.9
 },
 {
  "id": "2408.14837",
  "slug": "diffusion-models-are-real-time-game-engines",
  "title": "Diffusion Models Are Real-Time Game Engines",
  "abstract": "We present GameNGen, the first game engine powered entirely by a neural model that also enables real-time interaction with a complex environment over long trajectories at high quality. When trained on the classic game DOOM, GameNGen extracts gameplay and uses it to generate a playable environment that can interactively simulate new trajectories. GameNGen runs at 20 frames per second on a single TPU and remains stable over extended multi-minute play sessions. Next frame prediction achieves a PSNR of 29.4, comparable to lossy JPEG compression. Human raters are only slightly better than random chance at distinguishing short clips of the game from clips of the simulation, even after 5 minutes of auto-regressive generation. GameNGen is trained in two phases: (1) an RL-agent learns to play the game and the training sessions are recorded, and (2) a diffusion model is trained to produce the next frame, conditioned on the sequence of past frames and actions. Conditioning augmentations help ensure stable auto-regressive generation over long trajectories, and decoder fine-tuning improves the fidelity of visual details and text.",
  "published": "2024-08-27",
  "updated": "2025-04-24",
  "year": "2024",
  "authors": [
   "Dani Valevski",
   "Yaniv Leviathan",
   "Moab Arar",
   "Shlomi Fruchter"
  ],
  "author_count": 4,
  "categories": [
   "cs.LG",
   "cs.AI",
   "cs.CV"
  ],
  "primary_category": "cs.LG",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 277,
  "influential_citations": 26,
  "tldr": "GameNGen is the first game engine powered entirely by a neural model that also enables real-time interaction with a complex environment over long trajectories at high quality and conditioning augmentations help ensure stable auto-regressive generation over long trajectories.",
  "doi": "10.48550/arXiv.2408.14837",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Dani Valevski",
    "id": "2188055998",
    "h_index": 7,
    "papers": 7
   },
   {
    "name": "Yaniv Leviathan",
    "id": "2188067081",
    "h_index": 11,
    "papers": 12
   },
   {
    "name": "Moab Arar",
    "id": "34905004",
    "h_index": 11,
    "papers": 18
   },
   {
    "name": "Shlomi Fruchter",
    "id": "2266840927",
    "h_index": 7,
    "papers": 7
   }
  ],
  "comment": "ICLR 2025. Project page: https://gamengen.github.io/",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2408.14837v2",
  "pdf_url": "https://arxiv.org/pdf/2408.14837v2",
  "html_url": "https://arxiv.org/html/2408.14837v2",
  "code_url": "https://gamengen.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.94
 },
 {
  "id": "2408.14037",
  "slug": "re-mix-optimizing-data-mixtures-for-large-scale-imitation-learning",
  "title": "Re-Mix: Optimizing Data Mixtures for Large Scale Imitation Learning",
  "abstract": "Increasingly large imitation learning datasets are being collected with the goal of training foundation models for robotics. However, despite the fact that data selection has been of utmost importance in vision and natural language processing, little work in robotics has questioned what data such models should actually be trained on. In this work we investigate how to weigh different subsets or ``domains'' of robotics datasets for robot foundation model pre-training. Concrete, we use distributionally robust optimization (DRO) to maximize worst-case performance across all possible downstream domains. Our method, Re-Mix, addresses the wide range of challenges that arise when applying DRO to robotics datasets including variability in action spaces and dynamics across different datasets. Re-Mix employs early stopping, action normalization, and discretization to counteract these issues. Through extensive experimentation on the largest open-source robot manipulation dataset, the Open X-Embodiment dataset, we demonstrate that data curation can have an outsized impact on downstream performance. Specifically, domain weights learned by Re-Mix outperform uniform weights by 38\\% on average and outperform human-selected weights by 32\\% on datasets used to train existing generalist robot policies, specifically the RT-X models.",
  "published": "2024-08-26",
  "updated": "2024-08-26",
  "year": "2024",
  "authors": [
   "Joey Hejna",
   "Chethan Bhateja",
   "Yichen Jiang",
   "Karl Pertsch",
   "Dorsa Sadigh"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 59,
  "influential_citations": 4,
  "tldr": "",
  "doi": "10.48550/arXiv.2408.14037",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Joey Hejna",
    "id": "2122700519",
    "h_index": 17,
    "papers": 27
   },
   {
    "name": "Chethan Bhateja",
    "id": "2214093446",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Yichen Jiang",
    "id": "2373987498",
    "h_index": 4,
    "papers": 31
   },
   {
    "name": "Karl Pertsch",
    "id": "31719101",
    "h_index": 32,
    "papers": 56
   },
   {
    "name": "Dorsa Sadigh",
    "id": "1779671",
    "h_index": 69,
    "papers": 226
   }
  ],
  "comment": "",
  "topics": [
   "imitation-diffusion",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2408.14037v1",
  "pdf_url": "https://arxiv.org/pdf/2408.14037v1",
  "html_url": "https://arxiv.org/html/2408.14037v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.28
 },
 {
  "id": "2408.12588",
  "slug": "real-time-video-generation-with-pyramid-attention-broadcast",
  "title": "Real-Time Video Generation with Pyramid Attention Broadcast",
  "abstract": "We present Pyramid Attention Broadcast (PAB), a real-time, high quality and training-free approach for DiT-based video generation. Our method is founded on the observation that attention difference in the diffusion process exhibits a U-shaped pattern, indicating significant redundancy. We mitigate this by broadcasting attention outputs to subsequent steps in a pyramid style. It applies different broadcast strategies to each attention based on their variance for best efficiency. We further introduce broadcast sequence parallel for more efficient distributed inference. PAB demonstrates up to 10.5x speedup across three models compared to baselines, achieving real-time generation for up to 720p videos. We anticipate that our simple yet effective method will serve as a robust baseline and facilitate future research and application for video generation.",
  "published": "2024-08-22",
  "updated": "2025-02-27",
  "year": "2024",
  "authors": [
   "Xuanlei Zhao",
   "Xiaolong Jin",
   "Kai Wang",
   "Yang You"
  ],
  "author_count": 4,
  "categories": [
   "cs.CV",
   "cs.DC"
  ],
  "primary_category": "cs.CV",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 144,
  "influential_citations": 24,
  "tldr": "",
  "doi": "10.48550/arXiv.2408.12588",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xu Zhao",
    "id": "2238738307",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Xiaolong Jin",
    "id": "2314891898",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Kai Wang",
    "id": "2314739123",
    "h_index": 7,
    "papers": 15
   },
   {
    "name": "Yang You",
    "id": "2314810632",
    "h_index": 8,
    "papers": 22
   }
  ],
  "comment": "ICLR 2025",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2408.12588v3",
  "pdf_url": "https://arxiv.org/pdf/2408.12588v3",
  "html_url": "https://arxiv.org/html/2408.12588v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.66
 },
 {
  "id": "2408.11812",
  "slug": "scaling-cross-embodied-learning-one-policy-for-manipulation-navigation",
  "title": "Scaling Cross-Embodied Learning: One Policy for Manipulation, Navigation, Locomotion and Aviation",
  "abstract": "Modern machine learning systems rely on large datasets to attain broad generalization, and this often poses a challenge in robot learning, where each robotic platform and task might have only a small dataset. By training a single policy across many different kinds of robots, a robot learning method can leverage much broader and more diverse datasets, which in turn can lead to better generalization and robustness. However, training a single policy on multi-robot data is challenging because robots can have widely varying sensors, actuators, and control frequencies. We propose CrossFormer, a scalable and flexible transformer-based policy that can consume data from any embodiment. We train CrossFormer on the largest and most diverse dataset to date, 900K trajectories across 20 different robot embodiments. We demonstrate that the same network weights can control vastly different robots, including single and dual arm manipulation systems, wheeled robots, quadcopters, and quadrupeds. Unlike prior work, our model does not require manual alignment of the observation or action spaces. Extensive experiments in the real world show that our method matches the performance of specialist policies tailored for each embodiment, while also significantly outperforming the prior state of the art in cross-embodiment learning.",
  "published": "2024-08-21",
  "updated": "2024-08-21",
  "year": "2024",
  "authors": [
   "Ria Doshi",
   "Homer Walke",
   "Oier Mees",
   "Sudeep Dasari",
   "Sergey Levine"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 157,
  "influential_citations": 8,
  "tldr": "This work proposes CrossFormer, a scalable and flexible transformer-based policy that can consume data from any embodiment, and demonstrates that the same network weights can control vastly different robots, including single and dual arm manipulation systems, wheeled robots, quadcopters, and quadrupeds.",
  "doi": "10.48550/arXiv.2408.11812",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ria Doshi",
    "id": "2197078118",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "H. Walke",
    "id": "2029241116",
    "h_index": 17,
    "papers": 23
   },
   {
    "name": "Oier Mees",
    "id": "7264115",
    "h_index": 28,
    "papers": 45
   },
   {
    "name": "S. Dasari",
    "id": "36076404",
    "h_index": 23,
    "papers": 39
   },
   {
    "name": "Sergey Levine",
    "id": "2249615151",
    "h_index": 32,
    "papers": 53
   }
  ],
  "comment": "Project website at https://crossformer-model.github.io/",
  "topics": [
   "humanoids",
   "navigation",
   "foundation-pretraining",
   "hardware-codesign",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2408.11812v1",
  "pdf_url": "https://arxiv.org/pdf/2408.11812v1",
  "html_url": "https://arxiv.org/html/2408.11812v1",
  "code_url": "https://crossformer-model.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.7
 },
 {
  "id": "2408.11805",
  "slug": "ace-a-cross-platform-visual-exoskeletons-system-for-low-cost-dexterous",
  "title": "ACE: A Cross-Platform Visual-Exoskeletons System for Low-Cost Dexterous Teleoperation",
  "abstract": "Learning from demonstrations has shown to be an effective approach to robotic manipulation, especially with the recently collected large-scale robot data with teleoperation systems. Building an efficient teleoperation system across diverse robot platforms has become more crucial than ever. However, there is a notable lack of cost-effective and user-friendly teleoperation systems for different end-effectors, e.g., anthropomorphic robot hands and grippers, that can operate across multiple platforms. To address this issue, we develop ACE, a cross-platform visual-exoskeleton system for low-cost dexterous teleoperation. Our system utilizes a hand-facing camera to capture 3D hand poses and an exoskeleton mounted on a portable base, enabling accurate real-time capture of both finger and wrist poses. Compared to previous systems, which often require hardware customization according to different robots, our single system can generalize to humanoid hands, arm-hands, arm-gripper, and quadruped-gripper systems with high-precision teleoperation. This enables imitation learning for complex manipulation tasks on diverse platforms.",
  "published": "2024-08-21",
  "updated": "2024-08-21",
  "year": "2024",
  "authors": [
   "Shiqi Yang",
   "Minghuan Liu",
   "Yuzhe Qin",
   "Runyu Ding",
   "Jialong Li",
   "Xuxin Cheng",
   "Ruihan Yang",
   "Sha Yi",
   "Xiaolong Wang"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 73,
  "influential_citations": 2,
  "tldr": "ACE, a cross-platform visual-exoskeleton system for low-cost dexterous teleoperation that can generalize to humanoid hands, arm-hands, arm-gripper, and quadruped-gripper systems with high-precision teleoperation, enables imitation learning for complex manipulation tasks on diverse platforms.",
  "doi": "10.48550/arXiv.2408.11805",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Shiqi Yang",
    "id": "2309666838",
    "h_index": 8,
    "papers": 9
   },
   {
    "name": "Minghuan Liu",
    "id": "2293431642",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Yuzhe Qin",
    "id": "12701031",
    "h_index": 24,
    "papers": 34
   },
   {
    "name": "Runyu Ding",
    "id": "2316564055",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Jialong Li",
    "id": "2309196968",
    "h_index": 7,
    "papers": 7
   },
   {
    "name": "Xuxin Cheng",
    "id": "2287822264",
    "h_index": 8,
    "papers": 9
   },
   {
    "name": "Ruihan Yang",
    "id": "143955842",
    "h_index": 18,
    "papers": 22
   },
   {
    "name": "Sha Yi",
    "id": "2316591556",
    "h_index": 5,
    "papers": 15
   },
   {
    "name": "Xiaolong Wang",
    "id": "2239141122",
    "h_index": 12,
    "papers": 13
   }
  ],
  "comment": "Webpage: https://ace-teleop.github.io/",
  "topics": [
   "dexterous-manipulation",
   "humanoids",
   "imitation-diffusion",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2408.11805v1",
  "pdf_url": "https://arxiv.org/pdf/2408.11805v1",
  "html_url": "https://arxiv.org/html/2408.11805v1",
  "code_url": "https://ace-teleop.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.87
 },
 {
  "id": "2408.06506",
  "slug": "tacsl-a-library-for-visuotactile-sensor-simulation-and-learning",
  "title": "TacSL: A Library for Visuotactile Sensor Simulation and Learning",
  "abstract": "For both humans and robots, the sense of touch, known as tactile sensing, is critical for performing contact-rich manipulation tasks. Three key challenges in robotic tactile sensing are 1) interpreting sensor signals, 2) generating sensor signals in novel scenarios, and 3) learning sensor-based policies. For visuotactile sensors, interpretation has been facilitated by their close relationship with vision sensors (e.g., RGB cameras). However, generation is still difficult, as visuotactile sensors typically involve contact, deformation, illumination, and imaging, all of which are expensive to simulate; in turn, policy learning has been challenging, as simulation cannot be leveraged for large-scale data collection. We present TacSL (taxel), a library for GPU-based visuotactile sensor simulation and learning. TacSL can be used to simulate visuotactile images and extract contact-force distributions over $200\\times$ faster than the prior state-of-the-art, all within the widely-used Isaac Simulator. Furthermore, TacSL provides a learning toolkit containing multiple sensor models, contact-intensive training environments, and online/offline algorithms that can facilitate policy learning for sim-to-real applications. On the algorithmic side, we introduce a novel online reinforcement-learning algorithm called asymmetric actor-critic distillation (AACD), designed to effectively and efficiently learn tactile-based policies in simulation that can transfer to the real world. Finally, we demonstrate the utility of our library and algorithms by evaluating the benefits of distillation and multimodal sensing for contact-rich manipulation tasks, and most critically, performing sim-to-real transfer. Supplementary videos and results are at https://iakinola23.github.io/tacsl/.",
  "published": "2024-08-12",
  "updated": "2025-03-08",
  "year": "2024",
  "authors": [
   "Iretiayo Akinola",
   "Jie Xu",
   "Jan Carius",
   "Dieter Fox",
   "Yashraj Narang"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "T-RO",
  "venue_source": "semantic-scholar",
  "citations": 64,
  "influential_citations": 5,
  "tldr": "TacSL (taxel), a library for GPU-based visuotactile sensor simulation and learning, and a novel online reinforcement-learning algorithm called asymmetric actor-critic distillation, designed to effectively and efficiently learn tactile-based policies in simulation that can transfer to the real world.",
  "doi": "10.1109/TRO.2025.3547267",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Iretiayo Akinola",
    "id": "2856639",
    "h_index": 19,
    "papers": 38
   },
   {
    "name": "Jie Xu",
    "id": "2273556820",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Jan Carius",
    "id": "30474787",
    "h_index": 13,
    "papers": 17
   },
   {
    "name": "Dieter Fox",
    "id": "2294877174",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Yashraj S. Narang",
    "id": "2387216730",
    "h_index": 12,
    "papers": 16
   }
  ],
  "comment": "",
  "topics": [
   "tactile",
   "sim2real",
   "rl-control",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2408.06506v2",
  "pdf_url": "https://arxiv.org/pdf/2408.06506v2",
  "html_url": "https://arxiv.org/html/2408.06506v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.31
 },
 {
  "id": "2408.06481",
  "slug": "unit-data-efficient-tactile-representation-with-generalization-to-unse",
  "title": "UniT: Data Efficient Tactile Representation with Generalization to Unseen Objects",
  "abstract": "UniT is an approach to tactile representation learning, using VQGAN to learn a compact latent space and serve as the tactile representation. It uses tactile images obtained from a single simple object to train the representation with generalizability. This tactile representation can be zero-shot transferred to various downstream tasks, including perception tasks and manipulation policy learning. Our benchmarkings on in-hand 3D pose and 6D pose estimation tasks and a tactile classification task show that UniT outperforms existing visual and tactile representation learning methods. Additionally, UniT's effectiveness in policy learning is demonstrated across three real-world tasks involving diverse manipulated objects and complex robot-object-environment interactions. Through extensive experimentation, UniT is shown to be a simple-to-train, plug-and-play, yet widely effective method for tactile representation learning. For more details, please refer to our open-source repository https://github.com/ZhengtongXu/UniT and the project website https://zhengtongxu.github.io/unit-website/.",
  "published": "2024-08-12",
  "updated": "2025-04-01",
  "year": "2024",
  "authors": [
   "Zhengtong Xu",
   "Raghava Uppuluri",
   "Xinwei Zhang",
   "Cael Fitch",
   "Philip Glen Crandall",
   "Wan Shou",
   "Dongyi Wang",
   "Yu She"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "RA-L",
  "venue_source": "semantic-scholar",
  "citations": 47,
  "influential_citations": 1,
  "tldr": "Through extensive experimentation, UniT is shown to be a simple-to-train, plug-and-play, yet widely effective method for tactile representation learning.",
  "doi": "10.1109/LRA.2025.3559835",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhengtong Xu",
    "id": "2238437162",
    "h_index": 8,
    "papers": 26
   },
   {
    "name": "Raghava Uppuluri",
    "id": "2312642919",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Xinwei Zhang",
    "id": "2316008439",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Cael Fitch",
    "id": "2315980939",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "P. Crandall",
    "id": "2315980927",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Wan Shou",
    "id": "2279750044",
    "h_index": 2,
    "papers": 11
   },
   {
    "name": "Dongyi Wang",
    "id": "2312647938",
    "h_index": 3,
    "papers": 18
   },
   {
    "name": "Yu She",
    "id": "2279750111",
    "h_index": 4,
    "papers": 13
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2408.06481v2",
  "pdf_url": "https://arxiv.org/pdf/2408.06481v2",
  "html_url": "https://arxiv.org/html/2408.06481v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.18
 },
 {
  "id": "2408.06265",
  "slug": "eyesight-hand-design-of-a-fully-actuated-dexterous-robot-hand-with-int",
  "title": "EyeSight Hand: Design of a Fully-Actuated Dexterous Robot Hand with Integrated Vision-Based Tactile Sensors and Compliant Actuation",
  "abstract": "In this work, we introduce the EyeSight Hand, a novel 7 degrees of freedom (DoF) humanoid hand featuring integrated vision-based tactile sensors tailored for enhanced whole-hand manipulation. Additionally, we introduce an actuation scheme centered around quasi-direct drive actuation to achieve human-like strength and speed while ensuring robustness for large-scale data collection. We evaluate the EyeSight Hand on three challenging tasks: bottle opening, plasticine cutting, and plate pick and place, which require a blend of complex manipulation, tool use, and precise force application. Imitation learning models trained on these tasks, with a novel vision dropout strategy, showcase the benefits of tactile feedback in enhancing task success rates. Our results reveal that the integration of tactile sensing dramatically improves task performance, underscoring the critical role of tactile information in dexterous manipulation.",
  "published": "2024-08-12",
  "updated": "2024-08-12",
  "year": "2024",
  "authors": [
   "Branden Romero",
   "Hao-Shu Fang",
   "Pulkit Agrawal",
   "Edward Adelson"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS",
  "venue_source": "semantic-scholar",
  "citations": 45,
  "influential_citations": 2,
  "tldr": "The EyeSight Hand, a 7 degrees of freedom humanoid hand featuring integrated vision-based tactile sensors tailored for enhanced whole-hand manipulation, is introduced, underscoring the critical role of tactile information in dexterous manipulation.",
  "doi": "10.1109/IROS58592.2024.10802778",
  "oa_pdf": "http://arxiv.org/pdf/2408.06265",
  "s2_authors": [
   {
    "name": "Branden Romero",
    "id": "21201572",
    "h_index": 9,
    "papers": 10
   },
   {
    "name": "Haoshu Fang",
    "id": "2316016989",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Pulkit Agrawal",
    "id": "2257003971",
    "h_index": 18,
    "papers": 41
   },
   {
    "name": "Edward H. Adelson",
    "id": "2293172170",
    "h_index": 8,
    "papers": 18
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "humanoids",
   "tactile",
   "imitation-diffusion",
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2408.06265v1",
  "pdf_url": "https://arxiv.org/pdf/2408.06265v1",
  "html_url": "https://arxiv.org/html/2408.06265v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.16
 },
 {
  "id": "2408.06072",
  "slug": "cogvideox-text-to-video-diffusion-models-with-an-expert-transformer",
  "title": "CogVideoX: Text-to-Video Diffusion Models with An Expert Transformer",
  "abstract": "We present CogVideoX, a large-scale text-to-video generation model based on diffusion transformer, which can generate 10-second continuous videos aligned with text prompt, with a frame rate of 16 fps and resolution of 768 * 1360 pixels. Previous video generation models often had limited movement and short durations, and is difficult to generate videos with coherent narratives based on text. We propose several designs to address these issues. First, we propose a 3D Variational Autoencoder (VAE) to compress videos along both spatial and temporal dimensions, to improve both compression rate and video fidelity. Second, to improve the text-video alignment, we propose an expert transformer with the expert adaptive LayerNorm to facilitate the deep fusion between the two modalities. Third, by employing a progressive training and multi-resolution frame pack technique, CogVideoX is adept at producing coherent, long-duration, different shape videos characterized by significant motions. In addition, we develop an effective text-video data processing pipeline that includes various data preprocessing strategies and a video captioning method, greatly contributing to the generation quality and semantic alignment. Results show that CogVideoX demonstrates state-of-the-art performance across both multiple machine metrics and human evaluations. The model weight of both 3D Causal VAE, Video caption model and CogVideoX are publicly available at https://github.com/THUDM/CogVideo.",
  "published": "2024-08-12",
  "updated": "2025-03-26",
  "year": "2024",
  "authors": [
   "Zhuoyi Yang",
   "Jiayan Teng",
   "Wendi Zheng",
   "Ming Ding",
   "Shiyu Huang",
   "Jiazheng Xu",
   "Yuanming Yang",
   "Wenyi Hong",
   "Xiaohan Zhang",
   "Guanyu Feng",
   "Da Yin",
   "Yuxuan Zhang",
   "Weihan Wang",
   "Yean Cheng",
   "Bin Xu",
   "Xiaotao Gu",
   "Yuxiao Dong",
   "Jie Tang"
  ],
  "author_count": 18,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 2297,
  "influential_citations": 445,
  "tldr": "CogVideoX is adept at producing coherent, long-duration, different shape videos characterized by significant motions, and develops an effective text-video data processing pipeline that includes various data preprocessing strategies and a video captioning method, greatly contributing to the generation quality and semantic alignment.",
  "doi": "10.48550/arXiv.2408.06072",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhuoyi Yang",
    "id": "2303231681",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Jiayan Teng",
    "id": "2238205354",
    "h_index": 8,
    "papers": 11
   },
   {
    "name": "Wendi Zheng",
    "id": "2163967642",
    "h_index": 9,
    "papers": 13
   },
   {
    "name": "Ming Ding",
    "id": "2055623340",
    "h_index": 23,
    "papers": 30
   },
   {
    "name": "Shiyu Huang",
    "id": "2305795673",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "Jiazheng Xu",
    "id": "2214082934",
    "h_index": 10,
    "papers": 14
   },
   {
    "name": "Yuanming Yang",
    "id": "2315948290",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Wenyi Hong",
    "id": "2105844599",
    "h_index": 16,
    "papers": 25
   },
   {
    "name": "Xiaohan Zhang",
    "id": "2268628279",
    "h_index": 12,
    "papers": 15
   },
   {
    "name": "Guanyu Feng",
    "id": "2307077651",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Da Yin",
    "id": "2307075814",
    "h_index": 8,
    "papers": 11
   },
   {
    "name": "Xiaotao Gu",
    "id": "2290625851",
    "h_index": 15,
    "papers": 31
   },
   {
    "name": "Yuxuan Zhang",
    "id": "2316099643",
    "h_index": 7,
    "papers": 9
   },
   {
    "name": "Weihan Wang",
    "id": "2265518149",
    "h_index": 12,
    "papers": 19
   },
   {
    "name": "Yean Cheng",
    "id": "2306161782",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "Ting Liu",
    "id": "2325000243",
    "h_index": 6,
    "papers": 6
   },
   {
    "name": "Bin Xu",
    "id": "2288066971",
    "h_index": 11,
    "papers": 20
   },
   {
    "name": "Yuxiao Dong",
    "id": "2243402027",
    "h_index": 41,
    "papers": 94
   },
   {
    "name": "Jie Tang",
    "id": "2238207092",
    "h_index": 16,
    "papers": 21
   }
  ],
  "comment": "Accepted by ICLR2025",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2408.06072v3",
  "pdf_url": "https://arxiv.org/pdf/2408.06072v3",
  "html_url": "https://arxiv.org/html/2408.06072v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "2408.00714",
  "slug": "sam-2-segment-anything-in-images-and-videos",
  "title": "SAM 2: Segment Anything in Images and Videos",
  "abstract": "We present Segment Anything Model 2 (SAM 2), a foundation model towards solving promptable visual segmentation in images and videos. We build a data engine, which improves model and data via user interaction, to collect the largest video segmentation dataset to date. Our model is a simple transformer architecture with streaming memory for real-time video processing. SAM 2 trained on our data provides strong performance across a wide range of tasks. In video segmentation, we observe better accuracy, using 3x fewer interactions than prior approaches. In image segmentation, our model is more accurate and 6x faster than the Segment Anything Model (SAM). We believe that our data, model, and insights will serve as a significant milestone for video segmentation and related perception tasks. We are releasing our main model, dataset, as well as code for model training and our demo.",
  "published": "2024-08-01",
  "updated": "2024-10-28",
  "year": "2024",
  "authors": [
   "Nikhila Ravi",
   "Valentin Gabeur",
   "Yuan-Ting Hu",
   "Ronghang Hu",
   "Chaitanya Ryali",
   "Tengyu Ma",
   "Haitham Khedr",
   "Roman R\u00e4dle",
   "Chloe Rolland",
   "Laura Gustafson",
   "Eric Mintun",
   "Junting Pan",
   "Kalyan Vasudev Alwala",
   "Nicolas Carion",
   "Chao-Yuan Wu",
   "Ross Girshick",
   "Piotr Doll\u00e1r",
   "Christoph Feichtenhofer"
  ],
  "author_count": 18,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.CV",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 4022,
  "influential_citations": 584,
  "tldr": "A data engine is built, which improves model and data via user interaction, to collect the largest video segmentation dataset to date, and a simple transformer architecture with streaming memory for real-time video processing.",
  "doi": "10.48550/arXiv.2408.00714",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Nikhila Ravi",
    "id": "2308102494",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Valentin Gabeur",
    "id": "151352107",
    "h_index": 7,
    "papers": 9
   },
   {
    "name": "Yuan-Ting Hu",
    "id": "2314514593",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Ronghang Hu",
    "id": "2314374069",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Chaitanya K. Ryali",
    "id": "52190116",
    "h_index": 10,
    "papers": 21
   },
   {
    "name": "Tengyu Ma",
    "id": "2314306272",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Haitham Khedr",
    "id": "2314116200",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Roman R\u00e4dle",
    "id": "2320227019",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Chlo\u00e9 Rolland",
    "id": "2213549340",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Laura Gustafson",
    "id": "2314116847",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Eric Mintun",
    "id": "13131689",
    "h_index": 11,
    "papers": 18
   },
   {
    "name": "Junting Pan",
    "id": "7588865",
    "h_index": 19,
    "papers": 28
   },
   {
    "name": "Kalyan Vasudev Alwala",
    "id": "3085301",
    "h_index": 12,
    "papers": 13
   },
   {
    "name": "Nicolas Carion",
    "id": "3422899",
    "h_index": 11,
    "papers": 17
   },
   {
    "name": "Chao Wu",
    "id": "2331855091",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Ross B. Girshick",
    "id": "2257205047",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Piotr Doll'ar",
    "id": "2065731243",
    "h_index": 7,
    "papers": 7
   },
   {
    "name": "Christoph Feichtenhofer",
    "id": "2322150",
    "h_index": 36,
    "papers": 57
   }
  ],
  "comment": "Website: https://ai.meta.com/sam2",
  "topics": [
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2408.00714v2",
  "pdf_url": "https://arxiv.org/pdf/2408.00714v2",
  "html_url": "https://arxiv.org/html/2408.00714v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "2408.00118",
  "slug": "gemma-2-improving-open-language-models-at-a-practical-size",
  "title": "Gemma 2: Improving Open Language Models at a Practical Size",
  "abstract": "In this work, we introduce Gemma 2, a new addition to the Gemma family of lightweight, state-of-the-art open models, ranging in scale from 2 billion to 27 billion parameters. In this new version, we apply several known technical modifications to the Transformer architecture, such as interleaving local-global attentions (Beltagy et al., 2020a) and group-query attention (Ainslie et al., 2023). We also train the 2B and 9B models with knowledge distillation (Hinton et al., 2015) instead of next token prediction. The resulting models deliver the best performance for their size, and even offer competitive alternatives to models that are 2-3 times bigger. We release all our models to the community.",
  "published": "2024-07-31",
  "updated": "2024-10-02",
  "year": "2024",
  "authors": [
   " Gemma Team",
   "Morgane Riviere",
   "Shreya Pathak",
   "Pier Giuseppe Sessa",
   "Cassidy Hardin",
   "Surya Bhupatiraju",
   "L\u00e9onard Hussenot",
   "Thomas Mesnard",
   "Bobak Shahriari",
   "Alexandre Ram\u00e9",
   "Johan Ferret",
   "Peter Liu",
   "Pouya Tafti",
   "Abe Friesen",
   "Michelle Casbon",
   "Sabela Ramos",
   "Ravin Kumar",
   "Charline Le Lan",
   "Sammy Jerome",
   "Anton Tsitsulin",
   "Nino Vieillard",
   "Piotr Stanczyk",
   "Sertan Girgin",
   "Nikola Momchev",
   "Matt Hoffman",
   "Shantanu Thakoor",
   "Jean-Bastien Grill",
   "Behnam Neyshabur",
   "Olivier Bachem",
   "Alanna Walton",
   "Aliaksei Severyn",
   "Alicia Parrish",
   "Aliya Ahmad",
   "Allen Hutchison",
   "Alvin Abdagic",
   "Amanda Carl",
   "Amy Shen",
   "Andy Brock",
   "Andy Coenen",
   "Anthony Laforge",
   "Antonia Paterson",
   "Ben Bastian",
   "Bilal Piot",
   "Bo Wu",
   "Brandon Royal",
   "Charlie Chen",
   "Chintu Kumar",
   "Chris Perry",
   "Chris Welty",
   "Christopher A. Choquette-Choo",
   "Danila Sinopalnikov",
   "David Weinberger",
   "Dimple Vijaykumar",
   "Dominika Rogozi\u0144ska",
   "Dustin Herbison",
   "Elisa Bandy",
   "Emma Wang",
   "Eric Noland",
   "Erica Moreira",
   "Evan Senter",
   "Evgenii Eltyshev",
   "Francesco Visin",
   "Gabriel Rasskin",
   "Gary Wei",
   "Glenn Cameron",
   "Gus Martins",
   "Hadi Hashemi",
   "Hanna Klimczak-Pluci\u0144ska",
   "Harleen Batra",
   "Harsh Dhand",
   "Ivan Nardini",
   "Jacinda Mein",
   "Jack Zhou",
   "James Svensson",
   "Jeff Stanway",
   "Jetha Chan",
   "Jin Peng Zhou",
   "Joana Carrasqueira",
   "Joana Iljazi",
   "Jocelyn Becker",
   "Joe Fernandez",
   "Joost van Amersfoort",
   "Josh Gordon",
   "Josh Lipschultz",
   "Josh Newlan",
   "Ju-yeong Ji",
   "Kareem Mohamed",
   "Kartikeya Badola",
   "Kat Black",
   "Katie Millican",
   "Keelin McDonell",
   "Kelvin Nguyen",
   "Kiranbir Sodhia",
   "Kish Greene",
   "Lars Lowe Sjoesund",
   "Lauren Usui",
   "Laurent Sifre",
   "Lena Heuermann",
   "Leticia Lago",
   "Lilly McNealus",
   "Livio Baldini Soares",
   "Logan Kilpatrick",
   "Lucas Dixon",
   "Luciano Martins",
   "Machel Reid",
   "Manvinder Singh",
   "Mark Iverson",
   "Martin G\u00f6rner",
   "Mat Velloso",
   "Mateo Wirth",
   "Matt Davidow",
   "Matt Miller",
   "Matthew Rahtz",
   "Matthew Watson",
   "Meg Risdal",
   "Mehran Kazemi",
   "Michael Moynihan",
   "Ming Zhang",
   "Minsuk Kahng",
   "Minwoo Park",
   "Mofi Rahman",
   "Mohit Khatwani",
   "Natalie Dao",
   "Nenshad Bardoliwalla",
   "Nesh Devanathan",
   "Neta Dumai",
   "Nilay Chauhan",
   "Oscar Wahltinez",
   "Pankil Botarda",
   "Parker Barnes",
   "Paul Barham",
   "Paul Michel",
   "Pengchong Jin",
   "Petko Georgiev",
   "Phil Culliton",
   "Pradeep Kuppala",
   "Ramona Comanescu",
   "Ramona Merhej",
   "Reena Jana",
   "Reza Ardeshir Rokni",
   "Rishabh Agarwal",
   "Ryan Mullins",
   "Samaneh Saadat",
   "Sara Mc Carthy",
   "Sarah Cogan",
   "Sarah Perrin",
   "S\u00e9bastien M. R. Arnold",
   "Sebastian Krause",
   "Shengyang Dai",
   "Shruti Garg",
   "Shruti Sheth",
   "Sue Ronstrom",
   "Susan Chan",
   "Timothy Jordan",
   "Ting Yu",
   "Tom Eccles",
   "Tom Hennigan",
   "Tomas Kocisky",
   "Tulsee Doshi",
   "Vihan Jain",
   "Vikas Yadav",
   "Vilobh Meshram",
   "Vishal Dharmadhikari",
   "Warren Barkley",
   "Wei Wei",
   "Wenming Ye",
   "Woohyun Han",
   "Woosuk Kwon",
   "Xiang Xu",
   "Zhe Shen",
   "Zhitao Gong",
   "Zichuan Wei",
   "Victor Cotruta",
   "Phoebe Kirk",
   "Anand Rao",
   "Minh Giang",
   "Ludovic Peran",
   "Tris Warkentin",
   "Eli Collins",
   "Joelle Barral",
   "Zoubin Ghahramani",
   "Raia Hadsell",
   "D. Sculley",
   "Jeanine Banks",
   "Anca Dragan",
   "Slav Petrov",
   "Oriol Vinyals",
   "Jeff Dean",
   "Demis Hassabis",
   "Koray Kavukcuoglu",
   "Clement Farabet",
   "Elena Buchatskaya",
   "Sebastian Borgeaud",
   "Noah Fiedel",
   "Armand Joulin",
   "Kathleen Kenealy",
   "Robert Dadashi",
   "Alek Andreev"
  ],
  "author_count": 198,
  "categories": [
   "cs.CL",
   "cs.AI"
  ],
  "primary_category": "cs.CL",
  "venue": "",
  "venue_source": "",
  "citations": 2339,
  "influential_citations": 383,
  "tldr": "Gemma 2, a new addition to the Gemma family of lightweight, state-of-the-art open models, ranging in scale from 2 billion to 27 billion parameters, delivers the best performance for their size, and even offers competitive alternatives to models that are 2-3 times bigger.",
  "doi": "10.48550/arXiv.2408.00118",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Gemma Team Morgane Riviere",
    "id": "2314116504",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Shreya Pathak",
    "id": "2273651441",
    "h_index": 9,
    "papers": 32
   },
   {
    "name": "Pier Giuseppe Sessa",
    "id": "7281978",
    "h_index": 16,
    "papers": 36
   },
   {
    "name": "Cassidy Hardin",
    "id": "2275186843",
    "h_index": 9,
    "papers": 29
   },
   {
    "name": "Surya Bhupatiraju",
    "id": "9692128",
    "h_index": 11,
    "papers": 37
   },
   {
    "name": "L'eonard Hussenot",
    "id": "2312322565",
    "h_index": 9,
    "papers": 24
   },
   {
    "name": "Thomas Mesnard",
    "id": "2237423540",
    "h_index": 16,
    "papers": 42
   },
   {
    "name": "Bobak Shahriari",
    "id": "2067577",
    "h_index": 15,
    "papers": 42
   },
   {
    "name": "Alexandre Ram'e",
    "id": "2280134846",
    "h_index": 9,
    "papers": 12
   },
   {
    "name": "Johan Ferret",
    "id": "151047979",
    "h_index": 18,
    "papers": 51
   },
   {
    "name": "Peter Liu",
    "id": "2314572038",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "P. Tafti",
    "id": "1775270",
    "h_index": 15,
    "papers": 64
   },
   {
    "name": "Abe Friesen",
    "id": "2312322254",
    "h_index": 6,
    "papers": 19
   },
   {
    "name": "M. Casbon",
    "id": "2314115981",
    "h_index": 6,
    "papers": 17
   },
   {
    "name": "Sabela Ramos",
    "id": "2253595555",
    "h_index": 8,
    "papers": 25
   },
   {
    "name": "Ravin Kumar",
    "id": "2290629265",
    "h_index": 9,
    "papers": 27
   },
   {
    "name": "Charline Le Lan",
    "id": "153892869",
    "h_index": 16,
    "papers": 37
   },
   {
    "name": "Sammy Jerome",
    "id": "2287843663",
    "h_index": 6,
    "papers": 17
   },
   {
    "name": "Anton Tsitsulin",
    "id": "40900939",
    "h_index": 16,
    "papers": 39
   },
   {
    "name": "Nino Vieillard",
    "id": "2308037523",
    "h_index": 10,
    "papers": 31
   },
   {
    "name": "P. Sta\u0144czyk",
    "id": "2067024583",
    "h_index": 13,
    "papers": 41
   },
   {
    "name": "Sertan Girgin",
    "id": "35022714",
    "h_index": 23,
    "papers": 78
   },
   {
    "name": "Nikola Momchev",
    "id": "1470531643",
    "h_index": 12,
    "papers": 26
   },
   {
    "name": "Matt Hoffman",
    "id": "2312323782",
    "h_index": 6,
    "papers": 14
   },
   {
    "name": "S. Thakoor",
    "id": "41037204",
    "h_index": 16,
    "papers": 42
   },
   {
    "name": "Jean-Bastien Grill",
    "id": "145840757",
    "h_index": 15,
    "papers": 34
   },
   {
    "name": "Behnam Neyshabur",
    "id": "3007442",
    "h_index": 46,
    "papers": 79
   },
   {
    "name": "Alanna Walton",
    "id": "2275185047",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "A. Severyn",
    "id": "3091861",
    "h_index": 35,
    "papers": 81
   },
   {
    "name": "Alicia Parrish",
    "id": "2292197465",
    "h_index": 5,
    "papers": 17
   },
   {
    "name": "Aliya Ahmad",
    "id": "2314901120",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Allen Hutchison",
    "id": "2314109748",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Alvin Abdagic",
    "id": "3026185",
    "h_index": 6,
    "papers": 16
   },
   {
    "name": "Amanda Carl",
    "id": "2314115816",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Amy Shen",
    "id": "2314112131",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Andy Brock",
    "id": "2065040422",
    "h_index": 9,
    "papers": 17
   },
   {
    "name": "Andy Coenen",
    "id": "1388485917",
    "h_index": 13,
    "papers": 18
   },
   {
    "name": "Anthony Laforge",
    "id": "2314116317",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Antonia Paterson",
    "id": "2291068272",
    "h_index": 6,
    "papers": 18
   },
   {
    "name": "Ben Bastian",
    "id": "2314111874",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Bilal Piot",
    "id": "1808897",
    "h_index": 44,
    "papers": 80
   },
   {
    "name": "Boxi Wu",
    "id": "2275291909",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Brandon Royal",
    "id": "2314111478",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Charlie Chen",
    "id": "2182971260",
    "h_index": 10,
    "papers": 22
   },
   {
    "name": "Chintu Kumar",
    "id": "2275197748",
    "h_index": 4,
    "papers": 16
   },
   {
    "name": "Chris Perry",
    "id": "2314116521",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Christoper A. Welty",
    "id": "2062396538",
    "h_index": 7,
    "papers": 27
   },
   {
    "name": "Christopher A. Choquette-Choo",
    "id": "2314115870",
    "h_index": 14,
    "papers": 38
   },
   {
    "name": "Danila Sinopalnikov",
    "id": "1470524352",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "David Weinberger",
    "id": "2314109694",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "D. Vijaykumar",
    "id": "2314116311",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Dominika Rogozi'nska",
    "id": "2275184739",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "D. Herbison",
    "id": "2314116307",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Elisa Bandy",
    "id": "2295989389",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Emma Wang",
    "id": "2314334001",
    "h_index": 6,
    "papers": 16
   },
   {
    "name": "Eric Noland",
    "id": "51210148",
    "h_index": 12,
    "papers": 33
   },
   {
    "name": "Erica Moreira",
    "id": "2275185558",
    "h_index": 8,
    "papers": 29
   },
   {
    "name": "Evan Senter",
    "id": "2268665228",
    "h_index": 9,
    "papers": 21
   },
   {
    "name": "Evgenii Eltyshev",
    "id": "2275187189",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Francesco Visin",
    "id": "2077146",
    "h_index": 16,
    "papers": 28
   },
   {
    "name": "Gabriel Rasskin",
    "id": "2303849717",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Gary Wei",
    "id": "2315902795",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Glenn Cameron",
    "id": "2295989714",
    "h_index": 6,
    "papers": 16
   },
   {
    "name": "Gus Martins",
    "id": "2314112038",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Hadi Hashemi",
    "id": "2070487928",
    "h_index": 8,
    "papers": 30
   },
   {
    "name": "Hanna Klimczak-Pluci'nska",
    "id": "2275187187",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Harleen Batra",
    "id": "2232330934",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "H. Dhand",
    "id": "2061871",
    "h_index": 6,
    "papers": 19
   },
   {
    "name": "Ivan Nardini",
    "id": "2314115441",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Jacinda Mein",
    "id": "2314111466",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Jack Zhou",
    "id": "2314333945",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "James Svensson",
    "id": "2275188153",
    "h_index": 6,
    "papers": 18
   },
   {
    "name": "J. Stanway",
    "id": "35149729",
    "h_index": 10,
    "papers": 32
   },
   {
    "name": "Jetha Chan",
    "id": "2314546523",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Jin Zhou",
    "id": "2314553207",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Joana Carrasqueira",
    "id": "2334688409",
    "h_index": 5,
    "papers": 15
   },
   {
    "name": "Joana Iljazi",
    "id": "2543878",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Jocelyn Becker",
    "id": "2314113456",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Joe Fernandez",
    "id": "2314081561",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Joost R. van Amersfoort",
    "id": "3038326",
    "h_index": 20,
    "papers": 32
   },
   {
    "name": "J. Gordon",
    "id": "2314118228",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Josh Lipschultz",
    "id": "2290487597",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Joshua Newlan",
    "id": "2160888100",
    "h_index": 7,
    "papers": 17
   },
   {
    "name": "Junsong Ji",
    "id": "2299505170",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Kareem Mohamed",
    "id": "2314113987",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Kartikeya Badola",
    "id": "2051018967",
    "h_index": 9,
    "papers": 19
   },
   {
    "name": "Kat Black",
    "id": "2314110240",
    "h_index": 7,
    "papers": 19
   },
   {
    "name": "Katie Millican",
    "id": "2143434227",
    "h_index": 11,
    "papers": 29
   },
   {
    "name": "K. McDonell",
    "id": "2314111460",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "K. Nguyen",
    "id": "2314888051",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Kiranbir Sodhia",
    "id": "2303850028",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Kish Greene",
    "id": "2314114429",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Lars Lowe Sjoesund",
    "id": "2291065624",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Lauren Usui",
    "id": "2314111775",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "L. Sifre",
    "id": "2175946",
    "h_index": 28,
    "papers": 42
   },
   {
    "name": "Lena Heuermann",
    "id": "2314115691",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Leti-cia Lago",
    "id": "2314109377",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Lilly McNealus",
    "id": "2314114318",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Livio Baldini Soares",
    "id": "7353832",
    "h_index": 23,
    "papers": 39
   },
   {
    "name": "Logan Kilpatrick",
    "id": "2314112114",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Lucas Dixon",
    "id": "2275186508",
    "h_index": 8,
    "papers": 17
   },
   {
    "name": "Luciano Martins",
    "id": "2314335768",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Machel Reid",
    "id": "1557386977",
    "h_index": 19,
    "papers": 35
   },
   {
    "name": "Manvinder Singh",
    "id": "2314652123",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Mark Iverson",
    "id": "2091391378",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Martin Gorner",
    "id": "2314112064",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "M. Velloso",
    "id": "2314115650",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Mateo Wirth",
    "id": "2275185968",
    "h_index": 8,
    "papers": 16
   },
   {
    "name": "Matt Davidow",
    "id": "2314109345",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Matt Miller",
    "id": "2314523909",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Matthew Rahtz",
    "id": "3183032",
    "h_index": 10,
    "papers": 25
   },
   {
    "name": "Matthew Watson",
    "id": "2303852016",
    "h_index": 5,
    "papers": 17
   },
   {
    "name": "Meg Risdal",
    "id": "2267728069",
    "h_index": 6,
    "papers": 14
   },
   {
    "name": "Mehran Kazemi",
    "id": "2283308055",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Michael Moynihan",
    "id": "2314111806",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Ming Zhang",
    "id": "2290594698",
    "h_index": 6,
    "papers": 19
   },
   {
    "name": "Minsuk Kahng",
    "id": "1768057",
    "h_index": 23,
    "papers": 84
   },
   {
    "name": "Minwoo Park",
    "id": "2314516113",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Mofi Rahman",
    "id": "2314153942",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "M. Khatwani",
    "id": "15439864",
    "h_index": 8,
    "papers": 17
   },
   {
    "name": "Natalie Dao",
    "id": "2314116003",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Nen-shad Bardoliwalla",
    "id": "8100379",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "N. Devanathan",
    "id": "2295988937",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Neta Dumai",
    "id": "2314114325",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Nilay Chauhan",
    "id": "2295989865",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "Oscar Wahltinez",
    "id": "2038523726",
    "h_index": 7,
    "papers": 21
   },
   {
    "name": "Pankil Botarda",
    "id": "2314113467",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Parker Barnes",
    "id": "80940648",
    "h_index": 8,
    "papers": 21
   },
   {
    "name": "P. Barham",
    "id": "152399055",
    "h_index": 12,
    "papers": 32
   },
   {
    "name": "Paul Michel",
    "id": "2275185732",
    "h_index": 9,
    "papers": 24
   },
   {
    "name": "Peng-chong Jin",
    "id": "2314115279",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Petko Georgiev",
    "id": "1737522",
    "h_index": 22,
    "papers": 46
   },
   {
    "name": "Phil Culliton",
    "id": "40579094",
    "h_index": 6,
    "papers": 16
   },
   {
    "name": "Pradeep Kuppala",
    "id": "2314111697",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "R. Comanescu",
    "id": "89066101",
    "h_index": 12,
    "papers": 35
   },
   {
    "name": "Ramona Merhej",
    "id": "2092094191",
    "h_index": 7,
    "papers": 23
   },
   {
    "name": "Reena Jana",
    "id": "2291065688",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "R. Rokni",
    "id": "15170591",
    "h_index": 5,
    "papers": 18
   },
   {
    "name": "Rishabh Agarwal",
    "id": "2258553001",
    "h_index": 7,
    "papers": 18
   },
   {
    "name": "Ryan Mullins",
    "id": "2291067498",
    "h_index": 10,
    "papers": 31
   },
   {
    "name": "Samaneh Saadat",
    "id": "50592937",
    "h_index": 11,
    "papers": 29
   },
   {
    "name": "S. M. Carthy",
    "id": "2255299263",
    "h_index": 7,
    "papers": 18
   },
   {
    "name": "Sarah Perrin",
    "id": "2312326042",
    "h_index": 8,
    "papers": 22
   },
   {
    "name": "S\u00e9bastien M. R. Arnold",
    "id": "2275186656",
    "h_index": 6,
    "papers": 17
   },
   {
    "name": "Se-bastian Krause",
    "id": "2275186493",
    "h_index": 4,
    "papers": 14
   },
   {
    "name": "Shengyang Dai",
    "id": "2314112926",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "S. Garg",
    "id": "2119475982",
    "h_index": 12,
    "papers": 69
   },
   {
    "name": "Shruti Sheth",
    "id": "35702090",
    "h_index": 7,
    "papers": 20
   },
   {
    "name": "S. Ronstrom",
    "id": "47568290",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Susan Chan",
    "id": "2314882045",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Timothy Jordan",
    "id": "2314109322",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Ting Yu",
    "id": "2314504391",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "Tom Eccles",
    "id": "2314114940",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "T. Hennigan",
    "id": "2146532222",
    "h_index": 9,
    "papers": 33
   },
   {
    "name": "Tom\u00e1s Kocisk\u00fd",
    "id": "2367821",
    "h_index": 17,
    "papers": 41
   },
   {
    "name": "Tulsee Doshi",
    "id": "2314111664",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Vihan Jain",
    "id": "2314113713",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Vikas Yadav",
    "id": "2125258677",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Vilobh Meshram",
    "id": "2314109173",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Vishal Dharmadhikari",
    "id": "2314115551",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Warren Barkley",
    "id": "2314109429",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Wei Wei",
    "id": "2314672859",
    "h_index": 6,
    "papers": 17
   },
   {
    "name": "Wenming Ye",
    "id": "2314114264",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Woohyun Han",
    "id": "2215449616",
    "h_index": 7,
    "papers": 15
   },
   {
    "name": "Woosuk Kwon",
    "id": "2314115546",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Xiang Xu",
    "id": "2314330161",
    "h_index": 5,
    "papers": 15
   },
   {
    "name": "Zhe Shen",
    "id": "2314348713",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Zhitao Gong",
    "id": "2275190668",
    "h_index": 7,
    "papers": 28
   },
   {
    "name": "Zichuan Wei",
    "id": "2314328859",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "Victor Cotruta",
    "id": "2275189218",
    "h_index": 6,
    "papers": 20
   },
   {
    "name": "Phoebe Kirk",
    "id": "2314107430",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Anand Rao",
    "id": "2314521404",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Minh Giang",
    "id": "2275187490",
    "h_index": 8,
    "papers": 25
   },
   {
    "name": "Ludovic Peran",
    "id": "2291065300",
    "h_index": 7,
    "papers": 16
   },
   {
    "name": "Tris Warkentin",
    "id": "1986491804",
    "h_index": 13,
    "papers": 29
   },
   {
    "name": "Eli Collins",
    "id": "2275181648",
    "h_index": 8,
    "papers": 23
   },
   {
    "name": "Joelle Barral",
    "id": "2254701020",
    "h_index": 6,
    "papers": 17
   },
   {
    "name": "Z. Ghahramani",
    "id": "1983575",
    "h_index": 22,
    "papers": 61
   },
   {
    "name": "R. Hadsell",
    "id": "2315504",
    "h_index": 51,
    "papers": 111
   },
   {
    "name": "D. Sculley",
    "id": "2265885590",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Jeanine Banks",
    "id": "2314116903",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Anca Dragan",
    "id": "2064066935",
    "h_index": 18,
    "papers": 51
   },
   {
    "name": "Slav Petrov",
    "id": "2257293575",
    "h_index": 6,
    "papers": 17
   },
   {
    "name": "O. Vinyals",
    "id": "1689108",
    "h_index": 103,
    "papers": 204
   },
   {
    "name": "Jeffrey Dean",
    "id": "2265529729",
    "h_index": 8,
    "papers": 16
   },
   {
    "name": "D. Hassabis",
    "id": "48987704",
    "h_index": 92,
    "papers": 160
   },
   {
    "name": "K. Kavukcuoglu",
    "id": "2645384",
    "h_index": 76,
    "papers": 124
   },
   {
    "name": "C. Farabet",
    "id": "2256269",
    "h_index": 27,
    "papers": 50
   },
   {
    "name": "Elena Buchatskaya",
    "id": "118801223",
    "h_index": 12,
    "papers": 36
   },
   {
    "name": "Sebastian Borgeaud",
    "id": "148016269",
    "h_index": 20,
    "papers": 52
   },
   {
    "name": "Noah Fiedel",
    "id": "22640071",
    "h_index": 20,
    "papers": 42
   },
   {
    "name": "Armand Joulin",
    "id": "2319608",
    "h_index": 72,
    "papers": 151
   },
   {
    "name": "Kathleen Kenealy",
    "id": "1914502282",
    "h_index": 9,
    "papers": 26
   },
   {
    "name": "Robert Dadashi",
    "id": "51914693",
    "h_index": 20,
    "papers": 40
   },
   {
    "name": "Alek Andreev",
    "id": "2290741315",
    "h_index": 8,
    "papers": 20
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2408.00118v3",
  "pdf_url": "https://arxiv.org/pdf/2408.00118v3",
  "html_url": "https://arxiv.org/html/2408.00118v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.0
 },
 {
  "id": "2407.20635",
  "slug": "autonomous-improvement-of-instruction-following-skills-via-foundation",
  "title": "Autonomous Improvement of Instruction Following Skills via Foundation Models",
  "abstract": "Intelligent instruction-following robots capable of improving from autonomously collected experience have the potential to transform robot learning: instead of collecting costly teleoperated demonstration data, large-scale deployment of fleets of robots can quickly collect larger quantities of autonomous data that can collectively improve their performance. However, autonomous improvement requires solving two key problems: (i) fully automating a scalable data collection procedure that can collect diverse and semantically meaningful robot data and (ii) learning from non-optimal, autonomous data with no human annotations. To this end, we propose a novel approach that addresses these challenges, allowing instruction-following policies to improve from autonomously collected data without human supervision. Our framework leverages vision-language models to collect and evaluate semantically meaningful experiences in new environments, and then utilizes a decomposition of instruction following tasks into (semantic) language-conditioned image generation and (non-semantic) goal reaching, which makes it significantly more practical to improve from this autonomously collected data without any human annotations. We carry out extensive experiments in the real world to demonstrate the effectiveness of our approach, and find that in a suite of unseen environments, the robot policy can be improved 2x with autonomously collected data. We open-source the code for our semantic autonomous improvement pipeline, as well as our autonomous dataset of 30.5K trajectories collected across five tabletop environments.",
  "published": "2024-07-30",
  "updated": "2024-10-15",
  "year": "2024",
  "authors": [
   "Zhiyuan Zhou",
   "Pranav Atreya",
   "Abraham Lee",
   "Homer Walke",
   "Oier Mees",
   "Sergey Levine"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 35,
  "influential_citations": 3,
  "tldr": "This framework leverages vision-language models to collect and evaluate semantically meaningful experiences in new environments, and utilizes a decomposition of instruction following tasks into (semantic) language-conditioned image generation and (non-semantic) goal reaching to improve from autonomously collected data without any human annotations.",
  "doi": "10.48550/arXiv.2407.20635",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhiyuan Zhou",
    "id": "2314073778",
    "h_index": 6,
    "papers": 16
   },
   {
    "name": "P. Atreya",
    "id": "1643955309",
    "h_index": 9,
    "papers": 13
   },
   {
    "name": "Abraham Lee",
    "id": "2233425761",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "H. Walke",
    "id": "2029241116",
    "h_index": 17,
    "papers": 23
   },
   {
    "name": "Oier Mees",
    "id": "7264115",
    "h_index": 28,
    "papers": 45
   },
   {
    "name": "Sergey Levine",
    "id": "2249615151",
    "h_index": 32,
    "papers": 53
   }
  ],
  "comment": "2024 Conference on Robot Learning (CoRL)",
  "topics": [
   "vla",
   "foundation-pretraining",
   "data-teleop",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2407.20635v2",
  "pdf_url": "https://arxiv.org/pdf/2407.20635v2",
  "html_url": "https://arxiv.org/html/2407.20635v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.06
 },
 {
  "id": "2407.18911",
  "slug": "hrp-human-affordances-for-robotic-pre-training",
  "title": "HRP: Human Affordances for Robotic Pre-Training",
  "abstract": "In order to *generalize* to various tasks in the wild, robotic agents will need a suitable representation (i.e., vision network) that enables the robot to predict optimal actions given high dimensional vision inputs. However, learning such a representation requires an extreme amount of diverse training data, which is prohibitively expensive to collect on a real robot. How can we overcome this problem? Instead of collecting more robot data, this paper proposes using internet-scale, human videos to extract \"affordances,\" both at the environment and agent level, and distill them into a pre-trained representation. We present a simple framework for pre-training representations on hand, object, and contact \"affordance labels\" that highlight relevant objects in images and how to interact with them. These affordances are automatically extracted from human video data (with the help of off-the-shelf computer vision modules) and used to fine-tune existing representations. Our approach can efficiently fine-tune *any* existing representation, and results in models with stronger downstream robotic performance across the board. We experimentally demonstrate (using 3000+ robot trials) that this affordance pre-training scheme boosts performance by a minimum of 15% on 5 real-world tasks, which consider three diverse robot morphologies (including a dexterous hand). Unlike prior works in the space, these representations improve performance across 3 different camera views. Quantitatively, we find that our approach leads to higher levels of generalization in out-of-distribution settings. For code, weights, and data check: https://hrp-robot.github.io",
  "published": "2024-07-26",
  "updated": "2024-07-26",
  "year": "2024",
  "authors": [
   "Mohan Kumar Srirama",
   "Sudeep Dasari",
   "Shikhar Bahl",
   "Abhinav Gupta"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "RSS",
  "venue_source": "semantic-scholar",
  "citations": 50,
  "influential_citations": 3,
  "tldr": "This paper presents a simple framework for pre-training representations on hand, object, and contact that highlight relevant objects in images and how to interact with them and finds that this approach leads to higher levels of generalization in out-of-distribution settings.",
  "doi": "10.48550/arXiv.2407.18911",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "M. K. Srirama",
    "id": "2193493900",
    "h_index": 12,
    "papers": 14
   },
   {
    "name": "S. Dasari",
    "id": "36076404",
    "h_index": 23,
    "papers": 39
   },
   {
    "name": "Shikhar Bahl",
    "id": "8527563",
    "h_index": 20,
    "papers": 38
   },
   {
    "name": "Abhinav Gupta",
    "id": "2258675386",
    "h_index": 4,
    "papers": 7
   }
  ],
  "comment": "Accepted to Robotics Science and Systems 2024",
  "topics": [
   "dexterous-manipulation",
   "egocentric-data",
   "foundation-pretraining",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2407.18911v1",
  "pdf_url": "https://arxiv.org/pdf/2407.18911v1",
  "html_url": "https://arxiv.org/html/2407.18911v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.21
 },
 {
  "id": "2407.18902",
  "slug": "lessons-from-learning-to-spin-pens",
  "title": "Lessons from Learning to Spin \"Pens\"",
  "abstract": "In-hand manipulation of pen-like objects is an important skill in our daily lives, as many tools such as hammers and screwdrivers are similarly shaped. However, current learning-based methods struggle with this task due to a lack of high-quality demonstrations and the significant gap between simulation and the real world. In this work, we push the boundaries of learning-based in-hand manipulation systems by demonstrating the capability to spin pen-like objects. We first use reinforcement learning to train an oracle policy with privileged information and generate a high-fidelity trajectory dataset in simulation. This serves two purposes: 1) pre-training a sensorimotor policy in simulation; 2) conducting open-loop trajectory replay in the real world. We then fine-tune the sensorimotor policy using these real-world trajectories to adapt it to the real world dynamics. With less than 50 trajectories, our policy learns to rotate more than ten pen-like objects with different physical properties for multiple revolutions. We present a comprehensive analysis of our design choices and share the lessons learned during development.",
  "published": "2024-07-26",
  "updated": "2024-10-23",
  "year": "2024",
  "authors": [
   "Jun Wang",
   "Ying Yuan",
   "Haichuan Che",
   "Haozhi Qi",
   "Yi Ma",
   "Jitendra Malik",
   "Xiaolong Wang"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 66,
  "influential_citations": 4,
  "tldr": "This work uses reinforcement learning to train an oracle policy with privileged information and generates a high-fidelity trajectory dataset in simulation for pre-training a sensorimotor policy in simulation and fine-tune the sensorimotor policy using these real-world trajectories to adapt it to the real world dynamics.",
  "doi": "10.48550/arXiv.2407.18902",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jun Wang",
    "id": "2313695727",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Ying Yuan",
    "id": "2269689594",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Haichuan Che",
    "id": "2269471060",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Haozhi Qi",
    "id": "2247951244",
    "h_index": 11,
    "papers": 18
   },
   {
    "name": "Yi Ma",
    "id": "2324935696",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Jitendra Malik",
    "id": "2242761335",
    "h_index": 15,
    "papers": 32
   },
   {
    "name": "Xiaolong Wang",
    "id": "2255251113",
    "h_index": 10,
    "papers": 17
   }
  ],
  "comment": "CoRL 2024. Website: https://penspin.github.io/",
  "topics": [
   "dexterous-manipulation",
   "rl-control",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2407.18902v2",
  "pdf_url": "https://arxiv.org/pdf/2407.18902v2",
  "html_url": "https://arxiv.org/html/2407.18902v2",
  "code_url": "https://penspin.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.33
 },
 {
  "id": "2407.17470",
  "slug": "sv4d-dynamic-3d-content-generation-with-multi-frame-and-multi-view-con",
  "title": "SV4D: Dynamic 3D Content Generation with Multi-Frame and Multi-View Consistency",
  "abstract": "We present Stable Video 4D (SV4D), a latent video diffusion model for multi-frame and multi-view consistent dynamic 3D content generation. Unlike previous methods that rely on separately trained generative models for video generation and novel view synthesis, we design a unified diffusion model to generate novel view videos of dynamic 3D objects. Specifically, given a monocular reference video, SV4D generates novel views for each video frame that are temporally consistent. We then use the generated novel view videos to optimize an implicit 4D representation (dynamic NeRF) efficiently, without the need for cumbersome SDS-based optimization used in most prior works. To train our unified novel view video generation model, we curate a dynamic 3D object dataset from the existing Objaverse dataset. Extensive experimental results on multiple datasets and user studies demonstrate SV4D's state-of-the-art performance on novel-view video synthesis as well as 4D generation compared to prior works.",
  "published": "2024-07-24",
  "updated": "2025-02-27",
  "year": "2024",
  "authors": [
   "Yiming Xie",
   "Chun-Han Yao",
   "Vikram Voleti",
   "Huaizu Jiang",
   "Varun Jampani"
  ],
  "author_count": 5,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 162,
  "influential_citations": 19,
  "tldr": "Stable Video 4D is presented, a latent video diffusion model for multi-frame and multi-view consistent dynamic 3D content generation and state-of-the-art performance on novel-view video synthesis as well as 4D generation compared to prior works.",
  "doi": "10.48550/arXiv.2407.17470",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yiming Xie",
    "id": "46268911",
    "h_index": 13,
    "papers": 25
   },
   {
    "name": "C. Yao",
    "id": "2292210811",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Vikram S. Voleti",
    "id": "2961618",
    "h_index": 17,
    "papers": 45
   },
   {
    "name": "Huaizu Jiang",
    "id": "2254230515",
    "h_index": 11,
    "papers": 18
   },
   {
    "name": "Varun Jampani",
    "id": "2131639924",
    "h_index": 34,
    "papers": 88
   }
  ],
  "comment": "Project page: https://sv4d.github.io/",
  "topics": [
   "spatial-3d",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2407.17470v2",
  "pdf_url": "https://arxiv.org/pdf/2407.17470v2",
  "html_url": "https://arxiv.org/html/2407.17470v2",
  "code_url": "https://sv4d.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.71
 },
 {
  "id": "2407.15208",
  "slug": "flow-as-the-cross-domain-manipulation-interface",
  "title": "Flow as the Cross-Domain Manipulation Interface",
  "abstract": "We present Im2Flow2Act, a scalable learning framework that enables robots to acquire real-world manipulation skills without the need of real-world robot training data. The key idea behind Im2Flow2Act is to use object flow as the manipulation interface, bridging domain gaps between different embodiments (i.e., human and robot) and training environments (i.e., real-world and simulated). Im2Flow2Act comprises two components: a flow generation network and a flow-conditioned policy. The flow generation network, trained on human demonstration videos, generates object flow from the initial scene image, conditioned on the task description. The flow-conditioned policy, trained on simulated robot play data, maps the generated object flow to robot actions to realize the desired object movements. By using flow as input, this policy can be directly deployed in the real world with a minimal sim-to-real gap. By leveraging real-world human videos and simulated robot play data, we bypass the challenges of teleoperating physical robots in the real world, resulting in a scalable system for diverse tasks. We demonstrate Im2Flow2Act's capabilities in a variety of real-world tasks, including the manipulation of rigid, articulated, and deformable objects.",
  "published": "2024-07-21",
  "updated": "2024-10-04",
  "year": "2024",
  "authors": [
   "Mengda Xu",
   "Zhenjia Xu",
   "Yinghao Xu",
   "Cheng Chi",
   "Gordon Wetzstein",
   "Manuela Veloso",
   "Shuran Song"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 175,
  "influential_citations": 17,
  "tldr": "Im2Flow2Act is presented, a scalable learning framework that enables robots to acquire real-world manipulation skills without the need of real-world robot training data and its capabilities in a variety of real-world tasks, including the manipulation of rigid, articulated, and deformable objects.",
  "doi": "10.48550/arXiv.2407.15208",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Mengda Xu",
    "id": "2110683030",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "Zhenjia Xu",
    "id": "74498275",
    "h_index": 15,
    "papers": 22
   },
   {
    "name": "Yinghao Xu",
    "id": "121983635",
    "h_index": 35,
    "papers": 74
   },
   {
    "name": "Cheng Chi",
    "id": "2253746565",
    "h_index": 8,
    "papers": 11
   },
   {
    "name": "Gordon Wetzstein",
    "id": "2256985147",
    "h_index": 21,
    "papers": 43
   },
   {
    "name": "Manuela Veloso",
    "id": "2312325086",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Shuran Song",
    "id": "2254874914",
    "h_index": 12,
    "papers": 12
   }
  ],
  "comment": "Conference on Robot Learning 2024",
  "topics": [
   "egocentric-data",
   "sim2real"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2407.15208v2",
  "pdf_url": "https://arxiv.org/pdf/2407.15208v2",
  "html_url": "https://arxiv.org/html/2407.15208v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.75
 },
 {
  "id": "2407.13764",
  "slug": "shape-of-motion-4d-reconstruction-from-a-single-video",
  "title": "Shape of Motion: 4D Reconstruction from a Single Video",
  "abstract": "Monocular dynamic reconstruction is a challenging and long-standing vision problem due to the highly ill-posed nature of the task. Existing approaches depend on templates, are effective only in quasi-static scenes, or fail to model 3D motion explicitly. We introduce a method for reconstructing generic dynamic scenes, featuring explicit, persistent 3D motion trajectories in the world coordinate frame, from casually captured monocular videos. We tackle the problem with two key insights: First, we exploit the low-dimensional structure of 3D motion by representing scene motion with a compact set of SE(3) motion bases. Each point's motion is expressed as a linear combination of these bases, facilitating soft decomposition of the scene into multiple rigidly-moving groups. Second, we take advantage of off-the-shelf data-driven priors such as monocular depth maps and long-range 2D tracks, and devise a method to effectively consolidate these noisy supervisory signals, resulting in a globally consistent representation of the dynamic scene. Experiments show that our method achieves state-of-the-art performance for both long-range 3D/2D motion estimation and novel view synthesis on dynamic scenes. Project Page: https://shape-of-motion.github.io/",
  "published": "2024-07-18",
  "updated": "2025-10-16",
  "year": "2024",
  "authors": [
   "Qianqian Wang",
   "Vickie Ye",
   "Hang Gao",
   "Weijia Zeng",
   "Jake Austin",
   "Zhengqi Li",
   "Angjoo Kanazawa"
  ],
  "author_count": 7,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "ICCV",
  "venue_source": "semantic-scholar",
  "citations": 279,
  "influential_citations": 59,
  "tldr": "This work introduces a method for reconstructing generic dynamic scenes, featuring explicit, persistent 3D motion trajectories in the world coordinate frame, from casually captured monocular videos, with state-of-the-art performance for both long-range 3D/2D motion estimation and novel view synthesis on dynamic scenes.",
  "doi": "10.1109/ICCV51701.2025.00901",
  "oa_pdf": "https://arxiv.org/pdf/2407.13764",
  "s2_authors": [
   {
    "name": "Qianqian Wang",
    "id": "2311992344",
    "h_index": 8,
    "papers": 10
   },
   {
    "name": "Vickie Ye",
    "id": "31541718",
    "h_index": 12,
    "papers": 15
   },
   {
    "name": "Hang Gao",
    "id": "2312002037",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Jake Austin",
    "id": "2311880590",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Zhengqi Li",
    "id": "2145369560",
    "h_index": 23,
    "papers": 30
   },
   {
    "name": "Angjoo Kanazawa",
    "id": "20615377",
    "h_index": 60,
    "papers": 126
   }
  ],
  "comment": "ICCV 2025",
  "topics": [
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2407.13764v2",
  "pdf_url": "https://arxiv.org/pdf/2407.13764v2",
  "html_url": "https://arxiv.org/html/2407.13764v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.95
 },
 {
  "id": "2407.12957",
  "slug": "r-x-retrieval-and-execution-from-everyday-human-videos",
  "title": "R+X: Retrieval and Execution from Everyday Human Videos",
  "abstract": "We present R+X, a framework which enables robots to learn skills from long, unlabelled, first-person videos of humans performing everyday tasks. Given a language command from a human, R+X first retrieves short video clips containing relevant behaviour, and then executes the skill by conditioning an in-context imitation learning method (KAT) on this behaviour. By leveraging a Vision Language Model (VLM) for retrieval, R+X does not require any manual annotation of the videos, and by leveraging in-context learning for execution, robots can perform commanded skills immediately, without requiring a period of training on the retrieved videos. Experiments studying a range of everyday household tasks show that R+X succeeds at translating unlabelled human videos into robust robot skills, and that R+X outperforms several recent alternative methods. Videos and code are available at https://www.robot-learning.uk/r-plus-x.",
  "published": "2024-07-17",
  "updated": "2025-04-03",
  "year": "2024",
  "authors": [
   "Georgios Papagiannis",
   "Norman Di Palo",
   "Pietro Vitiello",
   "Edward Johns"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA",
  "venue_source": "semantic-scholar",
  "citations": 51,
  "influential_citations": 4,
  "tldr": "Experiments showing that a framework which enables robots to learn skills from long, unlabelled, first-person videos of humans performing everyday tasks succeeds at translating unlabelled human videos into robust robot skills and outperforms several recent alternative methods are shown.",
  "doi": "10.1109/ICRA55743.2025.11128322",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Georgios Papagiannis",
    "id": "121828617",
    "h_index": 9,
    "papers": 14
   },
   {
    "name": "Norman Di Palo",
    "id": "52218759",
    "h_index": 14,
    "papers": 20
   },
   {
    "name": "Pietro Vitiello",
    "id": "2259940808",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Edward Johns",
    "id": "2253752957",
    "h_index": 10,
    "papers": 14
   }
  ],
  "comment": "Published at the IEEE International Conference on Robotics and Automation (ICRA) 2025",
  "topics": [
   "egocentric-data",
   "imitation-diffusion",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2407.12957v2",
  "pdf_url": "https://arxiv.org/pdf/2407.12957v2",
  "html_url": "https://arxiv.org/html/2407.12957v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.22
 },
 {
  "id": "2407.12684",
  "slug": "4dynamic-text-to-4d-generation-with-hybrid-priors",
  "title": "4Dynamic: Text-to-4D Generation with Hybrid Priors",
  "abstract": "Due to the fascinating generative performance of text-to-image diffusion models, growing text-to-3D generation works explore distilling the 2D generative priors into 3D, using the score distillation sampling (SDS) loss, to bypass the data scarcity problem. The existing text-to-3D methods have achieved promising results in realism and 3D consistency, but text-to-4D generation still faces challenges, including lack of realism and insufficient dynamic motions. In this paper, we propose a novel method for text-to-4D generation, which ensures the dynamic amplitude and authenticity through direct supervision provided by a video prior. Specifically, we adopt a text-to-video diffusion model to generate a reference video and divide 4D generation into two stages: static generation and dynamic generation. The static 3D generation is achieved under the guidance of the input text and the first frame of the reference video, while in the dynamic generation stage, we introduce a customized SDS loss to ensure multi-view consistency, a video-based SDS loss to improve temporal consistency, and most importantly, direct priors from the reference video to ensure the quality of geometry and texture. Moreover, we design a prior-switching training strategy to avoid conflicts between different priors and fully leverage the benefits of each prior. In addition, to enrich the generated motion, we further introduce a dynamic modeling representation composed of a deformation network and a topology network, which ensures dynamic continuity while modeling topological changes. Our method not only supports text-to-4D generation but also enables 4D generation from monocular videos. The comparison experiments demonstrate the superiority of our method compared to existing methods.",
  "published": "2024-07-17",
  "updated": "2024-07-17",
  "year": "2024",
  "authors": [
   "Yu-Jie Yuan",
   "Leif Kobbelt",
   "Jiwen Liu",
   "Yuan Zhang",
   "Pengfei Wan",
   "Yu-Kun Lai",
   "Lin Gao"
  ],
  "author_count": 7,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 19,
  "influential_citations": 0,
  "tldr": "This paper proposes a novel method for text-to-4D generation, which ensures the dynamic amplitude and authenticity through direct supervision provided by a video prior, and divides 4D generation into two stages: static generation and dynamic generation.",
  "doi": "10.48550/arXiv.2407.12684",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yu-Jie Yuan",
    "id": "49521199",
    "h_index": 13,
    "papers": 18
   },
   {
    "name": "Leif Kobbelt",
    "id": "2241428629",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Jiwen Liu",
    "id": "2363686534",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Yuan Zhang",
    "id": "2311915614",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Pengfei Wan",
    "id": "2261751278",
    "h_index": 9,
    "papers": 34
   },
   {
    "name": "Yu-Kun Lai",
    "id": "2239059773",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Lin Gao",
    "id": "2312098760",
    "h_index": 3,
    "papers": 6
   }
  ],
  "comment": "",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2407.12684v1",
  "pdf_url": "https://arxiv.org/pdf/2407.12684v1",
  "html_url": "https://arxiv.org/html/2407.12684v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.3
 },
 {
  "id": "2407.11398",
  "slug": "animate3d-animating-any-3d-model-with-multi-view-video-diffusion",
  "title": "Animate3D: Animating Any 3D Model with Multi-view Video Diffusion",
  "abstract": "Recent advances in 4D generation mainly focus on generating 4D content by distilling pre-trained text or single-view image-conditioned models. It is inconvenient for them to take advantage of various off-the-shelf 3D assets with multi-view attributes, and their results suffer from spatiotemporal inconsistency owing to the inherent ambiguity in the supervision signals. In this work, we present Animate3D, a novel framework for animating any static 3D model. The core idea is two-fold: 1) We propose a novel multi-view video diffusion model (MV-VDM) conditioned on multi-view renderings of the static 3D object, which is trained on our presented large-scale multi-view video dataset (MV-Video). 2) Based on MV-VDM, we introduce a framework combining reconstruction and 4D Score Distillation Sampling (4D-SDS) to leverage the multi-view video diffusion priors for animating 3D objects. Specifically, for MV-VDM, we design a new spatiotemporal attention module to enhance spatial and temporal consistency by integrating 3D and video diffusion models. Additionally, we leverage the static 3D model's multi-view renderings as conditions to preserve its identity. For animating 3D models, an effective two-stage pipeline is proposed: we first reconstruct motions directly from generated multi-view videos, followed by the introduced 4D-SDS to refine both appearance and motion. Benefiting from accurate motion learning, we could achieve straightforward mesh animation. Qualitative and quantitative experiments demonstrate that Animate3D significantly outperforms previous approaches. Data, code, and models will be open-released.",
  "published": "2024-07-16",
  "updated": "2024-09-09",
  "year": "2024",
  "authors": [
   "Yanqin Jiang",
   "Chaohui Yu",
   "Chenjie Cao",
   "Fan Wang",
   "Weiming Hu",
   "Jin Gao"
  ],
  "author_count": 6,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 61,
  "influential_citations": 6,
  "tldr": "A novel multi-view video diffusion model (MV-VDM) conditioned on multi-view renderings of the static 3D object, which is trained on the authors' presented large-scale multi-view video dataset (MV-Video) and a framework combining reconstruction and 4D Score Distillation Sampling (4D-SDS) to leverage the multi-view video diffusion priors for animating 3D objects.",
  "doi": "10.48550/arXiv.2407.11398",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yanqin Jiang",
    "id": "2265542118",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Chaohui Yu",
    "id": "2110961040",
    "h_index": 16,
    "papers": 45
   },
   {
    "name": "Chenjie Cao",
    "id": "2296071044",
    "h_index": 7,
    "papers": 14
   },
   {
    "name": "Fan Wang",
    "id": "2257894784",
    "h_index": 8,
    "papers": 18
   },
   {
    "name": "Weiming Hu",
    "id": "2240303745",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Jin Gao",
    "id": "2265517996",
    "h_index": 2,
    "papers": 9
   }
  ],
  "comment": "Project Page: https://animate3d.github.io/",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2407.11398v2",
  "pdf_url": "https://arxiv.org/pdf/2407.11398v2",
  "html_url": "https://arxiv.org/html/2407.11398v2",
  "code_url": "https://animate3d.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.29
 },
 {
  "id": "2407.10353",
  "slug": "umi-on-legs-making-manipulation-policies-mobile-with-manipulation-cent",
  "title": "UMI on Legs: Making Manipulation Policies Mobile with Manipulation-Centric Whole-body Controllers",
  "abstract": "We introduce UMI-on-Legs, a new framework that combines real-world and simulation data for quadruped manipulation systems. We scale task-centric data collection in the real world using a hand-held gripper (UMI), providing a cheap way to demonstrate task-relevant manipulation skills without a robot. Simultaneously, we scale robot-centric data in simulation by training whole-body controller for task-tracking without task simulation setups. The interface between these two policies is end-effector trajectories in the task frame, inferred by the manipulation policy and passed to the whole-body controller for tracking. We evaluate UMI-on-Legs on prehensile, non-prehensile, and dynamic manipulation tasks, and report over 70% success rate on all tasks. Lastly, we demonstrate the zero-shot cross-embodiment deployment of a pre-trained manipulation policy checkpoint from prior work, originally intended for a fixed-base robot arm, on our quadruped system. We believe this framework provides a scalable path towards learning expressive manipulation skills on dynamic robot embodiments. Please checkout our website for robot videos, code, and data: https://umi-on-legs.github.io",
  "published": "2024-07-14",
  "updated": "2024-07-14",
  "year": "2024",
  "authors": [
   "Huy Ha",
   "Yihuai Gao",
   "Zipeng Fu",
   "Jie Tan",
   "Shuran Song"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 137,
  "influential_citations": 12,
  "tldr": "This work introduces UMI-on-Legs, a new framework that combines real-world and simulation data for quadruped manipulation systems, and believes this framework provides a scalable path towards learning expressive manipulation skills on dynamic robot embodiments.",
  "doi": "10.48550/arXiv.2407.10353",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Huy Ha",
    "id": "2291134164",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Yihuai Gao",
    "id": "2311944190",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Zipeng Fu",
    "id": "2307996544",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Jie Tan",
    "id": "2311681974",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Shuran Song",
    "id": "2289085682",
    "h_index": 7,
    "papers": 13
   }
  ],
  "comment": "18 pages, 7 figures, website: https://umi-on-legs.github.io/",
  "topics": [
   "humanoids",
   "foundation-pretraining",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2407.10353v1",
  "pdf_url": "https://arxiv.org/pdf/2407.10353v1",
  "html_url": "https://arxiv.org/html/2407.10353v1",
  "code_url": "https://umi-on-legs.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.64
 },
 {
  "id": "2407.08737",
  "slug": "video-diffusion-alignment-via-reward-gradients",
  "title": "Video Diffusion Alignment via Reward Gradients",
  "abstract": "We have made significant progress towards building foundational video diffusion models. As these models are trained using large-scale unsupervised data, it has become crucial to adapt these models to specific downstream tasks. Adapting these models via supervised fine-tuning requires collecting target datasets of videos, which is challenging and tedious. In this work, we utilize pre-trained reward models that are learned via preferences on top of powerful vision discriminative models to adapt video diffusion models. These models contain dense gradient information with respect to generated RGB pixels, which is critical to efficient learning in complex search spaces, such as videos. We show that backpropagating gradients from these reward models to a video diffusion model can allow for compute and sample efficient alignment of the video diffusion model. We show results across a variety of reward models and video diffusion models, demonstrating that our approach can learn much more efficiently in terms of reward queries and computation than prior gradient-free approaches. Our code, model weights,and more visualization are available at https://vader-vid.github.io.",
  "published": "2024-07-11",
  "updated": "2024-07-11",
  "year": "2024",
  "authors": [
   "Mihir Prabhudesai",
   "Russell Mendonca",
   "Zheyang Qin",
   "Katerina Fragkiadaki",
   "Deepak Pathak"
  ],
  "author_count": 5,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.LG",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 98,
  "influential_citations": 8,
  "tldr": "This work utilizes pre-trained reward models that are learned via preferences on top of powerful vision discriminative models to adapt video diffusion models, showing that backpropagating gradients from these reward models to a video diffusion model can allow for compute and sample efficient alignment of the video diffusion model.",
  "doi": "10.48550/arXiv.2407.08737",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Mihir Prabhudesai",
    "id": "1381902419",
    "h_index": 14,
    "papers": 23
   },
   {
    "name": "R. Mendonca",
    "id": "35509365",
    "h_index": 15,
    "papers": 24
   },
   {
    "name": "Zheyang Qin",
    "id": "2311373133",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Katerina Fragkiadaki",
    "id": "1705557",
    "h_index": 32,
    "papers": 82
   },
   {
    "name": "Deepak Pathak",
    "id": "2254258594",
    "h_index": 9,
    "papers": 13
   }
  ],
  "comment": "Project Webpage: https://vader-vid.github.io; Code available at: https://github.com/mihirp1998/VADER",
  "topics": [
   "rl-control",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2407.08737v1",
  "pdf_url": "https://arxiv.org/pdf/2407.08737v1",
  "html_url": "https://arxiv.org/html/2407.08737v1",
  "code_url": "https://vader-vid.github.io",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.0
 },
 {
  "id": "2407.03162",
  "slug": "bunny-visionpro-real-time-bimanual-dexterous-teleoperation-for-imitati",
  "title": "Bunny-VisionPro: Real-Time Bimanual Dexterous Teleoperation for Imitation Learning",
  "abstract": "Teleoperation is a crucial tool for collecting human demonstrations, but controlling robots with bimanual dexterous hands remains a challenge. Existing teleoperation systems struggle to handle the complexity of coordinating two hands for intricate manipulations. We introduce Bunny-VisionPro, a real-time bimanual dexterous teleoperation system that leverages a VR headset. Unlike previous vision-based teleoperation systems, we design novel low-cost devices to provide haptic feedback to the operator, enhancing immersion. Our system prioritizes safety by incorporating collision and singularity avoidance while maintaining real-time performance through innovative designs. Bunny-VisionPro outperforms prior systems on a standard task suite, achieving higher success rates and reduced task completion times. Moreover, the high-quality teleoperation demonstrations improve downstream imitation learning performance, leading to better generalizability. Notably, Bunny-VisionPro enables imitation learning with challenging multi-stage, long-horizon dexterous manipulation tasks, which have rarely been addressed in previous work. Our system's ability to handle bimanual manipulations while prioritizing safety and real-time performance makes it a powerful tool for advancing dexterous manipulation and imitation learning.",
  "published": "2024-07-03",
  "updated": "2024-07-03",
  "year": "2024",
  "authors": [
   "Runyu Ding",
   "Yuzhe Qin",
   "Jiyue Zhu",
   "Chengzhe Jia",
   "Shiqi Yang",
   "Ruihan Yang",
   "Xiaojuan Qi",
   "Xiaolong Wang"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS",
  "venue_source": "semantic-scholar",
  "citations": 157,
  "influential_citations": 7,
  "tldr": "Bunny-VisionPro is introduced, a real-time bimanual dexterous teleoperation system that leverages a VR headset that prioritizes safety by incorporating collision and singularity avoidance while maintaining real-time performance through innovative designs.",
  "doi": "10.1109/IROS60139.2025.11247017",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Runyu Ding",
    "id": "2066798330",
    "h_index": 9,
    "papers": 12
   },
   {
    "name": "Yuzhe Qin",
    "id": "12701031",
    "h_index": 24,
    "papers": 34
   },
   {
    "name": "Jiyue Zhu",
    "id": "2309670850",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Chengzhe Jia",
    "id": "2292143295",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Shiqi Yang",
    "id": "2309666838",
    "h_index": 8,
    "papers": 9
   },
   {
    "name": "Ruihan Yang",
    "id": "143955842",
    "h_index": 18,
    "papers": 22
   },
   {
    "name": "Xiaojuan Qi",
    "id": "2283520411",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Xiaolong Wang",
    "id": "2239141122",
    "h_index": 12,
    "papers": 13
   }
  ],
  "comment": "project page: https://dingry.github.io/projects/bunny_visionpro.html",
  "topics": [
   "dexterous-manipulation",
   "egocentric-data",
   "imitation-diffusion",
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2407.03162v1",
  "pdf_url": "https://arxiv.org/pdf/2407.03162v1",
  "html_url": "https://arxiv.org/html/2407.03162v1",
  "code_url": "https://dingry.github.io/projects/bunny_visionpro.html",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.7
 },
 {
  "id": "2407.02371",
  "slug": "openvid-1m-a-large-scale-high-quality-dataset-for-text-to-video-genera",
  "title": "OpenVid-1M: A Large-Scale High-Quality Dataset for Text-to-video Generation",
  "abstract": "Text-to-video (T2V) generation has recently garnered significant attention thanks to the large multi-modality model Sora. However, T2V generation still faces two important challenges: 1) Lacking a precise open sourced high-quality dataset. The previous popular video datasets, e.g. WebVid-10M and Panda-70M, are either with low quality or too large for most research institutions. Therefore, it is challenging but crucial to collect a precise high-quality text-video pairs for T2V generation. 2) Ignoring to fully utilize textual information. Recent T2V methods have focused on vision transformers, using a simple cross attention module for video generation, which falls short of thoroughly extracting semantic information from text prompt. To address these issues, we introduce OpenVid-1M, a precise high-quality dataset with expressive captions. This open-scenario dataset contains over 1 million text-video pairs, facilitating research on T2V generation. Furthermore, we curate 433K 1080p videos from OpenVid-1M to create OpenVidHD-0.4M, advancing high-definition video generation. Additionally, we propose a novel Multi-modal Video Diffusion Transformer (MVDiT) capable of mining both structure information from visual tokens and semantic information from text tokens. Extensive experiments and ablation studies verify the superiority of OpenVid-1M over previous datasets and the effectiveness of our MVDiT.",
  "published": "2024-07-02",
  "updated": "2025-02-13",
  "year": "2024",
  "authors": [
   "Kepan Nan",
   "Rui Xie",
   "Penghao Zhou",
   "Tiehan Fan",
   "Zhenheng Yang",
   "Zhijie Chen",
   "Xiang Li",
   "Jian Yang",
   "Ying Tai"
  ],
  "author_count": 9,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 340,
  "influential_citations": 33,
  "tldr": "OpenVid-1M, a precise high-quality dataset with expressive captions capable of mining both structure information from visual tokens and semantic information from text tokens, and a novel Multi-modal Video Diffusion Transformer (MVDiT) capable of mining both structure information from visual tokens and semantic information from text tokens are proposed.",
  "doi": "10.48550/arXiv.2407.02371",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Kepan Nan",
    "id": "2207118366",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Rui Xie",
    "id": "2294568380",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Penghao Zhou",
    "id": "2309360728",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Tiehan Fan",
    "id": "2309244412",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Zhenheng Yang",
    "id": "2309246065",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Zhijie Chen",
    "id": "2316662510",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Xiang Li",
    "id": "2284824283",
    "h_index": 8,
    "papers": 14
   },
   {
    "name": "Jian Yang",
    "id": "2309321403",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Ying Tai",
    "id": "2284862495",
    "h_index": 7,
    "papers": 22
   }
  ],
  "comment": "20 pages, 15 figures, Published as a conference paper at ICLR 2025",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2407.02371v3",
  "pdf_url": "https://arxiv.org/pdf/2407.02371v3",
  "html_url": "https://arxiv.org/html/2407.02371v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.03
 },
 {
  "id": "2407.02274",
  "slug": "dextrah-g-pixels-to-action-dexterous-arm-hand-grasping-with-geometric",
  "title": "DextrAH-G: Pixels-to-Action Dexterous Arm-Hand Grasping with Geometric Fabrics",
  "abstract": "A pivotal challenge in robotics is achieving fast, safe, and robust dexterous grasping across a diverse range of objects, an important goal within industrial applications. However, existing methods often have very limited speed, dexterity, and generality, along with limited or no hardware safety guarantees. In this work, we introduce DextrAH-G, a depth-based dexterous grasping policy trained entirely in simulation that combines reinforcement learning, geometric fabrics, and teacher-student distillation. We address key challenges in joint arm-hand policy learning, such as high-dimensional observation and action spaces, the sim2real gap, collision avoidance, and hardware constraints. DextrAH-G enables a 23 motor arm-hand robot to safely and continuously grasp and transport a large variety of objects at high speed using multi-modal inputs including depth images, allowing generalization across object geometry. Videos at https://sites.google.com/view/dextrah-g.",
  "published": "2024-07-02",
  "updated": "2024-10-15",
  "year": "2024",
  "authors": [
   "Tyler Ga Wei Lum",
   "Martin Matak",
   "Viktor Makoviychuk",
   "Ankur Handa",
   "Arthur Allshire",
   "Tucker Hermans",
   "Nathan D. Ratliff",
   "Karl Van Wyk"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 65,
  "influential_citations": 4,
  "tldr": "DextrAH-G is introduced, a depth-based dexterous grasping policy trained entirely in simulation that combines reinforcement learning, geometric fabrics, and teacher-student distillation that enables a 23 motor arm-hand robot to safely and continuously grasp and transport a large variety of objects at high speed using multi-modal inputs including depth images.",
  "doi": "10.48550/arXiv.2407.02274",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Tyler Ga Wei Lum",
    "id": "2309245667",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Martin Matak",
    "id": "1381574177",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Viktor Makoviychuk",
    "id": "79875630",
    "h_index": 18,
    "papers": 21
   },
   {
    "name": "Ankur Handa",
    "id": "34653454",
    "h_index": 33,
    "papers": 55
   },
   {
    "name": "Arthur Allshire",
    "id": "2061149217",
    "h_index": 9,
    "papers": 12
   },
   {
    "name": "Tucker Hermans",
    "id": "145767346",
    "h_index": 34,
    "papers": 107
   },
   {
    "name": "Nathan D. Ratliff",
    "id": "2240527931",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "Karl Van Wyk",
    "id": "2423933",
    "h_index": 20,
    "papers": 54
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "sim2real",
   "rl-control",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2407.02274v3",
  "pdf_url": "https://arxiv.org/pdf/2407.02274v3",
  "html_url": "https://arxiv.org/html/2407.02274v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.32
 },
 {
  "id": "2407.01812",
  "slug": "equivariant-diffusion-policy",
  "title": "Equivariant Diffusion Policy",
  "abstract": "Recent work has shown diffusion models are an effective approach to learning the multimodal distributions arising from demonstration data in behavior cloning. However, a drawback of this approach is the need to learn a denoising function, which is significantly more complex than learning an explicit policy. In this work, we propose Equivariant Diffusion Policy, a novel diffusion policy learning method that leverages domain symmetries to obtain better sample efficiency and generalization in the denoising function. We theoretically analyze the $\\mathrm{SO}(2)$ symmetry of full 6-DoF control and characterize when a diffusion model is $\\mathrm{SO}(2)$-equivariant. We furthermore evaluate the method empirically on a set of 12 simulation tasks in MimicGen, and show that it obtains a success rate that is, on average, 21.9% higher than the baseline Diffusion Policy. We also evaluate the method on a real-world system to show that effective policies can be learned with relatively few training samples, whereas the baseline Diffusion Policy cannot.",
  "published": "2024-07-01",
  "updated": "2024-10-15",
  "year": "2024",
  "authors": [
   "Dian Wang",
   "Stephen Hart",
   "David Surovik",
   "Tarik Kelestemur",
   "Haojie Huang",
   "Haibo Zhao",
   "Mark Yeatman",
   "Jiuguang Wang",
   "Robin Walters",
   "Robert Platt"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 87,
  "influential_citations": 12,
  "tldr": "This work proposes Equivariant Diffusion Policy, a novel diffusion policy learning method that leverages domain symmetries to obtain better sample efficiency and generalization in the denoising function.",
  "doi": "10.48550/arXiv.2407.01812",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Di Wang",
    "id": "2145383788",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Stephen M. Hart",
    "id": "2054396731",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "David Surovik",
    "id": "2309247336",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Tarik Kelestemur",
    "id": "2309246993",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Hao Huang",
    "id": "87356471",
    "h_index": 1,
    "papers": 11
   },
   {
    "name": "Haibo Zhao",
    "id": "2112675896",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Mark R. Yeatman",
    "id": "2094406399",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Jiu-yao Wang",
    "id": "2110179117",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Robin Walters",
    "id": "2066259129",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Robert C. Platt",
    "id": "2068058014",
    "h_index": 1,
    "papers": 3
   }
  ],
  "comment": "Conference on Robot Learning 2024, Oral Presentation",
  "topics": [
   "imitation-diffusion",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2407.01812v3",
  "pdf_url": "https://arxiv.org/pdf/2407.01812v3",
  "html_url": "https://arxiv.org/html/2407.01812v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.44
 },
 {
  "id": "2407.01512",
  "slug": "open-television-teleoperation-with-immersive-active-visual-feedback",
  "title": "Open-TeleVision: Teleoperation with Immersive Active Visual Feedback",
  "abstract": "Teleoperation serves as a powerful method for collecting on-robot data essential for robot learning from demonstrations. The intuitiveness and ease of use of the teleoperation system are crucial for ensuring high-quality, diverse, and scalable data. To achieve this, we propose an immersive teleoperation system Open-TeleVision that allows operators to actively perceive the robot's surroundings in a stereoscopic manner. Additionally, the system mirrors the operator's arm and hand movements on the robot, creating an immersive experience as if the operator's mind is transmitted to a robot embodiment. We validate the effectiveness of our system by collecting data and training imitation learning policies on four long-horizon, precise tasks (Can Sorting, Can Insertion, Folding, and Unloading) for 2 different humanoid robots and deploy them in the real world. The system is open-sourced at: https://robot-tv.github.io/",
  "published": "2024-07-01",
  "updated": "2024-07-08",
  "year": "2024",
  "authors": [
   "Xuxin Cheng",
   "Jialong Li",
   "Shiqi Yang",
   "Ge Yang",
   "Xiaolong Wang"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.HC",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 319,
  "influential_citations": 21,
  "tldr": "An immersive teleoperation system Open-TeleVision that allows operators to actively perceive the robot's surroundings in a stereoscopic manner and mirrors the operator's arm and hand movements on the robot, creating an immersive experience as if the operator's mind is transmitted to a robot embodiment.",
  "doi": "10.48550/arXiv.2407.01512",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xuxin Cheng",
    "id": "2287822264",
    "h_index": 8,
    "papers": 9
   },
   {
    "name": "Jialong Li",
    "id": "2309196968",
    "h_index": 7,
    "papers": 7
   },
   {
    "name": "Shiqi Yang",
    "id": "2309666838",
    "h_index": 8,
    "papers": 9
   },
   {
    "name": "Ge Yang",
    "id": "2288147740",
    "h_index": 9,
    "papers": 10
   },
   {
    "name": "Xiaolong Wang",
    "id": "2294782536",
    "h_index": 12,
    "papers": 17
   }
  ],
  "comment": "Website: https://robot-tv.github.io/",
  "topics": [
   "humanoids",
   "imitation-diffusion",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2407.01512v2",
  "pdf_url": "https://arxiv.org/pdf/2407.01512v2",
  "html_url": "https://arxiv.org/html/2407.01512v2",
  "code_url": "https://robot-tv.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.01
 },
 {
  "id": "2407.01392",
  "slug": "diffusion-forcing-next-token-prediction-meets-full-sequence-diffusion",
  "title": "Diffusion Forcing: Next-token Prediction Meets Full-Sequence Diffusion",
  "abstract": "This paper presents Diffusion Forcing, a new training paradigm where a diffusion model is trained to denoise a set of tokens with independent per-token noise levels. We apply Diffusion Forcing to sequence generative modeling by training a causal next-token prediction model to generate one or several future tokens without fully diffusing past ones. Our approach is shown to combine the strengths of next-token prediction models, such as variable-length generation, with the strengths of full-sequence diffusion models, such as the ability to guide sampling to desirable trajectories. Our method offers a range of additional capabilities, such as (1) rolling-out sequences of continuous tokens, such as video, with lengths past the training horizon, where baselines diverge and (2) new sampling and guiding schemes that uniquely profit from Diffusion Forcing's variable-horizon and causal architecture, and which lead to marked performance gains in decision-making and planning tasks. In addition to its empirical success, our method is proven to optimize a variational lower bound on the likelihoods of all subsequences of tokens drawn from the true joint distribution. Project website: https://boyuan.space/diffusion-forcing",
  "published": "2024-07-01",
  "updated": "2024-12-10",
  "year": "2024",
  "authors": [
   "Boyuan Chen",
   "Diego Marti Monso",
   "Yilun Du",
   "Max Simchowitz",
   "Russ Tedrake",
   "Vincent Sitzmann"
  ],
  "author_count": 6,
  "categories": [
   "cs.LG",
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.LG",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 646,
  "influential_citations": 81,
  "tldr": "This paper presents Diffusion Forcing, a new training paradigm where a diffusion model is trained to denoise a set of tokens with independent per-token noise levels, and is proven to optimize a variational lower bound on the likelihoods of all subsequences of tokens drawn from the true joint distribution.",
  "doi": "10.48550/arXiv.2407.01392",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Boyuan Chen",
    "id": "8786274",
    "h_index": 10,
    "papers": 15
   },
   {
    "name": "Diego Marti Monso",
    "id": "2309174104",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Yilun Du",
    "id": "15394275",
    "h_index": 48,
    "papers": 86
   },
   {
    "name": "Max Simchowitz",
    "id": "2243186471",
    "h_index": 9,
    "papers": 12
   },
   {
    "name": "Russ Tedrake",
    "id": "2263905014",
    "h_index": 14,
    "papers": 36
   },
   {
    "name": "Vincent Sitzmann",
    "id": "2280906248",
    "h_index": 7,
    "papers": 8
   }
  ],
  "comment": "Project website: https://boyuan.space/diffusion-forcing",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2407.01392v4",
  "pdf_url": "https://arxiv.org/pdf/2407.01392v4",
  "html_url": "https://arxiv.org/html/2407.01392v4",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.31
 },
 {
  "id": "2406.19464",
  "slug": "maniwav-learning-robot-manipulation-from-in-the-wild-audio-visual-data",
  "title": "ManiWAV: Learning Robot Manipulation from In-the-Wild Audio-Visual Data",
  "abstract": "Audio signals provide rich information for the robot interaction and object properties through contact. This information can surprisingly ease the learning of contact-rich robot manipulation skills, especially when the visual information alone is ambiguous or incomplete. However, the usage of audio data in robot manipulation has been constrained to teleoperated demonstrations collected by either attaching a microphone to the robot or object, which significantly limits its usage in robot learning pipelines. In this work, we introduce ManiWAV: an 'ear-in-hand' data collection device to collect in-the-wild human demonstrations with synchronous audio and visual feedback, and a corresponding policy interface to learn robot manipulation policy directly from the demonstrations. We demonstrate the capabilities of our system through four contact-rich manipulation tasks that require either passively sensing the contact events and modes, or actively sensing the object surface materials and states. In addition, we show that our system can generalize to unseen in-the-wild environments by learning from diverse in-the-wild human demonstrations.",
  "published": "2024-06-27",
  "updated": "2024-11-04",
  "year": "2024",
  "authors": [
   "Zeyi Liu",
   "Cheng Chi",
   "Eric Cousineau",
   "Naveen Kuppuswamy",
   "Benjamin Burchfiel",
   "Shuran Song"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.SD",
   "eess.AS"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 73,
  "influential_citations": 9,
  "tldr": "ManiWAV is introduced: an 'ear-in-hand' data collection device to collect in-the-wild human demonstrations with synchronous audio and visual feedback, and a corresponding policy interface to learn robot manipulation policy directly from the demonstrations.",
  "doi": "10.48550/arXiv.2406.19464",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zeyi Liu",
    "id": "2176845464",
    "h_index": 9,
    "papers": 11
   },
   {
    "name": "Cheng Chi",
    "id": "2253746565",
    "h_index": 8,
    "papers": 11
   },
   {
    "name": "Eric Cousineau",
    "id": "2090529",
    "h_index": 12,
    "papers": 16
   },
   {
    "name": "Naveen Kuppuswamy",
    "id": "2275353318",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "B. Burchfiel",
    "id": "2302757",
    "h_index": 18,
    "papers": 30
   },
   {
    "name": "Shuran Song",
    "id": "2254874914",
    "h_index": 12,
    "papers": 12
   }
  ],
  "comment": "Conference on Robot Learning (CoRL) 2024; Project website: https://maniwav.github.io/",
  "topics": [
   "dexterous-manipulation",
   "egocentric-data",
   "tactile",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2406.19464v2",
  "pdf_url": "https://arxiv.org/pdf/2406.19464v2",
  "html_url": "https://arxiv.org/html/2406.19464v2",
  "code_url": "https://maniwav.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.37
 },
 {
  "id": "2406.17840",
  "slug": "human-object-interaction-from-human-level-instructions",
  "title": "Human-Object Interaction from Human-Level Instructions",
  "abstract": "Intelligent agents must autonomously interact with the environments to perform daily tasks based on human-level instructions. They need a foundational understanding of the world to accurately interpret these instructions, along with precise low-level movement and interaction skills to execute the derived actions. In this work, we propose the first complete system for synthesizing physically plausible, long-horizon human-object interactions for object manipulation in contextual environments, driven by human-level instructions. We leverage large language models (LLMs) to interpret the input instructions into detailed execution plans. Unlike prior work, our system is capable of generating detailed finger-object interactions, in seamless coordination with full-body movements. We also train a policy to track generated motions in physics simulation via reinforcement learning (RL) to ensure physical plausibility of the motion. Our experiments demonstrate the effectiveness of our system in synthesizing realistic interactions with diverse objects in complex environments, highlighting its potential for real-world applications.",
  "published": "2024-06-25",
  "updated": "2025-08-21",
  "year": "2024",
  "authors": [
   "Zhen Wu",
   "Jiaman Li",
   "Pei Xu",
   "C. Karen Liu"
  ],
  "author_count": 4,
  "categories": [
   "cs.AI",
   "cs.CV"
  ],
  "primary_category": "cs.AI",
  "venue": "ICCV",
  "venue_source": "semantic-scholar",
  "citations": 70,
  "influential_citations": 2,
  "tldr": "This work proposes the first complete system for synthesizing physically plausible, long-horizon human-object interactions for object manipulation in contextual environments, driven by human-level instructions, and uses large language models to interpret the input instructions into detailed execution plans.",
  "doi": "10.1109/ICCV51701.2025.01040",
  "oa_pdf": "https://arxiv.org/pdf/2406.17840",
  "s2_authors": [
   {
    "name": "Zhen Wu",
    "id": "2308574851",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Jiaman Li",
    "id": "22133106",
    "h_index": 16,
    "papers": 22
   },
   {
    "name": "Pei Xu",
    "id": "2335601372",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "C. K. Liu",
    "id": "2247934447",
    "h_index": 9,
    "papers": 11
   }
  ],
  "comment": "ICCV 2025, project page: https://hoifhli.github.io/",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2406.17840v3",
  "pdf_url": "https://arxiv.org/pdf/2406.17840v3",
  "html_url": "https://arxiv.org/html/2406.17840v3",
  "code_url": "https://hoifhli.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.35
 },
 {
  "id": "2406.16862",
  "slug": "dreamitate-real-world-visuomotor-policy-learning-via-video-generation",
  "title": "Dreamitate: Real-World Visuomotor Policy Learning via Video Generation",
  "abstract": "A key challenge in manipulation is learning a policy that can robustly generalize to diverse visual environments. A promising mechanism for learning robust policies is to leverage video generative models, which are pretrained on large-scale datasets of internet videos. In this paper, we propose a visuomotor policy learning framework that fine-tunes a video diffusion model on human demonstrations of a given task. At test time, we generate an example of an execution of the task conditioned on images of a novel scene, and use this synthesized execution directly to control the robot. Our key insight is that using common tools allows us to effortlessly bridge the embodiment gap between the human hand and the robot manipulator. We evaluate our approach on four tasks of increasing complexity and demonstrate that harnessing internet-scale generative models allows the learned policy to achieve a significantly higher degree of generalization than existing behavior cloning approaches.",
  "published": "2024-06-24",
  "updated": "2024-06-24",
  "year": "2024",
  "authors": [
   "Junbang Liang",
   "Ruoshi Liu",
   "Ege Ozguroglu",
   "Sruthi Sudhakar",
   "Achal Dave",
   "Pavel Tokmakov",
   "Shuran Song",
   "Carl Vondrick"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 113,
  "influential_citations": 7,
  "tldr": "A visuomotor policy learning framework that fine-tunes a video diffusion model on human demonstrations of a given task and demonstrates that harnessing internet-scale generative models allows the learned policy to achieve a significantly higher degree of generalization than existing behavior cloning approaches.",
  "doi": "10.48550/arXiv.2406.16862",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Junbang Liang",
    "id": "2291325095",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Ruoshi Liu",
    "id": "2143183492",
    "h_index": 14,
    "papers": 24
   },
   {
    "name": "Ege Ozguroglu",
    "id": "2281036251",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Sruthi Sudhakar",
    "id": "2291134651",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Achal Dave",
    "id": "2298523",
    "h_index": 23,
    "papers": 41
   },
   {
    "name": "P. Tokmakov",
    "id": "2931554",
    "h_index": 24,
    "papers": 72
   },
   {
    "name": "Shuran Song",
    "id": "2289085682",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Carl Vondrick",
    "id": "1856025",
    "h_index": 47,
    "papers": 116
   }
  ],
  "comment": "Project page: https://dreamitate.cs.columbia.edu/",
  "topics": [
   "egocentric-data",
   "imitation-diffusion",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2406.16862v1",
  "pdf_url": "https://arxiv.org/pdf/2406.16862v1",
  "html_url": "https://arxiv.org/html/2406.16862v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.56
 },
 {
  "id": "2406.16860",
  "slug": "cambrian-1-a-fully-open-vision-centric-exploration-of-multimodal-llms",
  "title": "Cambrian-1: A Fully Open, Vision-Centric Exploration of Multimodal LLMs",
  "abstract": "We introduce Cambrian-1, a family of multimodal LLMs (MLLMs) designed with a vision-centric approach. While stronger language models can enhance multimodal capabilities, the design choices for vision components are often insufficiently explored and disconnected from visual representation learning research. This gap hinders accurate sensory grounding in real-world scenarios. Our study uses LLMs and visual instruction tuning as an interface to evaluate various visual representations, offering new insights into different models and architectures -- self-supervised, strongly supervised, or combinations thereof -- based on experiments with over 20 vision encoders. We critically examine existing MLLM benchmarks, address the difficulties involved in consolidating and interpreting results from various tasks, and introduce a new vision-centric benchmark, CV-Bench. To further improve visual grounding, we propose the Spatial Vision Aggregator (SVA), a dynamic and spatially-aware connector that integrates high-resolution vision features with LLMs while reducing the number of tokens. Additionally, we discuss the curation of high-quality visual instruction-tuning data from publicly available sources, emphasizing the importance of data source balancing and distribution ratio. Collectively, Cambrian-1 not only achieves state-of-the-art performance but also serves as a comprehensive, open cookbook for instruction-tuned MLLMs. We provide model weights, code, supporting tools, datasets, and detailed instruction-tuning and evaluation recipes. We hope our release will inspire and accelerate advancements in multimodal systems and visual representation learning.",
  "published": "2024-06-24",
  "updated": "2024-12-04",
  "year": "2024",
  "authors": [
   "Shengbang Tong",
   "Ellis Brown",
   "Penghao Wu",
   "Sanghyun Woo",
   "Manoj Middepogu",
   "Sai Charitha Akula",
   "Jihan Yang",
   "Shusheng Yang",
   "Adithya Iyer",
   "Xichen Pan",
   "Ziteng Wang",
   "Rob Fergus",
   "Yann LeCun",
   "Saining Xie"
  ],
  "author_count": 14,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 945,
  "influential_citations": 145,
  "tldr": "This study uses LLMs and visual instruction tuning as an interface to evaluate various visual representations, offering new insights into different models and architectures -- self-supervised, strongly supervised, or combinations thereof -- based on experiments with over 20 vision encoders.",
  "doi": "10.48550/arXiv.2406.16860",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Shengbang Tong",
    "id": "2143202419",
    "h_index": 20,
    "papers": 28
   },
   {
    "name": "Ellis Brown",
    "id": "113880966",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "Penghao Wu",
    "id": "2275969099",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Sanghyun Woo",
    "id": "2324788683",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Manoj Middepogu",
    "id": "2308037570",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Sai Akula",
    "id": "2308036736",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Jihan Yang",
    "id": "47987749",
    "h_index": 17,
    "papers": 21
   },
   {
    "name": "Shusheng Yang",
    "id": "2237626195",
    "h_index": 9,
    "papers": 31
   },
   {
    "name": "Adithya Iyer",
    "id": "2393205310",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Xichen Pan",
    "id": "2158877024",
    "h_index": 11,
    "papers": 17
   },
   {
    "name": "Austin Wang",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Rob Fergus",
    "id": "2308037493",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Yann LeCun",
    "id": "2265899558",
    "h_index": 22,
    "papers": 47
   },
   {
    "name": "Saining Xie",
    "id": "2282918539",
    "h_index": 13,
    "papers": 16
   }
  ],
  "comment": "NeurIPS 2024 (Oral). Website at https://cambrian-mllm.github.io",
  "topics": [
   "navigation",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2406.16860v2",
  "pdf_url": "https://arxiv.org/pdf/2406.16860v2",
  "html_url": "https://arxiv.org/html/2406.16860v2",
  "code_url": "https://cambrian-mllm.github.io",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.48
 },
 {
  "id": "2406.15252",
  "slug": "videoscore-building-automatic-metrics-to-simulate-fine-grained-human-f",
  "title": "VideoScore: Building Automatic Metrics to Simulate Fine-grained Human Feedback for Video Generation",
  "abstract": "The recent years have witnessed great advances in video generation. However, the development of automatic video metrics is lagging significantly behind. None of the existing metric is able to provide reliable scores over generated videos. The main barrier is the lack of large-scale human-annotated dataset. In this paper, we release VideoFeedback, the first large-scale dataset containing human-provided multi-aspect score over 37.6K synthesized videos from 11 existing video generative models. We train VideoScore (initialized from Mantis) based on VideoFeedback to enable automatic video quality assessment. Experiments show that the Spearman correlation between VideoScore and humans can reach 77.1 on VideoFeedback-test, beating the prior best metrics by about 50 points. Further result on other held-out EvalCrafter, GenAI-Bench, and VBench show that VideoScore has consistently much higher correlation with human judges than other metrics. Due to these results, we believe VideoScore can serve as a great proxy for human raters to (1) rate different video models to track progress (2) simulate fine-grained human feedback in Reinforcement Learning with Human Feedback (RLHF) to improve current video generation models.",
  "published": "2024-06-21",
  "updated": "2024-10-14",
  "year": "2024",
  "authors": [
   "Xuan He",
   "Dongfu Jiang",
   "Ge Zhang",
   "Max Ku",
   "Achint Soni",
   "Sherman Siu",
   "Haonan Chen",
   "Abhranil Chandra",
   "Ziyan Jiang",
   "Aaran Arulraj",
   "Kai Wang",
   "Quy Duc Do",
   "Yuansheng Ni",
   "Bohan Lyu",
   "Yaswanth Narsupalli",
   "Rongqi Fan",
   "Zhiheng Lyu",
   "Yuchen Lin",
   "Wenhu Chen"
  ],
  "author_count": 19,
  "categories": [
   "cs.CV",
   "cs.AI"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 201,
  "influential_citations": 36,
  "tldr": "VideoFeedback, the first large-scale dataset containing human-provided multi-aspect score over 37.6K synthesized videos from 11 existing video generative models, is released and it is believed that VideoScore can serve as a great proxy for human raters to rate different video models to track progress.",
  "doi": "10.48550/arXiv.2406.15252",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xuan He",
    "id": "2299486403",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Dongfu Jiang",
    "id": "2197076899",
    "h_index": 14,
    "papers": 33
   },
   {
    "name": "Ge Zhang",
    "id": "2319593516",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Max W.F. Ku",
    "id": "2218153604",
    "h_index": 14,
    "papers": 21
   },
   {
    "name": "Achint Soni",
    "id": "2307914737",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Sherman Siu",
    "id": "6680481",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Haonan Chen",
    "id": "2268767321",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Abhranil Chandra",
    "id": "2304449194",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Ziyan Jiang",
    "id": "2112347577",
    "h_index": 14,
    "papers": 32
   },
   {
    "name": "Aaran Arulraj",
    "id": "2304445005",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Kai Wang",
    "id": "2304713202",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Quy Duc Do",
    "id": "2294572942",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Yuansheng Ni",
    "id": "2268493966",
    "h_index": 11,
    "papers": 21
   },
   {
    "name": "B. Lyu",
    "id": "2307914708",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Yaswanth Narsupalli",
    "id": "2258712590",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Rongqi \"Richard\" Fan",
    "id": "2304426528",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Z. Lyu",
    "id": "2307914614",
    "h_index": 7,
    "papers": 16
   },
   {
    "name": "Yuchen Lin",
    "id": "2281613034",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Wenhu Chen",
    "id": "2253811180",
    "h_index": 16,
    "papers": 23
   }
  ],
  "comment": "",
  "topics": [
   "rl-control",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2406.15252v3",
  "pdf_url": "https://arxiv.org/pdf/2406.15252v3",
  "html_url": "https://arxiv.org/html/2406.15252v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.31
 },
 {
  "id": "2406.14655",
  "slug": "hypermotion-learning-hybrid-behavior-planning-for-autonomous-loco-mani",
  "title": "HYPERmotion: Learning Hybrid Behavior Planning for Autonomous Loco-manipulation",
  "abstract": "Enabling robots to autonomously perform hybrid motions in diverse environments can be beneficial for long-horizon tasks such as material handling, household chores, and work assistance. This requires extensive exploitation of intrinsic motion capabilities, extraction of affordances from rich environmental information, and planning of physical interaction behaviors. Despite recent progress has demonstrated impressive humanoid whole-body control abilities, they struggle to achieve versatility and adaptability for new tasks. In this work, we propose HYPERmotion, a framework that learns, selects and plans behaviors based on tasks in different scenarios. We combine reinforcement learning with whole-body optimization to generate motion for 38 actuated joints and create a motion library to store the learned skills. We apply the planning and reasoning features of the large language models (LLMs) to complex loco-manipulation tasks, constructing a hierarchical task graph that comprises a series of primitive behaviors to bridge lower-level execution with higher-level planning. By leveraging the interaction of distilled spatial geometry and 2D observation with a visual language model (VLM) to ground knowledge into a robotic morphology selector to choose appropriate actions in single- or dual-arm, legged or wheeled locomotion. Experiments in simulation and real-world show that learned motions can efficiently adapt to new tasks, demonstrating high autonomy from free-text commands in unstructured scenes. Videos and website: hy-motion.github.io/",
  "published": "2024-06-20",
  "updated": "2024-06-20",
  "year": "2024",
  "authors": [
   "Jin Wang",
   "Rui Dai",
   "Weijie Wang",
   "Luca Rossini",
   "Francesco Ruscelli",
   "Nikos Tsagarakis"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 17,
  "influential_citations": 1,
  "tldr": "This work combines reinforcement learning with whole-body optimization to generate motion for 38 actuated joints and create a motion library to store the learned skills, and proposes HYPERmotion, a framework that learns, selects and plans behaviors based on tasks in different scenarios.",
  "doi": "10.48550/arXiv.2406.14655",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jin Wang",
    "id": "2307932685",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Rui Dai",
    "id": "2323511277",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Weijie Wang",
    "id": "2307982102",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Luca Rossini",
    "id": "2399788479",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Francesco Ruscelli",
    "id": "8273495",
    "h_index": 7,
    "papers": 24
   },
   {
    "name": "Nikos G. Tsagarakis",
    "id": "2307917489",
    "h_index": 2,
    "papers": 11
   }
  ],
  "comment": "Project page: https://hy-motion.github.io/",
  "topics": [
   "humanoids",
   "rl-control",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2406.14655v1",
  "pdf_url": "https://arxiv.org/pdf/2406.14655v1",
  "html_url": "https://arxiv.org/html/2406.14655v1",
  "code_url": "https://hy-motion.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.76
 },
 {
  "id": "2406.13640",
  "slug": "transferable-tactile-transformers-for-representation-learning-across-d",
  "title": "Transferable Tactile Transformers for Representation Learning Across Diverse Sensors and Tasks",
  "abstract": "This paper presents T3: Transferable Tactile Transformers, a framework for tactile representation learning that scales across multi-sensors and multi-tasks. T3 is designed to overcome the contemporary issue that camera-based tactile sensing is extremely heterogeneous, i.e. sensors are built into different form factors, and existing datasets were collected for disparate tasks. T3 captures the shared latent information across different sensor-task pairings by constructing a shared trunk transformer with sensor-specific encoders and task-specific decoders. The pre-training of T3 utilizes a novel Foundation Tactile (FoTa) dataset, which is aggregated from several open-sourced datasets and it contains over 3 million data points gathered from 13 sensors and 11 tasks. FoTa is the largest and most diverse dataset in tactile sensing to date and it is made publicly available in a unified format. Across various sensors and tasks, experiments show that T3 pre-trained with FoTa achieved zero-shot transferability in certain sensor-task pairings, can be further fine-tuned with small amounts of domain-specific data, and its performance scales with bigger network sizes. T3 is also effective as a tactile encoder for long horizon contact-rich manipulation. Results from sub-millimeter multi-pin electronics insertion tasks show that T3 achieved a task success rate 25% higher than that of policies trained with tactile encoders trained from scratch, or 53% higher than without tactile sensing. Data, code, and model checkpoints are open-sourced at https://t3.alanz.info",
  "published": "2024-06-19",
  "updated": "2024-10-06",
  "year": "2024",
  "authors": [
   "Jialiang Zhao",
   "Yuxiang Ma",
   "Lirui Wang",
   "Edward H. Adelson"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 84,
  "influential_citations": 11,
  "tldr": "Results from sub-millimeter multi-pin electronics insertion tasks show that T3 achieved a task success rate 25% higher than that of policies trained with tactile encoders trained from scratch, or 53% higher than without tactile sensing.",
  "doi": "10.48550/arXiv.2406.13640",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jialiang Zhao",
    "id": "2243705199",
    "h_index": 12,
    "papers": 18
   },
   {
    "name": "Yuxiang Ma",
    "id": "2144365142",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Lirui Wang",
    "id": "2253973819",
    "h_index": 6,
    "papers": 6
   },
   {
    "name": "Edward H. Adelson",
    "id": "2293172170",
    "h_index": 8,
    "papers": 18
   }
  ],
  "comment": "Accepted to 2024 Conference on Robot Learning (CoRL)",
  "topics": [
   "tactile",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2406.13640v3",
  "pdf_url": "https://arxiv.org/pdf/2406.13640v3",
  "html_url": "https://arxiv.org/html/2406.13640v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.43
 },
 {
  "id": "2406.11838",
  "slug": "autoregressive-image-generation-without-vector-quantization",
  "title": "Autoregressive Image Generation without Vector Quantization",
  "abstract": "Conventional wisdom holds that autoregressive models for image generation are typically accompanied by vector-quantized tokens. We observe that while a discrete-valued space can facilitate representing a categorical distribution, it is not a necessity for autoregressive modeling. In this work, we propose to model the per-token probability distribution using a diffusion procedure, which allows us to apply autoregressive models in a continuous-valued space. Rather than using categorical cross-entropy loss, we define a Diffusion Loss function to model the per-token probability. This approach eliminates the need for discrete-valued tokenizers. We evaluate its effectiveness across a wide range of cases, including standard autoregressive models and generalized masked autoregressive (MAR) variants. By removing vector quantization, our image generator achieves strong results while enjoying the speed advantage of sequence modeling. We hope this work will motivate the use of autoregressive generation in other continuous-valued domains and applications. Code is available at: https://github.com/LTH14/mar.",
  "published": "2024-06-17",
  "updated": "2024-11-01",
  "year": "2024",
  "authors": [
   "Tianhong Li",
   "Yonglong Tian",
   "He Li",
   "Mingyang Deng",
   "Kaiming He"
  ],
  "author_count": 5,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 714,
  "influential_citations": 140,
  "tldr": "This work proposes to model the per-token probability distribution using a diffusion procedure, which allows to apply autoregressive models in a continuous-valued space and evaluates its effectiveness across a wide range of cases, including standard autoregressive models and generalized masked autoregressive (MAR) variants.",
  "doi": "10.48550/arXiv.2406.11838",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Tianhong Li",
    "id": "2307269819",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Yonglong Tian",
    "id": "2307043887",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "He Li",
    "id": "2307146098",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Mingyang Deng",
    "id": "2306970309",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Kaiming He",
    "id": "2270025109",
    "h_index": 6,
    "papers": 8
   }
  ],
  "comment": "Neurips 2024 (Spotlight). Code: https://github.com/LTH14/mar",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2406.11838v3",
  "pdf_url": "https://arxiv.org/pdf/2406.11838v3",
  "html_url": "https://arxiv.org/html/2406.11838v3",
  "code_url": "https://github.com/LTH14/mar",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.35
 },
 {
  "id": "2406.11815",
  "slug": "llarva-vision-action-instruction-tuning-enhances-robot-learning",
  "title": "LLARVA: Vision-Action Instruction Tuning Enhances Robot Learning",
  "abstract": "In recent years, instruction-tuned Large Multimodal Models (LMMs) have been successful at several tasks, including image captioning and visual question answering; yet leveraging these models remains an open question for robotics. Prior LMMs for robotics applications have been extensively trained on language and action data, but their ability to generalize in different settings has often been less than desired. To address this, we introduce LLARVA, a model trained with a novel instruction tuning method that leverages structured prompts to unify a range of robotic learning tasks, scenarios, and environments. Additionally, we show that predicting intermediate 2-D representations, which we refer to as \"visual traces\", can help further align vision and action spaces for robot learning. We generate 8.5M image-visual trace pairs from the Open X-Embodiment dataset in order to pre-train our model, and we evaluate on 12 different tasks in the RLBench simulator as well as a physical Franka Emika Panda 7-DoF robot. Our experiments yield strong performance, demonstrating that LLARVA - using 2-D and language representations - performs well compared to several contemporary baselines, and can generalize across various robot environments and configurations.",
  "published": "2024-06-17",
  "updated": "2024-06-17",
  "year": "2024",
  "authors": [
   "Dantong Niu",
   "Yuvan Sharma",
   "Giscard Biamby",
   "Jerome Quenum",
   "Yutong Bai",
   "Baifeng Shi",
   "Trevor Darrell",
   "Roei Herzig"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 93,
  "influential_citations": 6,
  "tldr": "This work introduces LLARVA, a model trained with a novel instruction tuning method that leverages structured prompts to unify a range of robotic learning tasks, scenarios, and environments and shows that predicting intermediate 2-D representations, which are referred to as \"visual traces\", can help further align vision and action spaces for robot learning.",
  "doi": "10.48550/arXiv.2406.11815",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Dantong Niu",
    "id": "2268757542",
    "h_index": 8,
    "papers": 17
   },
   {
    "name": "Yuvan Sharma",
    "id": "2307007378",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Giscard Biamby",
    "id": "1380219651",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "Jerome Quenum",
    "id": "2307004702",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Yutong Bai",
    "id": "48442730",
    "h_index": 20,
    "papers": 34
   },
   {
    "name": "Baifeng Shi",
    "id": "1596823732",
    "h_index": 17,
    "papers": 24
   },
   {
    "name": "Trevor Darrell",
    "id": "2257973285",
    "h_index": 11,
    "papers": 27
   },
   {
    "name": "Roei Herzig",
    "id": "46796686",
    "h_index": 22,
    "papers": 56
   }
  ],
  "comment": "",
  "topics": [
   "sim2real"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2406.11815v1",
  "pdf_url": "https://arxiv.org/pdf/2406.11815v1",
  "html_url": "https://arxiv.org/html/2406.11815v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.47
 },
 {
  "id": "2406.10759",
  "slug": "humanoid-parkour-learning",
  "title": "Humanoid Parkour Learning",
  "abstract": "Parkour is a grand challenge for legged locomotion, even for quadruped robots, requiring active perception and various maneuvers to overcome multiple challenging obstacles. Existing methods for humanoid locomotion either optimize a trajectory for a single parkour track or train a reinforcement learning policy only to walk with a significant amount of motion references. In this work, we propose a framework for learning an end-to-end vision-based whole-body-control parkour policy for humanoid robots that overcomes multiple parkour skills without any motion prior. Using the parkour policy, the humanoid robot can jump on a 0.42m platform, leap over hurdles, 0.8m gaps, and much more. It can also run at 1.8m/s in the wild and walk robustly on different terrains. We test our policy in indoor and outdoor environments to demonstrate that it can autonomously select parkour skills while following the rotation command of the joystick. We override the arm actions and show that this framework can easily transfer to humanoid mobile manipulation tasks. Videos can be found at https://humanoid4parkour.github.io",
  "published": "2024-06-15",
  "updated": "2024-09-26",
  "year": "2024",
  "authors": [
   "Ziwen Zhuang",
   "Shenzhe Yao",
   "Hang Zhao"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 189,
  "influential_citations": 9,
  "tldr": "This work proposes a framework for learning an end-to-end vision-based whole-body-control parkour policy for humanoid robots that overcomes multiple parkour skills without any motion prior and shows that this framework can easily transfer to humanoid mobile manipulation tasks.",
  "doi": "10.48550/arXiv.2406.10759",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ziwen Zhuang",
    "id": "1972362408",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Shenzhe Yao",
    "id": "2307047859",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Hang Zhao",
    "id": "2239158612",
    "h_index": 5,
    "papers": 9
   }
  ],
  "comment": "Published on CoRL 2024",
  "topics": [
   "humanoids",
   "rl-control",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2406.10759v2",
  "pdf_url": "https://arxiv.org/pdf/2406.10759v2",
  "html_url": "https://arxiv.org/html/2406.10759v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.78
 },
 {
  "id": "2406.10454",
  "slug": "humanplus-humanoid-shadowing-and-imitation-from-humans",
  "title": "HumanPlus: Humanoid Shadowing and Imitation from Humans",
  "abstract": "One of the key arguments for building robots that have similar form factors to human beings is that we can leverage the massive human data for training. Yet, doing so has remained challenging in practice due to the complexities in humanoid perception and control, lingering physical gaps between humanoids and humans in morphologies and actuation, and lack of a data pipeline for humanoids to learn autonomous skills from egocentric vision. In this paper, we introduce a full-stack system for humanoids to learn motion and autonomous skills from human data. We first train a low-level policy in simulation via reinforcement learning using existing 40-hour human motion datasets. This policy transfers to the real world and allows humanoid robots to follow human body and hand motion in real time using only a RGB camera, i.e. shadowing. Through shadowing, human operators can teleoperate humanoids to collect whole-body data for learning different tasks in the real world. Using the data collected, we then perform supervised behavior cloning to train skill policies using egocentric vision, allowing humanoids to complete different tasks autonomously by imitating human skills. We demonstrate the system on our customized 33-DoF 180cm humanoid, autonomously completing tasks such as wearing a shoe to stand up and walk, unloading objects from warehouse racks, folding a sweatshirt, rearranging objects, typing, and greeting another robot with 60-100% success rates using up to 40 demonstrations. Project website: https://humanoid-ai.github.io/",
  "published": "2024-06-15",
  "updated": "2024-06-15",
  "year": "2024",
  "authors": [
   "Zipeng Fu",
   "Qingqing Zhao",
   "Qi Wu",
   "Gordon Wetzstein",
   "Chelsea Finn"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 342,
  "influential_citations": 24,
  "tldr": "A full-stack system for humanoids to learn motion and autonomous skills from human data, allowing humanoids to complete different tasks autonomously by imitating human skills.",
  "doi": "10.48550/arXiv.2406.10454",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zipeng Fu",
    "id": "2307996544",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Qingqing Zhao",
    "id": "2263960925",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Qi Wu",
    "id": "2306951336",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Gordon Wetzstein",
    "id": "2304557070",
    "h_index": 10,
    "papers": 23
   },
   {
    "name": "Chelsea Finn",
    "id": "2239104473",
    "h_index": 7,
    "papers": 7
   }
  ],
  "comment": "project website: https://humanoid-ai.github.io/",
  "topics": [
   "humanoids",
   "egocentric-data",
   "imitation-diffusion",
   "rl-control",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2406.10454v1",
  "pdf_url": "https://arxiv.org/pdf/2406.10454v1",
  "html_url": "https://arxiv.org/html/2406.10454v1",
  "code_url": "https://humanoid-ai.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.04
 },
 {
  "id": "2406.10324",
  "slug": "l4gm-large-4d-gaussian-reconstruction-model",
  "title": "L4GM: Large 4D Gaussian Reconstruction Model",
  "abstract": "We present L4GM, the first 4D Large Reconstruction Model that produces animated objects from a single-view video input -- in a single feed-forward pass that takes only a second. Key to our success is a novel dataset of multiview videos containing curated, rendered animated objects from Objaverse. This dataset depicts 44K diverse objects with 110K animations rendered in 48 viewpoints, resulting in 12M videos with a total of 300M frames. We keep our L4GM simple for scalability and build directly on top of LGM, a pretrained 3D Large Reconstruction Model that outputs 3D Gaussian ellipsoids from multiview image input. L4GM outputs a per-frame 3D Gaussian Splatting representation from video frames sampled at a low fps and then upsamples the representation to a higher fps to achieve temporal smoothness. We add temporal self-attention layers to the base LGM to help it learn consistency across time, and utilize a per-timestep multiview rendering loss to train the model. The representation is upsampled to a higher framerate by training an interpolation model which produces intermediate 3D Gaussian representations. We showcase that L4GM that is only trained on synthetic data generalizes extremely well on in-the-wild videos, producing high quality animated 3D assets.",
  "published": "2024-06-14",
  "updated": "2024-06-14",
  "year": "2024",
  "authors": [
   "Jiawei Ren",
   "Kevin Xie",
   "Ashkan Mirzaei",
   "Hanxue Liang",
   "Xiaohui Zeng",
   "Karsten Kreis",
   "Ziwei Liu",
   "Antonio Torralba",
   "Sanja Fidler",
   "Seung Wook Kim",
   "Huan Ling"
  ],
  "author_count": 11,
  "categories": [
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.CV",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 162,
  "influential_citations": 26,
  "tldr": "",
  "doi": "10.48550/arXiv.2406.10324",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jiawei Ren",
    "id": "1820909323",
    "h_index": 20,
    "papers": 30
   },
   {
    "name": "Kevin Xie",
    "id": "2265532984",
    "h_index": 8,
    "papers": 14
   },
   {
    "name": "Ashkan Mirzaei",
    "id": "2174737496",
    "h_index": 13,
    "papers": 25
   },
   {
    "name": "Hanxue Liang",
    "id": "2307772593",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Xiaohui Zeng",
    "id": "1751476",
    "h_index": 12,
    "papers": 14
   },
   {
    "name": "Karsten Kreis",
    "id": "32113848",
    "h_index": 29,
    "papers": 52
   },
   {
    "name": "Ziwei Liu",
    "id": "2308229419",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Antonio Torralba",
    "id": "2262186466",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Sanja Fidler",
    "id": "2261282058",
    "h_index": 20,
    "papers": 41
   },
   {
    "name": "S. Kim",
    "id": "2262213592",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "Huan Ling",
    "id": "18900686",
    "h_index": 28,
    "papers": 44
   }
  ],
  "comment": "Project page: https://research.nvidia.com/labs/toronto-ai/l4gm",
  "topics": [
   "spatial-3d"
  ],
  "orgs": [
   "NVIDIA"
  ],
  "abs_url": "https://arxiv.org/abs/2406.10324v1",
  "pdf_url": "https://arxiv.org/pdf/2406.10324v1",
  "html_url": "https://arxiv.org/html/2406.10324v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.21
 },
 {
  "id": "2406.09905",
  "slug": "nymeria-a-massive-collection-of-multimodal-egocentric-daily-motion-in",
  "title": "Nymeria: A Massive Collection of Multimodal Egocentric Daily Motion in the Wild",
  "abstract": "We introduce Nymeria - a large-scale, diverse, richly annotated human motion dataset collected in the wild with multiple multimodal egocentric devices. The dataset comes with a) full-body ground-truth motion; b) multiple multimodal egocentric data from Project Aria devices with videos, eye tracking, IMUs and etc; and c) a third-person perspective by an additional observer. All devices are precisely synchronized and localized in on metric 3D world. We derive hierarchical protocol to add in-context language descriptions of human motion, from fine-grain motion narration, to simplified atomic action and high-level activity summarization. To the best of our knowledge, Nymeria dataset is the world's largest collection of human motion in the wild; first of its kind to provide synchronized and localized multi-device multimodal egocentric data; and the world's largest motion-language dataset. It provides 300 hours of daily activities from 264 participants across 50 locations, total travelling distance over 399Km. The language descriptions contain 301.5K sentences in 8.64M words from a vocabulary size of 6545. To demonstrate the potential of the dataset, we evaluate several SOTA algorithms for egocentric body tracking, motion synthesis, and action recognition. Data and code are open-sourced for research (c.f. https://www.projectaria.com/datasets/nymeria).",
  "published": "2024-06-14",
  "updated": "2024-09-20",
  "year": "2024",
  "authors": [
   "Lingni Ma",
   "Yuting Ye",
   "Fangzhou Hong",
   "Vladimir Guzov",
   "Yifeng Jiang",
   "Rowan Postyeni",
   "Luis Pesqueira",
   "Alexander Gamino",
   "Vijay Baiyya",
   "Hyo Jin Kim",
   "Kevin Bailey",
   "David Soriano Fosas",
   "C. Karen Liu",
   "Ziwei Liu",
   "Jakob Engel",
   "Renzo De Nardi",
   "Richard Newcombe"
  ],
  "author_count": 17,
  "categories": [
   "cs.CV",
   "cs.GR"
  ],
  "primary_category": "cs.CV",
  "venue": "ECCV",
  "venue_source": "semantic-scholar",
  "citations": 126,
  "influential_citations": 15,
  "tldr": "Nymeria dataset is introduced, a large-scale, diverse, richly annotated human motion dataset collected in the wild with multiple multimodal egocentric devices, and several SOTA algorithms for egocentric body tracking, motion synthesis, and action recognition are evaluated.",
  "doi": "10.48550/arXiv.2406.09905",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Lingni Ma",
    "id": "2284187142",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Yuting Ye",
    "id": "2307408723",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Fangzhou Hong",
    "id": "1568986485",
    "h_index": 24,
    "papers": 47
   },
   {
    "name": "Vladimir Guzov",
    "id": "2306788171",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Yifeng Jiang",
    "id": "2146419599",
    "h_index": 10,
    "papers": 14
   },
   {
    "name": "Rowan Postyeni",
    "id": "2306784373",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Luis Pesqueira",
    "id": "1416771217",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Alexander Gamino",
    "id": "2284862866",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Vijay Baiyya",
    "id": "2268759525",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "H. Kim",
    "id": "2307016154",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Kevin Bailey",
    "id": "2055721419",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "David Soriano Fosas",
    "id": "2306785653",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "C. K. Liu",
    "id": "2242331607",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Ziwei Liu",
    "id": "2294735512",
    "h_index": 10,
    "papers": 16
   },
   {
    "name": "J. Engel",
    "id": "2241357086",
    "h_index": 9,
    "papers": 13
   },
   {
    "name": "R. D. Nardi",
    "id": "1769365",
    "h_index": 15,
    "papers": 24
   },
   {
    "name": "Richard A. Newcombe",
    "id": "2292257340",
    "h_index": 9,
    "papers": 20
   }
  ],
  "comment": "",
  "topics": [
   "egocentric-data"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2406.09905v2",
  "pdf_url": "https://arxiv.org/pdf/2406.09905v2",
  "html_url": "https://arxiv.org/html/2406.09905v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.6
 },
 {
  "id": "2406.09598",
  "slug": "introducing-hot3d-an-egocentric-dataset-for-3d-hand-and-object-trackin",
  "title": "Introducing HOT3D: An Egocentric Dataset for 3D Hand and Object Tracking",
  "abstract": "We introduce HOT3D, a publicly available dataset for egocentric hand and object tracking in 3D. The dataset offers over 833 minutes (more than 3.7M images) of multi-view RGB/monochrome image streams showing 19 subjects interacting with 33 diverse rigid objects, multi-modal signals such as eye gaze or scene point clouds, as well as comprehensive ground truth annotations including 3D poses of objects, hands, and cameras, and 3D models of hands and objects. In addition to simple pick-up/observe/put-down actions, HOT3D contains scenarios resembling typical actions in a kitchen, office, and living room environment. The dataset is recorded by two head-mounted devices from Meta: Project Aria, a research prototype of light-weight AR/AI glasses, and Quest 3, a production VR headset sold in millions of units. Ground-truth poses were obtained by a professional motion-capture system using small optical markers attached to hands and objects. Hand annotations are provided in the UmeTrack and MANO formats and objects are represented by 3D meshes with PBR materials obtained by an in-house scanner. We aim to accelerate research on egocentric hand-object interaction by making the HOT3D dataset publicly available and by co-organizing public challenges on the dataset at ECCV 2024. The dataset can be downloaded from the project website: https://facebookresearch.github.io/hot3d/.",
  "published": "2024-06-13",
  "updated": "2024-06-13",
  "year": "2024",
  "authors": [
   "Prithviraj Banerjee",
   "Sindi Shkodrani",
   "Pierre Moulon",
   "Shreyas Hampali",
   "Fan Zhang",
   "Jade Fountain",
   "Edward Miller",
   "Selen Basol",
   "Richard Newcombe",
   "Robert Wang",
   "Jakob Julian Engel",
   "Tomas Hodan"
  ],
  "author_count": 12,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 56,
  "influential_citations": 7,
  "tldr": "HOT3D, a publicly available dataset for egocentric hand and object tracking in 3D, aims to accelerate research on egocentric hand-object interaction by making the HOT3D dataset publicly available and by co-organizing public challenges on the dataset at ECCV 2024.",
  "doi": "10.48550/arXiv.2406.09598",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Prithviraj Banerjee",
    "id": "2306783561",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Sindi Shkodrani",
    "id": "51208845",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Pierre Moulon",
    "id": "2284863436",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Shreyas Hampali",
    "id": "150296901",
    "h_index": 11,
    "papers": 20
   },
   {
    "name": "Fan Zhang",
    "id": "2307188098",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Jade Fountain",
    "id": "2306783804",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Edward Miller",
    "id": "2234024715",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Selen Basol",
    "id": "2904166",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Richard A. Newcombe",
    "id": "2292257340",
    "h_index": 9,
    "papers": 20
   },
   {
    "name": "Robert Wang",
    "id": "2307017739",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "J. Engel",
    "id": "2241357086",
    "h_index": 9,
    "papers": 13
   },
   {
    "name": "Tomas Hodan",
    "id": "2396902",
    "h_index": 21,
    "papers": 33
   }
  ],
  "comment": "",
  "topics": [
   "egocentric-data",
   "spatial-3d",
   "data-teleop"
  ],
  "orgs": [
   "Meta FAIR"
  ],
  "abs_url": "https://arxiv.org/abs/2406.09598v1",
  "pdf_url": "https://arxiv.org/pdf/2406.09598v1",
  "html_url": "https://arxiv.org/html/2406.09598v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.26
 },
 {
  "id": "2406.09416",
  "slug": "alleviating-distortion-in-image-generation-via-multi-resolution-diffus",
  "title": "Alleviating Distortion in Image Generation via Multi-Resolution Diffusion Models and Time-Dependent Layer Normalization",
  "abstract": "This paper presents innovative enhancements to diffusion models by integrating a novel multi-resolution network and time-dependent layer normalization. Diffusion models have gained prominence for their effectiveness in high-fidelity image generation. While conventional approaches rely on convolutional U-Net architectures, recent Transformer-based designs have demonstrated superior performance and scalability. However, Transformer architectures, which tokenize input data (via \"patchification\"), face a trade-off between visual fidelity and computational complexity due to the quadratic nature of self-attention operations concerning token length. While larger patch sizes enable attention computation efficiency, they struggle to capture fine-grained visual details, leading to image distortions. To address this challenge, we propose augmenting the Diffusion model with the Multi-Resolution network (DiMR), a framework that refines features across multiple resolutions, progressively enhancing detail from low to high resolution. Additionally, we introduce Time-Dependent Layer Normalization (TD-LN), a parameter-efficient approach that incorporates time-dependent parameters into layer normalization to inject time information and achieve superior performance. Our method's efficacy is demonstrated on the class-conditional ImageNet generation benchmark, where DiMR-XL variants outperform prior diffusion models, setting new state-of-the-art FID scores of 1.70 on ImageNet 256 x 256 and 2.89 on ImageNet 512 x 512. Project page: https://qihao067.github.io/projects/DiMR",
  "published": "2024-06-13",
  "updated": "2024-11-28",
  "year": "2024",
  "authors": [
   "Qihao Liu",
   "Zhanpeng Zeng",
   "Ju He",
   "Qihang Yu",
   "Xiaohui Shen",
   "Liang-Chieh Chen"
  ],
  "author_count": 6,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 39,
  "influential_citations": 0,
  "tldr": "Time-Dependent Layer Normalization (TD-LN), a parameter-efficient approach that incorporates time-dependent parameters into layer normalization to inject time information and achieve superior performance is introduced.",
  "doi": "10.48550/arXiv.2406.09416",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Qihao Liu",
    "id": "2306461720",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Zhanpeng Zeng",
    "id": "2337338246",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Ju He",
    "id": "2306090627",
    "h_index": 8,
    "papers": 14
   },
   {
    "name": "Qihang Yu",
    "id": "2304633845",
    "h_index": 10,
    "papers": 19
   },
   {
    "name": "Xiaohui Shen",
    "id": "2266472250",
    "h_index": 11,
    "papers": 18
   },
   {
    "name": "Liang-Chieh Chen",
    "id": "2266697544",
    "h_index": 13,
    "papers": 26
   }
  ],
  "comment": "Introducing DiMR, a new diffusion backbone that surpasses all existing image generation models of various sizes on ImageNet 256 with only 505M parameters. Project page: https://qihao067.github.io/projects/DiMR",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2406.09416v2",
  "pdf_url": "https://arxiv.org/pdf/2406.09416v2",
  "html_url": "https://arxiv.org/html/2406.09416v2",
  "code_url": "https://qihao067.github.io/projects/DiMR",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.1
 },
 {
  "id": "2406.09414",
  "slug": "depth-anything-v2",
  "title": "Depth Anything V2",
  "abstract": "This work presents Depth Anything V2. Without pursuing fancy techniques, we aim to reveal crucial findings to pave the way towards building a powerful monocular depth estimation model. Notably, compared with V1, this version produces much finer and more robust depth predictions through three key practices: 1) replacing all labeled real images with synthetic images, 2) scaling up the capacity of our teacher model, and 3) teaching student models via the bridge of large-scale pseudo-labeled real images. Compared with the latest models built on Stable Diffusion, our models are significantly more efficient (more than 10x faster) and more accurate. We offer models of different scales (ranging from 25M to 1.3B params) to support extensive scenarios. Benefiting from their strong generalization capability, we fine-tune them with metric depth labels to obtain our metric depth models. In addition to our models, considering the limited diversity and frequent noise in current test sets, we construct a versatile evaluation benchmark with precise annotations and diverse scenes to facilitate future research.",
  "published": "2024-06-13",
  "updated": "2024-06-13",
  "year": "2024",
  "authors": [
   "Lihe Yang",
   "Bingyi Kang",
   "Zilong Huang",
   "Zhen Zhao",
   "Xiaogang Xu",
   "Jiashi Feng",
   "Hengshuang Zhao"
  ],
  "author_count": 7,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 2095,
  "influential_citations": 308,
  "tldr": "This work presents Depth Anything V2, which produces much finer and more robust depth predictions through three key practices: replacing all labeled real images with synthetic images, scaling up the capacity of the teacher model, and teaching student models via the bridge of large-scale pseudo-labeled real images.",
  "doi": "10.48550/arXiv.2406.09414",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Lihe Yang",
    "id": "2268796616",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Bingyi Kang",
    "id": "2261363653",
    "h_index": 8,
    "papers": 10
   },
   {
    "name": "Zilong Huang",
    "id": "2276315322",
    "h_index": 11,
    "papers": 19
   },
   {
    "name": "Zhen Zhao",
    "id": "145737114",
    "h_index": 16,
    "papers": 37
   },
   {
    "name": "Xiaogang Xu",
    "id": "2261385713",
    "h_index": 8,
    "papers": 11
   },
   {
    "name": "Jiashi Feng",
    "id": "2276580610",
    "h_index": 16,
    "papers": 24
   },
   {
    "name": "Hengshuang Zhao",
    "id": "2253834598",
    "h_index": 11,
    "papers": 19
   }
  ],
  "comment": "Project page: https://depth-anything-v2.github.io",
  "topics": [
   "spatial-3d",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2406.09414v1",
  "pdf_url": "https://arxiv.org/pdf/2406.09414v1",
  "html_url": "https://arxiv.org/html/2406.09414v1",
  "code_url": "https://depth-anything-v2.github.io",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.5
 },
 {
  "id": "2406.09246",
  "slug": "openvla-an-open-source-vision-language-action-model",
  "title": "OpenVLA: An Open-Source Vision-Language-Action Model",
  "abstract": "Large policies pretrained on a combination of Internet-scale vision-language data and diverse robot demonstrations have the potential to change how we teach robots new skills: rather than training new behaviors from scratch, we can fine-tune such vision-language-action (VLA) models to obtain robust, generalizable policies for visuomotor control. Yet, widespread adoption of VLAs for robotics has been challenging as 1) existing VLAs are largely closed and inaccessible to the public, and 2) prior work fails to explore methods for efficiently fine-tuning VLAs for new tasks, a key component for adoption. Addressing these challenges, we introduce OpenVLA, a 7B-parameter open-source VLA trained on a diverse collection of 970k real-world robot demonstrations. OpenVLA builds on a Llama 2 language model combined with a visual encoder that fuses pretrained features from DINOv2 and SigLIP. As a product of the added data diversity and new model components, OpenVLA demonstrates strong results for generalist manipulation, outperforming closed models such as RT-2-X (55B) by 16.5% in absolute task success rate across 29 tasks and multiple robot embodiments, with 7x fewer parameters. We further show that we can effectively fine-tune OpenVLA for new settings, with especially strong generalization results in multi-task environments involving multiple objects and strong language grounding abilities, and outperform expressive from-scratch imitation learning methods such as Diffusion Policy by 20.4%. We also explore compute efficiency; as a separate contribution, we show that OpenVLA can be fine-tuned on consumer GPUs via modern low-rank adaptation methods and served efficiently via quantization without a hit to downstream success rate. Finally, we release model checkpoints, fine-tuning notebooks, and our PyTorch codebase with built-in support for training VLAs at scale on Open X-Embodiment datasets.",
  "published": "2024-06-13",
  "updated": "2024-09-05",
  "year": "2024",
  "authors": [
   "Moo Jin Kim",
   "Karl Pertsch",
   "Siddharth Karamcheti",
   "Ted Xiao",
   "Ashwin Balakrishna",
   "Suraj Nair",
   "Rafael Rafailov",
   "Ethan Foster",
   "Grace Lam",
   "Pannag Sanketi",
   "Quan Vuong",
   "Thomas Kollar",
   "Benjamin Burchfiel",
   "Russ Tedrake",
   "Dorsa Sadigh",
   "Sergey Levine",
   "Percy Liang",
   "Chelsea Finn"
  ],
  "author_count": 18,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 3113,
  "influential_citations": 484,
  "tldr": "OpenVLA, a 7B-parameter open-source VLA trained on a diverse collection of 970k real-world robot demonstrations, is introduced and it is shown that it can effectively fine-tune OpenVLA for new settings, with especially strong generalization results in multi-task environments involving multiple objects and strong language grounding abilities.",
  "doi": "10.48550/arXiv.2406.09246",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Moo Jin Kim",
    "id": "2159987907",
    "h_index": 10,
    "papers": 10
   },
   {
    "name": "Karl Pertsch",
    "id": "31719101",
    "h_index": 32,
    "papers": 56
   },
   {
    "name": "Siddharth Karamcheti",
    "id": "10737060",
    "h_index": 21,
    "papers": 38
   },
   {
    "name": "Ted Xiao",
    "id": "9961095",
    "h_index": 33,
    "papers": 45
   },
   {
    "name": "A. Balakrishna",
    "id": "3117588",
    "h_index": 24,
    "papers": 65
   },
   {
    "name": "Suraj Nair",
    "id": "2286638954",
    "h_index": 12,
    "papers": 19
   },
   {
    "name": "Rafael Rafailov",
    "id": "102801230",
    "h_index": 25,
    "papers": 44
   },
   {
    "name": "E. Foster",
    "id": "2306251665",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Grace Lam",
    "id": "2330190570",
    "h_index": 8,
    "papers": 9
   },
   {
    "name": "Pannag R. Sanketi",
    "id": "2840758",
    "h_index": 22,
    "papers": 44
   },
   {
    "name": "Quan Vuong",
    "id": "2288210223",
    "h_index": 19,
    "papers": 29
   },
   {
    "name": "Thomas Kollar",
    "id": "2283843631",
    "h_index": 10,
    "papers": 17
   },
   {
    "name": "Benjamin Burchfiel",
    "id": "2319412766",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Russ Tedrake",
    "id": "1726802",
    "h_index": 76,
    "papers": 270
   },
   {
    "name": "Dorsa Sadigh",
    "id": "1779671",
    "h_index": 69,
    "papers": 226
   },
   {
    "name": "Sergey Levine",
    "id": "2249615151",
    "h_index": 32,
    "papers": 53
   },
   {
    "name": "Percy Liang",
    "id": "2283843948",
    "h_index": 6,
    "papers": 6
   },
   {
    "name": "Chelsea Finn",
    "id": "2284774407",
    "h_index": 11,
    "papers": 13
   }
  ],
  "comment": "Website: https://openvla.github.io/",
  "topics": [
   "vla",
   "imitation-diffusion"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2406.09246v3",
  "pdf_url": "https://arxiv.org/pdf/2406.09246v3",
  "html_url": "https://arxiv.org/html/2406.09246v3",
  "code_url": "https://openvla.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.5
 },
 {
  "id": "2406.08858",
  "slug": "omnih2o-universal-and-dexterous-human-to-humanoid-whole-body-teleopera",
  "title": "OmniH2O: Universal and Dexterous Human-to-Humanoid Whole-Body Teleoperation and Learning",
  "abstract": "We present OmniH2O (Omni Human-to-Humanoid), a learning-based system for whole-body humanoid teleoperation and autonomy. Using kinematic pose as a universal control interface, OmniH2O enables various ways for a human to control a full-sized humanoid with dexterous hands, including using real-time teleoperation through VR headset, verbal instruction, and RGB camera. OmniH2O also enables full autonomy by learning from teleoperated demonstrations or integrating with frontier models such as GPT-4. OmniH2O demonstrates versatility and dexterity in various real-world whole-body tasks through teleoperation or autonomy, such as playing multiple sports, moving and manipulating objects, and interacting with humans. We develop an RL-based sim-to-real pipeline, which involves large-scale retargeting and augmentation of human motion datasets, learning a real-world deployable policy with sparse sensor input by imitating a privileged teacher policy, and reward designs to enhance robustness and stability. We release the first humanoid whole-body control dataset, OmniH2O-6, containing six everyday tasks, and demonstrate humanoid whole-body skill learning from teleoperated datasets.",
  "published": "2024-06-13",
  "updated": "2024-06-13",
  "year": "2024",
  "authors": [
   "Tairan He",
   "Zhengyi Luo",
   "Xialin He",
   "Wenli Xiao",
   "Chong Zhang",
   "Weinan Zhang",
   "Kris Kitani",
   "Changliu Liu",
   "Guanya Shi"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO",
   "cs.CV",
   "cs.LG",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 344,
  "influential_citations": 37,
  "tldr": "An RL-based sim-to-real pipeline, which involves large-scale retargeting and augmentation of human motion datasets, learning a real-world deployable policy with sparse sensor input by imitating a privileged teacher policy, and reward designs to enhance robustness and stability is developed.",
  "doi": "10.48550/arXiv.2406.08858",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Tairan He",
    "id": "2055132189",
    "h_index": 20,
    "papers": 29
   },
   {
    "name": "Zhengyi Luo",
    "id": "2566332",
    "h_index": 15,
    "papers": 20
   },
   {
    "name": "Xialin He",
    "id": "2195895437",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Wenli Xiao",
    "id": "2147212066",
    "h_index": 15,
    "papers": 32
   },
   {
    "name": "Chong Zhang",
    "id": "2290240944",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Weinan Zhang",
    "id": "2314883167",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Kris Kitani",
    "id": "2256989593",
    "h_index": 10,
    "papers": 20
   },
   {
    "name": "Changliu Liu",
    "id": "2282143156",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Guanya Shi",
    "id": "2249759531",
    "h_index": 20,
    "papers": 31
   }
  ],
  "comment": "Project page: https://omni.human2humanoid.com/",
  "topics": [
   "dexterous-manipulation",
   "humanoids",
   "sim2real",
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2406.08858v1",
  "pdf_url": "https://arxiv.org/pdf/2406.08858v1",
  "html_url": "https://arxiv.org/html/2406.08858v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.04
 },
 {
  "id": "2406.07550",
  "slug": "an-image-is-worth-32-tokens-for-reconstruction-and-generation",
  "title": "An Image is Worth 32 Tokens for Reconstruction and Generation",
  "abstract": "Recent advancements in generative models have highlighted the crucial role of image tokenization in the efficient synthesis of high-resolution images. Tokenization, which transforms images into latent representations, reduces computational demands compared to directly processing pixels and enhances the effectiveness and efficiency of the generation process. Prior methods, such as VQGAN, typically utilize 2D latent grids with fixed downsampling factors. However, these 2D tokenizations face challenges in managing the inherent redundancies present in images, where adjacent regions frequently display similarities. To overcome this issue, we introduce Transformer-based 1-Dimensional Tokenizer (TiTok), an innovative approach that tokenizes images into 1D latent sequences. TiTok provides a more compact latent representation, yielding substantially more efficient and effective representations than conventional techniques. For example, a 256 x 256 x 3 image can be reduced to just 32 discrete tokens, a significant reduction from the 256 or 1024 tokens obtained by prior methods. Despite its compact nature, TiTok achieves competitive performance to state-of-the-art approaches. Specifically, using the same generator framework, TiTok attains 1.97 gFID, outperforming MaskGIT baseline significantly by 4.21 at ImageNet 256 x 256 benchmark. The advantages of TiTok become even more significant when it comes to higher resolution. At ImageNet 512 x 512 benchmark, TiTok not only outperforms state-of-the-art diffusion model DiT-XL/2 (gFID 2.74 vs. 3.04), but also reduces the image tokens by 64x, leading to 410x faster generation process. Our best-performing variant can significantly surpasses DiT-XL/2 (gFID 2.13 vs. 3.04) while still generating high-quality samples 74x faster.",
  "published": "2024-06-11",
  "updated": "2024-06-11",
  "year": "2024",
  "authors": [
   "Qihang Yu",
   "Mark Weber",
   "Xueqing Deng",
   "Xiaohui Shen",
   "Daniel Cremers",
   "Liang-Chieh Chen"
  ],
  "author_count": 6,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 312,
  "influential_citations": 53,
  "tldr": "Transformer-based 1-Dimensional Tokenizer (TiTok), an innovative approach that tokenizes images into 1D latent sequences, provides a more compact latent representation, yielding substantially more efficient and effective representations than conventional techniques.",
  "doi": "10.48550/arXiv.2406.07550",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Qihang Yu",
    "id": "2304633845",
    "h_index": 10,
    "papers": 19
   },
   {
    "name": "Mark Weber",
    "id": "2110605521",
    "h_index": 8,
    "papers": 10
   },
   {
    "name": "Xueqing Deng",
    "id": "2269123207",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "Xiaohui Shen",
    "id": "2266472250",
    "h_index": 11,
    "papers": 18
   },
   {
    "name": "Daniel Cremers",
    "id": "2311880050",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Liang-Chieh Chen",
    "id": "2266697544",
    "h_index": 13,
    "papers": 26
   }
  ],
  "comment": "A compact 1D Image Tokenization method, leading to SOTA generation performance while being substantially faster. Project page at https://yucornetto.github.io/projects/titok.html",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2406.07550v1",
  "pdf_url": "https://arxiv.org/pdf/2406.07550v1",
  "html_url": "https://arxiv.org/html/2406.07550v1",
  "code_url": "https://yucornetto.github.io/projects/titok.html",
  "presented": [
   {
    "club": "saturday-robotics",
    "club_name": "Saturday Robotics",
    "session_number": 9,
    "session_title": "Robotics & World Models Reading Club 09: CVPR Warm-up & Founders Spotlight \u2014 DeltaWorld + VisuoTactile Dexterous Hands | San Francisco 0523",
    "date_text": "Saturday, May 23, 2026 | 2:00 PM \u2013 5:00 PM",
    "city": "San Francisco, CA",
    "url": "https://lu.ma/wooiz0bf",
    "listed_as": "A Frame is Worth One Token: Efficient Generative World Modeling with Delta Tokens"
   }
  ],
  "club_note": "Keynote 2 by by Arjun Subramaniam (Factory Intelligence)",
  "featured": true,
  "signal": 8.0
 },
 {
  "id": "2406.07539",
  "slug": "baku-an-efficient-transformer-for-multi-task-policy-learning",
  "title": "BAKU: An Efficient Transformer for Multi-Task Policy Learning",
  "abstract": "Training generalist agents capable of solving diverse tasks is challenging, often requiring large datasets of expert demonstrations. This is particularly problematic in robotics, where each data point requires physical execution of actions in the real world. Thus, there is a pressing need for architectures that can effectively leverage the available training data. In this work, we present BAKU, a simple transformer architecture that enables efficient learning of multi-task robot policies. BAKU builds upon recent advancements in offline imitation learning and meticulously combines observation trunks, action chunking, multi-sensory observations, and action heads to substantially improve upon prior work. Our experiments on 129 simulated tasks across LIBERO, Meta-World suite, and the Deepmind Control suite exhibit an overall 18% absolute improvement over RT-1 and MT-ACT, with a 36% improvement on the harder LIBERO benchmark. On 30 real-world manipulation tasks, given an average of just 17 demonstrations per task, BAKU achieves a 91% success rate. Videos of the robot are best viewed at https://baku-robot.github.io/.",
  "published": "2024-06-11",
  "updated": "2024-07-16",
  "year": "2024",
  "authors": [
   "Siddhant Haldar",
   "Zhuoran Peng",
   "Lerrel Pinto"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 110,
  "influential_citations": 16,
  "tldr": "BAKU is presented, a simple transformer architecture that enables efficient learning of multi-task robot policies and meticulously combines observation trunks, action chunking, multi-sensory observations, and action heads to substantially improve upon prior work.",
  "doi": "10.48550/arXiv.2406.07539",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Siddhant Haldar",
    "id": "51445278",
    "h_index": 13,
    "papers": 21
   },
   {
    "name": "Zhuoran Peng",
    "id": "2291037143",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Lerrel Pinto",
    "id": "2253567347",
    "h_index": 15,
    "papers": 25
   }
  ],
  "comment": "",
  "topics": [
   "imitation-diffusion"
  ],
  "orgs": [
   "Google DeepMind"
  ],
  "abs_url": "https://arxiv.org/abs/2406.07539v2",
  "pdf_url": "https://arxiv.org/pdf/2406.07539v2",
  "html_url": "https://arxiv.org/html/2406.07539v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.05
 },
 {
  "id": "2406.07472",
  "slug": "4real-towards-photorealistic-4d-scene-generation-via-video-diffusion-m",
  "title": "4Real: Towards Photorealistic 4D Scene Generation via Video Diffusion Models",
  "abstract": "Existing dynamic scene generation methods mostly rely on distilling knowledge from pre-trained 3D generative models, which are typically fine-tuned on synthetic object datasets. As a result, the generated scenes are often object-centric and lack photorealism. To address these limitations, we introduce a novel pipeline designed for photorealistic text-to-4D scene generation, discarding the dependency on multi-view generative models and instead fully utilizing video generative models trained on diverse real-world datasets. Our method begins by generating a reference video using the video generation model. We then learn the canonical 3D representation of the video using a freeze-time video, delicately generated from the reference video. To handle inconsistencies in the freeze-time video, we jointly learn a per-frame deformation to model these imperfections. We then learn the temporal deformation based on the canonical representation to capture dynamic interactions in the reference video. The pipeline facilitates the generation of dynamic scenes with enhanced photorealism and structural integrity, viewable from multiple perspectives, thereby setting a new standard in 4D scene generation.",
  "published": "2024-06-11",
  "updated": "2024-11-20",
  "year": "2024",
  "authors": [
   "Heng Yu",
   "Chaoyang Wang",
   "Peiye Zhuang",
   "Willi Menapace",
   "Aliaksandr Siarohin",
   "Junli Cao",
   "Laszlo A Jeni",
   "Sergey Tulyakov",
   "Hsin-Ying Lee"
  ],
  "author_count": 9,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 73,
  "influential_citations": 4,
  "tldr": "A novel pipeline designed for photorealistic text-to-4D scene generation, discarding the dependency on multi-view generative models and instead fully utilizing video generative models trained on diverse real-world datasets, thereby setting a new standard in 4D scene generation.",
  "doi": "10.48550/arXiv.2406.07472",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Heng Yu",
    "id": "2261248154",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Chaoyang Wang",
    "id": "50097023",
    "h_index": 20,
    "papers": 82
   },
   {
    "name": "Peiye Zhuang",
    "id": "2274105478",
    "h_index": 9,
    "papers": 11
   },
   {
    "name": "W. Menapace",
    "id": "1698103472",
    "h_index": 22,
    "papers": 53
   },
   {
    "name": "Aliaksandr Siarohin",
    "id": "10753214",
    "h_index": 31,
    "papers": 76
   },
   {
    "name": "Junli Cao",
    "id": "2109829649",
    "h_index": 10,
    "papers": 15
   },
   {
    "name": "L\u00e1szl\u00f3 A. Jeni",
    "id": "2331345",
    "h_index": 30,
    "papers": 90
   },
   {
    "name": "S. Tulyakov",
    "id": "145582202",
    "h_index": 48,
    "papers": 171
   },
   {
    "name": "Hsin-Ying Lee",
    "id": "2257364073",
    "h_index": 13,
    "papers": 20
   }
  ],
  "comment": "NeurIPS 2024",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2406.07472v2",
  "pdf_url": "https://arxiv.org/pdf/2406.07472v2",
  "html_url": "https://arxiv.org/html/2406.07472v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.37
 },
 {
  "id": "2406.06424",
  "slug": "margin-aware-preference-optimization-for-aligning-diffusion-models-wit",
  "title": "Margin-aware Preference Optimization for Aligning Diffusion Models without Reference",
  "abstract": "Modern preference alignment methods, such as DPO, rely on divergence regularization to a reference model for training stability-but this creates a fundamental problem we call \"reference mismatch.\" In this paper, we investigate the negative impacts of reference mismatch in aligning text-to-image (T2I) diffusion models, showing that larger reference mismatch hinders effective adaptation given the same amount of data, e.g., as when learning new artistic styles, or personalizing to specific objects. We demonstrate this phenomenon across text-to-image (T2I) diffusion models and introduce margin-aware preference optimization (MaPO), a reference-agnostic approach that breaks free from this constraint. By directly optimizing the likelihood margin between preferred and dispreferred outputs under the Bradley-Terry model without anchoring to a reference, MaPO transforms diverse T2I tasks into unified pairwise preference optimization. We validate MaPO's versatility across five challenging domains: (1) safe generation, (2) style adaptation, (3) cultural representation, (4) personalization, and (5) general preference alignment. Our results reveal that MaPO's advantage grows dramatically with reference mismatch severity, outperforming both DPO and specialized methods like DreamBooth while reducing training time by 15%. MaPO thus emerges as a versatile and memory-efficient method for generic T2I adaptation tasks.",
  "published": "2024-06-10",
  "updated": "2025-12-03",
  "year": "2024",
  "authors": [
   "Jiwoo Hong",
   "Sayak Paul",
   "Noah Lee",
   "Kashif Rasul",
   "James Thorne",
   "Jongheon Jeong"
  ],
  "author_count": 6,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "AAAI 2026",
  "venue_source": "arxiv-comment",
  "citations": 54,
  "influential_citations": 4,
  "tldr": "MaPO emerges as a versatile and memory-efficient method for generic T2I adaptation tasks, outperforming both DPO and specialized methods like DreamBooth while reducing training time by 15%.",
  "doi": "10.48550/arXiv.2406.06424",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jiwoo Hong",
    "id": "2290955335",
    "h_index": 11,
    "papers": 14
   },
   {
    "name": "Sayak Paul",
    "id": "2279263703",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Noah Lee",
    "id": "2291076200",
    "h_index": 7,
    "papers": 17
   },
   {
    "name": "Kashif Rasul",
    "id": "4565995",
    "h_index": 17,
    "papers": 40
   },
   {
    "name": "James Thorne",
    "id": "2290905396",
    "h_index": 6,
    "papers": 6
   },
   {
    "name": "Jongheon Jeong",
    "id": "2358428630",
    "h_index": 3,
    "papers": 7
   }
  ],
  "comment": "Accepted to AAAI 2026 Main Technical Track",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2406.06424v2",
  "pdf_url": "https://arxiv.org/pdf/2406.06424v2",
  "html_url": "https://arxiv.org/html/2406.06424v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.24
 },
 {
  "id": "2406.04325",
  "slug": "sharegpt4video-improving-video-understanding-and-generation-with-bette",
  "title": "ShareGPT4Video: Improving Video Understanding and Generation with Better Captions",
  "abstract": "We present the ShareGPT4Video series, aiming to facilitate the video understanding of large video-language models (LVLMs) and the video generation of text-to-video models (T2VMs) via dense and precise captions. The series comprises: 1) ShareGPT4Video, 40K GPT4V annotated dense captions of videos with various lengths and sources, developed through carefully designed data filtering and annotating strategy. 2) ShareCaptioner-Video, an efficient and capable captioning model for arbitrary videos, with 4.8M high-quality aesthetic videos annotated by it. 3) ShareGPT4Video-8B, a simple yet superb LVLM that reached SOTA performance on three advancing video benchmarks. To achieve this, taking aside the non-scalable costly human annotators, we find using GPT4V to caption video with a naive multi-frame or frame-concatenation input strategy leads to less detailed and sometimes temporal-confused results. We argue the challenge of designing a high-quality video captioning strategy lies in three aspects: 1) Inter-frame precise temporal change understanding. 2) Intra-frame detailed content description. 3) Frame-number scalability for arbitrary-length videos. To this end, we meticulously designed a differential video captioning strategy, which is stable, scalable, and efficient for generating captions for videos with arbitrary resolution, aspect ratios, and length. Based on it, we construct ShareGPT4Video, which contains 40K high-quality videos spanning a wide range of categories, and the resulting captions encompass rich world knowledge, object attributes, camera movements, and crucially, detailed and precise temporal descriptions of events. Based on ShareGPT4Video, we further develop ShareCaptioner-Video, a superior captioner capable of efficiently generating high-quality captions for arbitrary videos...",
  "published": "2024-06-06",
  "updated": "2024-06-06",
  "year": "2024",
  "authors": [
   "Lin Chen",
   "Xilin Wei",
   "Jinsong Li",
   "Xiaoyi Dong",
   "Pan Zhang",
   "Yuhang Zang",
   "Zehui Chen",
   "Haodong Duan",
   "Bin Lin",
   "Zhenyu Tang",
   "Li Yuan",
   "Yu Qiao",
   "Dahua Lin",
   "Feng Zhao",
   "Jiaqi Wang"
  ],
  "author_count": 15,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 473,
  "influential_citations": 63,
  "tldr": "The ShareGPT4Video series, aiming to facilitate the video understanding of large video-language models (LVLMs) and the video generation of text-to-video models (T2VMs) via dense and precise captions via dense and precise captions, is presented.",
  "doi": "10.48550/arXiv.2406.04325",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Lin Chen",
    "id": "2267778503",
    "h_index": 9,
    "papers": 24
   },
   {
    "name": "Xilin Wei",
    "id": "2281913882",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Jinsong Li",
    "id": "2267504431",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "Xiao-wen Dong",
    "id": "2118187561",
    "h_index": 33,
    "papers": 77
   },
   {
    "name": "Pan Zhang",
    "id": "2271462894",
    "h_index": 23,
    "papers": 43
   },
   {
    "name": "Yuhang Zang",
    "id": "12862495",
    "h_index": 37,
    "papers": 110
   },
   {
    "name": "Zehui Chen",
    "id": "2293554731",
    "h_index": 11,
    "papers": 31
   },
   {
    "name": "Haodong Duan",
    "id": "31463937",
    "h_index": 38,
    "papers": 77
   },
   {
    "name": "Bin Lin",
    "id": "2254329478",
    "h_index": 12,
    "papers": 16
   },
   {
    "name": "Zhenyu Tang",
    "id": "2275126715",
    "h_index": 11,
    "papers": 17
   },
   {
    "name": "Li Yuan",
    "id": "2280992738",
    "h_index": 8,
    "papers": 10
   },
   {
    "name": "Yu Qiao",
    "id": "2270064219",
    "h_index": 9,
    "papers": 11
   },
   {
    "name": "Dahua Lin",
    "id": "2237734015",
    "h_index": 31,
    "papers": 86
   },
   {
    "name": "Feng Zhao",
    "id": "2295588235",
    "h_index": 9,
    "papers": 24
   },
   {
    "name": "Jiaqi Wang",
    "id": "2267494294",
    "h_index": 31,
    "papers": 76
   }
  ],
  "comment": "Project Page: https://sharegpt4video.github.io/",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2406.04325v1",
  "pdf_url": "https://arxiv.org/pdf/2406.04325v1",
  "html_url": "https://arxiv.org/html/2406.04325v1",
  "code_url": "https://sharegpt4video.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.18
 },
 {
  "id": "2406.00439",
  "slug": "learning-manipulation-by-predicting-interaction",
  "title": "Learning Manipulation by Predicting Interaction",
  "abstract": "Representation learning approaches for robotic manipulation have boomed in recent years. Due to the scarcity of in-domain robot data, prevailing methodologies tend to leverage large-scale human video datasets to extract generalizable features for visuomotor policy learning. Despite the progress achieved, prior endeavors disregard the interactive dynamics that capture behavior patterns and physical interaction during the manipulation process, resulting in an inadequate understanding of the relationship between objects and the environment. To this end, we propose a general pre-training pipeline that learns Manipulation by Predicting the Interaction (MPI) and enhances the visual representation.Given a pair of keyframes representing the initial and final states, along with language instructions, our algorithm predicts the transition frame and detects the interaction object, respectively. These two learning objectives achieve superior comprehension towards \"how-to-interact\" and \"where-to-interact\". We conduct a comprehensive evaluation of several challenging robotic tasks.The experimental results demonstrate that MPI exhibits remarkable improvement by 10% to 64% compared with previous state-of-the-art in real-world robot platforms as well as simulation environments. Code and checkpoints are publicly shared at https://github.com/OpenDriveLab/MPI.",
  "published": "2024-06-01",
  "updated": "2024-06-01",
  "year": "2024",
  "authors": [
   "Jia Zeng",
   "Qingwen Bu",
   "Bangjun Wang",
   "Wenke Xia",
   "Li Chen",
   "Hao Dong",
   "Haoming Song",
   "Dong Wang",
   "Di Hu",
   "Ping Luo",
   "Heming Cui",
   "Bin Zhao",
   "Xuelong Li",
   "Yu Qiao",
   "Hongyang Li"
  ],
  "author_count": 15,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "RSS",
  "venue_source": "semantic-scholar",
  "citations": 51,
  "influential_citations": 1,
  "tldr": "A general pre-training pipeline that learns Manipulation by Predicting the Interaction (MPI) and enhances the visual representation and achieves superior comprehension towards \"how-to-interact\" and \"where-to-interact\".",
  "doi": "10.48550/arXiv.2406.00439",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jia Zeng",
    "id": "2290180535",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Qingwen Bu",
    "id": "2290184536",
    "h_index": 13,
    "papers": 19
   },
   {
    "name": "Bangjun Wang",
    "id": "2262615613",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Wenke Xia",
    "id": "2201319923",
    "h_index": 8,
    "papers": 16
   },
   {
    "name": "Li Chen",
    "id": "2254272547",
    "h_index": 8,
    "papers": 13
   },
   {
    "name": "Hao Dong",
    "id": "2305454696",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Haoming Song",
    "id": "2304550745",
    "h_index": 9,
    "papers": 19
   },
   {
    "name": "Dong Wang",
    "id": "2152692487",
    "h_index": 15,
    "papers": 33
   },
   {
    "name": "Di Hu",
    "id": "2265546488",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Ping Luo",
    "id": "2262515628",
    "h_index": 14,
    "papers": 21
   },
   {
    "name": "Heming Cui",
    "id": "2275279817",
    "h_index": 11,
    "papers": 25
   },
   {
    "name": "Bin Zhao",
    "id": "2256773314",
    "h_index": 16,
    "papers": 52
   },
   {
    "name": "Xuelong Li",
    "id": "2192821449",
    "h_index": 21,
    "papers": 62
   },
   {
    "name": "Yu Qiao",
    "id": "2290187706",
    "h_index": 6,
    "papers": 6
   },
   {
    "name": "Hongyang Li",
    "id": "2290243003",
    "h_index": 11,
    "papers": 19
   }
  ],
  "comment": "Accepted to RSS 2024. Project page: https://github.com/OpenDriveLab/MPI",
  "topics": [
   "egocentric-data",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2406.00439v1",
  "pdf_url": "https://arxiv.org/pdf/2406.00439v1",
  "html_url": "https://arxiv.org/html/2406.00439v1",
  "code_url": "https://github.com/OpenDriveLab/MPI",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.22
 },
 {
  "id": "2405.18132",
  "slug": "eg4d-explicit-generation-of-4d-object-without-score-distillation",
  "title": "EG4D: Explicit Generation of 4D Object without Score Distillation",
  "abstract": "In recent years, the increasing demand for dynamic 3D assets in design and gaming applications has given rise to powerful generative pipelines capable of synthesizing high-quality 4D objects. Previous methods generally rely on score distillation sampling (SDS) algorithm to infer the unseen views and motion of 4D objects, thus leading to unsatisfactory results with defects like over-saturation and Janus problem. Therefore, inspired by recent progress of video diffusion models, we propose to optimize a 4D representation by explicitly generating multi-view videos from one input image. However, it is far from trivial to handle practical challenges faced by such a pipeline, including dramatic temporal inconsistency, inter-frame geometry and texture diversity, and semantic defects brought by video generation results. To address these issues, we propose DG4D, a novel multi-stage framework that generates high-quality and consistent 4D assets without score distillation. Specifically, collaborative techniques and solutions are developed, including an attention injection strategy to synthesize temporal-consistent multi-view videos, a robust and efficient dynamic reconstruction method based on Gaussian Splatting, and a refinement stage with diffusion prior for semantic restoration. The qualitative results and user preference study demonstrate that our framework outperforms the baselines in generation quality by a considerable margin. Code will be released at \\url{https://github.com/jasongzy/EG4D}.",
  "published": "2024-05-28",
  "updated": "2024-05-28",
  "year": "2024",
  "authors": [
   "Qi Sun",
   "Zhiyang Guo",
   "Ziyu Wan",
   "Jing Nathan Yan",
   "Shengming Yin",
   "Wengang Zhou",
   "Jing Liao",
   "Houqiang Li"
  ],
  "author_count": 8,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 50,
  "influential_citations": 3,
  "tldr": "D4D, a novel multi-stage framework that generates high-quality and consistent 4D assets without score distillation, is proposed, including an attention injection strategy to synthesize temporal-consistent multi-view videos, a robust and efficient dynamic reconstruction method based on Gaussian Splatting, and a refinement stage with diffusion prior for semantic restoration.",
  "doi": "10.48550/arXiv.2405.18132",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Qi Sun",
    "id": "2279540699",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Zhiyang Guo",
    "id": "2286026746",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Ziyu Wan",
    "id": "66892346",
    "h_index": 18,
    "papers": 31
   },
   {
    "name": "J. Yan",
    "id": "2408722939",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Shengming Yin",
    "id": "2376421876",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Wen-gang Zhou",
    "id": "38272296",
    "h_index": 71,
    "papers": 389
   },
   {
    "name": "Jing Liao",
    "id": "2273664789",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Houqiang Li",
    "id": "2210048071",
    "h_index": 20,
    "papers": 119
   }
  ],
  "comment": "",
  "topics": [
   "spatial-3d",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2405.18132v1",
  "pdf_url": "https://arxiv.org/pdf/2405.18132v1",
  "html_url": "https://arxiv.org/html/2405.18132v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.21
 },
 {
  "id": "2405.17421",
  "slug": "mosca-dynamic-gaussian-fusion-from-casual-videos-via-4d-motion-scaffol",
  "title": "MoSca: Dynamic Gaussian Fusion from Casual Videos via 4D Motion Scaffolds",
  "abstract": "We introduce 4D Motion Scaffolds (MoSca), a modern 4D reconstruction system designed to reconstruct and synthesize novel views of dynamic scenes from monocular videos captured casually in the wild. To address such a challenging and ill-posed inverse problem, we leverage prior knowledge from foundational vision models and lift the video data to a novel Motion Scaffold (MoSca) representation, which compactly and smoothly encodes the underlying motions/deformations. The scene geometry and appearance are then disentangled from the deformation field and are encoded by globally fusing the Gaussians anchored onto the MoSca and optimized via Gaussian Splatting. Additionally, camera focal length and poses can be solved using bundle adjustment without the need of any other pose estimation tools. Experiments demonstrate state-of-the-art performance on dynamic rendering benchmarks and its effectiveness on real videos.",
  "published": "2024-05-27",
  "updated": "2024-11-29",
  "year": "2024",
  "authors": [
   "Jiahui Lei",
   "Yijia Weng",
   "Adam Harley",
   "Leonidas Guibas",
   "Kostas Daniilidis"
  ],
  "author_count": 5,
  "categories": [
   "cs.CV",
   "cs.GR"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 189,
  "influential_citations": 33,
  "tldr": "",
  "doi": "10.1109/CVPR52734.2025.00578",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jiahui Lei",
    "id": "2052835670",
    "h_index": 13,
    "papers": 41
   },
   {
    "name": "Yijia Weng",
    "id": "1693059352",
    "h_index": 9,
    "papers": 15
   },
   {
    "name": "Adam W. Harley",
    "id": "2292198078",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Leonidas J. Guibas",
    "id": "2287942488",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Kostas Daniilidis",
    "id": "2065557091",
    "h_index": 25,
    "papers": 89
   }
  ],
  "comment": "project page: https://www.cis.upenn.edu/~leijh/projects/mosca code release: https://github.com/JiahuiLei/MoSca",
  "topics": [
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2405.17421v2",
  "pdf_url": "https://arxiv.org/pdf/2405.17421v2",
  "html_url": "https://arxiv.org/html/2405.17421v2",
  "code_url": "https://github.com/JiahuiLei/MoSca",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.78
 },
 {
  "id": "2405.16822",
  "slug": "vidu4d-single-generated-video-to-high-fidelity-4d-reconstruction-with",
  "title": "Vidu4D: Single Generated Video to High-Fidelity 4D Reconstruction with Dynamic Gaussian Surfels",
  "abstract": "Video generative models are receiving particular attention given their ability to generate realistic and imaginative frames. Besides, these models are also observed to exhibit strong 3D consistency, significantly enhancing their potential to act as world simulators. In this work, we present Vidu4D, a novel reconstruction model that excels in accurately reconstructing 4D (i.e., sequential 3D) representations from single generated videos, addressing challenges associated with non-rigidity and frame distortion. This capability is pivotal for creating high-fidelity virtual contents that maintain both spatial and temporal coherence. At the core of Vidu4D is our proposed Dynamic Gaussian Surfels (DGS) technique. DGS optimizes time-varying warping functions to transform Gaussian surfels (surface elements) from a static state to a dynamically warped state. This transformation enables a precise depiction of motion and deformation over time. To preserve the structural integrity of surface-aligned Gaussian surfels, we design the warped-state geometric regularization based on continuous warping fields for estimating normals. Additionally, we learn refinements on rotation and scaling parameters of Gaussian surfels, which greatly alleviates texture flickering during the warping process and enhances the capture of fine-grained appearance details. Vidu4D also contains a novel initialization state that provides a proper start for the warping fields in DGS. Equipping Vidu4D with an existing video generative model, the overall framework demonstrates high-fidelity text-to-4D generation in both appearance and geometry.",
  "published": "2024-05-27",
  "updated": "2024-05-27",
  "year": "2024",
  "authors": [
   "Yikai Wang",
   "Xinzhou Wang",
   "Zilong Chen",
   "Zhengyi Wang",
   "Fuchun Sun",
   "Jun Zhu"
  ],
  "author_count": 6,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 39,
  "influential_citations": 1,
  "tldr": "Vidu4D is presented, a novel reconstruction model that excels in accurately reconstructing 4D representations from single generated videos, addressing challenges associated with non-rigidity and frame distortion.",
  "doi": "10.48550/arXiv.2405.16822",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yikai Wang",
    "id": "2271890043",
    "h_index": 13,
    "papers": 20
   },
   {
    "name": "Xinzhou Wang",
    "id": "2196924058",
    "h_index": 11,
    "papers": 21
   },
   {
    "name": "Zilong Chen",
    "id": "2248512843",
    "h_index": 8,
    "papers": 14
   },
   {
    "name": "Zhengyi Wang",
    "id": "2272269919",
    "h_index": 11,
    "papers": 17
   },
   {
    "name": "Fuchun Sun",
    "id": "2271723773",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Jun Zhu",
    "id": "2265522339",
    "h_index": 12,
    "papers": 18
   }
  ],
  "comment": "Project page: https://vidu4d-dgs.github.io",
  "topics": [
   "sim2real"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2405.16822v1",
  "pdf_url": "https://arxiv.org/pdf/2405.16822v1",
  "html_url": "https://arxiv.org/html/2405.16822v1",
  "code_url": "https://vidu4d-dgs.github.io",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.1
 },
 {
  "id": "2405.16645",
  "slug": "diffusion4d-fast-spatial-temporal-consistent-4d-generation-via-video-d",
  "title": "Diffusion4D: Fast Spatial-temporal Consistent 4D Generation via Video Diffusion Models",
  "abstract": "The availability of large-scale multimodal datasets and advancements in diffusion models have significantly accelerated progress in 4D content generation. Most prior approaches rely on multiple image or video diffusion models, utilizing score distillation sampling for optimization or generating pseudo novel views for direct supervision. However, these methods are hindered by slow optimization speeds and multi-view inconsistency issues. Spatial and temporal consistency in 4D geometry has been extensively explored respectively in 3D-aware diffusion models and traditional monocular video diffusion models. Building on this foundation, we propose a strategy to migrate the temporal consistency in video diffusion models to the spatial-temporal consistency required for 4D generation. Specifically, we present a novel framework, \\textbf{Diffusion4D}, for efficient and scalable 4D content generation. Leveraging a meticulously curated dynamic 3D dataset, we develop a 4D-aware video diffusion model capable of synthesizing orbital views of dynamic 3D assets. To control the dynamic strength of these assets, we introduce a 3D-to-4D motion magnitude metric as guidance. Additionally, we propose a novel motion magnitude reconstruction loss and 3D-aware classifier-free guidance to refine the learning and generation of motion dynamics. After obtaining orbital views of the 4D asset, we perform explicit 4D construction with Gaussian splatting in a coarse-to-fine manner. The synthesized multi-view consistent 4D image set enables us to swiftly generate high-fidelity and diverse 4D assets within just several minutes. Extensive experiments demonstrate that our method surpasses prior state-of-the-art techniques in terms of generation efficiency and 4D geometry consistency across various prompt modalities.",
  "published": "2024-05-26",
  "updated": "2024-05-26",
  "year": "2024",
  "authors": [
   "Hanwen Liang",
   "Yuyang Yin",
   "Dejia Xu",
   "Hanxue Liang",
   "Zhangyang Wang",
   "Konstantinos N. Plataniotis",
   "Yao Zhao",
   "Yunchao Wei"
  ],
  "author_count": 8,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 114,
  "influential_citations": 9,
  "tldr": "A novel framework to migrate the temporal consistency in video diffusion models to the spatial-temporal consistency required for 4D generation is presented and surpasses prior state-of-the-art techniques in terms of generation efficiency and 4D geometry consistency across various prompt modalities.",
  "doi": "10.48550/arXiv.2405.16645",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hanwen Liang",
    "id": "2293415032",
    "h_index": 4,
    "papers": 18
   },
   {
    "name": "Yuyang Yin",
    "id": "2109472890",
    "h_index": 7,
    "papers": 16
   },
   {
    "name": "Dejia Xu",
    "id": "1575684088",
    "h_index": 26,
    "papers": 54
   },
   {
    "name": "Hanxue Liang",
    "id": "2152874039",
    "h_index": 9,
    "papers": 14
   },
   {
    "name": "Zhangyang Wang",
    "id": "2269758990",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Konstantinos N. Plataniotis",
    "id": "2281992361",
    "h_index": 8,
    "papers": 43
   },
   {
    "name": "Yao Zhao",
    "id": "2303409157",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Yunchao Wei",
    "id": "2238213573",
    "h_index": 20,
    "papers": 73
   }
  ],
  "comment": "Project page: https://vita-group.github.io/Diffusion4D",
  "topics": [
   "spatial-3d",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2405.16645v1",
  "pdf_url": "https://arxiv.org/pdf/2405.16645v1",
  "html_url": "https://arxiv.org/html/2405.16645v1",
  "code_url": "https://vita-group.github.io/Diffusion4D",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.56
 },
 {
  "id": "2405.12399",
  "slug": "diffusion-for-world-modeling-visual-details-matter-in-atari",
  "title": "Diffusion for World Modeling: Visual Details Matter in Atari",
  "abstract": "World models constitute a promising approach for training reinforcement learning agents in a safe and sample-efficient manner. Recent world models predominantly operate on sequences of discrete latent variables to model environment dynamics. However, this compression into a compact discrete representation may ignore visual details that are important for reinforcement learning. Concurrently, diffusion models have become a dominant approach for image generation, challenging well-established methods modeling discrete latents. Motivated by this paradigm shift, we introduce DIAMOND (DIffusion As a Model Of eNvironment Dreams), a reinforcement learning agent trained in a diffusion world model. We analyze the key design choices that are required to make diffusion suitable for world modeling, and demonstrate how improved visual details can lead to improved agent performance. DIAMOND achieves a mean human normalized score of 1.46 on the competitive Atari 100k benchmark; a new best for agents trained entirely within a world model. We further demonstrate that DIAMOND's diffusion world model can stand alone as an interactive neural game engine by training on static Counter-Strike: Global Offensive gameplay. To foster future research on diffusion for world modeling, we release our code, agents, videos and playable world models at https://diamond-wm.github.io.",
  "published": "2024-05-20",
  "updated": "2024-10-30",
  "year": "2024",
  "authors": [
   "Eloi Alonso",
   "Adam Jelley",
   "Vincent Micheli",
   "Anssi Kanervisto",
   "Amos Storkey",
   "Tim Pearce",
   "Fran\u00e7ois Fleuret"
  ],
  "author_count": 7,
  "categories": [
   "cs.LG",
   "cs.AI",
   "cs.CV"
  ],
  "primary_category": "cs.LG",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 307,
  "influential_citations": 41,
  "tldr": "This work introduces DIAMOND (DIffusion As a Model Of eNvironment Dreams), a reinforcement learning agent trained in a diffusion world model, and analyzes the key design choices that are required to make diffusion suitable for world modeling, and demonstrates how improved visual details can lead to improved agent performance.",
  "doi": "10.48550/arXiv.2405.12399",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Eloi Alonso",
    "id": "144370326",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "Adam Jelley",
    "id": "2297770590",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Vincent Micheli",
    "id": "1491750155",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "A. Kanervisto",
    "id": "3469155",
    "h_index": 18,
    "papers": 43
   },
   {
    "name": "A. Storkey",
    "id": "1728216",
    "h_index": 48,
    "papers": 290
   },
   {
    "name": "Tim Pearce",
    "id": "2329555537",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Franccois Fleuret",
    "id": "116272138",
    "h_index": 19,
    "papers": 45
   }
  ],
  "comment": "NeurIPS 2024 (Spotlight)",
  "topics": [
   "world-models",
   "rl-control",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2405.12399v2",
  "pdf_url": "https://arxiv.org/pdf/2405.12399v2",
  "html_url": "https://arxiv.org/html/2405.12399v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.99
 },
 {
  "id": "2405.12213",
  "slug": "octo-an-open-source-generalist-robot-policy",
  "title": "Octo: An Open-Source Generalist Robot Policy",
  "abstract": "Large policies pretrained on diverse robot datasets have the potential to transform robotic learning: instead of training new policies from scratch, such generalist robot policies may be finetuned with only a little in-domain data, yet generalize broadly. However, to be widely applicable across a range of robotic learning scenarios, environments, and tasks, such policies need to handle diverse sensors and action spaces, accommodate a variety of commonly used robotic platforms, and finetune readily and efficiently to new domains. In this work, we aim to lay the groundwork for developing open-source, widely applicable, generalist policies for robotic manipulation. As a first step, we introduce Octo, a large transformer-based policy trained on 800k trajectories from the Open X-Embodiment dataset, the largest robot manipulation dataset to date. It can be instructed via language commands or goal images and can be effectively finetuned to robot setups with new sensory inputs and action spaces within a few hours on standard consumer GPUs. In experiments across 9 robotic platforms, we demonstrate that Octo serves as a versatile policy initialization that can be effectively finetuned to new observation and action spaces. We also perform detailed ablations of design decisions for the Octo model, from architecture to training data, to guide future research on building generalist robot models.",
  "published": "2024-05-20",
  "updated": "2024-05-26",
  "year": "2024",
  "authors": [
   " Octo Model Team",
   "Dibya Ghosh",
   "Homer Walke",
   "Karl Pertsch",
   "Kevin Black",
   "Oier Mees",
   "Sudeep Dasari",
   "Joey Hejna",
   "Tobias Kreiman",
   "Charles Xu",
   "Jianlan Luo",
   "You Liang Tan",
   "Lawrence Yunliang Chen",
   "Pannag Sanketi",
   "Quan Vuong",
   "Ted Xiao",
   "Dorsa Sadigh",
   "Chelsea Finn",
   "Sergey Levine"
  ],
  "author_count": 19,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "RSS",
  "venue_source": "semantic-scholar",
  "citations": 1705,
  "influential_citations": 133,
  "tldr": "This work introduces Octo, a large transformer-based policy trained on 800k trajectories from the Open X-Embodiment dataset, the largest robot manipulation dataset to date, and demonstrates that Octo serves as a versatile policy initialization that can be effectively finetuned to new observation and action spaces.",
  "doi": "10.48550/arXiv.2405.12213",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "O. Team",
    "id": "2302331721",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Dibya Ghosh",
    "id": "8021910",
    "h_index": 18,
    "papers": 24
   },
   {
    "name": "H. Walke",
    "id": "2029241116",
    "h_index": 17,
    "papers": 23
   },
   {
    "name": "Karl Pertsch",
    "id": "31719101",
    "h_index": 32,
    "papers": 56
   },
   {
    "name": "Kevin Black",
    "id": "2258959388",
    "h_index": 13,
    "papers": 15
   },
   {
    "name": "Oier Mees",
    "id": "7264115",
    "h_index": 28,
    "papers": 45
   },
   {
    "name": "S. Dasari",
    "id": "36076404",
    "h_index": 23,
    "papers": 39
   },
   {
    "name": "Joey Hejna",
    "id": "2122700519",
    "h_index": 17,
    "papers": 27
   },
   {
    "name": "Tobias Kreiman",
    "id": "2286880491",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Charles Xu",
    "id": "2254150689",
    "h_index": 10,
    "papers": 14
   },
   {
    "name": "Jianlan Luo",
    "id": "2238220544",
    "h_index": 13,
    "papers": 21
   },
   {
    "name": "You Liang Tan",
    "id": "2281816408",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Pannag R. Sanketi",
    "id": "2840758",
    "h_index": 22,
    "papers": 44
   },
   {
    "name": "Quan Vuong",
    "id": "2288210223",
    "h_index": 19,
    "papers": 29
   },
   {
    "name": "Ted Xiao",
    "id": "9961095",
    "h_index": 33,
    "papers": 45
   },
   {
    "name": "Dorsa Sadigh",
    "id": "1779671",
    "h_index": 69,
    "papers": 226
   },
   {
    "name": "Chelsea Finn",
    "id": "2257346440",
    "h_index": 23,
    "papers": 32
   },
   {
    "name": "Sergey Levine",
    "id": "2249615151",
    "h_index": 32,
    "papers": 53
   }
  ],
  "comment": "Project website: https://octo-models.github.io",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2405.12213v2",
  "pdf_url": "https://arxiv.org/pdf/2405.12213v2",
  "html_url": "https://arxiv.org/html/2405.12213v2",
  "code_url": "https://octo-models.github.io",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.5
 },
 {
  "id": "2405.10369",
  "slug": "reinforcement-learning",
  "title": "Reinforcement learning",
  "abstract": "Observing celestial objects and advancing our scientific knowledge about them involves tedious planning, scheduling, data collection and data post-processing. Many of these operational aspects of astronomy are guided and executed by expert astronomers. Reinforcement learning is a mechanism where we (as humans and astronomers) can teach agents of artificial intelligence to perform some of these tedious tasks. In this paper, we will present a state of the art overview of reinforcement learning and how it can benefit astronomy.",
  "published": "2024-05-16",
  "updated": "2024-05-16",
  "year": "2024",
  "authors": [
   "Sarod Yatawatta"
  ],
  "author_count": 1,
  "categories": [
   "astro-ph.IM",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "astro-ph.IM",
  "venue": "Scholarpedia",
  "venue_source": "semantic-scholar",
  "citations": 3422,
  "influential_citations": 268,
  "tldr": "The discussion here considers a much more common learning condition where an agent has to learn to make decisions in the environment from simple feedback, where feedback is provided only after periods of actions in the form of reward or punishment.",
  "doi": "10.4249/scholarpedia.1448",
  "oa_pdf": "https://onlinelibrary.wiley.com/doi/pdfdirect/10.1002/9780470512517.ch6",
  "s2_authors": [
   {
    "name": "F. W\u00f6rg\u00f6tter",
    "id": "1714016",
    "h_index": 51,
    "papers": 493
   },
   {
    "name": "B. Porr",
    "id": "2728662",
    "h_index": 21,
    "papers": 133
   }
  ],
  "comment": "To appear, Astronomy & Computing",
  "topics": [
   "rl-control",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2405.10369v1",
  "pdf_url": "https://arxiv.org/pdf/2405.10369v1",
  "html_url": "https://arxiv.org/html/2405.10369v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "2405.10020",
  "slug": "natural-language-can-help-bridge-the-sim2real-gap",
  "title": "Natural Language Can Help Bridge the Sim2Real Gap",
  "abstract": "The main challenge in learning image-conditioned robotic policies is acquiring a visual representation conducive to low-level control. Due to the high dimensionality of the image space, learning a good visual representation requires a considerable amount of visual data. However, when learning in the real world, data is expensive. Sim2Real is a promising paradigm for overcoming data scarcity in the real-world target domain by using a simulator to collect large amounts of cheap data closely related to the target task. However, it is difficult to transfer an image-conditioned policy from sim to real when the domains are very visually dissimilar. To bridge the sim2real visual gap, we propose using natural language descriptions of images as a unifying signal across domains that captures the underlying task-relevant semantics. Our key insight is that if two image observations from different domains are labeled with similar language, the policy should predict similar action distributions for both images. We demonstrate that training the image encoder to predict the language description or the distance between descriptions of a sim or real image serves as a useful, data-efficient pretraining step that helps learn a domain-invariant image representation. We can then use this image encoder as the backbone of an IL policy trained simultaneously on a large amount of simulated and a handful of real demonstrations. Our approach outperforms widely used prior sim2real methods and strong vision-language pretraining baselines like CLIP and R3M by 25 to 40%. See additional videos and materials at https://robin-lab.cs.utexas.edu/lang4sim2real/.",
  "published": "2024-05-16",
  "updated": "2024-07-02",
  "year": "2024",
  "authors": [
   "Albert Yu",
   "Adeline Foote",
   "Raymond Mooney",
   "Roberto Mart\u00edn-Mart\u00edn"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.CL",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "RSS",
  "venue_source": "semantic-scholar",
  "citations": 30,
  "influential_citations": 0,
  "tldr": "This work demonstrates that training the image encoder to predict the language description or the distance between descriptions of a sim or real image serves as a useful, data-efficient pretraining step that helps learn a domain-invariant image representation.",
  "doi": "10.48550/arXiv.2405.10020",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Albert Yu",
    "id": "2055846675",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Adeline Foote",
    "id": "2301455153",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Raymond Mooney",
    "id": "2301455218",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Roberto Mart\u00edn-Mart\u00edn",
    "id": "2316638007",
    "h_index": 6,
    "papers": 12
   }
  ],
  "comment": "To appear in RSS 2024. Project website at https://robin-lab.cs.utexas.edu/lang4sim2real/",
  "topics": [
   "sim2real",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2405.10020v2",
  "pdf_url": "https://arxiv.org/pdf/2405.10020v2",
  "html_url": "https://arxiv.org/html/2405.10020v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.99
 },
 {
  "id": "2405.07391",
  "slug": "anyrotate-gravity-invariant-in-hand-object-rotation-with-sim-to-real-t",
  "title": "AnyRotate: Gravity-Invariant In-Hand Object Rotation with Sim-to-Real Touch",
  "abstract": "Human hands are capable of in-hand manipulation in the presence of different hand motions. For a robot hand, harnessing rich tactile information to achieve this level of dexterity still remains a significant challenge. In this paper, we present AnyRotate, a system for gravity-invariant multi-axis in-hand object rotation using dense featured sim-to-real touch. We tackle this problem by training a dense tactile policy in simulation and present a sim-to-real method for rich tactile sensing to achieve zero-shot policy transfer. Our formulation allows the training of a unified policy to rotate unseen objects about arbitrary rotation axes in any hand direction. In our experiments, we highlight the benefit of capturing detailed contact information when handling objects of varying properties. Interestingly, we found rich multi-fingered tactile sensing can detect unstable grasps and provide a reactive behavior that improves the robustness of the policy. The project website can be found at https://maxyang27896.github.io/anyrotate/.",
  "published": "2024-05-12",
  "updated": "2024-11-03",
  "year": "2024",
  "authors": [
   "Max Yang",
   "Chenghua Lu",
   "Alex Church",
   "Yijiong Lin",
   "Chris Ford",
   "Haoran Li",
   "Efi Psomopoulou",
   "David A. W. Barton",
   "Nathan F. Lepora"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 62,
  "influential_citations": 1,
  "tldr": "This paper presents AnyRotate, a system for gravity-invariant multi-axis in-hand object rotation using dense featured sim-to-real touch and finds rich multi-fingered tactile sensing can detect unstable grasps and provide a reactive behavior that improves the robustness of the policy.",
  "doi": "10.48550/arXiv.2405.07391",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Max Yang",
    "id": "2157685104",
    "h_index": 7,
    "papers": 12
   },
   {
    "name": "Chenghua Lu",
    "id": "2248223914",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Alex Church",
    "id": "40976774",
    "h_index": 9,
    "papers": 15
   },
   {
    "name": "Yijiong Lin",
    "id": "81983572",
    "h_index": 10,
    "papers": 21
   },
   {
    "name": "Christopher J. Ford",
    "id": "2173605714",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Haoran Li",
    "id": "2296388921",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Efi Psomopoulou",
    "id": "1910114",
    "h_index": 11,
    "papers": 36
   },
   {
    "name": "David A. W. Barton",
    "id": "2301157295",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "N. Lepora",
    "id": "2467565",
    "h_index": 37,
    "papers": 214
   }
  ],
  "comment": "Project website can be found at https://maxyang27896.github.io/anyrotate/",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "sim2real",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2405.07391v3",
  "pdf_url": "https://arxiv.org/pdf/2405.07391v3",
  "html_url": "https://arxiv.org/html/2405.07391v3",
  "code_url": "https://maxyang27896.github.io/anyrotate/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.3
 },
 {
  "id": "2405.05941",
  "slug": "evaluating-real-world-robot-manipulation-policies-in-simulation",
  "title": "Evaluating Real-World Robot Manipulation Policies in Simulation",
  "abstract": "The field of robotics has made significant advances towards generalist robot manipulation policies. However, real-world evaluation of such policies is not scalable and faces reproducibility challenges, which are likely to worsen as policies broaden the spectrum of tasks they can perform. We identify control and visual disparities between real and simulated environments as key challenges for reliable simulated evaluation and propose approaches for mitigating these gaps without needing to craft full-fidelity digital twins of real-world environments. We then employ these approaches to create SIMPLER, a collection of simulated environments for manipulation policy evaluation on common real robot setups. Through paired sim-and-real evaluations of manipulation policies, we demonstrate strong correlation between policy performance in SIMPLER environments and in the real world. Additionally, we find that SIMPLER evaluations accurately reflect real-world policy behavior modes such as sensitivity to various distribution shifts. We open-source all SIMPLER environments along with our workflow for creating new environments at https://simpler-env.github.io to facilitate research on general-purpose manipulation policies and simulated evaluation frameworks.",
  "published": "2024-05-09",
  "updated": "2024-05-09",
  "year": "2024",
  "authors": [
   "Xuanlin Li",
   "Kyle Hsu",
   "Jiayuan Gu",
   "Karl Pertsch",
   "Oier Mees",
   "Homer Rich Walke",
   "Chuyuan Fu",
   "Ishikaa Lunawat",
   "Isabel Sieh",
   "Sean Kirmani",
   "Sergey Levine",
   "Jiajun Wu",
   "Chelsea Finn",
   "Hao Su",
   "Quan Vuong",
   "Ted Xiao"
  ],
  "author_count": 16,
  "categories": [
   "cs.RO",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 498,
  "influential_citations": 97,
  "tldr": "This work identifies control and visual disparities between real and simulated environments as key challenges for reliable simulated evaluation and proposes approaches for mitigating these gaps without needing to craft full-fidelity digital twins of real-world environments.",
  "doi": "10.48550/arXiv.2405.05941",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xuanlin Li",
    "id": "2108263986",
    "h_index": 17,
    "papers": 21
   },
   {
    "name": "Kyle Hsu",
    "id": "2286719009",
    "h_index": 8,
    "papers": 15
   },
   {
    "name": "Jiayuan Gu",
    "id": "2256468903",
    "h_index": 12,
    "papers": 15
   },
   {
    "name": "Karl Pertsch",
    "id": "31719101",
    "h_index": 32,
    "papers": 56
   },
   {
    "name": "Oier Mees",
    "id": "7264115",
    "h_index": 28,
    "papers": 45
   },
   {
    "name": "H. Walke",
    "id": "2029241116",
    "h_index": 17,
    "papers": 23
   },
   {
    "name": "Chuyuan Fu",
    "id": "3430433",
    "h_index": 15,
    "papers": 21
   },
   {
    "name": "Ishikaa Lunawat",
    "id": "2169163717",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Isabel Sieh",
    "id": "2297991920",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Sean Kirmani",
    "id": "51881277",
    "h_index": 21,
    "papers": 29
   },
   {
    "name": "Sergey Levine",
    "id": "2249615151",
    "h_index": 32,
    "papers": 53
   },
   {
    "name": "Jiajun Wu",
    "id": "2326444500",
    "h_index": 35,
    "papers": 214
   },
   {
    "name": "Chelsea Finn",
    "id": "2257346440",
    "h_index": 23,
    "papers": 32
   },
   {
    "name": "Hao Su",
    "id": "2255041135",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Q. Vuong",
    "id": "144579461",
    "h_index": 23,
    "papers": 40
   },
   {
    "name": "Ted Xiao",
    "id": "9961095",
    "h_index": 33,
    "papers": 45
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2405.05941v1",
  "pdf_url": "https://arxiv.org/pdf/2405.05941v1",
  "html_url": "https://arxiv.org/html/2405.05941v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.2
 },
 {
  "id": "2405.04378",
  "slug": "splat-mover-multi-stage-open-vocabulary-robotic-manipulation-via-edita",
  "title": "Splat-MOVER: Multi-Stage, Open-Vocabulary Robotic Manipulation via Editable Gaussian Splatting",
  "abstract": "We present Splat-MOVER, a modular robotics stack for open-vocabulary robotic manipulation, which leverages the editability of Gaussian Splatting (GSplat) scene representations to enable multi-stage manipulation tasks. Splat-MOVER consists of: (i) ASK-Splat, a GSplat representation that distills semantic and grasp affordance features into the 3D scene. ASK-Splat enables geometric, semantic, and affordance understanding of 3D scenes, which is critical in many robotics tasks; (ii) SEE-Splat, a real-time scene-editing module using 3D semantic masking and infilling to visualize the motions of objects that result from robot interactions in the real-world. SEE-Splat creates a \"digital twin\" of the evolving environment throughout the manipulation task; and (iii) Grasp-Splat, a grasp generation module that uses ASK-Splat and SEE-Splat to propose affordance-aligned candidate grasps for open-world objects. ASK-Splat is trained in real-time from RGB images in a brief scanning phase prior to operation, while SEE-Splat and Grasp-Splat run in real-time during operation. We demonstrate the superior performance of Splat-MOVER in hardware experiments on a Kinova robot compared to two recent baselines in four single-stage, open-vocabulary manipulation tasks and in four multi-stage manipulation tasks, using the edited scene to reflect changes due to prior manipulation stages, which is not possible with existing baselines. Video demonstrations and the code for the project are available at https://splatmover.github.io.",
  "published": "2024-05-07",
  "updated": "2024-09-26",
  "year": "2024",
  "authors": [
   "Ola Shorinwa",
   "Johnathan Tucker",
   "Aliyah Smith",
   "Aiden Swann",
   "Timothy Chen",
   "Roya Firoozi",
   "Monroe Kennedy",
   "Mac Schwager"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 68,
  "influential_citations": 1,
  "tldr": "The superior performance of Splat-MOVER is demonstrated in hardware experiments on a Kinova robot compared to two recent baselines in four single-stage, open-vocabulary manipulation tasks and in four multi-stage manipulation tasks, using the edited scene to reflect changes due to prior manipulation stages, which is not possible with existing baselines.",
  "doi": "10.48550/arXiv.2405.04378",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "O. Shorinwa",
    "id": "116069035",
    "h_index": 14,
    "papers": 37
   },
   {
    "name": "Johnathan Tucker",
    "id": "2273927013",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Aliyah Smith",
    "id": "2300397203",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Aiden Swann",
    "id": "2291964162",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Timothy Chen",
    "id": "2300331757",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Roya Firoozi",
    "id": "1416657816",
    "h_index": 9,
    "papers": 15
   },
   {
    "name": "Monroe Kennedy",
    "id": "2243201425",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Mac Schwager",
    "id": "2243338895",
    "h_index": 11,
    "papers": 27
   }
  ],
  "comment": "https://splatmover.github.io",
  "topics": [
   "dexterous-manipulation",
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2405.04378v4",
  "pdf_url": "https://arxiv.org/pdf/2405.04378v4",
  "html_url": "https://arxiv.org/html/2405.04378v4",
  "code_url": "https://splatmover.github.io",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.34
 },
 {
  "id": "2405.03520",
  "slug": "is-sora-a-world-simulator-a-comprehensive-survey-on-general-world-mode",
  "title": "Is Sora a World Simulator? A Comprehensive Survey on General World Models and Beyond",
  "abstract": "General world models represent a crucial pathway toward achieving Artificial General Intelligence (AGI), serving as the cornerstone for various applications ranging from virtual environments to decision-making systems. Recently, the emergence of the Sora model has attained significant attention due to its remarkable simulation capabilities, which exhibits an incipient comprehension of physical laws. In this survey, we embark on a comprehensive exploration of the latest advancements in world models. Our analysis navigates through the forefront of generative methodologies in video generation, where world models stand as pivotal constructs facilitating the synthesis of highly realistic visual content. Additionally, we scrutinize the burgeoning field of autonomous-driving world models, meticulously delineating their indispensable role in reshaping transportation and urban mobility. Furthermore, we delve into the intricacies inherent in world models deployed within autonomous agents, shedding light on their profound significance in enabling intelligent interactions within dynamic environmental contexts. At last, we examine challenges and limitations of world models, and discuss their potential future directions. We hope this survey can serve as a foundational reference for the research community and inspire continued innovation. This survey will be regularly updated at: https://github.com/GigaAI-research/General-World-Models-Survey.",
  "published": "2024-05-06",
  "updated": "2025-10-28",
  "year": "2024",
  "authors": [
   "Zheng Zhu",
   "Xiaofeng Wang",
   "Wangbo Zhao",
   "Chen Min",
   "Bohan Li",
   "Nianchen Deng",
   "Min Dou",
   "Yuqi Wang",
   "Botian Shi",
   "Kai Wang",
   "Chi Zhang",
   "Yang You",
   "Zhaoxiang Zhang",
   "Dawei Zhao",
   "Liang Xiao",
   "Jian Zhao",
   "Jiwen Lu",
   "Guan Huang"
  ],
  "author_count": 18,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 130,
  "influential_citations": 1,
  "tldr": "A comprehensive exploration of the latest advancements in world models, exploring the intricacies inherent in world models deployed within autonomous agents, shedding light on their profound significance in enabling intelligent interactions within dynamic environmental contexts.",
  "doi": "10.48550/arXiv.2405.03520",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zheng Zhu",
    "id": "2265968976",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Xiaofeng Wang",
    "id": "2242976725",
    "h_index": 17,
    "papers": 57
   },
   {
    "name": "Wangbo Zhao",
    "id": "2292217857",
    "h_index": 10,
    "papers": 54
   },
   {
    "name": "Chen Min",
    "id": "2061284983",
    "h_index": 9,
    "papers": 19
   },
   {
    "name": "Nianchen Deng",
    "id": "2282958636",
    "h_index": 8,
    "papers": 16
   },
   {
    "name": "Min Dou",
    "id": "2197075911",
    "h_index": 18,
    "papers": 27
   },
   {
    "name": "Yuqi Wang",
    "id": "2300166309",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Botian Shi",
    "id": "2278899936",
    "h_index": 13,
    "papers": 36
   },
   {
    "name": "Kai Wang",
    "id": "2292214744",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Chi Zhang",
    "id": "2300133692",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Yang You",
    "id": "2283134324",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Zhaoxiang Zhang",
    "id": "2300304635",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Dawei Zhao",
    "id": "2110816085",
    "h_index": 16,
    "papers": 40
   },
   {
    "name": "Liang Xiao",
    "id": "2300816725",
    "h_index": 5,
    "papers": 21
   },
   {
    "name": "Jian Zhao",
    "id": "2282237850",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Jiwen Lu",
    "id": "2243332262",
    "h_index": 6,
    "papers": 17
   },
   {
    "name": "Guan Huang",
    "id": "2256954306",
    "h_index": 16,
    "papers": 44
   }
  ],
  "comment": "This survey will be regularly updated at: https://github.com/GigaAI-research/General-World-Models-Survey",
  "topics": [
   "world-models",
   "sim2real",
   "navigation",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2405.03520v2",
  "pdf_url": "https://arxiv.org/pdf/2405.03520v2",
  "html_url": "https://arxiv.org/html/2405.03520v2",
  "code_url": "https://github.com/GigaAI-research/General-World-Models-Survey",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.12
 },
 {
  "id": "2405.02280",
  "slug": "dreamscene4d-dynamic-multi-object-scene-generation-from-monocular-vide",
  "title": "DreamScene4D: Dynamic Multi-Object Scene Generation from Monocular Videos",
  "abstract": "View-predictive generative models provide strong priors for lifting object-centric images and videos into 3D and 4D through rendering and score distillation objectives. A question then remains: what about lifting complete multi-object dynamic scenes? There are two challenges in this direction: First, rendering error gradients are often insufficient to recover fast object motion, and second, view predictive generative models work much better for objects than whole scenes, so, score distillation objectives cannot currently be applied at the scene level directly. We present DreamScene4D, the first approach to generate 3D dynamic scenes of multiple objects from monocular videos via 360-degree novel view synthesis. Our key insight is a \"decompose-recompose\" approach that factorizes the video scene into the background and object tracks, while also factorizing object motion into 3 components: object-centric deformation, object-to-world-frame transformation, and camera motion. Such decomposition permits rendering error gradients and object view-predictive models to recover object 3D completions and deformations while bounding box tracks guide the large object movements in the scene. We show extensive results on challenging DAVIS, Kubric, and self-captured videos with quantitative comparisons and a user preference study. Besides 4D scene generation, DreamScene4D obtains accurate 2D persistent point track by projecting the inferred 3D trajectories to 2D. We will release our code and hope our work will stimulate more research on fine-grained 4D understanding from videos.",
  "published": "2024-05-03",
  "updated": "2024-05-23",
  "year": "2024",
  "authors": [
   "Wen-Hsuan Chu",
   "Lei Ke",
   "Katerina Fragkiadaki"
  ],
  "author_count": 3,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 75,
  "influential_citations": 4,
  "tldr": "DreamScene4D is presented, the first approach to generate 3D dynamic scenes of multiple objects from monocular videos via 360-degree novel view synthesis with a \"decompose-recompose\" approach that factorizes the video scene into the background and object tracks, while also factorizing object motion into 3 components: object-centric deformation, object-to-world-frame transformation, and camera motion.",
  "doi": "10.48550/arXiv.2405.02280",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Wen-Hsuan Chu",
    "id": "2257037951",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Lei Ke",
    "id": "2265229",
    "h_index": 23,
    "papers": 29
   },
   {
    "name": "Katerina Fragkiadaki",
    "id": "1705557",
    "h_index": 32,
    "papers": 82
   }
  ],
  "comment": "Project page: https://dreamscene4d.github.io/",
  "topics": [
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2405.02280v2",
  "pdf_url": "https://arxiv.org/pdf/2405.02280v2",
  "html_url": "https://arxiv.org/html/2405.02280v2",
  "code_url": "https://dreamscene4d.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.38
 },
 {
  "id": "2405.01527",
  "slug": "track2act-predicting-point-tracks-from-internet-videos-enables-general",
  "title": "Track2Act: Predicting Point Tracks from Internet Videos enables Generalizable Robot Manipulation",
  "abstract": "We seek to learn a generalizable goal-conditioned policy that enables zero-shot robot manipulation: interacting with unseen objects in novel scenes without test-time adaptation. While typical approaches rely on a large amount of demonstration data for such generalization, we propose an approach that leverages web videos to predict plausible interaction plans and learns a task-agnostic transformation to obtain robot actions in the real world. Our framework,Track2Act predicts tracks of how points in an image should move in future time-steps based on a goal, and can be trained with diverse videos on the web including those of humans and robots manipulating everyday objects. We use these 2D track predictions to infer a sequence of rigid transforms of the object to be manipulated, and obtain robot end-effector poses that can be executed in an open-loop manner. We then refine this open-loop plan by predicting residual actions through a closed loop policy trained with a few embodiment-specific demonstrations. We show that this approach of combining scalably learned track prediction with a residual policy requiring minimal in-domain robot-specific data enables diverse generalizable robot manipulation, and present a wide array of real-world robot manipulation results across unseen tasks, objects, and scenes. https://homangab.github.io/track2act/",
  "published": "2024-05-02",
  "updated": "2024-08-08",
  "year": "2024",
  "authors": [
   "Homanga Bharadhwaj",
   "Roozbeh Mottaghi",
   "Abhinav Gupta",
   "Shubham Tulsiani"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "ECCV",
  "venue_source": "semantic-scholar",
  "citations": 155,
  "influential_citations": 17,
  "tldr": "It is shown that this approach of combining scalably learned track prediction with a residual policy requiring minimal in-domain robot-specific data enables diverse generalizable robot manipulation, and present a wide array of real-world robot manipulation results across unseen tasks, objects, and scenes.",
  "doi": "10.1007/978-3-031-73116-7_18",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Homanga Bharadhwaj",
    "id": "51113848",
    "h_index": 23,
    "papers": 59
   },
   {
    "name": "Roozbeh Mottaghi",
    "id": "3012475",
    "h_index": 46,
    "papers": 102
   },
   {
    "name": "Abhinav Gupta",
    "id": "2240431852",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Shubham Tulsiani",
    "id": "2757335",
    "h_index": 45,
    "papers": 98
   }
  ],
  "comment": "ECCV 2024. Last 3 authors contributed equally",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2405.01527v2",
  "pdf_url": "https://arxiv.org/pdf/2405.01527v2",
  "html_url": "https://arxiv.org/html/2405.01527v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.69
 },
 {
  "id": "2404.19664",
  "slug": "towards-generalist-robot-learning-from-internet-video-a-survey",
  "title": "Towards Generalist Robot Learning from Internet Video: A Survey",
  "abstract": "Scaling deep learning to massive and diverse internet data has driven remarkable breakthroughs in domains such as video generation and natural language processing. Robot learning, however, has thus far failed to replicate this success and remains constrained by a scarcity of available data. Learning from videos (LfV) methods aim to address this data bottleneck by augmenting traditional robot data with large-scale internet video. This video data provides foundational information regarding physical dynamics, behaviours, and tasks, and can be highly informative for general-purpose robots. This survey systematically examines the emerging field of LfV. We first outline essential concepts, including detailing fundamental LfV challenges such as distribution shift and missing action labels in video data. Next, we comprehensively review current methods for extracting knowledge from large-scale internet video, overcoming LfV challenges, and improving robot learning through video-informed training. The survey concludes with a critical discussion of future opportunities. Here, we emphasize the need for scalable foundation model approaches that can leverage the full range of available internet video and enhance the learning of robot policies and dynamics models. Overall, the survey aims to inform and catalyse future LfV research, driving progress towards general-purpose robots.",
  "published": "2024-04-30",
  "updated": "2025-07-23",
  "year": "2024",
  "authors": [
   "Robert McCarthy",
   "Daniel C. H. Tan",
   "Dominik Schmidt",
   "Fernando Acero",
   "Nathan Herr",
   "Yilun Du",
   "Thomas G. Thuruthel",
   "Zhibin Li"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 59,
  "influential_citations": 0,
  "tldr": "This survey systematically examines the emerging field of Learning from Videos (LfV), comprehensively review current methods for extracting knowledge from large-scale internet video, overcoming LfV challenges, and improving robot learning through video-informed training.",
  "doi": "10.1613/jair.1.17400",
  "oa_pdf": "https://jair.org/index.php/jair/article/download/17400/27192",
  "s2_authors": [
   {
    "name": "Robert McCarthy",
    "id": "144722083",
    "h_index": 9,
    "papers": 24
   },
   {
    "name": "D. Tan",
    "id": "2219272080",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Dominik Schmidt",
    "id": "2275159707",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Fernando Acero",
    "id": "2126051157",
    "h_index": 7,
    "papers": 14
   },
   {
    "name": "Nathan Herr",
    "id": "2298969368",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Yilun Du",
    "id": "2315467731",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "T. G. Thuruthel",
    "id": "3455927",
    "h_index": 19,
    "papers": 72
   },
   {
    "name": "Zhibin Li",
    "id": "2247864302",
    "h_index": 6,
    "papers": 13
   }
  ],
  "comment": "",
  "topics": [
   "foundation-pretraining",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2404.19664v5",
  "pdf_url": "https://arxiv.org/pdf/2404.19664v5",
  "html_url": "https://arxiv.org/html/2404.19664v5",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.78
 },
 {
  "id": "2404.17521",
  "slug": "ag2manip-learning-novel-manipulation-skills-with-agent-agnostic-visual",
  "title": "Ag2Manip: Learning Novel Manipulation Skills with Agent-Agnostic Visual and Action Representations",
  "abstract": "Autonomous robotic systems capable of learning novel manipulation tasks are poised to transform industries from manufacturing to service automation. However, modern methods (e.g., VIP and R3M) still face significant hurdles, notably the domain gap among robotic embodiments and the sparsity of successful task executions within specific action spaces, resulting in misaligned and ambiguous task representations. We introduce Ag2Manip (Agent-Agnostic representations for Manipulation), a framework aimed at surmounting these challenges through two key innovations: a novel agent-agnostic visual representation derived from human manipulation videos, with the specifics of embodiments obscured to enhance generalizability; and an agent-agnostic action representation abstracting a robot's kinematics to a universal agent proxy, emphasizing crucial interactions between end-effector and object. Ag2Manip's empirical validation across simulated benchmarks like FrankaKitchen, ManiSkill, and PartManip shows a 325% increase in performance, achieved without domain-specific demonstrations. Ablation studies underline the essential contributions of the visual and action representations to this success. Extending our evaluations to the real world, Ag2Manip significantly improves imitation learning success rates from 50% to 77.5%, demonstrating its effectiveness and generalizability across both simulated and physical environments.",
  "published": "2024-04-26",
  "updated": "2024-04-26",
  "year": "2024",
  "authors": [
   "Puhao Li",
   "Tengyu Liu",
   "Yuyang Li",
   "Muzhi Han",
   "Haoran Geng",
   "Shu Wang",
   "Yixin Zhu",
   "Song-Chun Zhu",
   "Siyuan Huang"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS",
  "venue_source": "semantic-scholar",
  "citations": 33,
  "influential_citations": 0,
  "tldr": "Ag2Manip (Agent-Agnostic representations for Manipulation), a framework aimed at addressing challenges through two key innovations: an agent-agnostic visual representation derived from human manipulation videos, with the specifics of embodiments obscured to enhance generalizability, demonstrates its effectiveness and generalizability across both simulated and real environments.",
  "doi": "10.1109/IROS58592.2024.10801835",
  "oa_pdf": "https://arxiv.org/pdf/2404.17521",
  "s2_authors": [
   {
    "name": "Puhao Li",
    "id": "2145015272",
    "h_index": 13,
    "papers": 18
   },
   {
    "name": "Tengyu Liu",
    "id": "2110032600",
    "h_index": 21,
    "papers": 32
   },
   {
    "name": "Yuyang Li",
    "id": "2261448933",
    "h_index": 8,
    "papers": 11
   },
   {
    "name": "Muzhi Han",
    "id": "2044532694",
    "h_index": 8,
    "papers": 11
   },
   {
    "name": "Haoran Geng",
    "id": "2144742582",
    "h_index": 17,
    "papers": 32
   },
   {
    "name": "Shu Wang",
    "id": "2116931976",
    "h_index": 6,
    "papers": 6
   },
   {
    "name": "Yixin Zhu",
    "id": "2261513442",
    "h_index": 11,
    "papers": 20
   },
   {
    "name": "Song-Chun Zhu",
    "id": "2148575818",
    "h_index": 15,
    "papers": 32
   },
   {
    "name": "Siyuan Huang",
    "id": "2264375840",
    "h_index": 13,
    "papers": 27
   }
  ],
  "comment": "Project website and open-source code: https://xiaoyao-li.github.io/research/ag2manip",
  "topics": [
   "imitation-diffusion"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2404.17521v1",
  "pdf_url": "https://arxiv.org/pdf/2404.17521v1",
  "html_url": "https://arxiv.org/html/2404.17521v1",
  "code_url": "https://xiaoyao-li.github.io/research/ag2manip",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.03
 },
 {
  "id": "2404.16823",
  "slug": "learning-visuotactile-skills-with-two-multifingered-hands",
  "title": "Learning Visuotactile Skills with Two Multifingered Hands",
  "abstract": "Aiming to replicate human-like dexterity, perceptual experiences, and motion patterns, we explore learning from human demonstrations using a bimanual system with multifingered hands and visuotactile data. Two significant challenges exist: the lack of an affordable and accessible teleoperation system suitable for a dual-arm setup with multifingered hands, and the scarcity of multifingered hand hardware equipped with touch sensing. To tackle the first challenge, we develop HATO, a low-cost hands-arms teleoperation system that leverages off-the-shelf electronics, complemented with a software suite that enables efficient data collection; the comprehensive software suite also supports multimodal data processing, scalable policy learning, and smooth policy deployment. To tackle the latter challenge, we introduce a novel hardware adaptation by repurposing two prosthetic hands equipped with touch sensors for research. Using visuotactile data collected from our system, we learn skills to complete long-horizon, high-precision tasks which are difficult to achieve without multifingered dexterity and touch feedback. Furthermore, we empirically investigate the effects of dataset size, sensing modality, and visual input preprocessing on policy learning. Our results mark a promising step forward in bimanual multifingered manipulation from visuotactile data. Videos, code, and datasets can be found at https://toruowo.github.io/hato/ .",
  "published": "2024-04-25",
  "updated": "2024-05-22",
  "year": "2024",
  "authors": [
   "Toru Lin",
   "Yu Zhang",
   "Qiyang Li",
   "Haozhi Qi",
   "Brent Yi",
   "Sergey Levine",
   "Jitendra Malik"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA",
  "venue_source": "semantic-scholar",
  "citations": 148,
  "influential_citations": 4,
  "tldr": "HATO is developed, a low-cost hands-arms teleoperation system that leverages off-the-shelf electronics, complemented with a software suite that enables efficient data collection and empirically investigates the effects of dataset size, sensing modality, and visual input preprocessing on policy learning.",
  "doi": "10.1109/ICRA55743.2025.11128180",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Toru Lin",
    "id": "152997384",
    "h_index": 6,
    "papers": 6
   },
   {
    "name": "Yu Zhang",
    "id": "2329789543",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Qiyang Li",
    "id": "8194287",
    "h_index": 15,
    "papers": 34
   },
   {
    "name": "Haozhi Qi",
    "id": "2247951244",
    "h_index": 11,
    "papers": 18
   },
   {
    "name": "Brent Yi",
    "id": "2242880086",
    "h_index": 18,
    "papers": 27
   },
   {
    "name": "Sergey Levine",
    "id": "2254622427",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "Jitendra Malik",
    "id": "2242761335",
    "h_index": 15,
    "papers": 32
   }
  ],
  "comment": "Code and Project Website: https://toruowo.github.io/hato/",
  "topics": [
   "dexterous-manipulation",
   "egocentric-data",
   "tactile",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2404.16823v2",
  "pdf_url": "https://arxiv.org/pdf/2404.16823v2",
  "html_url": "https://arxiv.org/html/2404.16823v2",
  "code_url": "https://toruowo.github.io/hato/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.67
 },
 {
  "id": "2404.15709",
  "slug": "vividex-learning-vision-based-dexterous-manipulation-from-human-videos",
  "title": "ViViDex: Learning Vision-based Dexterous Manipulation from Human Videos",
  "abstract": "In this work, we aim to learn a unified vision-based policy for multi-fingered robot hands to manipulate a variety of objects in diverse poses. Though prior work has shown benefits of using human videos for policy learning, performance gains have been limited by the noise in estimated trajectories. Moreover, reliance on privileged object information such as ground-truth object states further limits the applicability in realistic scenarios. To address these limitations, we propose a new framework ViViDex to improve vision-based policy learning from human videos. It first uses reinforcement learning with trajectory guided rewards to train state-based policies for each video, obtaining both visually natural and physically plausible trajectories from the video. We then rollout successful episodes from state-based policies and train a unified visual policy without using any privileged information. We propose coordinate transformation to further enhance the visual point cloud representation, and compare behavior cloning and diffusion policy for the visual policy training. Experiments both in simulation and on the real robot demonstrate that ViViDex outperforms state-of-the-art approaches on three dexterous manipulation tasks.",
  "published": "2024-04-24",
  "updated": "2025-03-01",
  "year": "2024",
  "authors": [
   "Zerui Chen",
   "Shizhe Chen",
   "Etienne Arlaud",
   "Ivan Laptev",
   "Cordelia Schmid"
  ],
  "author_count": 5,
  "categories": [
   "cs.CV",
   "cs.LG",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "ICRA",
  "venue_source": "semantic-scholar",
  "citations": 56,
  "influential_citations": 4,
  "tldr": "This work proposes a new framework ViViDex to improve vision-based policy learning from human videos that outperforms state-of-theart approaches on three dexterous manipulation tasks and proposes coordinate transformation to further enhance the visual point cloud representation.",
  "doi": "10.1109/ICRA55743.2025.11127358",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zerui Chen",
    "id": "2298208737",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Shizhe Chen",
    "id": "2248998076",
    "h_index": 7,
    "papers": 21
   },
   {
    "name": "Cordelia Schmid",
    "id": "2248308045",
    "h_index": 7,
    "papers": 14
   },
   {
    "name": "I. Laptev",
    "id": "143991676",
    "h_index": 76,
    "papers": 193
   }
  ],
  "comment": "Accepted by ICRA 2025. Project Page: https://zerchen.github.io/projects/vividex.html",
  "topics": [
   "dexterous-manipulation",
   "egocentric-data",
   "imitation-diffusion",
   "rl-control",
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2404.15709v3",
  "pdf_url": "https://arxiv.org/pdf/2404.15709v3",
  "html_url": "https://arxiv.org/html/2404.15709v3",
  "code_url": "https://zerchen.github.io/projects/vividex.html",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.26
 },
 {
  "id": "2404.13026",
  "slug": "physdreamer-physics-based-interaction-with-3d-objects-via-video-genera",
  "title": "PhysDreamer: Physics-Based Interaction with 3D Objects via Video Generation",
  "abstract": "Realistic object interactions are crucial for creating immersive virtual experiences, yet synthesizing realistic 3D object dynamics in response to novel interactions remains a significant challenge. Unlike unconditional or text-conditioned dynamics generation, action-conditioned dynamics requires perceiving the physical material properties of objects and grounding the 3D motion prediction on these properties, such as object stiffness. However, estimating physical material properties is an open problem due to the lack of material ground-truth data, as measuring these properties for real objects is highly difficult. We present PhysDreamer, a physics-based approach that endows static 3D objects with interactive dynamics by leveraging the object dynamics priors learned by video generation models. By distilling these priors, PhysDreamer enables the synthesis of realistic object responses to novel interactions, such as external forces or agent manipulations. We demonstrate our approach on diverse examples of elastic objects and evaluate the realism of the synthesized interactions through a user study. PhysDreamer takes a step towards more engaging and realistic virtual experiences by enabling static 3D objects to dynamically respond to interactive stimuli in a physically plausible manner. See our project page at https://physdreamer.github.io/.",
  "published": "2024-04-19",
  "updated": "2024-10-07",
  "year": "2024",
  "authors": [
   "Tianyuan Zhang",
   "Hong-Xing Yu",
   "Rundi Wu",
   "Brandon Y. Feng",
   "Changxi Zheng",
   "Noah Snavely",
   "Jiajun Wu",
   "William T. Freeman"
  ],
  "author_count": 8,
  "categories": [
   "cs.CV",
   "cs.AI"
  ],
  "primary_category": "cs.CV",
  "venue": "ECCV",
  "venue_source": "semantic-scholar",
  "citations": 214,
  "influential_citations": 23,
  "tldr": "PhysDreamer is a physics-based approach that endows static 3D objects with interactive dynamics by leveraging the object dynamics priors learned by video generation models and enables the synthesis of realistic object responses to novel interactions, such as external forces or agent manipulations.",
  "doi": "10.48550/arXiv.2404.13026",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Tianyuan Zhang",
    "id": "2297868523",
    "h_index": 10,
    "papers": 13
   },
   {
    "name": "Hong-Xing Yu",
    "id": "2239448099",
    "h_index": 18,
    "papers": 32
   },
   {
    "name": "Rundi Wu",
    "id": "1406236938",
    "h_index": 13,
    "papers": 21
   },
   {
    "name": "Brandon Y. Feng",
    "id": "2297671955",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Changxi Zheng",
    "id": "2297884549",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Noah Snavely",
    "id": "1830653",
    "h_index": 75,
    "papers": 184
   },
   {
    "name": "Jiajun Wu",
    "id": "3045089",
    "h_index": 80,
    "papers": 228
   },
   {
    "name": "William T. Freeman",
    "id": "2271150979",
    "h_index": 3,
    "papers": 5
   }
  ],
  "comment": "Project website at: https://physdreamer.github.io/ Appear on ECCV 2024",
  "topics": [
   "world-models",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2404.13026v2",
  "pdf_url": "https://arxiv.org/pdf/2404.13026v2",
  "html_url": "https://arxiv.org/html/2404.13026v2",
  "code_url": "https://physdreamer.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.83
 },
 {
  "id": "2404.12383",
  "slug": "g-hop-generative-hand-object-prior-for-interaction-reconstruction-and",
  "title": "G-HOP: Generative Hand-Object Prior for Interaction Reconstruction and Grasp Synthesis",
  "abstract": "We propose G-HOP, a denoising diffusion based generative prior for hand-object interactions that allows modeling both the 3D object and a human hand, conditioned on the object category. To learn a 3D spatial diffusion model that can capture this joint distribution, we represent the human hand via a skeletal distance field to obtain a representation aligned with the (latent) signed distance field for the object. We show that this hand-object prior can then serve as generic guidance to facilitate other tasks like reconstruction from interaction clip and human grasp synthesis. We believe that our model, trained by aggregating seven diverse real-world interaction datasets spanning across 155 categories, represents a first approach that allows jointly generating both hand and object. Our empirical evaluations demonstrate the benefit of this joint prior in video-based reconstruction and human grasp synthesis, outperforming current task-specific baselines. Project website: https://judyye.github.io/ghop-www",
  "published": "2024-04-18",
  "updated": "2024-04-18",
  "year": "2024",
  "authors": [
   "Yufei Ye",
   "Abhinav Gupta",
   "Kris Kitani",
   "Shubham Tulsiani"
  ],
  "author_count": 4,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 53,
  "influential_citations": 8,
  "tldr": "G-HOP, a denoising diffusion based generative prior for hand-object interactions that allows modeling both the 3D object and a human hand, conditioned on the object category, represents a first approach that allows jointly generating both hand and object.",
  "doi": "10.1109/CVPR52733.2024.00187",
  "oa_pdf": "https://arxiv.org/pdf/2404.12383",
  "s2_authors": [
   {
    "name": "Yufei Ye",
    "id": "9653518",
    "h_index": 10,
    "papers": 18
   },
   {
    "name": "Abhinav Gupta",
    "id": "2240431852",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "K. Kitani",
    "id": "2297185992",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Shubham Tulsiani",
    "id": "2757335",
    "h_index": 45,
    "papers": 98
   }
  ],
  "comment": "accepted to CVPR2024; project page at https://judyye.github.io/ghop-www",
  "topics": [
   "dexterous-manipulation",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2404.12383v1",
  "pdf_url": "https://arxiv.org/pdf/2404.12383v1",
  "html_url": "https://arxiv.org/html/2404.12383v1",
  "code_url": "https://judyye.github.io/ghop-www",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.23
 },
 {
  "id": "2404.12377",
  "slug": "robodreamer-learning-compositional-world-models-for-robot-imagination",
  "title": "RoboDreamer: Learning Compositional World Models for Robot Imagination",
  "abstract": "Text-to-video models have demonstrated substantial potential in robotic decision-making, enabling the imagination of realistic plans of future actions as well as accurate environment simulation. However, one major issue in such models is generalization -- models are limited to synthesizing videos subject to language instructions similar to those seen at training time. This is heavily limiting in decision-making, where we seek a powerful world model to synthesize plans of unseen combinations of objects and actions in order to solve previously unseen tasks in new environments. To resolve this issue, we introduce RoboDreamer, an innovative approach for learning a compositional world model by factorizing the video generation. We leverage the natural compositionality of language to parse instructions into a set of lower-level primitives, which we condition a set of models on to generate videos. We illustrate how this factorization naturally enables compositional generalization, by allowing us to formulate a new natural language instruction as a combination of previously seen components. We further show how such a factorization enables us to add additional multimodal goals, allowing us to specify a video we wish to generate given both natural language instructions and a goal image. Our approach can successfully synthesize video plans on unseen goals in the RT-X, enables successful robot execution in simulation, and substantially outperforms monolithic baseline approaches to video generation.",
  "published": "2024-04-18",
  "updated": "2024-04-18",
  "year": "2024",
  "authors": [
   "Siyuan Zhou",
   "Yilun Du",
   "Jiaben Chen",
   "Yandong Li",
   "Dit-Yan Yeung",
   "Chuang Gan"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "ICML",
  "venue_source": "semantic-scholar",
  "citations": 199,
  "influential_citations": 6,
  "tldr": "RoboDreamer is introduced, an innovative approach for learning a compositional world model by factorizing the video generation, which can successfully synthesize video plans on unseen goals in the RT-X, enables successful robot execution in simulation, and substantially outperforms monolithic baseline approaches to video generation.",
  "doi": "10.48550/arXiv.2404.12377",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Siyuan Zhou",
    "id": "2258920872",
    "h_index": 9,
    "papers": 19
   },
   {
    "name": "Yilun Du",
    "id": "2258799458",
    "h_index": 12,
    "papers": 18
   },
   {
    "name": "Jiaben Chen",
    "id": "2120262069",
    "h_index": 11,
    "papers": 15
   },
   {
    "name": "Yandong Li",
    "id": "2271358134",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "D. Yeung",
    "id": "1739816",
    "h_index": 68,
    "papers": 257
   },
   {
    "name": "Chuang Gan",
    "id": "2280331955",
    "h_index": 9,
    "papers": 10
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2404.12377v1",
  "pdf_url": "https://arxiv.org/pdf/2404.12377v1",
  "html_url": "https://arxiv.org/html/2404.12377v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.8
 },
 {
  "id": "2404.03736",
  "slug": "sc4d-sparse-controlled-video-to-4d-generation-and-motion-transfer",
  "title": "SC4D: Sparse-Controlled Video-to-4D Generation and Motion Transfer",
  "abstract": "Recent advances in 2D/3D generative models enable the generation of dynamic 3D objects from a single-view video. Existing approaches utilize score distillation sampling to form the dynamic scene as dynamic NeRF or dense 3D Gaussians. However, these methods struggle to strike a balance among reference view alignment, spatio-temporal consistency, and motion fidelity under single-view conditions due to the implicit nature of NeRF or the intricate dense Gaussian motion prediction. To address these issues, this paper proposes an efficient, sparse-controlled video-to-4D framework named SC4D, that decouples motion and appearance to achieve superior video-to-4D generation. Moreover, we introduce Adaptive Gaussian (AG) initialization and Gaussian Alignment (GA) loss to mitigate shape degeneration issue, ensuring the fidelity of the learned motion and shape. Comprehensive experimental results demonstrate that our method surpasses existing methods in both quality and efficiency. In addition, facilitated by the disentangled modeling of motion and appearance of SC4D, we devise a novel application that seamlessly transfers the learned motion onto a diverse array of 4D entities according to textual descriptions.",
  "published": "2024-04-04",
  "updated": "2024-08-14",
  "year": "2024",
  "authors": [
   "Zijie Wu",
   "Chaohui Yu",
   "Yanqin Jiang",
   "Chenjie Cao",
   "Fan Wang",
   "Xiang Bai"
  ],
  "author_count": 6,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "ECCV",
  "venue_source": "semantic-scholar",
  "citations": 70,
  "influential_citations": 10,
  "tldr": "This paper proposes an efficient, sparse-controlled video-to-4D framework named SC4D, that decouples motion and appearance to achieve superior video-to-4D generation and introduces Adaptive Gaussian (AG) initialization and Gaussian Alignment (GA) loss to mitigate shape degeneration issue, ensuring the fidelity of the learned motion and shape.",
  "doi": "10.48550/arXiv.2404.03736",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zijie Wu",
    "id": "2187780453",
    "h_index": 4,
    "papers": 14
   },
   {
    "name": "Chaohui Yu",
    "id": "2110961040",
    "h_index": 16,
    "papers": 45
   },
   {
    "name": "Yanqin Jiang",
    "id": "2265542118",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Chenjie Cao",
    "id": "2296071044",
    "h_index": 7,
    "papers": 14
   },
   {
    "name": "Fan Wang",
    "id": "2257894784",
    "h_index": 8,
    "papers": 18
   },
   {
    "name": "Xiang Bai",
    "id": "2258401423",
    "h_index": 10,
    "papers": 18
   }
  ],
  "comment": "Accepted by ECCV2024! Project Page: https://sc4d.github.io/ Code is available at: https://github.com/JarrentWu1031/SC4D",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2404.03736v2",
  "pdf_url": "https://arxiv.org/pdf/2404.03736v2",
  "html_url": "https://arxiv.org/html/2404.03736v2",
  "code_url": "https://sc4d.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.35
 },
 {
  "id": "2404.02817",
  "slug": "a-survey-of-optimization-based-task-and-motion-planning-from-classical",
  "title": "A Survey of Optimization-based Task and Motion Planning: From Classical To Learning Approaches",
  "abstract": "Task and Motion Planning (TAMP) integrates high-level task planning and low-level motion planning to equip robots with the autonomy to effectively reason over long-horizon, dynamic tasks. Optimization-based TAMP focuses on hybrid optimization approaches that define goal conditions via objective functions and are capable of handling open-ended goals, robotic dynamics, and physical interaction between the robot and the environment. Therefore, optimization-based TAMP is particularly suited to solve highly complex, contact-rich locomotion and manipulation problems. This survey provides a comprehensive review on optimization-based TAMP, covering (i) planning domain representations, including action description languages and temporal logic, (ii) individual solution strategies for components of TAMP, including AI planning and trajectory optimization (TO), and (iii) the dynamic interplay between logic-based task planning and model-based TO. A particular focus of this survey is to highlight the algorithm structures to efficiently solve TAMP, especially hierarchical and distributed approaches. Additionally, the survey emphasizes the synergy between the classical methods and contemporary learning-based innovations such as large language models. Furthermore, the future research directions for TAMP is discussed in this survey, highlighting both algorithmic and application-specific challenges.",
  "published": "2024-04-03",
  "updated": "2024-10-07",
  "year": "2024",
  "authors": [
   "Zhigen Zhao",
   "Shuo Cheng",
   "Yan Ding",
   "Ziyi Zhou",
   "Shiqi Zhang",
   "Danfei Xu",
   "Ye Zhao"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 92,
  "influential_citations": 0,
  "tldr": "This survey provides a comprehensive review on optimization-based TAMP, covering first, planning domain representations, including action description languages and temporal logic, second, individual solution strategies for components of TAMP, including AI planning and trajectory optimization (TO), and finally, the dynamic interplay between logic-based task planning and model-based TO.",
  "doi": "10.1109/TMECH.2024.3452509",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhigen Zhao",
    "id": "2214963761",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Shuo Cheng",
    "id": "2232588215",
    "h_index": 6,
    "papers": 16
   },
   {
    "name": "Yan Ding",
    "id": "2110668031",
    "h_index": 11,
    "papers": 24
   },
   {
    "name": "Ziyi Zhou",
    "id": "2121298138",
    "h_index": 10,
    "papers": 30
   },
   {
    "name": "Shiqi Zhang",
    "id": "2257315750",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Danfei Xu",
    "id": "2260291195",
    "h_index": 7,
    "papers": 14
   },
   {
    "name": "Ye Zhao",
    "id": "2258781763",
    "h_index": 2,
    "papers": 5
   }
  ],
  "comment": "26 pages, 13 figures, published at IEEE/ASME Transactions on Mechatronics",
  "topics": [
   "humanoids",
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2404.02817v5",
  "pdf_url": "https://arxiv.org/pdf/2404.02817v5",
  "html_url": "https://arxiv.org/html/2404.02817v5",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.97
 },
 {
  "id": "2404.02148",
  "slug": "diffusion-2-dynamic-3d-content-generation-via-score-composition-of-vid",
  "title": "Diffusion$^2$: Dynamic 3D Content Generation via Score Composition of Video and Multi-view Diffusion Models",
  "abstract": "Recent advancements in 3D generation are predominantly propelled by improvements in 3D-aware image diffusion models. These models are pretrained on Internet-scale image data and fine-tuned on massive 3D data, offering the capability of producing highly consistent multi-view images. However, due to the scarcity of synchronized multi-view video data, it remains challenging to adapt this paradigm to 4D generation directly. Despite that, the available video and 3D data are adequate for training video and multi-view diffusion models separately that can provide satisfactory dynamic and geometric priors respectively. To take advantage of both, this paper presents Diffusion$^2$, a novel framework for dynamic 3D content creation that reconciles the knowledge about geometric consistency and temporal smoothness from these models to directly sample dense multi-view multi-frame images which can be employed to optimize continuous 4D representation. Specifically, we design a simple yet effective denoising strategy via score composition of pretrained video and multi-view diffusion models based on the probability structure of the target image array. To alleviate the potential conflicts between two heterogeneous scores, we further introduce variance-reducing sampling via interpolated steps, facilitating smooth and stable generation. Owing to the high parallelism of the proposed image generation process and the efficiency of the modern 4D reconstruction pipeline, our framework can generate 4D content within few minutes. Notably, our method circumvents the reliance on expensive and hard-to-scale 4D data, thereby having the potential to benefit from the scaling of the foundation video and multi-view diffusion models. Extensive experiments demonstrate the efficacy of our proposed framework in generating highly seamless and consistent 4D assets under various types of conditions.",
  "published": "2024-04-02",
  "updated": "2024-10-02",
  "year": "2024",
  "authors": [
   "Zeyu Yang",
   "Zijie Pan",
   "Chun Gu",
   "Li Zhang"
  ],
  "author_count": 4,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 26,
  "influential_citations": 2,
  "tldr": "A novel framework for dynamic 3D content creation that reconciles the knowledge about geometric consistency and temporal smoothness from these models to directly sample dense multi-view multi-frame images which can be employed to optimize continuous 4D representation is presented.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zeyu Yang",
    "id": "2259639017",
    "h_index": 9,
    "papers": 14
   },
   {
    "name": "Zijie Pan",
    "id": "2260343025",
    "h_index": 8,
    "papers": 11
   },
   {
    "name": "Chun Gu",
    "id": "2268399619",
    "h_index": 11,
    "papers": 19
   },
   {
    "name": "Li Zhang",
    "id": "2269750616",
    "h_index": 8,
    "papers": 15
   }
  ],
  "comment": "Technical Report",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2404.02148v4",
  "pdf_url": "https://arxiv.org/pdf/2404.02148v4",
  "html_url": "https://arxiv.org/html/2404.02148v4",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.93
 },
 {
  "id": "2404.02132",
  "slug": "vitamin-designing-scalable-vision-models-in-the-vision-language-era",
  "title": "ViTamin: Designing Scalable Vision Models in the Vision-Language Era",
  "abstract": "Recent breakthroughs in vision-language models (VLMs) start a new page in the vision community. The VLMs provide stronger and more generalizable feature embeddings compared to those from ImageNet-pretrained models, thanks to the training on the large-scale Internet image-text pairs. However, despite the amazing achievement from the VLMs, vanilla Vision Transformers (ViTs) remain the default choice for the image encoder. Although pure transformer proves its effectiveness in the text encoding area, it remains questionable whether it is also the case for image encoding, especially considering that various types of networks are proposed on the ImageNet benchmark, which, unfortunately, are rarely studied in VLMs. Due to small data/model scale, the original conclusions of model design on ImageNet can be limited and biased. In this paper, we aim at building an evaluation protocol of vision models in the vision-language era under the contrastive language-image pretraining (CLIP) framework. We provide a comprehensive way to benchmark different vision models, covering their zero-shot performance and scalability in both model and training data sizes. To this end, we introduce ViTamin, a new vision models tailored for VLMs. ViTamin-L significantly outperforms ViT-L by 2.0% ImageNet zero-shot accuracy, when using the same publicly available DataComp-1B dataset and the same OpenCLIP training scheme. ViTamin-L presents promising results on 60 diverse benchmarks, including classification, retrieval, open-vocabulary detection and segmentation, and large multi-modal models. When further scaling up the model size, our ViTamin-XL with only 436M parameters attains 82.9% ImageNet zero-shot accuracy, surpassing 82.0% achieved by EVA-E that has ten times more parameters (4.4B).",
  "published": "2024-04-02",
  "updated": "2024-04-03",
  "year": "2024",
  "authors": [
   "Jieneng Chen",
   "Qihang Yu",
   "Xiaohui Shen",
   "Alan Yuille",
   "Liang-Chieh Chen"
  ],
  "author_count": 5,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 66,
  "influential_citations": 7,
  "tldr": "An evaluation protocol of vision models in the vision-language era under the contrastive language-image pretraining (CLIP) framework is built, and ViTamin, a new vision models tailored for VLMs is introduced, with promising results on 60 diverse benchmarks.",
  "doi": "10.1109/CVPR52733.2024.01231",
  "oa_pdf": "https://arxiv.org/pdf/2404.02132",
  "s2_authors": [
   {
    "name": "Jieneng Chen",
    "id": "2305329570",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Qihang Yu",
    "id": "2156559",
    "h_index": 23,
    "papers": 41
   },
   {
    "name": "Xiaohui Shen",
    "id": "2266472250",
    "h_index": 11,
    "papers": 18
   },
   {
    "name": "Alan L. Yuille",
    "id": "2253485882",
    "h_index": 20,
    "papers": 78
   },
   {
    "name": "Liang-Chieh Chen",
    "id": "2266697544",
    "h_index": 13,
    "papers": 26
   }
  ],
  "comment": "CVPR 2024; https://github.com/Beckschen/ViTamin",
  "topics": [
   "sim2real",
   "foundation-pretraining",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2404.02132v2",
  "pdf_url": "https://arxiv.org/pdf/2404.02132v2",
  "html_url": "https://arxiv.org/html/2404.02132v2",
  "code_url": "https://github.com/Beckschen/ViTamin",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.33
 },
 {
  "id": "2403.19046",
  "slug": "lita-language-instructed-temporal-localization-assistant",
  "title": "LITA: Language Instructed Temporal-Localization Assistant",
  "abstract": "There has been tremendous progress in multimodal Large Language Models (LLMs). Recent works have extended these models to video input with promising instruction following capabilities. However, an important missing piece is temporal localization. These models cannot accurately answer the \"When?\" questions. We identify three key aspects that limit their temporal localization capabilities: (i) time representation, (ii) architecture, and (iii) data. We address these shortcomings by proposing Language Instructed Temporal-Localization Assistant (LITA) with the following features: (1) We introduce time tokens that encode timestamps relative to the video length to better represent time in videos. (2) We introduce SlowFast tokens in the architecture to capture temporal information at fine temporal resolution. (3) We emphasize temporal localization data for LITA. In addition to leveraging existing video datasets with timestamps, we propose a new task, Reasoning Temporal Localization (RTL), along with the dataset, ActivityNet-RTL, for learning and evaluating this task. Reasoning temporal localization requires both the reasoning and temporal localization of Video LLMs. LITA demonstrates strong performance on this challenging task, nearly doubling the temporal mean intersection-over-union (mIoU) of baselines. In addition, we show that our emphasis on temporal localization also substantially improves video-based text generation compared to existing Video LLMs, including a 36% relative improvement of Temporal Understanding. Code is available at: https://github.com/NVlabs/LITA",
  "published": "2024-03-27",
  "updated": "2024-03-27",
  "year": "2024",
  "authors": [
   "De-An Huang",
   "Shijia Liao",
   "Subhashree Radhakrishnan",
   "Hongxu Yin",
   "Pavlo Molchanov",
   "Zhiding Yu",
   "Jan Kautz"
  ],
  "author_count": 7,
  "categories": [
   "cs.CV",
   "cs.AI"
  ],
  "primary_category": "cs.CV",
  "venue": "ECCV",
  "venue_source": "semantic-scholar",
  "citations": 149,
  "influential_citations": 18,
  "tldr": "The proposed Language Instructed Temporal-Localization Assistant (LITA) with an emphasis on temporal localization also substantially improves video-based text generation compared to existing Video LLMs, including a 36% relative improvement of Temporal Understanding.",
  "doi": "10.48550/arXiv.2403.19046",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "De-An Huang",
    "id": "2293944351",
    "h_index": 12,
    "papers": 16
   },
   {
    "name": "Shijia Liao",
    "id": "2293723407",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Subhashree Radhakrishnan",
    "id": "2091913923",
    "h_index": 8,
    "papers": 14
   },
   {
    "name": "Hongxu Yin",
    "id": "1989015",
    "h_index": 40,
    "papers": 72
   },
   {
    "name": "Pavlo Molchanov",
    "id": "2824500",
    "h_index": 50,
    "papers": 147
   },
   {
    "name": "Zhiding Yu",
    "id": "2269841405",
    "h_index": 21,
    "papers": 38
   },
   {
    "name": "Jan Kautz",
    "id": "2273651410",
    "h_index": 44,
    "papers": 107
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [
   "NVIDIA"
  ],
  "abs_url": "https://arxiv.org/abs/2403.19046v1",
  "pdf_url": "https://arxiv.org/pdf/2403.19046v1",
  "html_url": "https://arxiv.org/html/2403.19046v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.18
 },
 {
  "id": "2403.17920",
  "slug": "tc4d-trajectory-conditioned-text-to-4d-generation",
  "title": "TC4D: Trajectory-Conditioned Text-to-4D Generation",
  "abstract": "Recent techniques for text-to-4D generation synthesize dynamic 3D scenes using supervision from pre-trained text-to-video models. However, existing representations for motion, such as deformation models or time-dependent neural representations, are limited in the amount of motion they can generate-they cannot synthesize motion extending far beyond the bounding box used for volume rendering. The lack of a more flexible motion model contributes to the gap in realism between 4D generation methods and recent, near-photorealistic video generation models. Here, we propose TC4D: trajectory-conditioned text-to-4D generation, which factors motion into global and local components. We represent the global motion of a scene's bounding box using rigid transformation along a trajectory parameterized by a spline. We learn local deformations that conform to the global trajectory using supervision from a text-to-video model. Our approach enables the synthesis of scenes animated along arbitrary trajectories, compositional scene generation, and significant improvements to the realism and amount of generated motion, which we evaluate qualitatively and through a user study. Video results can be viewed on our website: https://sherwinbahmani.github.io/tc4d.",
  "published": "2024-03-26",
  "updated": "2024-10-14",
  "year": "2024",
  "authors": [
   "Sherwin Bahmani",
   "Xian Liu",
   "Wang Yifan",
   "Ivan Skorokhodov",
   "Victor Rong",
   "Ziwei Liu",
   "Xihui Liu",
   "Jeong Joon Park",
   "Sergey Tulyakov",
   "Gordon Wetzstein",
   "Andrea Tagliasacchi",
   "David B. Lindell"
  ],
  "author_count": 12,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "ECCV",
  "venue_source": "semantic-scholar",
  "citations": 86,
  "influential_citations": 9,
  "tldr": "This work proposes TC4D: trajectory-conditioned text-to-4D generation, which factors motion into global and local components, and enables the synthesis of scenes animated along arbitrary trajectories, compositional scene generation, and significant improvements to the realism and amount of generated motion.",
  "doi": "10.48550/arXiv.2403.17920",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Sherwin Bahmani",
    "id": "2142454988",
    "h_index": 11,
    "papers": 14
   },
   {
    "name": "Xian Liu",
    "id": "2257708193",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Yifan Wang",
    "id": "2303611768",
    "h_index": 1,
    "papers": 5
   },
   {
    "name": "Ivan Skorokhodov",
    "id": "51118864",
    "h_index": 22,
    "papers": 45
   },
   {
    "name": "Victor Rong",
    "id": "2268759553",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Ziwei Liu",
    "id": "2145252993",
    "h_index": 12,
    "papers": 23
   },
   {
    "name": "Xihui Liu",
    "id": "2257370021",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "J. Park",
    "id": "2148838020",
    "h_index": 14,
    "papers": 18
   },
   {
    "name": "S. Tulyakov",
    "id": "145582202",
    "h_index": 48,
    "papers": 171
   },
   {
    "name": "Gordon Wetzstein",
    "id": "2256985147",
    "h_index": 21,
    "papers": 43
   },
   {
    "name": "Andrea Tagliasacchi",
    "id": "2237987366",
    "h_index": 13,
    "papers": 26
   },
   {
    "name": "David B. Lindell",
    "id": "2202838",
    "h_index": 28,
    "papers": 50
   }
  ],
  "comment": "ECCV 2024; Project Page: https://sherwinbahmani.github.io/tc4d",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2403.17920v3",
  "pdf_url": "https://arxiv.org/pdf/2403.17920v3",
  "html_url": "https://arxiv.org/html/2403.17920v3",
  "code_url": "https://sherwinbahmani.github.io/tc4d",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.44
 },
 {
  "id": "2403.14599",
  "slug": "myvlm-personalizing-vlms-for-user-specific-queries",
  "title": "MyVLM: Personalizing VLMs for User-Specific Queries",
  "abstract": "Recent large-scale vision-language models (VLMs) have demonstrated remarkable capabilities in understanding and generating textual descriptions for visual content. However, these models lack an understanding of user-specific concepts. In this work, we take a first step toward the personalization of VLMs, enabling them to learn and reason over user-provided concepts. For example, we explore whether these models can learn to recognize you in an image and communicate what you are doing, tailoring the model to reflect your personal experiences and relationships. To effectively recognize a variety of user-specific concepts, we augment the VLM with external concept heads that function as toggles for the model, enabling the VLM to identify the presence of specific target concepts in a given image. Having recognized the concept, we learn a new concept embedding in the intermediate feature space of the VLM. This embedding is tasked with guiding the language model to naturally integrate the target concept in its generated response. We apply our technique to BLIP-2 and LLaVA for personalized image captioning and further show its applicability for personalized visual question-answering. Our experiments demonstrate our ability to generalize to unseen images of learned concepts while preserving the model behavior on unrelated inputs.",
  "published": "2024-03-21",
  "updated": "2024-03-21",
  "year": "2024",
  "authors": [
   "Yuval Alaluf",
   "Elad Richardson",
   "Sergey Tulyakov",
   "Kfir Aberman",
   "Daniel Cohen-Or"
  ],
  "author_count": 5,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "ECCV",
  "venue_source": "semantic-scholar",
  "citations": 69,
  "influential_citations": 14,
  "tldr": "This work takes a first step toward the personalization of VLMs, enabling them to learn and reason over user-provided concepts, and demonstrates the ability to generalize to unseen images of learned concepts while preserving the model behavior on unrelated inputs.",
  "doi": "10.48550/arXiv.2403.14599",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yuval Alaluf",
    "id": "1850630812",
    "h_index": 18,
    "papers": 27
   },
   {
    "name": "Elad Richardson",
    "id": "2511847",
    "h_index": 14,
    "papers": 27
   },
   {
    "name": "Sergey Tulyakov",
    "id": "2292401534",
    "h_index": 17,
    "papers": 63
   },
   {
    "name": "Kfir Aberman",
    "id": "3451442",
    "h_index": 31,
    "papers": 60
   },
   {
    "name": "D. Cohen-Or",
    "id": "2257215972",
    "h_index": 16,
    "papers": 31
   }
  ],
  "comment": "Project page: https://snap-research.github.io/MyVLM/",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2403.14599v1",
  "pdf_url": "https://arxiv.org/pdf/2403.14599v1",
  "html_url": "https://arxiv.org/html/2403.14599v1",
  "code_url": "https://snap-research.github.io/MyVLM/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.35
 },
 {
  "id": "2403.12945",
  "slug": "droid-a-large-scale-in-the-wild-robot-manipulation-dataset",
  "title": "DROID: A Large-Scale In-The-Wild Robot Manipulation Dataset",
  "abstract": "The creation of large, diverse, high-quality robot manipulation datasets is an important stepping stone on the path toward more capable and robust robotic manipulation policies. However, creating such datasets is challenging: collecting robot manipulation data in diverse environments poses logistical and safety challenges and requires substantial investments in hardware and human labour. As a result, even the most general robot manipulation policies today are mostly trained on data collected in a small number of environments with limited scene and task diversity. In this work, we introduce DROID (Distributed Robot Interaction Dataset), a diverse robot manipulation dataset with 76k demonstration trajectories or 350 hours of interaction data, collected across 564 scenes and 84 tasks by 50 data collectors in North America, Asia, and Europe over the course of 12 months. We demonstrate that training with DROID leads to policies with higher performance and improved generalization ability. We open source the full dataset, policy learning code, and a detailed guide for reproducing our robot hardware setup.",
  "published": "2024-03-19",
  "updated": "2025-04-22",
  "year": "2024",
  "authors": [
   "Alexander Khazatsky",
   "Karl Pertsch",
   "Suraj Nair",
   "Ashwin Balakrishna",
   "Sudeep Dasari",
   "Siddharth Karamcheti",
   "Soroush Nasiriany",
   "Mohan Kumar Srirama",
   "Lawrence Yunliang Chen",
   "Kirsty Ellis",
   "Peter David Fagan",
   "Joey Hejna",
   "Masha Itkina",
   "Marion Lepert",
   "Yecheng Jason Ma",
   "Patrick Tree Miller",
   "Jimmy Wu",
   "Suneel Belkhale",
   "Shivin Dass",
   "Huy Ha",
   "Arhan Jain",
   "Abraham Lee",
   "Youngwoon Lee",
   "Marius Memmel",
   "Sungjae Park",
   "Ilija Radosavovic",
   "Kaiyuan Wang",
   "Albert Zhan",
   "Kevin Black",
   "Cheng Chi",
   "Kyle Beltran Hatch",
   "Shan Lin",
   "Jingpei Lu",
   "Jean Mercat",
   "Abdul Rehman",
   "Pannag R Sanketi",
   "Archit Sharma",
   "Cody Simpson",
   "Quan Vuong",
   "Homer Rich Walke",
   "Blake Wulfe",
   "Ted Xiao",
   "Jonathan Heewon Yang",
   "Arefeh Yavary",
   "Tony Z. Zhao",
   "Christopher Agia",
   "Rohan Baijal",
   "Mateo Guaman Castro",
   "Daphne Chen",
   "Qiuyu Chen",
   "Trinity Chung",
   "Jaimyn Drake",
   "Ethan Paul Foster",
   "Jensen Gao",
   "Vitor Guizilini",
   "David Antonio Herrera",
   "Minho Heo",
   "Kyle Hsu",
   "Jiaheng Hu",
   "Muhammad Zubair Irshad",
   "Donovon Jackson",
   "Charlotte Le",
   "Yunshuang Li",
   "Kevin Lin",
   "Roy Lin",
   "Zehan Ma",
   "Abhiram Maddukuri",
   "Suvir Mirchandani",
   "Daniel Morton",
   "Tony Nguyen",
   "Abigail O'Neill",
   "Rosario Scalise",
   "Derick Seale",
   "Victor Son",
   "Stephen Tian",
   "Emi Tran",
   "Andrew E. Wang",
   "Yilin Wu",
   "Annie Xie",
   "Jingyun Yang",
   "Patrick Yin",
   "Yunchu Zhang",
   "Osbert Bastani",
   "Glen Berseth",
   "Jeannette Bohg",
   "Ken Goldberg",
   "Abhinav Gupta",
   "Abhishek Gupta",
   "Dinesh Jayaraman",
   "Joseph J Lim",
   "Jitendra Malik",
   "Roberto Mart\u00edn-Mart\u00edn",
   "Subramanian Ramamoorthy",
   "Dorsa Sadigh",
   "Shuran Song",
   "Jiajun Wu",
   "Michael C. Yip",
   "Yuke Zhu",
   "Thomas Kollar",
   "Sergey Levine",
   "Chelsea Finn"
  ],
  "author_count": 101,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "RSS",
  "venue_source": "semantic-scholar",
  "citations": 991,
  "influential_citations": 86,
  "tldr": "This work introduces DROID (Distributed Robot Interaction Dataset), a diverse robot manipulation dataset with 76k demonstration trajectories or 350 hours of interaction data, collected across 564 scenes and 84 tasks by 50 data collectors in North America, Asia, and Europe over the course of 12 months.",
  "doi": "10.48550/arXiv.2403.12945",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Alexander Khazatsky",
    "id": "121873407",
    "h_index": 10,
    "papers": 11
   },
   {
    "name": "Karl Pertsch",
    "id": "31719101",
    "h_index": 32,
    "papers": 56
   },
   {
    "name": "S. Nair",
    "id": "2070674420",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Ashwin Balakrishna",
    "id": "2057483815",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "S. Dasari",
    "id": "36076404",
    "h_index": 23,
    "papers": 39
   },
   {
    "name": "Siddharth Karamcheti",
    "id": "10737060",
    "h_index": 21,
    "papers": 38
   },
   {
    "name": "Soroush Nasiriany",
    "id": "3457048",
    "h_index": 18,
    "papers": 24
   },
   {
    "name": "M. K. Srirama",
    "id": "2193493900",
    "h_index": 12,
    "papers": 14
   },
   {
    "name": "L. Chen",
    "id": "2143804724",
    "h_index": 14,
    "papers": 24
   },
   {
    "name": "Kirsty Ellis",
    "id": "2292155661",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "P. Fagan",
    "id": "2069362314",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Joey Hejna",
    "id": "2122700519",
    "h_index": 17,
    "papers": 27
   },
   {
    "name": "Masha Itkina",
    "id": "30112153",
    "h_index": 16,
    "papers": 31
   },
   {
    "name": "Marion Lepert",
    "id": "10710717",
    "h_index": 8,
    "papers": 11
   },
   {
    "name": "Y. Ma",
    "id": "2130215451",
    "h_index": 16,
    "papers": 23
   },
   {
    "name": "Patrick Miller",
    "id": "2292161974",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Jimmy Wu",
    "id": "2261393585",
    "h_index": 7,
    "papers": 8
   },
   {
    "name": "Suneel Belkhale",
    "id": "69879999",
    "h_index": 18,
    "papers": 25
   },
   {
    "name": "S. Dass",
    "id": "2053057249",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Huy Ha",
    "id": "2291134164",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Arhan Jain",
    "id": "2292386654",
    "h_index": 7,
    "papers": 9
   },
   {
    "name": "Abraham Lee",
    "id": "2233425761",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Youngwoon Lee",
    "id": "2437201516",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Marius Memmel",
    "id": "2120036970",
    "h_index": 11,
    "papers": 16
   },
   {
    "name": "Sungjae Park",
    "id": "2371412504",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Ilija Radosavovic",
    "id": "30407997",
    "h_index": 20,
    "papers": 22
   },
   {
    "name": "Kaiyuan Wang",
    "id": "2242971934",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Albert Zhan",
    "id": "1491380866",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Kevin Black",
    "id": "2292143938",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Cheng Chi",
    "id": "2052795966",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "K. Hatch",
    "id": "2054754929",
    "h_index": 8,
    "papers": 13
   },
   {
    "name": "Shan Lin",
    "id": "2242528006",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Jingpei Lu",
    "id": "153155220",
    "h_index": 12,
    "papers": 26
   },
   {
    "name": "Jean-Pierre Mercat",
    "id": "72847120",
    "h_index": 10,
    "papers": 33
   },
   {
    "name": "Abdul Rehman",
    "id": "2064048397",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Pannag R. Sanketi",
    "id": "2840758",
    "h_index": 22,
    "papers": 44
   },
   {
    "name": "Archit Sharma",
    "id": "50465276",
    "h_index": 22,
    "papers": 36
   },
   {
    "name": "C. Simpson",
    "id": "2064665262",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Q. V\u01b0\u01a1ng",
    "id": "2273025166",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "H. Walke",
    "id": "2029241116",
    "h_index": 17,
    "papers": 23
   },
   {
    "name": "Blake Wulfe",
    "id": "9414028",
    "h_index": 12,
    "papers": 17
   },
   {
    "name": "Ted Xiao",
    "id": "9961095",
    "h_index": 33,
    "papers": 45
   },
   {
    "name": "J. Yang",
    "id": "2274728028",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Arefeh Yavary",
    "id": "50811874",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Tony Zhao",
    "id": "145914976",
    "h_index": 17,
    "papers": 19
   },
   {
    "name": "Christopher Agia",
    "id": "2280140879",
    "h_index": 8,
    "papers": 14
   },
   {
    "name": "R. Baijal",
    "id": "2266194504",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Mateo Guaman Castro",
    "id": "1384145601",
    "h_index": 7,
    "papers": 12
   },
   {
    "name": "D. Chen",
    "id": "153642310",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Qiuyu Chen",
    "id": "2292183651",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "T. Chung",
    "id": "2265754339",
    "h_index": 4,
    "papers": 14
   },
   {
    "name": "Jaimyn Drake",
    "id": "2265754459",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "E. P. Foster",
    "id": "2130616918",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Jensen Gao",
    "id": "2238154243",
    "h_index": 8,
    "papers": 10
   },
   {
    "name": "D. Herrera",
    "id": "2064630980",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Minho Heo",
    "id": "2218050062",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Kyle Hsu",
    "id": "2286719009",
    "h_index": 8,
    "papers": 15
   },
   {
    "name": "Jiaheng Hu",
    "id": "81703072",
    "h_index": 13,
    "papers": 29
   },
   {
    "name": "Muhammad Zubair Irshad",
    "id": "147495445",
    "h_index": 17,
    "papers": 37
   },
   {
    "name": "Donovon Jackson",
    "id": "2292188317",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Charlotte Le",
    "id": "2265754058",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Yunshuang Li",
    "id": "2382432559",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "K. Lin",
    "id": "2107438214",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Roy Lin",
    "id": "2068172426",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Zehan Ma",
    "id": "2292215354",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Abhiram Maddukuri",
    "id": "2292182414",
    "h_index": 7,
    "papers": 9
   },
   {
    "name": "Suvir Mirchandani",
    "id": "2247302",
    "h_index": 13,
    "papers": 21
   },
   {
    "name": "D. Morton",
    "id": "2082234132",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Tony Nguyen",
    "id": "2349547163",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Abigail O'Neill",
    "id": "2292155800",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "R. Scalise",
    "id": "115694887",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Derick Seale",
    "id": "2292183089",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "V. Son",
    "id": "2292182411",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Stephen Tian",
    "id": "2264968694",
    "h_index": 9,
    "papers": 14
   },
   {
    "name": "E. Tran",
    "id": "2064452058",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Andrew Wang",
    "id": "2292259868",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Y. Wu",
    "id": "2237807662",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Annie Xie",
    "id": "14484808",
    "h_index": 21,
    "papers": 23
   },
   {
    "name": "Jingyun Yang",
    "id": "2261555773",
    "h_index": 9,
    "papers": 10
   },
   {
    "name": "Patrick Yin",
    "id": "2163582683",
    "h_index": 10,
    "papers": 12
   },
   {
    "name": "Yunchu Zhang",
    "id": "2260344636",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "O. Bastani",
    "id": "1697444",
    "h_index": 40,
    "papers": 186
   },
   {
    "name": "Glen Berseth",
    "id": "2312919053",
    "h_index": 15,
    "papers": 28
   },
   {
    "name": "Jeannette Bohg",
    "id": "1775407",
    "h_index": 50,
    "papers": 161
   },
   {
    "name": "Ken Goldberg",
    "id": "2253727638",
    "h_index": 8,
    "papers": 13
   },
   {
    "name": "Abhinav Gupta",
    "id": "2218443075",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Abhishek Gupta",
    "id": "2257350284",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "D. Jayaraman",
    "id": "2257346533",
    "h_index": 9,
    "papers": 12
   },
   {
    "name": "Joseph J. Lim",
    "id": "2253826001",
    "h_index": 7,
    "papers": 9
   },
   {
    "name": "Jitendra Malik",
    "id": "2285642433",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Roberto Mart'in-Mart'in",
    "id": "2065917078",
    "h_index": 20,
    "papers": 32
   },
   {
    "name": "S. Ramamoorthy",
    "id": "2067107661",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Dorsa Sadigh",
    "id": "1779671",
    "h_index": 69,
    "papers": 226
   },
   {
    "name": "Shuran Song",
    "id": "2110600471",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Jiajun Wu",
    "id": "2326444500",
    "h_index": 35,
    "papers": 214
   },
   {
    "name": "Michael C. Yip",
    "id": "2065643936",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Yuke Zhu",
    "id": "2253507326",
    "h_index": 16,
    "papers": 20
   },
   {
    "name": "T. Kollar",
    "id": "145141630",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Sergey Levine",
    "id": "2249615151",
    "h_index": 32,
    "papers": 53
   },
   {
    "name": "Chelsea Finn",
    "id": "2257346440",
    "h_index": 23,
    "papers": 32
   }
  ],
  "comment": "Project website: https://droid-dataset.github.io/",
  "topics": [
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2403.12945v2",
  "pdf_url": "https://arxiv.org/pdf/2403.12945v2",
  "html_url": "https://arxiv.org/html/2403.12945v2",
  "code_url": "https://droid-dataset.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.5
 },
 {
  "id": "2403.12943",
  "slug": "vid2robot-end-to-end-video-conditioned-policy-learning-with-cross-atte",
  "title": "Vid2Robot: End-to-end Video-conditioned Policy Learning with Cross-Attention Transformers",
  "abstract": "Large-scale multi-task robotic manipulation systems often rely on text to specify the task. In this work, we explore whether a robot can learn by observing humans. To do so, the robot must understand a person's intent and perform the inferred task despite differences in the embodiments and environments. We introduce Vid2Robot, an end-to-end video-conditioned policy that takes human videos demonstrating manipulation tasks as input and produces robot actions. Our model is trained with a large dataset of prompt video-robot trajectory pairs to learn unified representations of human and robot actions from videos. Vid2Robot uses cross-attention transformer layers between video features and the current robot state to produce the actions and perform the same task as shown in the video. We use auxiliary contrastive losses to align the prompt and robot video representations for better policies. We evaluate Vid2Robot on real-world robots and observe over 20% improvement over BC-Z when using human prompt videos. Further, we also show cross-object motion transfer ability that enables video-conditioned policies to transfer a motion observed on one object in the prompt video to another object in the robot's own environment. Videos available at https://vid2robot.github.io",
  "published": "2024-03-19",
  "updated": "2024-08-27",
  "year": "2024",
  "authors": [
   "Vidhi Jain",
   "Maria Attarian",
   "Nikhil J Joshi",
   "Ayzaan Wahid",
   "Danny Driess",
   "Quan Vuong",
   "Pannag R Sanketi",
   "Pierre Sermanet",
   "Stefan Welker",
   "Christine Chan",
   "Igor Gilitschenski",
   "Yonatan Bisk",
   "Debidatta Dwibedi"
  ],
  "author_count": 13,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "RSS",
  "venue_source": "semantic-scholar",
  "citations": 69,
  "influential_citations": 5,
  "tldr": "This work introduces Vid2Robot, an end-to-end video-conditioned policy that takes human videos demonstrating manipulation tasks as input and produces robot actions and shows cross-object motion transfer ability that enables video-conditioned policies to transfer a motion observed on one object in the prompt video to another object in the robot's own environment.",
  "doi": "10.48550/arXiv.2403.12943",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Vidhi Jain",
    "id": "2253472236",
    "h_index": 9,
    "papers": 17
   },
   {
    "name": "Maria Attarian",
    "id": "51922893",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "Nikhil J. Joshi",
    "id": "2052368480",
    "h_index": 11,
    "papers": 14
   },
   {
    "name": "Ayzaan Wahid",
    "id": "88728227",
    "h_index": 21,
    "papers": 27
   },
   {
    "name": "Danny Driess",
    "id": "2283848260",
    "h_index": 27,
    "papers": 35
   },
   {
    "name": "Quan Vuong",
    "id": "2288210223",
    "h_index": 19,
    "papers": 29
   },
   {
    "name": "Pannag R. Sanketi",
    "id": "2840758",
    "h_index": 22,
    "papers": 44
   },
   {
    "name": "P. Sermanet",
    "id": "3142556",
    "h_index": 39,
    "papers": 77
   },
   {
    "name": "Stefan Welker",
    "id": "69426588",
    "h_index": 13,
    "papers": 17
   },
   {
    "name": "Christine Chan",
    "id": "2256938625",
    "h_index": 8,
    "papers": 8
   },
   {
    "name": "Igor Gilitschenski",
    "id": "2261320095",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Yonatan Bisk",
    "id": "3312309",
    "h_index": 46,
    "papers": 144
   },
   {
    "name": "Debidatta Dwibedi",
    "id": "2420123",
    "h_index": 19,
    "papers": 35
   }
  ],
  "comment": "Robotics: Science & Systems (RSS) 2024. https://vid2robot.github.io/",
  "topics": [
   "egocentric-data"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2403.12943v2",
  "pdf_url": "https://arxiv.org/pdf/2403.12943v2",
  "html_url": "https://arxiv.org/html/2403.12943v2",
  "code_url": "https://vid2robot.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.35
 },
 {
  "id": "2403.12365",
  "slug": "gaussianflow-splatting-gaussian-dynamics-for-4d-content-creation",
  "title": "GaussianFlow: Splatting Gaussian Dynamics for 4D Content Creation",
  "abstract": "Creating 4D fields of Gaussian Splatting from images or videos is a challenging task due to its under-constrained nature. While the optimization can draw photometric reference from the input videos or be regulated by generative models, directly supervising Gaussian motions remains underexplored. In this paper, we introduce a novel concept, Gaussian flow, which connects the dynamics of 3D Gaussians and pixel velocities between consecutive frames. The Gaussian flow can be efficiently obtained by splatting Gaussian dynamics into the image space. This differentiable process enables direct dynamic supervision from optical flow. Our method significantly benefits 4D dynamic content generation and 4D novel view synthesis with Gaussian Splatting, especially for contents with rich motions that are hard to be handled by existing methods. The common color drifting issue that happens in 4D generation is also resolved with improved Guassian dynamics. Superior visual quality on extensive experiments demonstrates our method's effectiveness. Quantitative and qualitative evaluations show that our method achieves state-of-the-art results on both tasks of 4D generation and 4D novel view synthesis. Project page: https://zerg-overmind.github.io/GaussianFlow.github.io/",
  "published": "2024-03-19",
  "updated": "2024-05-13",
  "year": "2024",
  "authors": [
   "Quankai Gao",
   "Qiangeng Xu",
   "Zhe Cao",
   "Ben Mildenhall",
   "Wenchao Ma",
   "Le Chen",
   "Danhang Tang",
   "Ulrich Neumann"
  ],
  "author_count": 8,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "Trans. Mach. Learn. Res.",
  "venue_source": "semantic-scholar",
  "citations": 127,
  "influential_citations": 10,
  "tldr": "A novel concept, Gaussian flow, which connects the dynamics of 3D Gaussians and pixel velocities between consecutive frames between consecutive frames is introduced, which enables direct dynamic supervision from optical flow.",
  "doi": "10.48550/arXiv.2403.12365",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Quankai Gao",
    "id": "2052543032",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Qiangeng Xu",
    "id": "12601304",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Zhe Cao",
    "id": "2292259836",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "B. Mildenhall",
    "id": "2577533",
    "h_index": 37,
    "papers": 53
   },
   {
    "name": "Wenchao Ma",
    "id": "2292662135",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Le Chen",
    "id": "2146071097",
    "h_index": 8,
    "papers": 25
   },
   {
    "name": "Danhang Tang",
    "id": "2268494923",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Ulrich Neumann",
    "id": "2292198671",
    "h_index": 2,
    "papers": 5
   }
  ],
  "comment": "",
  "topics": [
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2403.12365v2",
  "pdf_url": "https://arxiv.org/pdf/2403.12365v2",
  "html_url": "https://arxiv.org/html/2403.12365v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.61
 },
 {
  "id": "2403.12037",
  "slug": "minedreamer-learning-to-follow-instructions-via-chain-of-imagination-f",
  "title": "MineDreamer: Learning to Follow Instructions via Chain-of-Imagination for Simulated-World Control",
  "abstract": "It is a long-lasting goal to design a generalist-embodied agent that can follow diverse instructions in human-like ways. However, existing approaches often fail to steadily follow instructions due to difficulties in understanding abstract and sequential natural language instructions. To this end, we introduce MineDreamer, an open-ended embodied agent built upon the challenging Minecraft simulator with an innovative paradigm that enhances instruction-following ability in low-level control signal generation. Specifically, MineDreamer is developed on top of recent advances in Multimodal Large Language Models (MLLMs) and diffusion models, and we employ a Chain-of-Imagination (CoI) mechanism to envision the step-by-step process of executing instructions and translating imaginations into more precise visual prompts tailored to the current state; subsequently, the agent generates keyboard-and-mouse actions to efficiently achieve these imaginations, steadily following the instructions at each step. Extensive experiments demonstrate that MineDreamer follows single and multi-step instructions steadily, significantly outperforming the best generalist agent baseline and nearly doubling its performance. Moreover, qualitative analysis of the agent's imaginative ability reveals its generalization and comprehension of the open world.",
  "published": "2024-03-18",
  "updated": "2024-03-19",
  "year": "2024",
  "authors": [
   "Enshen Zhou",
   "Yiran Qin",
   "Zhenfei Yin",
   "Yuzhou Huang",
   "Ruimao Zhang",
   "Lu Sheng",
   "Yu Qiao",
   "Jing Shao"
  ],
  "author_count": 8,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 59,
  "influential_citations": 2,
  "tldr": "MineDreamer is an open-ended embodied agent built upon the challenging Minecraft simulator with an innovative paradigm that enhances instruction-following ability in low-level control signal generation and qualitative analysis of the agent's imaginative ability reveals its generalization and comprehension of the open world.",
  "doi": "10.48550/arXiv.2403.12037",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Enshen Zhou",
    "id": "2273688063",
    "h_index": 9,
    "papers": 20
   },
   {
    "name": "Yiran Qin",
    "id": "2240266091",
    "h_index": 7,
    "papers": 20
   },
   {
    "name": "Zhen-fei Yin",
    "id": "13050405",
    "h_index": 19,
    "papers": 37
   },
   {
    "name": "Yuzhou Huang",
    "id": "2273812462",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Ruimao Zhang",
    "id": "2274031380",
    "h_index": 7,
    "papers": 15
   },
   {
    "name": "Lu Sheng",
    "id": "2290983876",
    "h_index": 8,
    "papers": 16
   },
   {
    "name": "Yu Qiao",
    "id": "2265493981",
    "h_index": 12,
    "papers": 14
   },
   {
    "name": "Jing Shao",
    "id": "2254280929",
    "h_index": 13,
    "papers": 17
   }
  ],
  "comment": "Project page: https://sites.google.com/view/minedreamer/main",
  "topics": [
   "world-models",
   "sim2real",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2403.12037v2",
  "pdf_url": "https://arxiv.org/pdf/2403.12037v2",
  "html_url": "https://arxiv.org/html/2403.12037v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.78
 },
 {
  "id": "2403.09631",
  "slug": "3d-vla-a-3d-vision-language-action-generative-world-model",
  "title": "3D-VLA: A 3D Vision-Language-Action Generative World Model",
  "abstract": "Recent vision-language-action (VLA) models rely on 2D inputs, lacking integration with the broader realm of the 3D physical world. Furthermore, they perform action prediction by learning a direct mapping from perception to action, neglecting the vast dynamics of the world and the relations between actions and dynamics. In contrast, human beings are endowed with world models that depict imagination about future scenarios to plan actions accordingly. To this end, we propose 3D-VLA by introducing a new family of embodied foundation models that seamlessly link 3D perception, reasoning, and action through a generative world model. Specifically, 3D-VLA is built on top of a 3D-based large language model (LLM), and a set of interaction tokens is introduced to engage with the embodied environment. Furthermore, to inject generation abilities into the model, we train a series of embodied diffusion models and align them into the LLM for predicting the goal images and point clouds. To train our 3D-VLA, we curate a large-scale 3D embodied instruction dataset by extracting vast 3D-related information from existing robotics datasets. Our experiments on held-in datasets demonstrate that 3D-VLA significantly improves the reasoning, multimodal generation, and planning capabilities in embodied environments, showcasing its potential in real-world applications.",
  "published": "2024-03-14",
  "updated": "2024-03-14",
  "year": "2024",
  "authors": [
   "Haoyu Zhen",
   "Xiaowen Qiu",
   "Peihao Chen",
   "Jincheng Yang",
   "Xin Yan",
   "Yilun Du",
   "Yining Hong",
   "Chuang Gan"
  ],
  "author_count": 8,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.CL",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "ICML",
  "venue_source": "semantic-scholar",
  "citations": 385,
  "influential_citations": 18,
  "tldr": "3D-VLA is proposed by introducing a new family of embodied foundation models that seamlessly link 3D perception, reasoning, and action through a generative world model and significantly improves the reasoning, multimodal generation, and planning capabilities in embodied environments.",
  "doi": "10.48550/arXiv.2403.09631",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Haoyu Zhen",
    "id": "2226286961",
    "h_index": 7,
    "papers": 14
   },
   {
    "name": "Xiaowen Qiu",
    "id": "2292176508",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Peihao Chen",
    "id": "2158502526",
    "h_index": 21,
    "papers": 34
   },
   {
    "name": "Jincheng Yang",
    "id": "2291198667",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Xin Yan",
    "id": "2291325237",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Yilun Du",
    "id": "2258799458",
    "h_index": 12,
    "papers": 18
   },
   {
    "name": "Yining Hong",
    "id": "2265627123",
    "h_index": 6,
    "papers": 6
   },
   {
    "name": "Chuang Gan",
    "id": "2266854520",
    "h_index": 5,
    "papers": 13
   }
  ],
  "comment": "Project page: https://vis-www.cs.umass.edu/3dvla/",
  "topics": [
   "world-models",
   "vla",
   "spatial-3d",
   "foundation-pretraining",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2403.09631v1",
  "pdf_url": "https://arxiv.org/pdf/2403.09631v1",
  "html_url": "https://arxiv.org/html/2403.09631v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.09
 },
 {
  "id": "2403.09227",
  "slug": "behavior-1k-a-human-centered-embodied-ai-benchmark-with-1-000-everyday",
  "title": "BEHAVIOR-1K: A Human-Centered, Embodied AI Benchmark with 1,000 Everyday Activities and Realistic Simulation",
  "abstract": "We present BEHAVIOR-1K, a comprehensive simulation benchmark for human-centered robotics. BEHAVIOR-1K includes two components, guided and motivated by the results of an extensive survey on \"what do you want robots to do for you?\". The first is the definition of 1,000 everyday activities, grounded in 50 scenes (houses, gardens, restaurants, offices, etc.) with more than 9,000 objects annotated with rich physical and semantic properties. The second is OMNIGIBSON, a novel simulation environment that supports these activities via realistic physics simulation and rendering of rigid bodies, deformable bodies, and liquids. Our experiments indicate that the activities in BEHAVIOR-1K are long-horizon and dependent on complex manipulation skills, both of which remain a challenge for even state-of-the-art robot learning solutions. To calibrate the simulation-to-reality gap of BEHAVIOR-1K, we provide an initial study on transferring solutions learned with a mobile manipulator in a simulated apartment to its real-world counterpart. We hope that BEHAVIOR-1K's human-grounded nature, diversity, and realism make it valuable for embodied AI and robot learning research. Project website: https://behavior.stanford.edu.",
  "published": "2024-03-14",
  "updated": "2024-03-14",
  "year": "2024",
  "authors": [
   "Chengshu Li",
   "Ruohan Zhang",
   "Josiah Wong",
   "Cem Gokmen",
   "Sanjana Srivastava",
   "Roberto Mart\u00edn-Mart\u00edn",
   "Chen Wang",
   "Gabrael Levine",
   "Wensi Ai",
   "Benjamin Martinez",
   "Hang Yin",
   "Michael Lingelbach",
   "Minjune Hwang",
   "Ayano Hiranaka",
   "Sujay Garlanka",
   "Arman Aydin",
   "Sharon Lee",
   "Jiankai Sun",
   "Mona Anvari",
   "Manasi Sharma",
   "Dhruva Bansal",
   "Samuel Hunter",
   "Kyu-Young Kim",
   "Alan Lou",
   "Caleb R Matthews",
   "Ivan Villa-Renteria",
   "Jerry Huayang Tang",
   "Claire Tang",
   "Fei Xia",
   "Yunzhu Li",
   "Silvio Savarese",
   "Hyowon Gweon",
   "C. Karen Liu",
   "Jiajun Wu",
   "Li Fei-Fei"
  ],
  "author_count": 35,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL 2022",
  "venue_source": "arxiv-comment",
  "citations": 169,
  "influential_citations": 17,
  "tldr": "BEHAVIOR-1K's human-grounded nature, diversity, and realism make it valuable for embodied AI and robot learning research, and it is hoped that its human-grounded nature, diversity, and realism make it valuable for embodied AI and robot learning research.",
  "doi": "10.48550/arXiv.2403.09227",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Chengshu Li",
    "id": "2128669132",
    "h_index": 16,
    "papers": 22
   },
   {
    "name": "Ruohan Zhang",
    "id": "2248277538",
    "h_index": 7,
    "papers": 16
   },
   {
    "name": "Josiah Wong",
    "id": "33808086",
    "h_index": 11,
    "papers": 13
   },
   {
    "name": "Cem Gokmen",
    "id": "46217329",
    "h_index": 10,
    "papers": 12
   },
   {
    "name": "S. Srivastava",
    "id": "18241595",
    "h_index": 12,
    "papers": 20
   },
   {
    "name": "Roberto Mart\u00edn-Mart\u00edn",
    "id": "1382655067",
    "h_index": 27,
    "papers": 39
   },
   {
    "name": "Chen Wang",
    "id": "2292199635",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Gabrael Levine",
    "id": "2152650854",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Wensi Ai",
    "id": "2248064864",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "B. Martinez",
    "id": "2292086939",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Hang Yin",
    "id": "2292126834",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Michael Lingelbach",
    "id": "1802680741",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Minjune Hwang",
    "id": "2052264905",
    "h_index": 7,
    "papers": 12
   },
   {
    "name": "Ayano Hiranaka",
    "id": "2198501824",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "Sujay Garlanka",
    "id": "2210189725",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Arman Aydin",
    "id": "2161564780",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Sharon Lee",
    "id": "2226133780",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Jiankai Sun",
    "id": "2282025",
    "h_index": 22,
    "papers": 61
   },
   {
    "name": "M. Anvari",
    "id": "2878331",
    "h_index": 14,
    "papers": 24
   },
   {
    "name": "Manasi Sharma",
    "id": "2198628811",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Dhruva Bansal",
    "id": "146465102",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Samuel Hunter",
    "id": "2198504233",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Kyu-Young Kim",
    "id": "2110659540",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Alan Lou",
    "id": "2121388205",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Caleb R. Matthews",
    "id": "1520021462",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Ivan Villa-Renteria",
    "id": "2130174838",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "J. Tang",
    "id": "2292090656",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Claire Tang",
    "id": "1799405777",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Fei Xia",
    "id": "2273899492",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Yunzhu Li",
    "id": "2294926592",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Silvio Savarese",
    "id": "2238207181",
    "h_index": 31,
    "papers": 114
   },
   {
    "name": "H. Gweon",
    "id": "2764049",
    "h_index": 32,
    "papers": 141
   },
   {
    "name": "C. K. Liu",
    "id": "2247934447",
    "h_index": 9,
    "papers": 11
   },
   {
    "name": "Jiajun Wu",
    "id": "3045089",
    "h_index": 80,
    "papers": 228
   },
   {
    "name": "Fei-Fei Li",
    "id": "2238030496",
    "h_index": 7,
    "papers": 8
   },
   {
    "name": "S. Research",
    "id": "2208653278",
    "h_index": 2,
    "papers": 3
   }
  ],
  "comment": "A preliminary version was published at 6th Conference on Robot Learning (CoRL 2022)",
  "topics": [
   "sim2real"
  ],
  "orgs": [
   "Stanford"
  ],
  "abs_url": "https://arxiv.org/abs/2403.09227v1",
  "pdf_url": "https://arxiv.org/pdf/2403.09227v1",
  "html_url": "https://arxiv.org/html/2403.09227v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.23
 },
 {
  "id": "2403.08716",
  "slug": "difftactile-a-physics-based-differentiable-tactile-simulator-for-conta",
  "title": "DIFFTACTILE: A Physics-based Differentiable Tactile Simulator for Contact-rich Robotic Manipulation",
  "abstract": "We introduce DIFFTACTILE, a physics-based differentiable tactile simulation system designed to enhance robotic manipulation with dense and physically accurate tactile feedback. In contrast to prior tactile simulators which primarily focus on manipulating rigid bodies and often rely on simplified approximations to model stress and deformations of materials in contact, DIFFTACTILE emphasizes physics-based contact modeling with high fidelity, supporting simulations of diverse contact modes and interactions with objects possessing a wide range of material properties. Our system incorporates several key components, including a Finite Element Method (FEM)-based soft body model for simulating the sensing elastomer, a multi-material simulator for modeling diverse object types (such as elastic, elastoplastic, cables) under manipulation, a penalty-based contact model for handling contact dynamics. The differentiable nature of our system facilitates gradient-based optimization for both 1) refining physical properties in simulation using real-world data, hence narrowing the sim-to-real gap and 2) efficient learning of tactile-assisted grasping and contact-rich manipulation skills. Additionally, we introduce a method to infer the optical response of our tactile sensor to contact using an efficient pixel-based neural module. We anticipate that DIFFTACTILE will serve as a useful platform for studying contact-rich manipulations, leveraging the benefits of dense tactile feedback and differentiable physics. Code and supplementary materials are available at the project website https://difftactile.github.io/.",
  "published": "2024-03-13",
  "updated": "2024-03-13",
  "year": "2024",
  "authors": [
   "Zilin Si",
   "Gu Zhang",
   "Qingwei Ben",
   "Branden Romero",
   "Zhou Xian",
   "Chao Liu",
   "Chuang Gan"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 57,
  "influential_citations": 4,
  "tldr": "DIFFTACTILE is introduced, a physics-based differentiable tactile simulation system designed to enhance robotic manipulation with dense and physically accurate tactile feedback, and a method to infer the optical response of the tactile sensor to contact using an efficient pixel-based neural module.",
  "doi": "10.48550/arXiv.2403.08716",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zilin Si",
    "id": "2257003048",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Gu Zhang",
    "id": "2291085400",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Qingwei Ben",
    "id": "2293395502",
    "h_index": 11,
    "papers": 16
   },
   {
    "name": "Branden Romero",
    "id": "21201572",
    "h_index": 9,
    "papers": 10
   },
   {
    "name": "Zhou Xian",
    "id": "2291070250",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Chao Liu",
    "id": "2291316108",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Chuang Gan",
    "id": "2291068301",
    "h_index": 6,
    "papers": 11
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "sim2real"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2403.08716v1",
  "pdf_url": "https://arxiv.org/pdf/2403.08716v1",
  "html_url": "https://arxiv.org/html/2403.08716v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.26
 },
 {
  "id": "2403.07870",
  "slug": "open-teach-a-versatile-teleoperation-system-for-robotic-manipulation",
  "title": "OPEN TEACH: A Versatile Teleoperation System for Robotic Manipulation",
  "abstract": "Open-sourced, user-friendly tools form the bedrock of scientific advancement across disciplines. The widespread adoption of data-driven learning has led to remarkable progress in multi-fingered dexterity, bimanual manipulation, and applications ranging from logistics to home robotics. However, existing data collection platforms are often proprietary, costly, or tailored to specific robotic morphologies. We present OPEN TEACH, a new teleoperation system leveraging VR headsets to immerse users in mixed reality for intuitive robot control. Built on the affordable Meta Quest 3, which costs $500, OPEN TEACH enables real-time control of various robots, including multi-fingered hands and bimanual arms, through an easy-to-use app. Using natural hand gestures and movements, users can manipulate robots at up to 90Hz with smooth visual feedback and interface widgets offering closeup environment views. We demonstrate the versatility of OPEN TEACH across 38 tasks on different robots. A comprehensive user study indicates significant improvement in teleoperation capability over the AnyTeleop framework. Further experiments exhibit that the collected data is compatible with policy learning on 10 dexterous and contact-rich manipulation tasks. Currently supporting Franka, xArm, Jaco, and Allegro platforms, OPEN TEACH is fully open-sourced to promote broader adoption. Videos are available at https://open-teach.github.io/.",
  "published": "2024-03-12",
  "updated": "2024-03-12",
  "year": "2024",
  "authors": [
   "Aadhithya Iyer",
   "Zhuoran Peng",
   "Yinlong Dai",
   "Irmak Guzey",
   "Siddhant Haldar",
   "Soumith Chintala",
   "Lerrel Pinto"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 159,
  "influential_citations": 9,
  "tldr": "This work presents OPEN TEACH, a new teleoperation system leveraging VR headsets to immerse users in mixed reality for intuitive robot control and demonstrates the versatility of OPEN TEACH across 38 tasks on different robots.",
  "doi": "10.48550/arXiv.2403.07870",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Aadhithya Iyer",
    "id": "2203910567",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Zhuoran Peng",
    "id": "2291037143",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Yinlong Dai",
    "id": "2243500130",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Irmak G\u00fczey",
    "id": "2212471096",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Siddhant Haldar",
    "id": "51445278",
    "h_index": 13,
    "papers": 21
   },
   {
    "name": "Soumith Chintala",
    "id": "2127604",
    "h_index": 32,
    "papers": 49
   },
   {
    "name": "Lerrel Pinto",
    "id": "2253567347",
    "h_index": 15,
    "papers": 25
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2403.07870v1",
  "pdf_url": "https://arxiv.org/pdf/2403.07870v1",
  "html_url": "https://arxiv.org/html/2403.07870v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.7
 },
 {
  "id": "2403.07869",
  "slug": "telemoma-a-modular-and-versatile-teleoperation-system-for-mobile-manip",
  "title": "TeleMoMa: A Modular and Versatile Teleoperation System for Mobile Manipulation",
  "abstract": "A critical bottleneck limiting imitation learning in robotics is the lack of data. This problem is more severe in mobile manipulation, where collecting demonstrations is harder than in stationary manipulation due to the lack of available and easy-to-use teleoperation interfaces. In this work, we demonstrate TeleMoMa, a general and modular interface for whole-body teleoperation of mobile manipulators. TeleMoMa unifies multiple human interfaces including RGB and depth cameras, virtual reality controllers, keyboard, joysticks, etc., and any combination thereof. In its more accessible version, TeleMoMa works using simply vision (e.g., an RGB-D camera), lowering the entry bar for humans to provide mobile manipulation demonstrations. We demonstrate the versatility of TeleMoMa by teleoperating several existing mobile manipulators - PAL Tiago++, Toyota HSR, and Fetch - in simulation and the real world. We demonstrate the quality of the demonstrations collected with TeleMoMa by training imitation learning policies for mobile manipulation tasks involving synchronized whole-body motion. Finally, we also show that TeleMoMa's teleoperation channel enables teleoperation on site, looking at the robot, or remote, sending commands and observations through a computer network, and perform user studies to evaluate how easy it is for novice users to learn to collect demonstrations with different combinations of human interfaces enabled by our system. We hope TeleMoMa becomes a helpful tool for the community enabling researchers to collect whole-body mobile manipulation demonstrations. For more information and video results, https://robin-lab.cs.utexas.edu/telemoma-web.",
  "published": "2024-03-12",
  "updated": "2024-03-21",
  "year": "2024",
  "authors": [
   "Shivin Dass",
   "Wensi Ai",
   "Yuqian Jiang",
   "Samik Singh",
   "Jiaheng Hu",
   "Ruohan Zhang",
   "Peter Stone",
   "Ben Abbatematteo",
   "Roberto Mart\u00edn-Mart\u00edn"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 54,
  "influential_citations": 4,
  "tldr": "This work demonstrates TeleMoMa, a general and modular interface for whole-body teleoperation of mobile manipulators, and unifies multiple human interfaces including RGB and depth cameras, virtual reality controllers, keyboard, joysticks, etc., and any combination thereof, which helps researchers to collect whole-body mobile manipulation demonstrations.",
  "doi": "10.48550/arXiv.2403.07869",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Shivin Dass",
    "id": "2193057311",
    "h_index": 9,
    "papers": 15
   },
   {
    "name": "Wensi Ai",
    "id": "2248064864",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Yuqian Jiang",
    "id": "48751979",
    "h_index": 14,
    "papers": 33
   },
   {
    "name": "Samik Singh",
    "id": "2291409259",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Jiaheng Hu",
    "id": "81703072",
    "h_index": 13,
    "papers": 29
   },
   {
    "name": "Ruohan Zhang",
    "id": "2248277538",
    "h_index": 7,
    "papers": 16
   },
   {
    "name": "Peter Stone",
    "id": "2290916532",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Ben Abbatematteo",
    "id": "2300094477",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Roberto Mart\u00edn-Mart\u00edn",
    "id": "1382655067",
    "h_index": 27,
    "papers": 39
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "imitation-diffusion",
   "navigation",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2403.07869v2",
  "pdf_url": "https://arxiv.org/pdf/2403.07869v2",
  "html_url": "https://arxiv.org/html/2403.07869v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.74
 },
 {
  "id": "2403.07788",
  "slug": "dexcap-scalable-and-portable-mocap-data-collection-system-for-dexterou",
  "title": "DexCap: Scalable and Portable Mocap Data Collection System for Dexterous Manipulation",
  "abstract": "Imitation learning from human hand motion data presents a promising avenue for imbuing robots with human-like dexterity in real-world manipulation tasks. Despite this potential, substantial challenges persist, particularly with the portability of existing hand motion capture (mocap) systems and the complexity of translating mocap data into effective robotic policies. To tackle these issues, we introduce DexCap, a portable hand motion capture system, alongside DexIL, a novel imitation algorithm for training dexterous robot skills directly from human hand mocap data. DexCap offers precise, occlusion-resistant tracking of wrist and finger motions based on SLAM and electromagnetic field together with 3D observations of the environment. Utilizing this rich dataset, DexIL employs inverse kinematics and point cloud-based imitation learning to seamlessly replicate human actions with robot hands. Beyond direct learning from human motion, DexCap also offers an optional human-in-the-loop correction mechanism during policy rollouts to refine and further improve task performance. Through extensive evaluation across six challenging dexterous manipulation tasks, our approach not only demonstrates superior performance but also showcases the system's capability to effectively learn from in-the-wild mocap data, paving the way for future data collection methods in the pursuit of human-level robot dexterity. More details can be found at https://dex-cap.github.io",
  "published": "2024-03-12",
  "updated": "2024-07-04",
  "year": "2024",
  "authors": [
   "Chen Wang",
   "Haochen Shi",
   "Weizhuo Wang",
   "Ruohan Zhang",
   "Li Fei-Fei",
   "C. Karen Liu"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "RSS",
  "venue_source": "semantic-scholar",
  "citations": 315,
  "influential_citations": 23,
  "tldr": "This work introduces DexCap, a portable hand motion capture system, alongside DexIL, a novel imitation algorithm for training dexterous robot skills directly from human hand mocap data, paving the way for future data collection methods in the pursuit of human-level robot dexterity.",
  "doi": "10.48550/arXiv.2403.07788",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Chen Wang",
    "id": "2261162246",
    "h_index": 7,
    "papers": 8
   },
   {
    "name": "Haochen Shi",
    "id": "2257452648",
    "h_index": 7,
    "papers": 9
   },
   {
    "name": "Weizhuo Wang",
    "id": "2290968965",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Ruohan Zhang",
    "id": "2285397162",
    "h_index": 8,
    "papers": 17
   },
   {
    "name": "Fei-Fei Li",
    "id": "2238030496",
    "h_index": 7,
    "papers": 8
   },
   {
    "name": "Karen Liu",
    "id": "2320718296",
    "h_index": 3,
    "papers": 7
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "egocentric-data",
   "imitation-diffusion",
   "spatial-3d",
   "data-teleop",
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2403.07788v2",
  "pdf_url": "https://arxiv.org/pdf/2403.07788v2",
  "html_url": "https://arxiv.org/html/2403.07788v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.0
 },
 {
  "id": "2403.05304",
  "slug": "spatiotemporal-predictive-pre-training-for-robotic-motor-control",
  "title": "Spatiotemporal Predictive Pre-training for Robotic Motor Control",
  "abstract": "Robotic motor control necessitates the ability to predict the dynamics of environments and interaction objects. However, advanced self-supervised pre-trained visual representations in robotic motor control, leveraging large-scale egocentric videos, often focus solely on learning the static content features. This neglects the crucial temporal motion clues in human video, which implicitly contain key knowledge about interacting and manipulating with the environments and objects. In this paper, we present a simple yet effective robotic motor control visual pre-training framework that jointly performs spatiotemporal prediction with dual decoders, utilizing large-scale video data, termed as STP. STP adheres to two key designs in a multi-task learning manner. First, we perform spatial prediction on the masked current frame for learning content features. Second, we utilize the future frame with an extremely high masking ratio as a condition, based on the masked current frame, to conduct temporal prediction for capturing motion features. The asymmetric masking and decoupled dual decoders ensure that our image representation focusing on motion information while capturing spatial details. Extensive simulation and real-world experiments demonstrate the effectiveness and generalization abilities of STP, especially in generalizing to unseen environments with more distractors. Additionally, further post-pre-training and hybrid pre-training unleash its generality and data efficiency. Our code and weights will be released for further applications.",
  "published": "2024-03-08",
  "updated": "2024-11-21",
  "year": "2024",
  "authors": [
   "Jiange Yang",
   "Bei Liu",
   "Jianlong Fu",
   "Bocheng Pan",
   "Gangshan Wu",
   "Limin Wang"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 26,
  "influential_citations": 0,
  "tldr": "A simple yet effective visual pre-training framework termed as STP for robotic motor control is presented, which jointly performs spatial and temporal prediction with different masking ratios and a dual-decoder design by utilizing large-scale egocentric video data.",
  "doi": "10.1007/s11263-026-02948-3",
  "oa_pdf": "https://arxiv.org/pdf/2403.05304",
  "s2_authors": [
   {
    "name": "Jiange Yang",
    "id": "2220590629",
    "h_index": 9,
    "papers": 12
   },
   {
    "name": "Bei Liu",
    "id": "2290519570",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Jianlong Fu",
    "id": "2265967195",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Bocheng Pan",
    "id": "2298476151",
    "h_index": 1,
    "papers": 8
   },
   {
    "name": "Gangshan Wu",
    "id": "2249716237",
    "h_index": 11,
    "papers": 34
   },
   {
    "name": "Limin Wang",
    "id": "2290519814",
    "h_index": 8,
    "papers": 9
   }
  ],
  "comment": "19 pages, 7 figures, 14 tables",
  "topics": [
   "egocentric-data",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2403.05304v4",
  "pdf_url": "https://arxiv.org/pdf/2403.05304v4",
  "html_url": "https://arxiv.org/html/2403.05304v4",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.43
 },
 {
  "id": "2403.05110",
  "slug": "efficient-data-collection-for-robotic-manipulation-via-compositional-g",
  "title": "Efficient Data Collection for Robotic Manipulation via Compositional Generalization",
  "abstract": "Data collection has become an increasingly important problem in robotic manipulation, yet there still lacks much understanding of how to effectively collect data to facilitate broad generalization. Recent works on large-scale robotic data collection typically vary many environmental factors of variation (e.g., object types, table textures) during data collection, to cover a diverse range of scenarios. However, they do not explicitly account for the possible compositional abilities of policies trained on the data. If robot policies can compose environmental factors from their data to succeed when encountering unseen factor combinations, we can exploit this to avoid collecting data for situations that composition would address. To investigate this possibility, we conduct thorough empirical studies both in simulation and on a real robot that compare data collection strategies and assess whether visual imitation learning policies can compose environmental factors. We find that policies do exhibit composition, although leveraging prior robotic datasets is critical for this on a real robot. We use these insights to propose better in-domain data collection strategies that exploit composition, which can induce better generalization than naive approaches for the same amount of effort during data collection. We further demonstrate that a real robot policy trained on data from such a strategy achieves a success rate of 77.5% when transferred to entirely new environments that encompass unseen combinations of environmental factors, whereas policies trained using data collected without accounting for environmental variation fail to transfer effectively, with a success rate of only 2.5%. We provide videos at http://iliad.stanford.edu/robot-data-comp/.",
  "published": "2024-03-08",
  "updated": "2024-05-21",
  "year": "2024",
  "authors": [
   "Jensen Gao",
   "Annie Xie",
   "Ted Xiao",
   "Chelsea Finn",
   "Dorsa Sadigh"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "RSS",
  "venue_source": "semantic-scholar",
  "citations": 56,
  "influential_citations": 3,
  "tldr": "It is demonstrated that a real robot policy trained on data from such a strategy achieves a success rate of 77.5% when transferred to entirely new environments that encompass unseen combinations of environmental factors, whereas policies trained using data collected without accounting for environmental variation fail to transfer effectively.",
  "doi": "10.48550/arXiv.2403.05110",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jensen Gao",
    "id": "2238154243",
    "h_index": 8,
    "papers": 10
   },
   {
    "name": "Annie Xie",
    "id": "14484808",
    "h_index": 21,
    "papers": 23
   },
   {
    "name": "Ted Xiao",
    "id": "9961095",
    "h_index": 33,
    "papers": 45
   },
   {
    "name": "Chelsea Finn",
    "id": "2257346440",
    "h_index": 23,
    "papers": 32
   },
   {
    "name": "Dorsa Sadigh",
    "id": "1779671",
    "h_index": 69,
    "papers": 226
   }
  ],
  "comment": "RSS 2024",
  "topics": [
   "imitation-diffusion",
   "data-teleop"
  ],
  "orgs": [
   "Stanford"
  ],
  "abs_url": "https://arxiv.org/abs/2403.05110v2",
  "pdf_url": "https://arxiv.org/pdf/2403.05110v2",
  "html_url": "https://arxiv.org/html/2403.05110v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.76
 },
 {
  "id": "2403.04934",
  "slug": "letac-mpc-learning-model-predictive-control-for-tactile-reactive-grasp",
  "title": "LeTac-MPC: Learning Model Predictive Control for Tactile-reactive Grasping",
  "abstract": "Grasping is a crucial task in robotics, necessitating tactile feedback and reactive grasping adjustments for robust grasping of objects under various conditions and with differing physical properties. In this paper, we introduce LeTac-MPC, a learning-based model predictive control (MPC) for tactile-reactive grasping. Our approach enables the gripper to grasp objects with different physical properties on dynamic and force-interactive tasks. We utilize a vision-based tactile sensor, GelSight, which is capable of perceiving high-resolution tactile feedback that contains information on the physical properties and states of the grasped object. LeTac-MPC incorporates a differentiable MPC layer designed to model the embeddings extracted by a neural network (NN) from tactile feedback. This design facilitates convergent and robust grasping control at a frequency of 25 Hz. We propose a fully automated data collection pipeline and collect a dataset only using standardized blocks with different physical properties. However, our trained controller can generalize to daily objects with different sizes, shapes, materials, and textures. The experimental results demonstrate the effectiveness and robustness of the proposed approach. We compare LeTac-MPC with two purely model-based tactile-reactive controllers (MPC and PD) and open-loop grasping. Our results show that LeTac-MPC has optimal performance in dynamic and force-interactive tasks and optimal generalizability. We release our code and dataset at https://github.com/ZhengtongXu/LeTac-MPC.",
  "published": "2024-03-07",
  "updated": "2024-09-07",
  "year": "2024",
  "authors": [
   "Zhengtong Xu",
   "Yu She"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "T-RO",
  "venue_source": "semantic-scholar",
  "citations": 29,
  "influential_citations": 0,
  "tldr": "This article introduces LeTac-MPC, a learning-based model predictive control for tactile-reactive grasping that incorporates a differentiable MPC layer designed to model the embeddings extracted by a neural network from tactile feedback.",
  "doi": "10.1109/TRO.2024.3463470",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhengtong Xu",
    "id": "2238437162",
    "h_index": 8,
    "papers": 26
   },
   {
    "name": "Yu She",
    "id": "2238342225",
    "h_index": 7,
    "papers": 24
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "rl-control",
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2403.04934v2",
  "pdf_url": "https://arxiv.org/pdf/2403.04934v2",
  "html_url": "https://arxiv.org/html/2403.04934v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.98
 },
 {
  "id": "2403.03954",
  "slug": "3d-diffusion-policy-generalizable-visuomotor-policy-learning-via-simpl",
  "title": "3D Diffusion Policy: Generalizable Visuomotor Policy Learning via Simple 3D Representations",
  "abstract": "Imitation learning provides an efficient way to teach robots dexterous skills; however, learning complex skills robustly and generalizablely usually consumes large amounts of human demonstrations. To tackle this challenging problem, we present 3D Diffusion Policy (DP3), a novel visual imitation learning approach that incorporates the power of 3D visual representations into diffusion policies, a class of conditional action generative models. The core design of DP3 is the utilization of a compact 3D visual representation, extracted from sparse point clouds with an efficient point encoder. In our experiments involving 72 simulation tasks, DP3 successfully handles most tasks with just 10 demonstrations and surpasses baselines with a 24.2% relative improvement. In 4 real robot tasks, DP3 demonstrates precise control with a high success rate of 85%, given only 40 demonstrations of each task, and shows excellent generalization abilities in diverse aspects, including space, viewpoint, appearance, and instance. Interestingly, in real robot experiments, DP3 rarely violates safety requirements, in contrast to baseline methods which frequently do, necessitating human intervention. Our extensive evaluation highlights the critical importance of 3D representations in real-world robot learning. Videos, code, and data are available on https://3d-diffusion-policy.github.io .",
  "published": "2024-03-06",
  "updated": "2024-09-27",
  "year": "2024",
  "authors": [
   "Yanjie Ze",
   "Gu Zhang",
   "Kangning Zhang",
   "Chenyuan Hu",
   "Muhan Wang",
   "Huazhe Xu"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "RSS",
  "venue_source": "semantic-scholar",
  "citations": 147,
  "influential_citations": 24,
  "tldr": "3D Diffusion Policy (DP3), a novel visual imitation learning approach that incorporates the power of 3D visual representations into diffusion policies, a class of conditional action generative models, is presented.",
  "doi": "10.48550/arXiv.2403.03954",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yanjie Ze",
    "id": "2151089356",
    "h_index": 12,
    "papers": 16
   },
   {
    "name": "G. Zhang",
    "id": "2169993972",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Kangning Zhang",
    "id": "2290435183",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Chenyuan Hu",
    "id": "2290162299",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Muhan Wang",
    "id": "2334434687",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Huazhe Xu",
    "id": "2265666936",
    "h_index": 4,
    "papers": 4
   }
  ],
  "comment": "Published at Robotics: Science and Systems (RSS) 2024. Videos, code, and data: https://3d-diffusion-policy.github.io",
  "topics": [
   "dexterous-manipulation",
   "egocentric-data",
   "imitation-diffusion",
   "spatial-3d",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2403.03954v7",
  "pdf_url": "https://arxiv.org/pdf/2403.03954v7",
  "html_url": "https://arxiv.org/html/2403.03954v7",
  "code_url": "https://3d-diffusion-policy.github.io",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.67
 },
 {
  "id": "2403.03206",
  "slug": "scaling-rectified-flow-transformers-for-high-resolution-image-synthesi",
  "title": "Scaling Rectified Flow Transformers for High-Resolution Image Synthesis",
  "abstract": "Diffusion models create data from noise by inverting the forward paths of data towards noise and have emerged as a powerful generative modeling technique for high-dimensional, perceptual data such as images and videos. Rectified flow is a recent generative model formulation that connects data and noise in a straight line. Despite its better theoretical properties and conceptual simplicity, it is not yet decisively established as standard practice. In this work, we improve existing noise sampling techniques for training rectified flow models by biasing them towards perceptually relevant scales. Through a large-scale study, we demonstrate the superior performance of this approach compared to established diffusion formulations for high-resolution text-to-image synthesis. Additionally, we present a novel transformer-based architecture for text-to-image generation that uses separate weights for the two modalities and enables a bidirectional flow of information between image and text tokens, improving text comprehension, typography, and human preference ratings. We demonstrate that this architecture follows predictable scaling trends and correlates lower validation loss to improved text-to-image synthesis as measured by various metrics and human evaluations. Our largest models outperform state-of-the-art models, and we will make our experimental data, code, and model weights publicly available.",
  "published": "2024-03-05",
  "updated": "2024-03-05",
  "year": "2024",
  "authors": [
   "Patrick Esser",
   "Sumith Kulal",
   "Andreas Blattmann",
   "Rahim Entezari",
   "Jonas M\u00fcller",
   "Harry Saini",
   "Yam Levi",
   "Dominik Lorenz",
   "Axel Sauer",
   "Frederic Boesel",
   "Dustin Podell",
   "Tim Dockhorn",
   "Zion English",
   "Kyle Lacey",
   "Alex Goodwin",
   "Yannik Marek",
   "Robin Rombach"
  ],
  "author_count": 17,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "ICML",
  "venue_source": "semantic-scholar",
  "citations": 4594,
  "influential_citations": 804,
  "tldr": "This work improves existing noise sampling techniques for training rectified flow models by biasing them towards perceptually relevant scales and presents a novel transformer-based architecture for text-to-image generation that uses separate weights for the two modalities and enables a bidirectional flow of information between image and text tokens.",
  "doi": "10.48550/arXiv.2403.03206",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Patrick Esser",
    "id": "35175531",
    "h_index": 17,
    "papers": 24
   },
   {
    "name": "Sumith Kulal",
    "id": "3411322",
    "h_index": 12,
    "papers": 19
   },
   {
    "name": "A. Blattmann",
    "id": "119843260",
    "h_index": 17,
    "papers": 25
   },
   {
    "name": "Rahim Entezari",
    "id": "2316859494",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Jonas Muller",
    "id": "2188737195",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Harry Saini",
    "id": "2289994508",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Yam Levi",
    "id": "2290013499",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Dominik Lorenz",
    "id": "2053482699",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "Axel Sauer",
    "id": "40562186",
    "h_index": 13,
    "papers": 19
   },
   {
    "name": "Frederic Boesel",
    "id": "2290014125",
    "h_index": 6,
    "papers": 6
   },
   {
    "name": "Dustin Podell",
    "id": "2221125727",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Tim Dockhorn",
    "id": "102541178",
    "h_index": 13,
    "papers": 17
   },
   {
    "name": "Zion English",
    "id": "2221127565",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Kyle Lacey",
    "id": "2221126982",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Alex Goodwin",
    "id": "2290014122",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Yannik Marek",
    "id": "2290014387",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Robin Rombach",
    "id": "1660819540",
    "h_index": 22,
    "papers": 29
   }
  ],
  "comment": "",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2403.03206v1",
  "pdf_url": "https://arxiv.org/pdf/2403.03206v1",
  "html_url": "https://arxiv.org/html/2403.03206v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "2403.03181",
  "slug": "behavior-generation-with-latent-actions",
  "title": "Behavior Generation with Latent Actions",
  "abstract": "Generative modeling of complex behaviors from labeled datasets has been a longstanding problem in decision making. Unlike language or image generation, decision making requires modeling actions - continuous-valued vectors that are multimodal in their distribution, potentially drawn from uncurated sources, where generation errors can compound in sequential prediction. A recent class of models called Behavior Transformers (BeT) addresses this by discretizing actions using k-means clustering to capture different modes. However, k-means struggles to scale for high-dimensional action spaces or long sequences, and lacks gradient information, and thus BeT suffers in modeling long-range actions. In this work, we present Vector-Quantized Behavior Transformer (VQ-BeT), a versatile model for behavior generation that handles multimodal action prediction, conditional generation, and partial observations. VQ-BeT augments BeT by tokenizing continuous actions with a hierarchical vector quantization module. Across seven environments including simulated manipulation, autonomous driving, and robotics, VQ-BeT improves on state-of-the-art models such as BeT and Diffusion Policies. Importantly, we demonstrate VQ-BeT's improved ability to capture behavior modes while accelerating inference speed 5x over Diffusion Policies. Videos and code can be found https://sjlee.cc/vq-bet",
  "published": "2024-03-05",
  "updated": "2024-06-28",
  "year": "2024",
  "authors": [
   "Seungjae Lee",
   "Yibin Wang",
   "Haritheja Etukuru",
   "H. Jin Kim",
   "Nur Muhammad Mahi Shafiullah",
   "Lerrel Pinto"
  ],
  "author_count": 6,
  "categories": [
   "cs.LG",
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.LG",
  "venue": "ICML",
  "venue_source": "semantic-scholar",
  "citations": 202,
  "influential_citations": 24,
  "tldr": "Vector-Quantized Behavior Transformer (VQ-BeT) is presented, a versatile model for behavior generation that handles multimodal action prediction, conditional generation, and partial observations and improves on state-of-the-art models such as BeT and Diffusion Policies.",
  "doi": "10.48550/arXiv.2403.03181",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Seungjae Lee",
    "id": "2153321718",
    "h_index": 6,
    "papers": 19
   },
   {
    "name": "Yibin Wang",
    "id": "2384504643",
    "h_index": 8,
    "papers": 11
   },
   {
    "name": "Haritheja Etukuru",
    "id": "2268398574",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "H. J. Kim",
    "id": "2272427800",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Nur Muhammad",
    "id": "2290028642",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Mahi Shafiullah",
    "id": "2290027018",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Lerrel Pinto",
    "id": "2253567347",
    "h_index": 15,
    "papers": 25
   }
  ],
  "comment": "Github repo: https://github.com/jayLEE0301/vq_bet_official",
  "topics": [
   "navigation",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2403.03181v2",
  "pdf_url": "https://arxiv.org/pdf/2403.03181v2",
  "html_url": "https://arxiv.org/html/2403.03181v2",
  "code_url": "https://github.com/jayLEE0301/vq_bet_official",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.81
 },
 {
  "id": "2403.02338",
  "slug": "twisting-lids-off-with-two-hands",
  "title": "Twisting Lids Off with Two Hands",
  "abstract": "Manipulating objects with two multi-fingered hands has been a long-standing challenge in robotics, due to the contact-rich nature of many manipulation tasks and the complexity inherent in coordinating a high-dimensional bimanual system. In this work, we share novel insights into physical modeling, real-time perception, and reward design that enable policies trained in simulation using deep reinforcement learning (RL) to be effectively and efficiently transferred to the real world. Specifically, we consider the problem of twisting lids of various bottle-like objects with two hands, demonstrating policies with generalization capabilities across a diverse set of unseen objects as well as dynamic and dexterous behaviors. To the best of our knowledge, this is the first sim-to-real RL system that enables such capabilities on bimanual multi-fingered hands.",
  "published": "2024-03-04",
  "updated": "2024-10-14",
  "year": "2024",
  "authors": [
   "Toru Lin",
   "Zhao-Heng Yin",
   "Haozhi Qi",
   "Pieter Abbeel",
   "Jitendra Malik"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 64,
  "influential_citations": 4,
  "tldr": "This work considers the problem of twisting lids of various bottle-like objects with two hands, demonstrating policies with generalization capabilities across a diverse set of unseen objects as well as dynamic and dexterous behaviors.",
  "doi": "10.48550/arXiv.2403.02338",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Toru Lin",
    "id": "152997384",
    "h_index": 6,
    "papers": 6
   },
   {
    "name": "Zhao-Heng Yin",
    "id": "2290035529",
    "h_index": 7,
    "papers": 12
   },
   {
    "name": "Haozhi Qi",
    "id": "2247951244",
    "h_index": 11,
    "papers": 18
   },
   {
    "name": "Pieter Abbeel",
    "id": "2257003229",
    "h_index": 11,
    "papers": 17
   },
   {
    "name": "Jitendra Malik",
    "id": "2242761335",
    "h_index": 15,
    "papers": 32
   }
  ],
  "comment": "Project page can be found at https://toruowo.github.io/bimanual-twist",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "sim2real",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2403.02338v2",
  "pdf_url": "https://arxiv.org/pdf/2403.02338v2",
  "html_url": "https://arxiv.org/html/2403.02338v2",
  "code_url": "https://toruowo.github.io/bimanual-twist",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.31
 },
 {
  "id": "2403.01823",
  "slug": "rt-h-action-hierarchies-using-language",
  "title": "RT-H: Action Hierarchies Using Language",
  "abstract": "Language provides a way to break down complex concepts into digestible pieces. Recent works in robot imitation learning use language-conditioned policies that predict actions given visual observations and the high-level task specified in language. These methods leverage the structure of natural language to share data between semantically similar tasks (e.g., \"pick coke can\" and \"pick an apple\") in multi-task datasets. However, as tasks become more semantically diverse (e.g., \"pick coke can\" and \"pour cup\"), sharing data between tasks becomes harder, so learning to map high-level tasks to actions requires much more demonstration data. To bridge tasks and actions, our insight is to teach the robot the language of actions, describing low-level motions with more fine-grained phrases like \"move arm forward\". Predicting these language motions as an intermediate step between tasks and actions forces the policy to learn the shared structure of low-level motions across seemingly disparate tasks. Furthermore, a policy that is conditioned on language motions can easily be corrected during execution through human-specified language motions. This enables a new paradigm for flexible policies that can learn from human intervention in language. Our method RT-H builds an action hierarchy using language motions: it first learns to predict language motions, and conditioned on this and the high-level task, it predicts actions, using visual context at all stages. We show that RT-H leverages this language-action hierarchy to learn policies that are more robust and flexible by effectively tapping into multi-task datasets. We show that these policies not only allow for responding to language interventions, but can also learn from such interventions and outperform methods that learn from teleoperated interventions. Our website and videos are found at https://rt-hierarchy.github.io.",
  "published": "2024-03-04",
  "updated": "2024-06-01",
  "year": "2024",
  "authors": [
   "Suneel Belkhale",
   "Tianli Ding",
   "Ted Xiao",
   "Pierre Sermanet",
   "Quon Vuong",
   "Jonathan Tompson",
   "Yevgen Chebotar",
   "Debidatta Dwibedi",
   "Dorsa Sadigh"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "RSS",
  "venue_source": "semantic-scholar",
  "citations": 232,
  "influential_citations": 13,
  "tldr": "The method RT-H builds an action hierarchy using language motions: it first learns to predict language motions, and conditioned on this and the high-level task, it predicts actions, using visual context at all stages, and it shows that RT-H leverages this language-action hierarchy to learn policies that are more robust and flexible by effectively tapping into multi-task datasets.",
  "doi": "10.48550/arXiv.2403.01823",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Suneel Belkhale",
    "id": "69879999",
    "h_index": 18,
    "papers": 25
   },
   {
    "name": "Tianli Ding",
    "id": "95691186",
    "h_index": 15,
    "papers": 24
   },
   {
    "name": "Ted Xiao",
    "id": "9961095",
    "h_index": 33,
    "papers": 45
   },
   {
    "name": "P. Sermanet",
    "id": "3142556",
    "h_index": 39,
    "papers": 77
   },
   {
    "name": "Quon Vuong",
    "id": "2290010517",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Jonathan Tompson",
    "id": "2704494",
    "h_index": 43,
    "papers": 71
   },
   {
    "name": "Yevgen Chebotar",
    "id": "2527420",
    "h_index": 33,
    "papers": 57
   },
   {
    "name": "Debidatta Dwibedi",
    "id": "2420123",
    "h_index": 19,
    "papers": 35
   },
   {
    "name": "Dorsa Sadigh",
    "id": "1779671",
    "h_index": 69,
    "papers": 226
   }
  ],
  "comment": "",
  "topics": [
   "imitation-diffusion"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2403.01823v2",
  "pdf_url": "https://arxiv.org/pdf/2403.01823v2",
  "html_url": "https://arxiv.org/html/2403.01823v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.87
 },
 {
  "id": "2403.01694",
  "slug": "tac-man-tactile-informed-prior-free-manipulation-of-articulated-object",
  "title": "Tac-Man: Tactile-Informed Prior-Free Manipulation of Articulated Objects",
  "abstract": "Integrating robots into human-centric environments such as homes, necessitates advanced manipulation skills as robotic devices will need to engage with articulated objects like doors and drawers. Key challenges in robotic manipulation of articulated objects are the unpredictability and diversity of these objects' internal structures, which render models based on object kinematics priors, both explicit and implicit, inadequate. Their reliability is significantly diminished by pre-interaction ambiguities, imperfect structural parameters, encounters with unknown objects, and unforeseen disturbances. Here, we present a prior-free strategy, Tac-Man, focusing on maintaining stable robot-object contact during manipulation. Without relying on object priors, Tac-Man leverages tactile feedback to enable robots to proficiently handle a variety of articulated objects, including those with complex joints, even when influenced by unexpected disturbances. Demonstrated in both real-world experiments and extensive simulations, it consistently achieves near-perfect success in dynamic and varied settings, outperforming existing methods. Our results indicate that tactile sensing alone suffices for managing diverse articulated objects, offering greater robustness and generalization than prior-based approaches. This underscores the importance of detailed contact modeling in complex manipulation tasks, especially with articulated objects. Advancements in tactile-informed approaches significantly expand the scope of robotic applications in human-centric environments, particularly where accurate models are difficult to obtain. See additional material at https://tacman-aom.github.io.",
  "published": "2024-03-04",
  "updated": "2025-09-12",
  "year": "2024",
  "authors": [
   "Zihang Zhao",
   "Yuyang Li",
   "Wanlin Li",
   "Zhenghao Qi",
   "Lecheng Ruan",
   "Yixin Zhu",
   "Kaspar Althoefer"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "T-RO",
  "venue_source": "semantic-scholar",
  "citations": 38,
  "influential_citations": 0,
  "tldr": "Tac-Man leverages tactile feedback to enable robots to proficiently handle a variety of articulated objects, including those with complex joints, even when influenced by unexpected disturbances, and indicates that tactile sensing alone suffices for managing diverse articulated objects, offering greater robustness and generalization than prior-based approaches.",
  "doi": "10.1109/TRO.2024.3508134",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zihang Zhao",
    "id": "2290006604",
    "h_index": 7,
    "papers": 12
   },
   {
    "name": "Yuyang Li",
    "id": "2261448933",
    "h_index": 8,
    "papers": 11
   },
   {
    "name": "Wanlin Li",
    "id": "2286300708",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Zhenghao Qi",
    "id": "2290170453",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Lecheng Ruan",
    "id": "2221288461",
    "h_index": 7,
    "papers": 26
   },
   {
    "name": "Yixin Zhu",
    "id": "2377257949",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "K. Althoefer",
    "id": "1682235",
    "h_index": 59,
    "papers": 568
   }
  ],
  "comment": "Accepted for publication in the IEEE Transactions on Robotics (T-RO)",
  "topics": [
   "tactile",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2403.01694v4",
  "pdf_url": "https://arxiv.org/pdf/2403.01694v4",
  "html_url": "https://arxiv.org/html/2403.01694v4",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.09
 },
 {
  "id": "2402.19479",
  "slug": "panda-70m-captioning-70m-videos-with-multiple-cross-modality-teachers",
  "title": "Panda-70M: Captioning 70M Videos with Multiple Cross-Modality Teachers",
  "abstract": "The quality of the data and annotation upper-bounds the quality of a downstream model. While there exist large text corpora and image-text pairs, high-quality video-text data is much harder to collect. First of all, manual labeling is more time-consuming, as it requires an annotator to watch an entire video. Second, videos have a temporal dimension, consisting of several scenes stacked together, and showing multiple actions. Accordingly, to establish a video dataset with high-quality captions, we propose an automatic approach leveraging multimodal inputs, such as textual video description, subtitles, and individual video frames. Specifically, we curate 3.8M high-resolution videos from the publicly available HD-VILA-100M dataset. We then split them into semantically consistent video clips, and apply multiple cross-modality teacher models to obtain captions for each video. Next, we finetune a retrieval model on a small subset where the best caption of each video is manually selected and then employ the model in the whole dataset to select the best caption as the annotation. In this way, we get 70M videos paired with high-quality text captions. We dub the dataset as Panda-70M. We show the value of the proposed dataset on three downstream tasks: video captioning, video and text retrieval, and text-driven video generation. The models trained on the proposed data score substantially better on the majority of metrics across all the tasks.",
  "published": "2024-02-29",
  "updated": "2024-02-29",
  "year": "2024",
  "authors": [
   "Tsai-Shien Chen",
   "Aliaksandr Siarohin",
   "Willi Menapace",
   "Ekaterina Deyneka",
   "Hsiang-wei Chao",
   "Byung Eun Jeon",
   "Yuwei Fang",
   "Hsin-Ying Lee",
   "Jian Ren",
   "Ming-Hsuan Yang",
   "Sergey Tulyakov"
  ],
  "author_count": 11,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 439,
  "influential_citations": 45,
  "tldr": "This work curates 3.8M high-resolution videos from the publicly available HD-VILA-100M dataset, and applies multiple cross-modality teacher models to obtain captions for each video, resulting in 70M videos paired with high-quality text captions, dubbed as Panda-70M.",
  "doi": "10.1109/CVPR52733.2024.01265",
  "oa_pdf": "http://arxiv.org/pdf/2402.19479",
  "s2_authors": [
   {
    "name": "Tsai-Shien Chen",
    "id": "151229337",
    "h_index": 10,
    "papers": 11
   },
   {
    "name": "Aliaksandr Siarohin",
    "id": "10753214",
    "h_index": 31,
    "papers": 76
   },
   {
    "name": "W. Menapace",
    "id": "1698103472",
    "h_index": 22,
    "papers": 53
   },
   {
    "name": "Ekaterina Deyneka",
    "id": "2284986967",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Hsiang-wei Chao",
    "id": "2288532748",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Byung Eun Jeon",
    "id": "2391586100",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Yuwei Fang",
    "id": "2282237113",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Hsin-Ying Lee",
    "id": "2257364073",
    "h_index": 13,
    "papers": 20
   },
   {
    "name": "Jian Ren",
    "id": "2286038173",
    "h_index": 8,
    "papers": 11
   },
   {
    "name": "Ming-Hsuan Yang",
    "id": "2268794991",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "S. Tulyakov",
    "id": "145582202",
    "h_index": 48,
    "papers": 171
   }
  ],
  "comment": "CVPR 2024. Project Page: https://snap-research.github.io/Panda-70M",
  "topics": [
   "data-teleop",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2402.19479v1",
  "pdf_url": "https://arxiv.org/pdf/2402.19479v1",
  "html_url": "https://arxiv.org/html/2402.19479v1",
  "code_url": "https://snap-research.github.io/Panda-70M",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.14
 },
 {
  "id": "2402.19432",
  "slug": "pushing-the-limits-of-cross-embodiment-learning-for-manipulation-and-n",
  "title": "Pushing the Limits of Cross-Embodiment Learning for Manipulation and Navigation",
  "abstract": "Recent years in robotics and imitation learning have shown remarkable progress in training large-scale foundation models by leveraging data across a multitude of embodiments. The success of such policies might lead us to wonder: just how diverse can the robots in the training set be while still facilitating positive transfer? In this work, we study this question in the context of heterogeneous embodiments, examining how even seemingly very different domains, such as robotic navigation and manipulation, can provide benefits when included in the training data for the same model. We train a single goal-conditioned policy that is capable of controlling robotic arms, quadcopters, quadrupeds, and mobile bases. We then investigate the extent to which transfer can occur across navigation and manipulation on these embodiments by framing them as a single goal-reaching task. We find that co-training with navigation data can enhance robustness and performance in goal-conditioned manipulation with a wrist-mounted camera. We then deploy our policy trained only from navigation-only and static manipulation-only data on a mobile manipulator, showing that it can control a novel embodiment in a zero-shot manner. These results provide evidence that large-scale robotic policies can benefit from data collected across various embodiments. Further information and robot videos can be found on our project website http://extreme-cross-embodiment.github.io.",
  "published": "2024-02-29",
  "updated": "2024-02-29",
  "year": "2024",
  "authors": [
   "Jonathan Yang",
   "Catherine Glossop",
   "Arjun Bhorkar",
   "Dhruv Shah",
   "Quan Vuong",
   "Chelsea Finn",
   "Dorsa Sadigh",
   "Sergey Levine"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "RSS",
  "venue_source": "semantic-scholar",
  "citations": 94,
  "influential_citations": 4,
  "tldr": "This work trains a single goal-conditioned policy that is capable of controlling robotic arms, quadcopters, quadrupeds, and mobile bases, and investigates the extent to which transfer can occur across navigation and manipulation on these embodiments by framing them as a single goal-reaching task.",
  "doi": "10.48550/arXiv.2402.19432",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jonathan Yang",
    "id": "2143067490",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Catherine Glossop",
    "id": "2257348904",
    "h_index": 9,
    "papers": 13
   },
   {
    "name": "Arjun Bhorkar",
    "id": "2187206730",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Dhruv Shah",
    "id": "2322628540",
    "h_index": 29,
    "papers": 63
   },
   {
    "name": "Quan Vuong",
    "id": "2288210223",
    "h_index": 19,
    "papers": 29
   },
   {
    "name": "Chelsea Finn",
    "id": "2257346440",
    "h_index": 23,
    "papers": 32
   },
   {
    "name": "Dorsa Sadigh",
    "id": "1779671",
    "h_index": 69,
    "papers": 226
   },
   {
    "name": "Sergey Levine",
    "id": "2249615151",
    "h_index": 32,
    "papers": 53
   }
  ],
  "comment": "16 pages, 9 figures",
  "topics": [
   "humanoids",
   "imitation-diffusion",
   "navigation",
   "foundation-pretraining",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2402.19432v1",
  "pdf_url": "https://arxiv.org/pdf/2402.19432v1",
  "html_url": "https://arxiv.org/html/2402.19432v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.48
 },
 {
  "id": "2402.19249",
  "slug": "mirage-cross-embodiment-zero-shot-policy-transfer-with-cross-painting",
  "title": "Mirage: Cross-Embodiment Zero-Shot Policy Transfer with Cross-Painting",
  "abstract": "The ability to reuse collected data and transfer trained policies between robots could alleviate the burden of additional data collection and training. While existing approaches such as pretraining plus finetuning and co-training show promise, they do not generalize to robots unseen in training. Focusing on common robot arms with similar workspaces and 2-jaw grippers, we investigate the feasibility of zero-shot transfer. Through simulation studies on 8 manipulation tasks, we find that state-based Cartesian control policies can successfully zero-shot transfer to a target robot after accounting for forward dynamics. To address robot visual disparities for vision-based policies, we introduce Mirage, which uses \"cross-painting\"--masking out the unseen target robot and inpainting the seen source robot--during execution in real time so that it appears to the policy as if the trained source robot were performing the task. Mirage applies to both first-person and third-person camera views and policies that take in both states and images as inputs or only images as inputs. Despite its simplicity, our extensive simulation and physical experiments provide strong evidence that Mirage can successfully zero-shot transfer between different robot arms and grippers with only minimal performance degradation on a variety of manipulation tasks such as picking, stacking, and assembly, significantly outperforming a generalist policy. Project website: https://robot-mirage.github.io/",
  "published": "2024-02-29",
  "updated": "2024-09-09",
  "year": "2024",
  "authors": [
   "Lawrence Yunliang Chen",
   "Kush Hari",
   "Karthik Dharmarajan",
   "Chenfeng Xu",
   "Quan Vuong",
   "Ken Goldberg"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "RSS",
  "venue_source": "semantic-scholar",
  "citations": 64,
  "influential_citations": 3,
  "tldr": "Mirage is introduced, which uses cross-painting during execution in real time to address robot visual disparities for vision-based policies, and can successfully zero-shot transfer between different robot arms and grippers with only minimal performance degradation on a variety of manipulation tasks.",
  "doi": "10.48550/arXiv.2402.19249",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "L. Chen",
    "id": "2143804724",
    "h_index": 14,
    "papers": 24
   },
   {
    "name": "Kush Hari",
    "id": "2265320656",
    "h_index": 8,
    "papers": 17
   },
   {
    "name": "K. Dharmarajan",
    "id": "1436202543",
    "h_index": 10,
    "papers": 19
   },
   {
    "name": "Chenfeng Xu",
    "id": "1490695028",
    "h_index": 25,
    "papers": 46
   },
   {
    "name": "Quan Vuong",
    "id": "2288210223",
    "h_index": 19,
    "papers": 29
   },
   {
    "name": "Kenneth Y. Goldberg",
    "id": "2310822486",
    "h_index": 4,
    "papers": 7
   }
  ],
  "comment": "RSS 2024. Project page: https://robot-mirage.github.io/",
  "topics": [
   "egocentric-data",
   "foundation-pretraining",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2402.19249v3",
  "pdf_url": "https://arxiv.org/pdf/2402.19249v3",
  "html_url": "https://arxiv.org/html/2402.19249v3",
  "code_url": "https://robot-mirage.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.31
 },
 {
  "id": "2402.17139",
  "slug": "video-as-the-new-language-for-real-world-decision-making",
  "title": "Video as the New Language for Real-World Decision Making",
  "abstract": "Both text and video data are abundant on the internet and support large-scale self-supervised learning through next token or frame prediction. However, they have not been equally leveraged: language models have had significant real-world impact, whereas video generation has remained largely limited to media entertainment. Yet video data captures important information about the physical world that is difficult to express in language. To address this gap, we discuss an under-appreciated opportunity to extend video generation to solve tasks in the real world. We observe how, akin to language, video can serve as a unified interface that can absorb internet knowledge and represent diverse tasks. Moreover, we demonstrate how, like language models, video generation can serve as planners, agents, compute engines, and environment simulators through techniques such as in-context learning, planning and reinforcement learning. We identify major impact opportunities in domains such as robotics, self-driving, and science, supported by recent work that demonstrates how such advanced capabilities in video generation are plausibly within reach. Lastly, we identify key challenges in video generation that mitigate progress. Addressing these challenges will enable video generation models to demonstrate unique value alongside language models in a wider array of AI applications.",
  "published": "2024-02-27",
  "updated": "2024-02-27",
  "year": "2024",
  "authors": [
   "Sherry Yang",
   "Jacob Walker",
   "Jack Parker-Holder",
   "Yilun Du",
   "Jake Bruce",
   "Andre Barreto",
   "Pieter Abbeel",
   "Dale Schuurmans"
  ],
  "author_count": 8,
  "categories": [
   "cs.CV",
   "cs.AI"
  ],
  "primary_category": "cs.CV",
  "venue": "ICML",
  "venue_source": "semantic-scholar",
  "citations": 98,
  "influential_citations": 2,
  "tldr": "This work observes how, akin to language, video can serve as a unified interface that can absorb internet knowledge and represent diverse tasks and demonstrates how video generation can serve as planners, agents, compute engines, and environment simulators through techniques such as in-context learning, planning and reinforcement learning.",
  "doi": "10.48550/arXiv.2402.17139",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Sherry Yang",
    "id": "2287810415",
    "h_index": 7,
    "papers": 8
   },
   {
    "name": "Jacob Walker",
    "id": "2336663153",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Jack Parker-Holder",
    "id": "1410302742",
    "h_index": 25,
    "papers": 60
   },
   {
    "name": "Yilun Du",
    "id": "15394275",
    "h_index": 48,
    "papers": 86
   },
   {
    "name": "Jake Bruce",
    "id": "2286344472",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Andr\u00e9 Barreto",
    "id": "2257207860",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Pieter Abbeel",
    "id": "2257003229",
    "h_index": 11,
    "papers": 17
   },
   {
    "name": "Dale Schuurmans",
    "id": "2265994815",
    "h_index": 11,
    "papers": 32
   }
  ],
  "comment": "",
  "topics": [
   "sim2real",
   "rl-control",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2402.17139v1",
  "pdf_url": "https://arxiv.org/pdf/2402.17139v1",
  "html_url": "https://arxiv.org/html/2402.17139v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.5
 },
 {
  "id": "2402.15487",
  "slug": "roboexp-action-conditioned-scene-graph-via-interactive-exploration-for",
  "title": "RoboEXP: Action-Conditioned Scene Graph via Interactive Exploration for Robotic Manipulation",
  "abstract": "We introduce the novel task of interactive scene exploration, wherein robots autonomously explore environments and produce an action-conditioned scene graph (ACSG) that captures the structure of the underlying environment. The ACSG accounts for both low-level information (geometry and semantics) and high-level information (action-conditioned relationships between different entities) in the scene. To this end, we present the Robotic Exploration (RoboEXP) system, which incorporates the Large Multimodal Model (LMM) and an explicit memory design to enhance our system's capabilities. The robot reasons about what and how to explore an object, accumulating new information through the interaction process and incrementally constructing the ACSG. Leveraging the constructed ACSG, we illustrate the effectiveness and efficiency of our RoboEXP system in facilitating a wide range of real-world manipulation tasks involving rigid, articulated objects, nested objects, and deformable objects.",
  "published": "2024-02-23",
  "updated": "2024-10-08",
  "year": "2024",
  "authors": [
   "Hanxiao Jiang",
   "Binghao Huang",
   "Ruihai Wu",
   "Zhuoran Li",
   "Shubham Garg",
   "Hooshang Nayyeri",
   "Shenlong Wang",
   "Yunzhu Li"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 79,
  "influential_citations": 4,
  "tldr": "The RoboEXP system is presented, which incorporates the Large Multimodal Model (LMM) and an explicit memory design to enhance the system's capabilities and illustrates the effectiveness and efficiency of the RoboEXP system in facilitating a wide range of real-world manipulation tasks involving rigid, articulated objects, nested objects, and deformable objects.",
  "doi": "10.48550/arXiv.2402.15487",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hanxiao Jiang",
    "id": "2288026132",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Binghao Huang",
    "id": "2287019710",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Ruihai Wu",
    "id": "9381012",
    "h_index": 18,
    "papers": 38
   },
   {
    "name": "Zhuoran Li",
    "id": "2247904852",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Shubham Garg",
    "id": "2286296854",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "H. Nayyeri",
    "id": "10412501",
    "h_index": 30,
    "papers": 113
   },
   {
    "name": "Shenlong Wang",
    "id": "2288429921",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Yunzhu Li",
    "id": "2352183976",
    "h_index": 4,
    "papers": 6
   }
  ],
  "comment": "Project Page: https://jianghanxiao.github.io/roboexp-web/",
  "topics": [
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2402.15487v2",
  "pdf_url": "https://arxiv.org/pdf/2402.15487v2",
  "html_url": "https://arxiv.org/html/2402.15487v2",
  "code_url": "https://jianghanxiao.github.io/roboexp-web/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.4
 },
 {
  "id": "2402.15391",
  "slug": "genie-generative-interactive-environments",
  "title": "Genie: Generative Interactive Environments",
  "abstract": "We introduce Genie, the first generative interactive environment trained in an unsupervised manner from unlabelled Internet videos. The model can be prompted to generate an endless variety of action-controllable virtual worlds described through text, synthetic images, photographs, and even sketches. At 11B parameters, Genie can be considered a foundation world model. It is comprised of a spatiotemporal video tokenizer, an autoregressive dynamics model, and a simple and scalable latent action model. Genie enables users to act in the generated environments on a frame-by-frame basis despite training without any ground-truth action labels or other domain-specific requirements typically found in the world model literature. Further the resulting learned latent action space facilitates training agents to imitate behaviors from unseen videos, opening the path for training generalist agents of the future.",
  "published": "2024-02-23",
  "updated": "2024-02-23",
  "year": "2024",
  "authors": [
   "Jake Bruce",
   "Michael Dennis",
   "Ashley Edwards",
   "Jack Parker-Holder",
   "Yuge Shi",
   "Edward Hughes",
   "Matthew Lai",
   "Aditi Mavalankar",
   "Richie Steigerwald",
   "Chris Apps",
   "Yusuf Aytar",
   "Sarah Bechtle",
   "Feryal Behbahani",
   "Stephanie Chan",
   "Nicolas Heess",
   "Lucy Gonzalez",
   "Simon Osindero",
   "Sherjil Ozair",
   "Scott Reed",
   "Jingwei Zhang",
   "Konrad Zolna",
   "Jeff Clune",
   "Nando de Freitas",
   "Satinder Singh",
   "Tim Rockt\u00e4schel"
  ],
  "author_count": 25,
  "categories": [
   "cs.LG",
   "cs.AI",
   "cs.CV"
  ],
  "primary_category": "cs.LG",
  "venue": "ICML",
  "venue_source": "semantic-scholar",
  "citations": 762,
  "influential_citations": 59,
  "tldr": "Genie is introduced, the first generative interactive environment trained in an unsupervised manner from unlabelled Internet videos, which enables users to act in the generated environments on a frame-by-frame basis despite training without any ground-truth action labels or other domain-specific requirements typically found in the world model literature.",
  "doi": "10.48550/arXiv.2402.15391",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jake Bruce",
    "id": "2286344472",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Michael Dennis",
    "id": "2286305249",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Ashley Edwards",
    "id": "2268732203",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Jack Parker-Holder",
    "id": "1410302742",
    "h_index": 25,
    "papers": 60
   },
   {
    "name": "Yuge Shi",
    "id": "2286668292",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Edward Hughes",
    "id": "2286317624",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "M. Lai",
    "id": "40227832",
    "h_index": 10,
    "papers": 15
   },
   {
    "name": "Aditi Mavalankar",
    "id": "38792754",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Richie Steigerwald",
    "id": "2188779988",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Chris Apps",
    "id": "2293394525",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Y. Aytar",
    "id": "3152281",
    "h_index": 30,
    "papers": 57
   },
   {
    "name": "Sarah Bechtle",
    "id": "33931249",
    "h_index": 10,
    "papers": 18
   },
   {
    "name": "Feryal M. P. Behbahani",
    "id": "145124447",
    "h_index": 14,
    "papers": 24
   },
   {
    "name": "Stephanie Chan",
    "id": "2297673118",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "N. Heess",
    "id": "2801204",
    "h_index": 73,
    "papers": 192
   },
   {
    "name": "Lucy Gonzalez",
    "id": "2287795688",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Simon Osindero",
    "id": "2217144",
    "h_index": 36,
    "papers": 85
   },
   {
    "name": "Sherjil Ozair",
    "id": "1955694",
    "h_index": 16,
    "papers": 32
   },
   {
    "name": "Scott E. Reed",
    "id": "2315316001",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Jingwei Zhang",
    "id": "2107958221",
    "h_index": 14,
    "papers": 22
   },
   {
    "name": "Konrad Zolna",
    "id": "7912420",
    "h_index": 21,
    "papers": 33
   },
   {
    "name": "Jeff Clune",
    "id": "2263705770",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Nando de Freitas",
    "id": "1737568",
    "h_index": 79,
    "papers": 193
   },
   {
    "name": "Satinder Singh",
    "id": "2286734040",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Tim Rocktaschel",
    "id": "1389854357",
    "h_index": 22,
    "papers": 42
   }
  ],
  "comment": "https://sites.google.com/corp/view/genie-2024/",
  "topics": [
   "world-models",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2402.15391v1",
  "pdf_url": "https://arxiv.org/pdf/2402.15391v1",
  "html_url": "https://arxiv.org/html/2402.15391v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.38
 },
 {
  "id": "2402.14797",
  "slug": "snap-video-scaled-spatiotemporal-transformers-for-text-to-video-synthe",
  "title": "Snap Video: Scaled Spatiotemporal Transformers for Text-to-Video Synthesis",
  "abstract": "Contemporary models for generating images show remarkable quality and versatility. Swayed by these advantages, the research community repurposes them to generate videos. Since video content is highly redundant, we argue that naively bringing advances of image models to the video generation domain reduces motion fidelity, visual quality and impairs scalability. In this work, we build Snap Video, a video-first model that systematically addresses these challenges. To do that, we first extend the EDM framework to take into account spatially and temporally redundant pixels and naturally support video generation. Second, we show that a U-Net - a workhorse behind image generation - scales poorly when generating videos, requiring significant computational overhead. Hence, we propose a new transformer-based architecture that trains 3.31 times faster than U-Nets (and is ~4.5 faster at inference). This allows us to efficiently train a text-to-video model with billions of parameters for the first time, reach state-of-the-art results on a number of benchmarks, and generate videos with substantially higher quality, temporal consistency, and motion complexity. The user studies showed that our model was favored by a large margin over the most recent methods. See our website at https://snap-research.github.io/snapvideo/.",
  "published": "2024-02-22",
  "updated": "2024-02-22",
  "year": "2024",
  "authors": [
   "Willi Menapace",
   "Aliaksandr Siarohin",
   "Ivan Skorokhodov",
   "Ekaterina Deyneka",
   "Tsai-Shien Chen",
   "Anil Kag",
   "Yuwei Fang",
   "Aleksei Stoliar",
   "Elisa Ricci",
   "Jian Ren",
   "Sergey Tulyakov"
  ],
  "author_count": 11,
  "categories": [
   "cs.CV",
   "cs.AI"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 121,
  "influential_citations": 5,
  "tldr": "Snap Video is built, a video-first model that systematically addresses the challenges of motion fidelity, visual quality and scalability in the video generation domain and proposes a new transformer-based architecture that trains 3.31 times faster than U-Nets (and is \u223c4.5 faster at inference).",
  "doi": "10.1109/CVPR52733.2024.00672",
  "oa_pdf": "https://arxiv.org/pdf/2402.14797",
  "s2_authors": [
   {
    "name": "W. Menapace",
    "id": "1698103472",
    "h_index": 22,
    "papers": 53
   },
   {
    "name": "Aliaksandr Siarohin",
    "id": "10753214",
    "h_index": 31,
    "papers": 76
   },
   {
    "name": "Ivan Skorokhodov",
    "id": "51118864",
    "h_index": 22,
    "papers": 45
   },
   {
    "name": "Ekaterina Deyneka",
    "id": "2284986967",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Tsai-Shien Chen",
    "id": "151229337",
    "h_index": 10,
    "papers": 11
   },
   {
    "name": "Anil Kag",
    "id": "2284982329",
    "h_index": 9,
    "papers": 23
   },
   {
    "name": "Yuwei Fang",
    "id": "2282237113",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "A. Stoliar",
    "id": "2079578028",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Elisa Ricci",
    "id": "2176320235",
    "h_index": 7,
    "papers": 17
   },
   {
    "name": "Jian Ren",
    "id": "2286038173",
    "h_index": 8,
    "papers": 11
   },
   {
    "name": "S. Tulyakov",
    "id": "145582202",
    "h_index": 48,
    "papers": 171
   }
  ],
  "comment": "",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2402.14797v1",
  "pdf_url": "https://arxiv.org/pdf/2402.14797v1",
  "html_url": "https://arxiv.org/html/2402.14797v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.59
 },
 {
  "id": "2404.08471",
  "slug": "revisiting-feature-prediction-for-learning-visual-representations-from",
  "title": "Revisiting Feature Prediction for Learning Visual Representations from Video",
  "abstract": "This paper explores feature prediction as a stand-alone objective for unsupervised learning from video and introduces V-JEPA, a collection of vision models trained solely using a feature prediction objective, without the use of pretrained image encoders, text, negative examples, reconstruction, or other sources of supervision. The models are trained on 2 million videos collected from public datasets and are evaluated on downstream image and video tasks. Our results show that learning by predicting video features leads to versatile visual representations that perform well on both motion and appearance-based tasks, without adaption of the model's parameters; e.g., using a frozen backbone. Our largest model, a ViT-H/16 trained only on videos, obtains 81.9% on Kinetics-400, 72.2% on Something-Something-v2, and 77.9% on ImageNet1K.",
  "published": "2024-02-15",
  "updated": "2024-02-15",
  "year": "2024",
  "authors": [
   "Adrien Bardes",
   "Quentin Garrido",
   "Jean Ponce",
   "Xinlei Chen",
   "Michael Rabbat",
   "Yann LeCun",
   "Mahmoud Assran",
   "Nicolas Ballas"
  ],
  "author_count": 8,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.CV",
  "venue": "Trans. Mach. Learn. Res.",
  "venue_source": "semantic-scholar",
  "citations": 463,
  "influential_citations": 71,
  "tldr": "",
  "doi": "10.48550/arXiv.2404.08471",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Adrien Bardes",
    "id": "1453740540",
    "h_index": 16,
    "papers": 23
   },
   {
    "name": "Q. Garrido",
    "id": "2048163343",
    "h_index": 16,
    "papers": 22
   },
   {
    "name": "Jean Ponce",
    "id": "2296601397",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Xinlei Chen",
    "id": "2296665228",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Michael G. Rabbat",
    "id": "2066127975",
    "h_index": 25,
    "papers": 42
   },
   {
    "name": "Yann LeCun",
    "id": "2265899558",
    "h_index": 22,
    "papers": 47
   },
   {
    "name": "Mahmoud Assran",
    "id": "38698856",
    "h_index": 15,
    "papers": 21
   },
   {
    "name": "Nicolas Ballas",
    "id": "2289844757",
    "h_index": 13,
    "papers": 19
   }
  ],
  "comment": "",
  "topics": [
   "world-models"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2404.08471v1",
  "pdf_url": "https://arxiv.org/pdf/2404.08471v1",
  "html_url": "https://arxiv.org/html/2404.08471v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.17
 },
 {
  "id": "2402.10329",
  "slug": "universal-manipulation-interface-in-the-wild-robot-teaching-without-in",
  "title": "Universal Manipulation Interface: In-The-Wild Robot Teaching Without In-The-Wild Robots",
  "abstract": "We present Universal Manipulation Interface (UMI) -- a data collection and policy learning framework that allows direct skill transfer from in-the-wild human demonstrations to deployable robot policies. UMI employs hand-held grippers coupled with careful interface design to enable portable, low-cost, and information-rich data collection for challenging bimanual and dynamic manipulation demonstrations. To facilitate deployable policy learning, UMI incorporates a carefully designed policy interface with inference-time latency matching and a relative-trajectory action representation. The resulting learned policies are hardware-agnostic and deployable across multiple robot platforms. Equipped with these features, UMI framework unlocks new robot manipulation capabilities, allowing zero-shot generalizable dynamic, bimanual, precise, and long-horizon behaviors, by only changing the training data for each task. We demonstrate UMI's versatility and efficacy with comprehensive real-world experiments, where policies learned via UMI zero-shot generalize to novel environments and objects when trained on diverse human demonstrations. UMI's hardware and software system is open-sourced at https://umi-gripper.github.io.",
  "published": "2024-02-15",
  "updated": "2024-03-06",
  "year": "2024",
  "authors": [
   "Cheng Chi",
   "Zhenjia Xu",
   "Chuer Pan",
   "Eric Cousineau",
   "Benjamin Burchfiel",
   "Siyuan Feng",
   "Russ Tedrake",
   "Shuran Song"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "RSS",
  "venue_source": "semantic-scholar",
  "citations": 646,
  "influential_citations": 70,
  "tldr": "Universal Manipulation Interface framework unlocks new robot manipulation capabilities, allowing zero-shot generalizable dynamic, bimanual, precise, and long-horizon behaviors, by only changing the training data for each task.",
  "doi": "10.48550/arXiv.2402.10329",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Cheng Chi",
    "id": "2253746565",
    "h_index": 8,
    "papers": 11
   },
   {
    "name": "Zhenjia Xu",
    "id": "74498275",
    "h_index": 15,
    "papers": 22
   },
   {
    "name": "Chuer Pan",
    "id": "2253801737",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Eric Cousineau",
    "id": "2090529",
    "h_index": 12,
    "papers": 16
   },
   {
    "name": "B. Burchfiel",
    "id": "2302757",
    "h_index": 18,
    "papers": 30
   },
   {
    "name": "Siyuan Feng",
    "id": "2284620540",
    "h_index": 8,
    "papers": 10
   },
   {
    "name": "Russ Tedrake",
    "id": "1726802",
    "h_index": 76,
    "papers": 270
   },
   {
    "name": "Shuran Song",
    "id": "2254874914",
    "h_index": 12,
    "papers": 12
   }
  ],
  "comment": "Project website: https://umi-gripper.github.io",
  "topics": [
   "egocentric-data",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2402.10329v3",
  "pdf_url": "https://arxiv.org/pdf/2402.10329v3",
  "html_url": "https://arxiv.org/html/2402.10329v3",
  "code_url": "https://umi-gripper.github.io",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.31
 },
 {
  "id": "2405.02292",
  "slug": "aloha-2-an-enhanced-low-cost-hardware-for-bimanual-teleoperation",
  "title": "ALOHA 2: An Enhanced Low-Cost Hardware for Bimanual Teleoperation",
  "abstract": "Diverse demonstration datasets have powered significant advances in robot learning, but the dexterity and scale of such data can be limited by the hardware cost, the hardware robustness, and the ease of teleoperation. We introduce ALOHA 2, an enhanced version of ALOHA that has greater performance, ergonomics, and robustness compared to the original design. To accelerate research in large-scale bimanual manipulation, we open source all hardware designs of ALOHA 2 with a detailed tutorial, together with a MuJoCo model of ALOHA 2 with system identification. See the project website at aloha-2.github.io.",
  "published": "2024-02-07",
  "updated": "2024-02-07",
  "year": "2024",
  "authors": [
   " ALOHA 2 Team",
   "Jorge Aldaco",
   "Travis Armstrong",
   "Robert Baruch",
   "Jeff Bingham",
   "Sanky Chan",
   "Kenneth Draper",
   "Debidatta Dwibedi",
   "Chelsea Finn",
   "Pete Florence",
   "Spencer Goodrich",
   "Wayne Gramlich",
   "Torr Hage",
   "Alexander Herzog",
   "Jonathan Hoech",
   "Thinh Nguyen",
   "Ian Storz",
   "Baruch Tabanpour",
   "Leila Takayama",
   "Jonathan Tompson",
   "Ayzaan Wahid",
   "Ted Wahrburg",
   "Sichun Xu",
   "Sergey Yaroshenko",
   "Kevin Zakka",
   "Tony Z. Zhao"
  ],
  "author_count": 26,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 116,
  "influential_citations": 7,
  "tldr": "This work introduces ALOHA 2, an enhanced version of ALOHA that has greater performance, ergonomics, and robustness compared to the original design and opens source all hardware designs of ALOHA 2.",
  "doi": "10.48550/arXiv.2405.02292",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "A. Team",
    "id": "144982403",
    "h_index": 15,
    "papers": 102
   },
   {
    "name": "J. Aldaco",
    "id": "118490061",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Travis Armstrong",
    "id": "2057103965",
    "h_index": 9,
    "papers": 11
   },
   {
    "name": "Robert Baruch",
    "id": "1659197538",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Jeff Bingham",
    "id": "1696556124",
    "h_index": 8,
    "papers": 9
   },
   {
    "name": "S. Chan",
    "id": "2300255934",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Kenneth L. Draper",
    "id": "103508342",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Debidatta Dwibedi",
    "id": "2420123",
    "h_index": 19,
    "papers": 35
   },
   {
    "name": "Chelsea Finn",
    "id": "2257346440",
    "h_index": 23,
    "papers": 32
   },
   {
    "name": "Pete Florence",
    "id": "2264974363",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Spencer Goodrich",
    "id": "2300097273",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Wayne Gramlich",
    "id": "2778410",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Torr Hage",
    "id": "2300095742",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Alex Herzog",
    "id": "2253642204",
    "h_index": 8,
    "papers": 14
   },
   {
    "name": "Jonathan Hoech",
    "id": "2300094924",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Thinh Nguyen",
    "id": "2277427185",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Ian Storz",
    "id": "2300094916",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Baruch Tabanpour",
    "id": "2215873836",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Leila Takayama",
    "id": "2281641163",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Jonathan Tompson",
    "id": "2704494",
    "h_index": 43,
    "papers": 71
   },
   {
    "name": "Ayzaan Wahid",
    "id": "88728227",
    "h_index": 21,
    "papers": 27
   },
   {
    "name": "Ted Wahrburg",
    "id": "2300096523",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Sichun Xu",
    "id": "3068504",
    "h_index": 11,
    "papers": 16
   },
   {
    "name": "Sergey Yaroshenko",
    "id": "2102918715",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Kevin Zakka",
    "id": "1390101014",
    "h_index": 9,
    "papers": 10
   },
   {
    "name": "Tony Zhao",
    "id": "145914976",
    "h_index": 17,
    "papers": 19
   }
  ],
  "comment": "Project website: aloha-2.github.io",
  "topics": [
   "dexterous-manipulation",
   "data-teleop",
   "hardware-codesign",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2405.02292v1",
  "pdf_url": "https://arxiv.org/pdf/2405.02292v1",
  "html_url": "https://arxiv.org/html/2405.02292v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.07
 },
 {
  "id": "2402.03300",
  "slug": "deepseekmath-pushing-the-limits-of-mathematical-reasoning-in-open-lang",
  "title": "DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models",
  "abstract": "Mathematical reasoning poses a significant challenge for language models due to its complex and structured nature. In this paper, we introduce DeepSeekMath 7B, which continues pre-training DeepSeek-Coder-Base-v1.5 7B with 120B math-related tokens sourced from Common Crawl, together with natural language and code data. DeepSeekMath 7B has achieved an impressive score of 51.7% on the competition-level MATH benchmark without relying on external toolkits and voting techniques, approaching the performance level of Gemini-Ultra and GPT-4. Self-consistency over 64 samples from DeepSeekMath 7B achieves 60.9% on MATH. The mathematical reasoning capability of DeepSeekMath is attributed to two key factors: First, we harness the significant potential of publicly available web data through a meticulously engineered data selection pipeline. Second, we introduce Group Relative Policy Optimization (GRPO), a variant of Proximal Policy Optimization (PPO), that enhances mathematical reasoning abilities while concurrently optimizing the memory usage of PPO.",
  "published": "2024-02-05",
  "updated": "2024-04-27",
  "year": "2024",
  "authors": [
   "Zhihong Shao",
   "Peiyi Wang",
   "Qihao Zhu",
   "Runxin Xu",
   "Junxiao Song",
   "Xiao Bi",
   "Haowei Zhang",
   "Mingchuan Zhang",
   "Y. K. Li",
   "Y. Wu",
   "Daya Guo"
  ],
  "author_count": 11,
  "categories": [
   "cs.CL",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.CL",
  "venue": "",
  "venue_source": "",
  "citations": 8388,
  "influential_citations": 2440,
  "tldr": "Group Relative Policy Optimization (GRPO), a variant of Proximal Policy Optimization (PPO), that enhances mathematical reasoning abilities while concurrently optimizing the memory usage of PPO is introduced.",
  "doi": "10.48550/arXiv.2402.03300",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhihong Shao",
    "id": "144485528",
    "h_index": 15,
    "papers": 27
   },
   {
    "name": "Peiyi Wang",
    "id": "144202874",
    "h_index": 16,
    "papers": 29
   },
   {
    "name": "Qihao Zhu",
    "id": "2278223869",
    "h_index": 13,
    "papers": 39
   },
   {
    "name": "R. Xu",
    "id": "2274917091",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Jun-Mei Song",
    "id": "2258088582",
    "h_index": 11,
    "papers": 34
   },
   {
    "name": "Mingchuan Zhang",
    "id": "1680884",
    "h_index": 21,
    "papers": 229
   },
   {
    "name": "Y. K. Li",
    "id": "2278599324",
    "h_index": 7,
    "papers": 18
   },
   {
    "name": "Yu Wu",
    "id": "49176273",
    "h_index": 33,
    "papers": 51
   },
   {
    "name": "Daya Guo",
    "id": "2278834796",
    "h_index": 26,
    "papers": 49
   }
  ],
  "comment": "",
  "topics": [
   "rl-control",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2402.03300v3",
  "pdf_url": "https://arxiv.org/pdf/2402.03300v3",
  "html_url": "https://arxiv.org/html/2402.03300v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.0
 },
 {
  "id": "2402.02511",
  "slug": "poco-policy-composition-from-and-for-heterogeneous-robot-learning",
  "title": "PoCo: Policy Composition from and for Heterogeneous Robot Learning",
  "abstract": "Training general robotic policies from heterogeneous data for different tasks is a significant challenge. Existing robotic datasets vary in different modalities such as color, depth, tactile, and proprioceptive information, and collected in different domains such as simulation, real robots, and human videos. Current methods usually collect and pool all data from one domain to train a single policy to handle such heterogeneity in tasks and domains, which is prohibitively expensive and difficult. In this work, we present a flexible approach, dubbed Policy Composition, to combine information across such diverse modalities and domains for learning scene-level and task-level generalized manipulation skills, by composing different data distributions represented with diffusion models. Our method can use task-level composition for multi-task manipulation and be composed with analytic cost functions to adapt policy behaviors at inference time. We train our method on simulation, human, and real robot data and evaluate in tool-use tasks. The composed policy achieves robust and dexterous performance under varying scenes and tasks and outperforms baselines from a single data source in both simulation and real-world experiments. See https://liruiw.github.io/policycomp for more details .",
  "published": "2024-02-04",
  "updated": "2024-12-01",
  "year": "2024",
  "authors": [
   "Lirui Wang",
   "Jialiang Zhao",
   "Yilun Du",
   "Edward H. Adelson",
   "Russ Tedrake"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "RSS",
  "venue_source": "semantic-scholar",
  "citations": 69,
  "influential_citations": 6,
  "tldr": "This work presents a flexible approach, dubbed Policy Composition, to combine information across such diverse modalities and domains for learning scene-level and task-level generalized manipulation skills, by composing different data distributions represented with diffusion models.",
  "doi": "10.48550/arXiv.2402.02511",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Lirui Wang",
    "id": "2253973819",
    "h_index": 6,
    "papers": 6
   },
   {
    "name": "Jialiang Zhao",
    "id": "2243705199",
    "h_index": 12,
    "papers": 18
   },
   {
    "name": "Yilun Du",
    "id": "15394275",
    "h_index": 48,
    "papers": 86
   },
   {
    "name": "E. Adelson",
    "id": "145358192",
    "h_index": 87,
    "papers": 272
   },
   {
    "name": "Russ Tedrake",
    "id": "2263905014",
    "h_index": 14,
    "papers": 36
   }
  ],
  "comment": "R:SS 2024",
  "topics": [
   "dexterous-manipulation",
   "egocentric-data",
   "tactile",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2402.02511v3",
  "pdf_url": "https://arxiv.org/pdf/2402.02511v3",
  "html_url": "https://arxiv.org/html/2402.02511v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.35
 },
 {
  "id": "2401.18084",
  "slug": "binding-touch-to-everything-learning-unified-multimodal-tactile-repres",
  "title": "Binding Touch to Everything: Learning Unified Multimodal Tactile Representations",
  "abstract": "The ability to associate touch with other modalities has huge implications for humans and computational systems. However, multimodal learning with touch remains challenging due to the expensive data collection process and non-standardized sensor outputs. We introduce UniTouch, a unified tactile model for vision-based touch sensors connected to multiple modalities, including vision, language, and sound. We achieve this by aligning our UniTouch embeddings to pretrained image embeddings already associated with a variety of other modalities. We further propose learnable sensor-specific tokens, allowing the model to learn from a set of heterogeneous tactile sensors, all at the same time. UniTouch is capable of conducting various touch sensing tasks in the zero-shot setting, from robot grasping prediction to touch image question answering. To the best of our knowledge, UniTouch is the first to demonstrate such capabilities. Project page: https://cfeng16.github.io/UniTouch/",
  "published": "2024-01-31",
  "updated": "2024-01-31",
  "year": "2024",
  "authors": [
   "Fengyu Yang",
   "Chao Feng",
   "Ziyang Chen",
   "Hyoungseob Park",
   "Daniel Wang",
   "Yiming Dou",
   "Ziyao Zeng",
   "Xien Chen",
   "Rit Gangopadhyay",
   "Andrew Owens",
   "Alex Wong"
  ],
  "author_count": 11,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 167,
  "influential_citations": 14,
  "tldr": "UniTouch is a unified model for vision-based touch sensors that connects their tactile signals to other modalities, including vision, language, and sound, by aligning their tactile embeddings to pretrained image embeddings already associated with a variety of other modalities.",
  "doi": "10.1109/CVPR52733.2024.02488",
  "oa_pdf": "https://arxiv.org/pdf/2401.18084",
  "s2_authors": [
   {
    "name": "Fengyu Yang",
    "id": "2281996692",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Chao Feng",
    "id": "2282280945",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Ziyang Chen",
    "id": "2294016182",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Hyoungseob Park",
    "id": "2258800605",
    "h_index": 8,
    "papers": 16
   },
   {
    "name": "Daniel Wang",
    "id": "2281966829",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Yiming Dou",
    "id": "2190753845",
    "h_index": 7,
    "papers": 12
   },
   {
    "name": "Ziyao Zeng",
    "id": "2282099512",
    "h_index": 6,
    "papers": 21
   },
   {
    "name": "Xien Chen",
    "id": "2282088951",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Rit Gangopadhyay",
    "id": "2281943421",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Andrew Owens",
    "id": "2281942935",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Alex Wong",
    "id": "2282549271",
    "h_index": 6,
    "papers": 13
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2401.18084v1",
  "pdf_url": "https://arxiv.org/pdf/2401.18084v1",
  "html_url": "https://arxiv.org/html/2401.18084v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.73
 },
 {
  "id": "2401.17583",
  "slug": "agile-but-safe-learning-collision-free-high-speed-legged-locomotion",
  "title": "Agile But Safe: Learning Collision-Free High-Speed Legged Locomotion",
  "abstract": "Legged robots navigating cluttered environments must be jointly agile for efficient task execution and safe to avoid collisions with obstacles or humans. Existing studies either develop conservative controllers (< 1.0 m/s) to ensure safety, or focus on agility without considering potentially fatal collisions. This paper introduces Agile But Safe (ABS), a learning-based control framework that enables agile and collision-free locomotion for quadrupedal robots. ABS involves an agile policy to execute agile motor skills amidst obstacles and a recovery policy to prevent failures, collaboratively achieving high-speed and collision-free navigation. The policy switch in ABS is governed by a learned control-theoretic reach-avoid value network, which also guides the recovery policy as an objective function, thereby safeguarding the robot in a closed loop. The training process involves the learning of the agile policy, the reach-avoid value network, the recovery policy, and an exteroception representation network, all in simulation. These trained modules can be directly deployed in the real world with onboard sensing and computation, leading to high-speed and collision-free navigation in confined indoor and outdoor spaces with both static and dynamic obstacles.",
  "published": "2024-01-31",
  "updated": "2024-05-21",
  "year": "2024",
  "authors": [
   "Tairan He",
   "Chong Zhang",
   "Wenli Xiao",
   "Guanqi He",
   "Changliu Liu",
   "Guanya Shi"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "RSS",
  "venue_source": "semantic-scholar",
  "citations": 156,
  "influential_citations": 10,
  "tldr": "Agile But Safe (ABS) is introduced, a learning-based control framework that enables agile and collision-free locomotion for quadrupedal robots and involves an agile policy to execute agile motor skills amidst obstacles and a recovery policy to prevent failures, collaboratively achieving high-speed and collision-free navigation.",
  "doi": "10.48550/arXiv.2401.17583",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Tairan He",
    "id": "2055132189",
    "h_index": 20,
    "papers": 29
   },
   {
    "name": "Chong Zhang",
    "id": "2254275727",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Wenli Xiao",
    "id": "2147212066",
    "h_index": 15,
    "papers": 32
   },
   {
    "name": "Guanqi He",
    "id": "2279862620",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "Changliu Liu",
    "id": "2282143156",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Guanya Shi",
    "id": "2249759531",
    "h_index": 20,
    "papers": 31
   }
  ],
  "comment": "Published at RSS 2024, Project website: https://agile-but-safe.github.io/",
  "topics": [
   "humanoids",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2401.17583v3",
  "pdf_url": "https://arxiv.org/pdf/2401.17583v3",
  "html_url": "https://arxiv.org/html/2401.17583v3",
  "code_url": "https://agile-but-safe.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.7
 },
 {
  "id": "2401.14403",
  "slug": "adaptive-mobile-manipulation-for-articulated-objects-in-the-open-world",
  "title": "Adaptive Mobile Manipulation for Articulated Objects In the Open World",
  "abstract": "Deploying robots in open-ended unstructured environments such as homes has been a long-standing research problem. However, robots are often studied only in closed-off lab settings, and prior mobile manipulation work is restricted to pick-move-place, which is arguably just the tip of the iceberg in this area. In this paper, we introduce Open-World Mobile Manipulation System, a full-stack approach to tackle realistic articulated object operation, e.g. real-world doors, cabinets, drawers, and refrigerators in open-ended unstructured environments. The robot utilizes an adaptive learning framework to initially learns from a small set of data through behavior cloning, followed by learning from online practice on novel objects that fall outside the training distribution. We also develop a low-cost mobile manipulation hardware platform capable of safe and autonomous online adaptation in unstructured environments with a cost of around 20,000 USD. In our experiments we utilize 20 articulate objects across 4 buildings in the CMU campus. With less than an hour of online learning for each object, the system is able to increase success rate from 50% of BC pre-training to 95% using online adaptation. Video results at https://open-world-mobilemanip.github.io/",
  "published": "2024-01-25",
  "updated": "2024-01-28",
  "year": "2024",
  "authors": [
   "Haoyu Xiong",
   "Russell Mendonca",
   "Kenneth Shaw",
   "Deepak Pathak"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 78,
  "influential_citations": 2,
  "tldr": "Open-World Mobile Manipulation System, a full-stack approach to tackle realistic articulated object operation, e.g. real-world doors, cabinets, drawers, and refrigerators in open-ended unstructured environments, is introduced.",
  "doi": "10.48550/arXiv.2401.14403",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Haoyu Xiong",
    "id": "2281036863",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "R. Mendonca",
    "id": "35509365",
    "h_index": 15,
    "papers": 24
   },
   {
    "name": "Kenneth Shaw",
    "id": "2263541750",
    "h_index": 10,
    "papers": 18
   },
   {
    "name": "Deepak Pathak",
    "id": "2269734979",
    "h_index": 11,
    "papers": 17
   }
  ],
  "comment": "Website at https://open-world-mobilemanip.github.io/",
  "topics": [
   "imitation-diffusion",
   "navigation",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2401.14403v2",
  "pdf_url": "https://arxiv.org/pdf/2401.14403v2",
  "html_url": "https://arxiv.org/html/2401.14403v2",
  "code_url": "https://open-world-mobilemanip.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.9
 },
 {
  "id": "2401.14391",
  "slug": "rethinking-patch-dependence-for-masked-autoencoders",
  "title": "Rethinking Patch Dependence for Masked Autoencoders",
  "abstract": "In this work, we examine the impact of inter-patch dependencies in the decoder of masked autoencoders (MAE) on representation learning. We decompose the decoding mechanism for masked reconstruction into self-attention between mask tokens and cross-attention between masked and visible tokens. Our findings reveal that MAE reconstructs coherent images from visible patches not through interactions between patches in the decoder but by learning a global representation within the encoder. This discovery leads us to propose a simple visual pretraining framework: cross-attention masked autoencoders (CrossMAE). This framework employs only cross-attention in the decoder to independently read out reconstructions for a small subset of masked patches from encoder outputs. This approach achieves comparable or superior performance to traditional MAE across models ranging from ViT-S to ViT-H and significantly reduces computational requirements. By its design, CrossMAE challenges the necessity of interaction between mask tokens for effective masked pretraining. Code and models are publicly available: https://crossmae.github.io",
  "published": "2024-01-25",
  "updated": "2025-04-10",
  "year": "2024",
  "authors": [
   "Letian Fu",
   "Long Lian",
   "Renhao Wang",
   "Baifeng Shi",
   "Xudong Wang",
   "Adam Yala",
   "Trevor Darrell",
   "Alexei A. Efros",
   "Ken Goldberg"
  ],
  "author_count": 9,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "Trans. Mach. Learn. Res.",
  "venue_source": "semantic-scholar",
  "citations": 46,
  "influential_citations": 9,
  "tldr": "It is revealed that MAE reconstructs coherent images from visible patches not through interactions between patches in the decoder but by learning a global representation within the encoder, which leads to a simple visual pretraining framework: CrossMAE.",
  "doi": "10.48550/arXiv.2401.14391",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Letian Fu",
    "id": "2087112499",
    "h_index": 12,
    "papers": 28
   },
   {
    "name": "Long Lian",
    "id": "2249587117",
    "h_index": 10,
    "papers": 18
   },
   {
    "name": "Renhao Wang",
    "id": "2281072309",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Baifeng Shi",
    "id": "1596823732",
    "h_index": 17,
    "papers": 24
   },
   {
    "name": "Xudong Wang",
    "id": "2276668087",
    "h_index": 11,
    "papers": 16
   },
   {
    "name": "A. Yala",
    "id": "2281035913",
    "h_index": 7,
    "papers": 18
   },
   {
    "name": "Trevor Darrell",
    "id": "2257173088",
    "h_index": 12,
    "papers": 18
   },
   {
    "name": "Alexei A. Efros",
    "id": "1763086",
    "h_index": 111,
    "papers": 241
   },
   {
    "name": "Ken Goldberg",
    "id": "2281037146",
    "h_index": 5,
    "papers": 11
   }
  ],
  "comment": "Transactions on Machine Learning Research (TMLR) 2025",
  "topics": [
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2401.14391v2",
  "pdf_url": "https://arxiv.org/pdf/2401.14391v2",
  "html_url": "https://arxiv.org/html/2401.14391v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.17
 },
 {
  "id": "2401.14159",
  "slug": "grounded-sam-assembling-open-world-models-for-diverse-visual-tasks",
  "title": "Grounded SAM: Assembling Open-World Models for Diverse Visual Tasks",
  "abstract": "We introduce Grounded SAM, which uses Grounding DINO as an open-set object detector to combine with the segment anything model (SAM). This integration enables the detection and segmentation of any regions based on arbitrary text inputs and opens a door to connecting various vision models. As shown in Fig.1, a wide range of vision tasks can be achieved by using the versatile Grounded SAM pipeline. For example, an automatic annotation pipeline based solely on input images can be realized by incorporating models such as BLIP and Recognize Anything. Additionally, incorporating Stable-Diffusion allows for controllable image editing, while the integration of OSX facilitates promptable 3D human motion analysis. Grounded SAM also shows superior performance on open-vocabulary benchmarks, achieving 48.7 mean AP on SegInW (Segmentation in the wild) zero-shot benchmark with the combination of Grounding DINO-Base and SAM-Huge models.",
  "published": "2024-01-25",
  "updated": "2024-01-25",
  "year": "2024",
  "authors": [
   "Tianhe Ren",
   "Shilong Liu",
   "Ailing Zeng",
   "Jing Lin",
   "Kunchang Li",
   "He Cao",
   "Jiayu Chen",
   "Xinyu Huang",
   "Yukang Chen",
   "Feng Yan",
   "Zhaoyang Zeng",
   "Hao Zhang",
   "Feng Li",
   "Jie Yang",
   "Hongyang Li",
   "Qing Jiang",
   "Lei Zhang"
  ],
  "author_count": 17,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 1268,
  "influential_citations": 125,
  "tldr": "This work introduces Grounded SAM, which uses Grounding DINO as an open-set object detector to combine with the segment anything model (SAM), which enables the detection and segmentation of any regions based on arbitrary text inputs and opens a door to connecting various vision models.",
  "doi": "10.48550/arXiv.2401.14159",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Tianhe Ren",
    "id": "2247536954",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Shilong Liu",
    "id": "8602739",
    "h_index": 20,
    "papers": 36
   },
   {
    "name": "Ailing Zeng",
    "id": "2238950645",
    "h_index": 13,
    "papers": 18
   },
   {
    "name": "Jing Lin",
    "id": "2318281153",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Kunchang Li",
    "id": "2335920477",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "He Cao",
    "id": "2153244338",
    "h_index": 10,
    "papers": 18
   },
   {
    "name": "Jiayu Chen",
    "id": "2365265012",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Xinyu Huang",
    "id": "2160051569",
    "h_index": 9,
    "papers": 26
   },
   {
    "name": "Yukang Chen",
    "id": "2109297557",
    "h_index": 31,
    "papers": 63
   },
   {
    "name": "Feng Yan",
    "id": "2345772022",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Zhaoyang Zeng",
    "id": "2267879736",
    "h_index": 12,
    "papers": 15
   },
   {
    "name": "Hao Zhang",
    "id": "2315254849",
    "h_index": 22,
    "papers": 36
   },
   {
    "name": "Feng Li",
    "id": "2152978390",
    "h_index": 26,
    "papers": 42
   },
   {
    "name": "Jie Yang",
    "id": "2279864422",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Hongyang Li",
    "id": "2267783957",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "Qing Jiang",
    "id": "2267869347",
    "h_index": 10,
    "papers": 19
   },
   {
    "name": "Lei Zhang",
    "id": "2272135297",
    "h_index": 3,
    "papers": 3
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2401.14159v1",
  "pdf_url": "https://arxiv.org/pdf/2401.14159v1",
  "html_url": "https://arxiv.org/html/2401.14159v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.0
 },
 {
  "id": "2401.12024",
  "slug": "multimodal-visual-tactile-representation-learning-through-self-supervi",
  "title": "Multimodal Visual-Tactile Representation Learning through Self-Supervised Contrastive Pre-Training",
  "abstract": "The rapidly evolving field of robotics necessitates methods that can facilitate the fusion of multiple modalities. Specifically, when it comes to interacting with tangible objects, effectively combining visual and tactile sensory data is key to understanding and navigating the complex dynamics of the physical world, enabling a more nuanced and adaptable response to changing environments. Nevertheless, much of the earlier work in merging these two sensory modalities has relied on supervised methods utilizing datasets labeled by humans.This paper introduces MViTac, a novel methodology that leverages contrastive learning to integrate vision and touch sensations in a self-supervised fashion. By availing both sensory inputs, MViTac leverages intra and inter-modality losses for learning representations, resulting in enhanced material property classification and more adept grasping prediction. Through a series of experiments, we showcase the effectiveness of our method and its superiority over existing state-of-the-art self-supervised and supervised techniques. In evaluating our methodology, we focus on two distinct tasks: material classification and grasping success prediction. Our results indicate that MViTac facilitates the development of improved modality encoders, yielding more robust representations as evidenced by linear probing assessments.",
  "published": "2024-01-22",
  "updated": "2024-01-22",
  "year": "2024",
  "authors": [
   "Vedant Dave",
   "Fotios Lygerakis",
   "Elmar Rueckert"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA",
  "venue_source": "semantic-scholar",
  "citations": 65,
  "influential_citations": 6,
  "tldr": "MViTac is introduced, a novel methodology that leverages contrastive learning to integrate vision and touch sensations in a self-supervised fashion and facilitates the development of improved modality encoders, yielding more robust representations as evidenced by linear probing assessments.",
  "doi": "10.1109/ICRA57147.2024.10610228",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Vedant Dave",
    "id": "2138217785",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Fotios Lygerakis",
    "id": "1500661955",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Elmar Rueckert",
    "id": "35739127",
    "h_index": 13,
    "papers": 73
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2401.12024v1",
  "pdf_url": "https://arxiv.org/pdf/2401.12024v1",
  "html_url": "https://arxiv.org/html/2401.12024v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.32
 },
 {
  "id": "2401.11439",
  "slug": "general-flow-as-foundation-affordance-for-scalable-robot-learning",
  "title": "General Flow as Foundation Affordance for Scalable Robot Learning",
  "abstract": "We address the challenge of acquiring real-world manipulation skills with a scalable framework. We hold the belief that identifying an appropriate prediction target capable of leveraging large-scale datasets is crucial for achieving efficient and universal learning. Therefore, we propose to utilize 3D flow, which represents the future trajectories of 3D points on objects of interest, as an ideal prediction target. To exploit scalable data resources, we turn our attention to human videos. We develop, for the first time, a language-conditioned 3D flow prediction model directly from large-scale RGBD human video datasets. Our predicted flow offers actionable guidance, thus facilitating zero-shot skill transfer in real-world scenarios. We deploy our method with a policy based on closed-loop flow prediction. Remarkably, without any in-domain finetuning, our method achieves an impressive 81\\% success rate in zero-shot human-to-robot skill transfer, covering 18 tasks in 6 scenes. Our framework features the following benefits: (1) scalability: leveraging cross-embodiment data resources; (2) wide application: multiple object categories, including rigid, articulated, and soft bodies; (3) stable skill transfer: providing actionable guidance with a small inference domain-gap. Code, data, and supplementary materials are available https://general-flow.github.io",
  "published": "2024-01-21",
  "updated": "2024-09-23",
  "year": "2024",
  "authors": [
   "Chengbo Yuan",
   "Chuan Wen",
   "Tong Zhang",
   "Yang Gao"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 114,
  "influential_citations": 20,
  "tldr": "This work develops, for the first time, a language-conditioned 3D flow prediction model directly from large-scale RGBD human video datasets, and achieves an impressive 81\\% success rate in zero-shot human-to-robot skill transfer.",
  "doi": "10.48550/arXiv.2401.11439",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Chengbo Yuan",
    "id": "2280187985",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Chuan Wen",
    "id": "2068033698",
    "h_index": 14,
    "papers": 25
   },
   {
    "name": "Tong Zhang",
    "id": "2270989945",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Yang Gao",
    "id": "2257027030",
    "h_index": 7,
    "papers": 10
   }
  ],
  "comment": "https://general-flow.github.io",
  "topics": [
   "egocentric-data",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2401.11439v2",
  "pdf_url": "https://arxiv.org/pdf/2401.11439v2",
  "html_url": "https://arxiv.org/html/2401.11439v2",
  "code_url": "https://general-flow.github.io",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.56
 },
 {
  "id": "2401.08541",
  "slug": "scalable-pre-training-of-large-autoregressive-image-models",
  "title": "Scalable Pre-training of Large Autoregressive Image Models",
  "abstract": "This paper introduces AIM, a collection of vision models pre-trained with an autoregressive objective. These models are inspired by their textual counterparts, i.e., Large Language Models (LLMs), and exhibit similar scaling properties. Specifically, we highlight two key findings: (1) the performance of the visual features scale with both the model capacity and the quantity of data, (2) the value of the objective function correlates with the performance of the model on downstream tasks. We illustrate the practical implication of these findings by pre-training a 7 billion parameter AIM on 2 billion images, that achieves 84.0% on ImageNet-1k with a frozen trunk. Interestingly, even at this scale, we observe no sign of saturation in performance, suggesting that AIM potentially represents a new frontier for training large-scale vision models. The pre-training of AIM is similar to the pre-training of LLMs, and does not require any image-specific strategy to stabilize the training at scale.",
  "published": "2024-01-16",
  "updated": "2024-01-16",
  "year": "2024",
  "authors": [
   "Alaaeldin El-Nouby",
   "Michal Klein",
   "Shuangfei Zhai",
   "Miguel Angel Bautista",
   "Alexander Toshev",
   "Vaishaal Shankar",
   "Joshua M Susskind",
   "Armand Joulin"
  ],
  "author_count": 8,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "ICML",
  "venue_source": "semantic-scholar",
  "citations": 145,
  "influential_citations": 12,
  "tldr": "Two key findings are highlighted: (1) the performance of the visual features scale with both the model capacity and the quantity of data, and (2) the value of the objective function correlates with the performance of the model on downstream tasks.",
  "doi": "10.48550/arXiv.2401.08541",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Alaaeldin El-Nouby",
    "id": "1388811741",
    "h_index": 20,
    "papers": 26
   },
   {
    "name": "Michal Klein",
    "id": "2279733001",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Shuangfei Zhai",
    "id": "2443456",
    "h_index": 28,
    "papers": 70
   },
   {
    "name": "Miguel Angel Bautista",
    "id": "2258553333",
    "h_index": 10,
    "papers": 20
   },
   {
    "name": "Alexander Toshev",
    "id": "1726415",
    "h_index": 42,
    "papers": 82
   },
   {
    "name": "Vaishaal Shankar",
    "id": "34961417",
    "h_index": 26,
    "papers": 51
   },
   {
    "name": "J. Susskind",
    "id": "49158771",
    "h_index": 36,
    "papers": 96
   },
   {
    "name": "Armand Joulin",
    "id": "2319608",
    "h_index": 72,
    "papers": 151
   }
  ],
  "comment": "https://github.com/apple/ml-aim",
  "topics": [
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2401.08541v1",
  "pdf_url": "https://arxiv.org/pdf/2401.08541v1",
  "html_url": "https://arxiv.org/html/2401.08541v1",
  "code_url": "https://github.com/apple/ml-aim",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.66
 },
 {
  "id": "2401.08399",
  "slug": "taco-benchmarking-generalizable-bimanual-tool-action-object-understand",
  "title": "TACO: Benchmarking Generalizable Bimanual Tool-ACtion-Object Understanding",
  "abstract": "Humans commonly work with multiple objects in daily life and can intuitively transfer manipulation skills to novel objects by understanding object functional regularities. However, existing technical approaches for analyzing and synthesizing hand-object manipulation are mostly limited to handling a single hand and object due to the lack of data support. To address this, we construct TACO, an extensive bimanual hand-object-interaction dataset spanning a large variety of tool-action-object compositions for daily human activities. TACO contains 2.5K motion sequences paired with third-person and egocentric views, precise hand-object 3D meshes, and action labels. To rapidly expand the data scale, we present a fully automatic data acquisition pipeline combining multi-view sensing with an optical motion capture system. With the vast research fields provided by TACO, we benchmark three generalizable hand-object-interaction tasks: compositional action recognition, generalizable hand-object motion forecasting, and cooperative grasp synthesis. Extensive experiments reveal new insights, challenges, and opportunities for advancing the studies of generalizable hand-object motion analysis and synthesis. Our data and code are available at https://taco2024.github.io.",
  "published": "2024-01-16",
  "updated": "2024-03-25",
  "year": "2024",
  "authors": [
   "Yun Liu",
   "Haolin Yang",
   "Xu Si",
   "Ling Liu",
   "Zipeng Li",
   "Yuxiang Zhang",
   "Yebin Liu",
   "Li Yi"
  ],
  "author_count": 8,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 103,
  "influential_citations": 10,
  "tldr": "TACO, an extensive bimanual hand-object-interaction dataset spanning a large variety of tool-action-object compositions for daily human activities, is constructed and benchmark three generalizable hand-object-interaction tasks: compositional action recognition, generalizable hand-object motion forecasting, and cooperative grasp synthesis.",
  "doi": "10.1109/CVPR52733.2024.02054",
  "oa_pdf": "https://arxiv.org/pdf/2401.08399",
  "s2_authors": [
   {
    "name": "Yun Liu",
    "id": "2279787165",
    "h_index": 10,
    "papers": 16
   },
   {
    "name": "Haolin Yang",
    "id": "2279759074",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Xu Si",
    "id": "2238706993",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Lingqin Liu",
    "id": "2400519193",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Zipeng Li",
    "id": "2268027582",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Yuxiang Zhang",
    "id": "2269686518",
    "h_index": 10,
    "papers": 15
   },
   {
    "name": "Yebin Liu",
    "id": "2268716372",
    "h_index": 10,
    "papers": 25
   },
   {
    "name": "Li Yi",
    "id": "2279808630",
    "h_index": 6,
    "papers": 9
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "egocentric-data"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2401.08399v2",
  "pdf_url": "https://arxiv.org/pdf/2401.08399v2",
  "html_url": "https://arxiv.org/html/2401.08399v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.52
 },
 {
  "id": "2401.06209",
  "slug": "eyes-wide-shut-exploring-the-visual-shortcomings-of-multimodal-llms",
  "title": "Eyes Wide Shut? Exploring the Visual Shortcomings of Multimodal LLMs",
  "abstract": "Is vision good enough for language? Recent advancements in multimodal models primarily stem from the powerful reasoning abilities of large language models (LLMs). However, the visual component typically depends only on the instance-level contrastive language-image pre-training (CLIP). Our research reveals that the visual capabilities in recent multimodal LLMs (MLLMs) still exhibit systematic shortcomings. To understand the roots of these errors, we explore the gap between the visual embedding space of CLIP and vision-only self-supervised learning. We identify ''CLIP-blind pairs'' - images that CLIP perceives as similar despite their clear visual differences. With these pairs, we construct the Multimodal Visual Patterns (MMVP) benchmark. MMVP exposes areas where state-of-the-art systems, including GPT-4V, struggle with straightforward questions across nine basic visual patterns, often providing incorrect answers and hallucinated explanations. We further evaluate various CLIP-based vision-and-language models and found a notable correlation between visual patterns that challenge CLIP models and those problematic for multimodal LLMs. As an initial effort to address these issues, we propose a Mixture of Features (MoF) approach, demonstrating that integrating vision self-supervised learning features with MLLMs can significantly enhance their visual grounding capabilities. Together, our research suggests visual representation learning remains an open challenge, and accurate visual grounding is crucial for future successful multimodal systems.",
  "published": "2024-01-11",
  "updated": "2024-04-25",
  "year": "2024",
  "authors": [
   "Shengbang Tong",
   "Zhuang Liu",
   "Yuexiang Zhai",
   "Yi Ma",
   "Yann LeCun",
   "Saining Xie"
  ],
  "author_count": 6,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 858,
  "influential_citations": 109,
  "tldr": "It is demonstrated that integrating vision self-supervised learning features with MLLMs can significantly enhance their visual grounding capabilities, suggesting visual representation learning remains an open challenge, and accurate visual grounding is crucial for future successful multimodal systems.",
  "doi": "10.1109/CVPR52733.2024.00914",
  "oa_pdf": "https://arxiv.org/pdf/2401.06209",
  "s2_authors": [
   {
    "name": "Shengbang Tong",
    "id": "2143202419",
    "h_index": 20,
    "papers": 28
   },
   {
    "name": "Zhuang Liu",
    "id": "2109168016",
    "h_index": 30,
    "papers": 40
   },
   {
    "name": "Yuexiang Zhai",
    "id": "119692515",
    "h_index": 19,
    "papers": 30
   },
   {
    "name": "Yi Ma",
    "id": "2255716385",
    "h_index": 7,
    "papers": 14
   },
   {
    "name": "Yann LeCun",
    "id": "2265899558",
    "h_index": 22,
    "papers": 47
   },
   {
    "name": "Saining Xie",
    "id": "2275940976",
    "h_index": 10,
    "papers": 13
   }
  ],
  "comment": "Project page: https://tsb0601.github.io/mmvp_blog/",
  "topics": [
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2401.06209v2",
  "pdf_url": "https://arxiv.org/pdf/2401.06209v2",
  "html_url": "https://arxiv.org/html/2401.06209v2",
  "code_url": "https://tsb0601.github.io/mmvp_blog/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.43
 },
 {
  "id": "2401.02117",
  "slug": "mobile-aloha-learning-bimanual-mobile-manipulation-with-low-cost-whole",
  "title": "Mobile ALOHA: Learning Bimanual Mobile Manipulation with Low-Cost Whole-Body Teleoperation",
  "abstract": "Imitation learning from human demonstrations has shown impressive performance in robotics. However, most results focus on table-top manipulation, lacking the mobility and dexterity necessary for generally useful tasks. In this work, we develop a system for imitating mobile manipulation tasks that are bimanual and require whole-body control. We first present Mobile ALOHA, a low-cost and whole-body teleoperation system for data collection. It augments the ALOHA system with a mobile base, and a whole-body teleoperation interface. Using data collected with Mobile ALOHA, we then perform supervised behavior cloning and find that co-training with existing static ALOHA datasets boosts performance on mobile manipulation tasks. With 50 demonstrations for each task, co-training can increase success rates by up to 90%, allowing Mobile ALOHA to autonomously complete complex mobile manipulation tasks such as sauteing and serving a piece of shrimp, opening a two-door wall cabinet to store heavy cooking pots, calling and entering an elevator, and lightly rinsing a used pan using a kitchen faucet. Project website: https://mobile-aloha.github.io",
  "published": "2024-01-04",
  "updated": "2024-01-04",
  "year": "2024",
  "authors": [
   "Zipeng Fu",
   "Tony Z. Zhao",
   "Chelsea Finn"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 741,
  "influential_citations": 36,
  "tldr": "This work develops a system for imitating mobile manipulation tasks that are bimanual and require whole-body control and presents Mobile ALOHA, a low-cost and whole-body teleoperation system for data collection and supervised behavior cloning.",
  "doi": "10.48550/arXiv.2401.02117",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zipeng Fu",
    "id": "89704471",
    "h_index": 13,
    "papers": 18
   },
   {
    "name": "Tony Zhao",
    "id": "145914976",
    "h_index": 17,
    "papers": 19
   },
   {
    "name": "Chelsea Finn",
    "id": "2239104473",
    "h_index": 7,
    "papers": 7
   }
  ],
  "comment": "Project website: https://mobile-aloha.github.io (Zipeng Fu and Tony Z. Zhao are project co-leads, Chelsea Finn is the advisor)",
  "topics": [
   "dexterous-manipulation",
   "humanoids",
   "egocentric-data",
   "imitation-diffusion",
   "navigation",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2401.02117v1",
  "pdf_url": "https://arxiv.org/pdf/2401.02117v1",
  "html_url": "https://arxiv.org/html/2401.02117v1",
  "code_url": "https://mobile-aloha.github.io",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.87
 },
 {
  "id": "2401.00025",
  "slug": "any-point-trajectory-modeling-for-policy-learning",
  "title": "Any-point Trajectory Modeling for Policy Learning",
  "abstract": "Learning from demonstration is a powerful method for teaching robots new skills, and having more demonstration data often improves policy learning. However, the high cost of collecting demonstration data is a significant bottleneck. Videos, as a rich data source, contain knowledge of behaviors, physics, and semantics, but extracting control-specific information from them is challenging due to the lack of action labels. In this work, we introduce a novel framework, Any-point Trajectory Modeling (ATM), that utilizes video demonstrations by pre-training a trajectory model to predict future trajectories of arbitrary points within a video frame. Once trained, these trajectories provide detailed control guidance, enabling the learning of robust visuomotor policies with minimal action-labeled data. Across over 130 language-conditioned tasks we evaluated in both simulation and the real world, ATM outperforms strong video pre-training baselines by 80% on average. Furthermore, we show effective transfer learning of manipulation skills from human videos and videos from a different robot morphology. Visualizations and code are available at: \\url{https://xingyu-lin.github.io/atm}.",
  "published": "2023-12-28",
  "updated": "2024-07-12",
  "year": "2023",
  "authors": [
   "Chuan Wen",
   "Xingyu Lin",
   "John So",
   "Kai Chen",
   "Qi Dou",
   "Yang Gao",
   "Pieter Abbeel"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "RSS",
  "venue_source": "semantic-scholar",
  "citations": 262,
  "influential_citations": 29,
  "tldr": "A novel framework, Any-point Trajectory Modeling (ATM), is introduced that utilizes video demonstrations by pre-training a trajectory model to predict future trajectories of arbitrary points within a video frame, enabling the learning of robust visuomotor policies with minimal action-labeled data.",
  "doi": "10.48550/arXiv.2401.00025",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Chuan Wen",
    "id": "2068033698",
    "h_index": 14,
    "papers": 25
   },
   {
    "name": "Xingyu Lin",
    "id": "2277977209",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "John So",
    "id": "2188834579",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Kai Chen",
    "id": "2157740829",
    "h_index": 14,
    "papers": 20
   },
   {
    "name": "Q. Dou",
    "id": "2254201408",
    "h_index": 5,
    "papers": 16
   },
   {
    "name": "Yang Gao",
    "id": "2257027030",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Pieter Abbeel",
    "id": "2265490900",
    "h_index": 8,
    "papers": 13
   }
  ],
  "comment": "18 pages, 15 figures",
  "topics": [
   "egocentric-data",
   "foundation-pretraining",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2401.00025v3",
  "pdf_url": "https://arxiv.org/pdf/2401.00025v3",
  "html_url": "https://arxiv.org/html/2401.00025v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.92
 },
 {
  "id": "2312.17142",
  "slug": "dreamgaussian4d-generative-4d-gaussian-splatting",
  "title": "DreamGaussian4D: Generative 4D Gaussian Splatting",
  "abstract": "4D content generation has achieved remarkable progress recently. However, existing methods suffer from long optimization times, a lack of motion controllability, and a low quality of details. In this paper, we introduce DreamGaussian4D (DG4D), an efficient 4D generation framework that builds on Gaussian Splatting (GS). Our key insight is that combining explicit modeling of spatial transformations with static GS makes an efficient and powerful representation for 4D generation. Moreover, video generation methods have the potential to offer valuable spatial-temporal priors, enhancing the high-quality 4D generation. Specifically, we propose an integral framework with two major modules: 1) Image-to-4D GS - we initially generate static GS with DreamGaussianHD, followed by HexPlane-based dynamic generation with Gaussian deformation; and 2) Video-to-Video Texture Refinement - we refine the generated UV-space texture maps and meanwhile enhance their temporal consistency by utilizing a pre-trained image-to-video diffusion model. Notably, DG4D reduces the optimization time from several hours to just a few minutes, allows the generated 3D motion to be visually controlled, and produces animated meshes that can be realistically rendered in 3D engines.",
  "published": "2023-12-28",
  "updated": "2024-06-10",
  "year": "2023",
  "authors": [
   "Jiawei Ren",
   "Liang Pan",
   "Jiaxiang Tang",
   "Chi Zhang",
   "Ang Cao",
   "Gang Zeng",
   "Ziwei Liu"
  ],
  "author_count": 7,
  "categories": [
   "cs.CV",
   "cs.GR"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 226,
  "influential_citations": 45,
  "tldr": "This paper introduces DreamGaussian4D (DG4D), an efficient 4D generation framework that builds on Gaussian Splatting (GS), and suggests that combining explicit modeling of spatial transformations with static GS makes an efficient and powerful representation for 4D generation.",
  "doi": "10.48550/arXiv.2312.17142",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jiawei Ren",
    "id": "1820909323",
    "h_index": 20,
    "papers": 30
   },
   {
    "name": "Liang Pan",
    "id": "2272233402",
    "h_index": 16,
    "papers": 31
   },
   {
    "name": "Jiaxiang Tang",
    "id": "1397711601",
    "h_index": 20,
    "papers": 35
   },
   {
    "name": "Chi Zhang",
    "id": "2276749531",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Ang Cao",
    "id": "2268400549",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Gang Zeng",
    "id": "2247995148",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Ziwei Liu",
    "id": "2249080787",
    "h_index": 9,
    "papers": 9
   }
  ],
  "comment": "Technical report. Project page is at https://jiawei-ren.github.io/projects/dreamgaussian4d Code is at https://github.com/jiawei-ren/dreamgaussian4d",
  "topics": [
   "spatial-3d",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2312.17142v3",
  "pdf_url": "https://arxiv.org/pdf/2312.17142v3",
  "html_url": "https://arxiv.org/html/2312.17142v3",
  "code_url": "https://jiawei-ren.github.io/projects/dreamgaussian4d",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.36
 },
 {
  "id": "2312.14134",
  "slug": "diffusion-reward-learning-rewards-via-conditional-video-diffusion",
  "title": "Diffusion Reward: Learning Rewards via Conditional Video Diffusion",
  "abstract": "Learning rewards from expert videos offers an affordable and effective solution to specify the intended behaviors for reinforcement learning (RL) tasks. In this work, we propose Diffusion Reward, a novel framework that learns rewards from expert videos via conditional video diffusion models for solving complex visual RL problems. Our key insight is that lower generative diversity is exhibited when conditioning diffusion on expert trajectories. Diffusion Reward is accordingly formalized by the negative of conditional entropy that encourages productive exploration of expert behaviors. We show the efficacy of our method over robotic manipulation tasks in both simulation platforms and the real world with visual input. Moreover, Diffusion Reward can even solve unseen tasks successfully and effectively, largely surpassing baseline methods. Project page and code: https://diffusion-reward.github.io.",
  "published": "2023-12-21",
  "updated": "2024-08-09",
  "year": "2023",
  "authors": [
   "Tao Huang",
   "Guangqi Jiang",
   "Yanjie Ze",
   "Huazhe Xu"
  ],
  "author_count": 4,
  "categories": [
   "cs.LG",
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.LG",
  "venue": "ECCV",
  "venue_source": "semantic-scholar",
  "citations": 62,
  "influential_citations": 8,
  "tldr": "The key insight is that lower generative diversity is exhibited when conditioning diffusion on expert trajectories, and Diffusion Reward is accordingly formalized by the negative of conditional entropy that encourages productive exploration of expert behaviors.",
  "doi": "10.48550/arXiv.2312.14134",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Tao Huang",
    "id": "2275761775",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Guangqi Jiang",
    "id": "2275609647",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Yanjie Ze",
    "id": "2151089356",
    "h_index": 12,
    "papers": 16
   },
   {
    "name": "Huazhe Xu",
    "id": "2255379295",
    "h_index": 14,
    "papers": 21
   }
  ],
  "comment": "Accepted to ECCV 2024. Project page and code: https://diffusion-reward.github.io/",
  "topics": [
   "rl-control",
   "navigation",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2312.14134v3",
  "pdf_url": "https://arxiv.org/pdf/2312.14134v3",
  "html_url": "https://arxiv.org/html/2312.14134v3",
  "code_url": "https://diffusion-reward.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.3
 },
 {
  "id": "2312.14125",
  "slug": "videopoet-a-large-language-model-for-zero-shot-video-generation",
  "title": "VideoPoet: A Large Language Model for Zero-Shot Video Generation",
  "abstract": "We present VideoPoet, a language model capable of synthesizing high-quality video, with matching audio, from a large variety of conditioning signals. VideoPoet employs a decoder-only transformer architecture that processes multimodal inputs -- including images, videos, text, and audio. The training protocol follows that of Large Language Models (LLMs), consisting of two stages: pretraining and task-specific adaptation. During pretraining, VideoPoet incorporates a mixture of multimodal generative objectives within an autoregressive Transformer framework. The pretrained LLM serves as a foundation that can be adapted for a range of video generation tasks. We present empirical results demonstrating the model's state-of-the-art capabilities in zero-shot video generation, specifically highlighting VideoPoet's ability to generate high-fidelity motions. Project page: http://sites.research.google/videopoet/",
  "published": "2023-12-21",
  "updated": "2024-06-04",
  "year": "2023",
  "authors": [
   "Dan Kondratyuk",
   "Lijun Yu",
   "Xiuye Gu",
   "Jos\u00e9 Lezama",
   "Jonathan Huang",
   "Grant Schindler",
   "Rachel Hornung",
   "Vighnesh Birodkar",
   "Jimmy Yan",
   "Ming-Chang Chiu",
   "Krishna Somandepalli",
   "Hassan Akbari",
   "Yair Alon",
   "Yong Cheng",
   "Josh Dillon",
   "Agrim Gupta",
   "Meera Hahn",
   "Anja Hauth",
   "David Hendon",
   "Alonso Martinez",
   "David Minnen",
   "Mikhail Sirotenko",
   "Kihyuk Sohn",
   "Xuan Yang",
   "Hartwig Adam",
   "Ming-Hsuan Yang",
   "Irfan Essa",
   "Huisheng Wang",
   "David A. Ross",
   "Bryan Seybold",
   "Lu Jiang"
  ],
  "author_count": 31,
  "categories": [
   "cs.CV",
   "cs.AI"
  ],
  "primary_category": "cs.CV",
  "venue": "ICML",
  "venue_source": "semantic-scholar",
  "citations": 544,
  "influential_citations": 24,
  "tldr": "Empirical results demonstrating the model's state-of-the-art capabilities in zero-shot video generation are presented, specifically highlighting VideoPoet's ability to generate high-fidelity motions.",
  "doi": "10.48550/arXiv.2312.14125",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "D. Kondratyuk",
    "id": "51208108",
    "h_index": 9,
    "papers": 15
   },
   {
    "name": "Lijun Yu",
    "id": "8547960",
    "h_index": 14,
    "papers": 24
   },
   {
    "name": "Xiuye Gu",
    "id": "2257336985",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Jos\u00e9 Lezama",
    "id": "2256999290",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "Jonathan Huang",
    "id": "2275643784",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Rachel Hornung",
    "id": "2275616325",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Hartwig Adam",
    "id": "2595180",
    "h_index": 44,
    "papers": 70
   },
   {
    "name": "Hassan Akbari",
    "id": "153769937",
    "h_index": 10,
    "papers": 15
   },
   {
    "name": "Y. Alon",
    "id": "1944358430",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Vighnesh Birodkar",
    "id": "3468723",
    "h_index": 11,
    "papers": 19
   },
   {
    "name": "Yong Cheng",
    "id": "2198464317",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Ming-Chang Chiu",
    "id": "116137272",
    "h_index": 5,
    "papers": 17
   },
   {
    "name": "Josh Dillon",
    "id": "2275564829",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Irfan Essa",
    "id": "145955800",
    "h_index": 22,
    "papers": 56
   },
   {
    "name": "Agrim Gupta",
    "id": "2265716291",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Meera Hahn",
    "id": "2268398776",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "A. Hauth",
    "id": "119556335",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "David Hendon",
    "id": "2275598646",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Alonso Martinez",
    "id": "2275749304",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "David C. Minnen",
    "id": "3144223",
    "h_index": 28,
    "papers": 49
   },
   {
    "name": "David A. Ross",
    "id": "2257003564",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Grant Schindler",
    "id": "2275605900",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Mikhail Sirotenko",
    "id": "89903811",
    "h_index": 9,
    "papers": 12
   },
   {
    "name": "Kihyuk Sohn",
    "id": "2256996545",
    "h_index": 10,
    "papers": 20
   },
   {
    "name": "Krishna Somandepalli",
    "id": "6079502",
    "h_index": 16,
    "papers": 51
   },
   {
    "name": "Huisheng Wang",
    "id": "2266273447",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Jimmy Yan",
    "id": "2275568368",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Ming Yang",
    "id": "2228360903",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Xuan Yang",
    "id": "2350843695",
    "h_index": 9,
    "papers": 24
   },
   {
    "name": "Bryan Seybold",
    "id": "2535887",
    "h_index": 11,
    "papers": 16
   },
   {
    "name": "Lu Jiang",
    "id": "39978626",
    "h_index": 48,
    "papers": 78
   }
  ],
  "comment": "To appear at ICML 2024; Project page: http://sites.research.google/videopoet/",
  "topics": [
   "foundation-pretraining",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2312.14125v4",
  "pdf_url": "https://arxiv.org/pdf/2312.14125v4",
  "html_url": "https://arxiv.org/html/2312.14125v4",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.24
 },
 {
  "id": "2312.13139",
  "slug": "unleashing-large-scale-video-generative-pre-training-for-visual-robot",
  "title": "Unleashing Large-Scale Video Generative Pre-training for Visual Robot Manipulation",
  "abstract": "Generative pre-trained models have demonstrated remarkable effectiveness in language and vision domains by learning useful representations. In this paper, we extend the scope of this effectiveness by showing that visual robot manipulation can significantly benefit from large-scale video generative pre-training. We introduce GR-1, a straightforward GPT-style model designed for multi-task language-conditioned visual robot manipulation. GR-1 takes as inputs a language instruction, a sequence of observation images, and a sequence of robot states. It predicts robot actions as well as future images in an end-to-end manner. Thanks to a flexible design, GR-1 can be seamlessly finetuned on robot data after pre-trained on a large-scale video dataset. We perform extensive experiments on the challenging CALVIN benchmark and a real robot. On CALVIN benchmark, our method outperforms state-of-the-art baseline methods and improves the success rate from 88.9% to 94.9%. In the setting of zero-shot unseen scene generalization, GR-1 improves the success rate from 53.3% to 85.4%. In real robot experiments, GR-1 also outperforms baseline methods and shows strong potentials in generalization to unseen scenes and objects. We provide inaugural evidence that a unified GPT-style transformer, augmented with large-scale video generative pre-training, exhibits remarkable generalization to multi-task visual robot manipulation. Project page: https://GR1-Manipulation.github.io",
  "published": "2023-12-20",
  "updated": "2023-12-21",
  "year": "2023",
  "authors": [
   "Hongtao Wu",
   "Ya Jing",
   "Chilam Cheang",
   "Guangzeng Chen",
   "Jiafeng Xu",
   "Xinghang Li",
   "Minghuan Liu",
   "Hang Li",
   "Tao Kong"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 418,
  "influential_citations": 25,
  "tldr": "The introduction of GR-1, a straightforward GPT-style model designed for multi-task language-conditioned visual robot manipulation, which outperforms baseline methods and shows strong potentials in generalization to unseen scenes and objects.",
  "doi": "10.48550/arXiv.2312.13139",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hongtao Wu",
    "id": "2264188661",
    "h_index": 9,
    "papers": 11
   },
   {
    "name": "Ya Jing",
    "id": "2264782777",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Chi-Hou Cheang",
    "id": "1481075732",
    "h_index": 8,
    "papers": 11
   },
   {
    "name": "Guangzeng Chen",
    "id": "2275562190",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Jiafeng Xu",
    "id": "2275277043",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "Xinghang Li",
    "id": "2155447887",
    "h_index": 8,
    "papers": 26
   },
   {
    "name": "Minghuan Liu",
    "id": "2275553483",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Hang Li",
    "id": "2265238860",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Tao Kong",
    "id": "2259922998",
    "h_index": 12,
    "papers": 18
   }
  ],
  "comment": "Project page: https://GR1-Manipulation.github.io",
  "topics": [
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2312.13139v2",
  "pdf_url": "https://arxiv.org/pdf/2312.13139v2",
  "html_url": "https://arxiv.org/html/2312.13139v2",
  "code_url": "https://GR1-Manipulation.github.io",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.12
 },
 {
  "id": "2312.12490",
  "slug": "instructvideo-instructing-video-diffusion-models-with-human-feedback",
  "title": "InstructVideo: Instructing Video Diffusion Models with Human Feedback",
  "abstract": "Diffusion models have emerged as the de facto paradigm for video generation. However, their reliance on web-scale data of varied quality often yields results that are visually unappealing and misaligned with the textual prompts. To tackle this problem, we propose InstructVideo to instruct text-to-video diffusion models with human feedback by reward fine-tuning. InstructVideo has two key ingredients: 1) To ameliorate the cost of reward fine-tuning induced by generating through the full DDIM sampling chain, we recast reward fine-tuning as editing. By leveraging the diffusion process to corrupt a sampled video, InstructVideo requires only partial inference of the DDIM sampling chain, reducing fine-tuning cost while improving fine-tuning efficiency. 2) To mitigate the absence of a dedicated video reward model for human preferences, we repurpose established image reward models, e.g., HPSv2. To this end, we propose Segmental Video Reward, a mechanism to provide reward signals based on segmental sparse sampling, and Temporally Attenuated Reward, a method that mitigates temporal modeling degradation during fine-tuning. Extensive experiments, both qualitative and quantitative, validate the practicality and efficacy of using image reward models in InstructVideo, significantly enhancing the visual quality of generated videos without compromising generalization capabilities. Code and models will be made publicly available.",
  "published": "2023-12-19",
  "updated": "2023-12-19",
  "year": "2023",
  "authors": [
   "Hangjie Yuan",
   "Shiwei Zhang",
   "Xiang Wang",
   "Yujie Wei",
   "Tao Feng",
   "Yining Pan",
   "Yingya Zhang",
   "Ziwei Liu",
   "Samuel Albanie",
   "Dong Ni"
  ],
  "author_count": 10,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.LG",
   "cs.MM"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 99,
  "influential_citations": 9,
  "tldr": "This work proposes Segmental Video Reward, a mechanism to provide reward signals based on segmental sparse sampling, and Temporally Attenuated Reward, a method that mitigates temporal modeling degradation during fine-tuning.",
  "doi": "10.1109/CVPR52733.2024.00618",
  "oa_pdf": "http://arxiv.org/pdf/2312.12490",
  "s2_authors": [
   {
    "name": "Hangjie Yuan",
    "id": "2151334837",
    "h_index": 18,
    "papers": 32
   },
   {
    "name": "Shiwei Zhang",
    "id": "2219846537",
    "h_index": 19,
    "papers": 33
   },
   {
    "name": "Xiang Wang",
    "id": "2265658953",
    "h_index": 15,
    "papers": 32
   },
   {
    "name": "Yujie Wei",
    "id": "2271878957",
    "h_index": 10,
    "papers": 28
   },
   {
    "name": "Tao Feng",
    "id": "2294365559",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Yining Pan",
    "id": "2115428342",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Yingya Zhang",
    "id": "2244766555",
    "h_index": 17,
    "papers": 35
   },
   {
    "name": "Ziwei Liu",
    "id": "2275647099",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Samuel Albanie",
    "id": "7641268",
    "h_index": 41,
    "papers": 109
   },
   {
    "name": "Dong Ni",
    "id": "2275206448",
    "h_index": 3,
    "papers": 6
   }
  ],
  "comment": "Project page: https://instructvideo.github.io/",
  "topics": [
   "rl-control",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2312.12490v1",
  "pdf_url": "https://arxiv.org/pdf/2312.12490v1",
  "html_url": "https://arxiv.org/html/2312.12490v1",
  "code_url": "https://instructvideo.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "2312.10240",
  "slug": "rich-human-feedback-for-text-to-image-generation",
  "title": "Rich Human Feedback for Text-to-Image Generation",
  "abstract": "Recent Text-to-Image (T2I) generation models such as Stable Diffusion and Imagen have made significant progress in generating high-resolution images based on text descriptions. However, many generated images still suffer from issues such as artifacts/implausibility, misalignment with text descriptions, and low aesthetic quality. Inspired by the success of Reinforcement Learning with Human Feedback (RLHF) for large language models, prior works collected human-provided scores as feedback on generated images and trained a reward model to improve the T2I generation. In this paper, we enrich the feedback signal by (i) marking image regions that are implausible or misaligned with the text, and (ii) annotating which words in the text prompt are misrepresented or missing on the image. We collect such rich human feedback on 18K generated images (RichHF-18K) and train a multimodal transformer to predict the rich feedback automatically. We show that the predicted rich human feedback can be leveraged to improve image generation, for example, by selecting high-quality training data to finetune and improve the generative models, or by creating masks with predicted heatmaps to inpaint the problematic regions. Notably, the improvements generalize to models (Muse) beyond those used to generate the images on which human feedback data were collected (Stable Diffusion variants). The RichHF-18K data set will be released in our GitHub repository: https://github.com/google-research/google-research/tree/master/richhf_18k.",
  "published": "2023-12-15",
  "updated": "2024-04-09",
  "year": "2023",
  "authors": [
   "Youwei Liang",
   "Junfeng He",
   "Gang Li",
   "Peizhao Li",
   "Arseniy Klimovskiy",
   "Nicholas Carolan",
   "Jiao Sun",
   "Jordi Pont-Tuset",
   "Sarah Young",
   "Feng Yang",
   "Junjie Ke",
   "Krishnamurthy Dj Dvijotham",
   "Katie Collins",
   "Yiwen Luo",
   "Yang Li",
   "Kai J Kohlhoff",
   "Deepak Ramachandran",
   "Vidhya Navalpakkam"
  ],
  "author_count": 18,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 172,
  "influential_citations": 13,
  "tldr": "It is shown that the predicted rich human feedback can be leveraged to improve image generation, for example, by selecting high-quality training data to finetune and improve the generative models, or by creating masks with predicted heatmaps to inpaint the problematic regions.",
  "doi": "10.1109/CVPR52733.2024.01835",
  "oa_pdf": "https://arxiv.org/pdf/2312.10240",
  "s2_authors": [
   {
    "name": "Youwei Liang",
    "id": "2275187005",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Junfeng He",
    "id": "2275304133",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Gang Li",
    "id": "2275603136",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Peizhao Li",
    "id": "2275181270",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Arseniy Klimovskiy",
    "id": "2275054278",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Nicholas Carolan",
    "id": "2275054275",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Jiao Sun",
    "id": "2268780887",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Jordi Pont-Tuset",
    "id": "2262217017",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "Sarah Young",
    "id": "2275056273",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Feng Yang",
    "id": "2275594056",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Junjie Ke",
    "id": "49287230",
    "h_index": 10,
    "papers": 22
   },
   {
    "name": "K. Dvijotham",
    "id": "1729912",
    "h_index": 36,
    "papers": 139
   },
   {
    "name": "Katie Collins",
    "id": "2055306721",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Yiwen Luo",
    "id": "2329110591",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Yang Li",
    "id": "2275058464",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Kai Kohlhoff",
    "id": "2065004224",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Deepak Ramachandran",
    "id": "143812128",
    "h_index": 14,
    "papers": 45
   },
   {
    "name": "Vidhya Navalpakkam",
    "id": "2575582",
    "h_index": 24,
    "papers": 60
   }
  ],
  "comment": "CVPR'24",
  "topics": [
   "rl-control",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2312.10240v2",
  "pdf_url": "https://arxiv.org/pdf/2312.10240v2",
  "html_url": "https://arxiv.org/html/2312.10240v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.74
 },
 {
  "id": "2312.10035",
  "slug": "point-transformer-v3-simpler-faster-stronger",
  "title": "Point Transformer V3: Simpler, Faster, Stronger",
  "abstract": "This paper is not motivated to seek innovation within the attention mechanism. Instead, it focuses on overcoming the existing trade-offs between accuracy and efficiency within the context of point cloud processing, leveraging the power of scale. Drawing inspiration from recent advances in 3D large-scale representation learning, we recognize that model performance is more influenced by scale than by intricate design. Therefore, we present Point Transformer V3 (PTv3), which prioritizes simplicity and efficiency over the accuracy of certain mechanisms that are minor to the overall performance after scaling, such as replacing the precise neighbor search by KNN with an efficient serialized neighbor mapping of point clouds organized with specific patterns. This principle enables significant scaling, expanding the receptive field from 16 to 1024 points while remaining efficient (a 3x increase in processing speed and a 10x improvement in memory efficiency compared with its predecessor, PTv2). PTv3 attains state-of-the-art results on over 20 downstream tasks that span both indoor and outdoor scenarios. Further enhanced with multi-dataset joint training, PTv3 pushes these results to a higher level.",
  "published": "2023-12-15",
  "updated": "2024-03-25",
  "year": "2023",
  "authors": [
   "Xiaoyang Wu",
   "Li Jiang",
   "Peng-Shuai Wang",
   "Zhijian Liu",
   "Xihui Liu",
   "Yu Qiao",
   "Wanli Ouyang",
   "Tong He",
   "Hengshuang Zhao"
  ],
  "author_count": 9,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 1048,
  "influential_citations": 141,
  "tldr": "This paper presents Point Transformer V3 (PTv3), which prioritizes simplicity and efficiency over the accuracy of certain mechanisms that are minor to the over-all performance after scaling, such as replacing the precise neighbor search by KNN with an efficient serialized neighbor mapping of point clouds organized with specific patterns.",
  "doi": "10.1109/CVPR52733.2024.00463",
  "oa_pdf": "https://arxiv.org/pdf/2312.10035",
  "s2_authors": [
   {
    "name": "Xiaoyang Wu",
    "id": "2257364811",
    "h_index": 13,
    "papers": 19
   },
   {
    "name": "Li Jiang",
    "id": "2274967780",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Peng-Shuai Wang",
    "id": "2051737794",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Zhijian Liu",
    "id": "2316442094",
    "h_index": 12,
    "papers": 14
   },
   {
    "name": "Xihui Liu",
    "id": "2271385386",
    "h_index": 7,
    "papers": 8
   },
   {
    "name": "Yu Qiao",
    "id": "2257348324",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Wanli Ouyang",
    "id": "2253464521",
    "h_index": 20,
    "papers": 81
   },
   {
    "name": "Tong He",
    "id": "2270015436",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Hengshuang Zhao",
    "id": "2237588370",
    "h_index": 17,
    "papers": 32
   }
  ],
  "comment": "CVPR 2024, code available at Pointcept (https://github.com/Pointcept/PointTransformerV3)",
  "topics": [
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2312.10035v2",
  "pdf_url": "https://arxiv.org/pdf/2312.10035v2",
  "html_url": "https://arxiv.org/html/2312.10035v2",
  "code_url": "https://github.com/Pointcept/PointTransformerV3",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.5
 },
 {
  "id": "2312.09109",
  "slug": "videolcm-video-latent-consistency-model",
  "title": "VideoLCM: Video Latent Consistency Model",
  "abstract": "Consistency models have demonstrated powerful capability in efficient image generation and allowed synthesis within a few sampling steps, alleviating the high computational cost in diffusion models. However, the consistency model in the more challenging and resource-consuming video generation is still less explored. In this report, we present the VideoLCM framework to fill this gap, which leverages the concept of consistency models from image generation to efficiently synthesize videos with minimal steps while maintaining high quality. VideoLCM builds upon existing latent video diffusion models and incorporates consistency distillation techniques for training the latent consistency model. Experimental results reveal the effectiveness of our VideoLCM in terms of computational efficiency, fidelity and temporal consistency. Notably, VideoLCM achieves high-fidelity and smooth video synthesis with only four sampling steps, showcasing the potential for real-time synthesis. We hope that VideoLCM can serve as a simple yet effective baseline for subsequent research. The source code and models will be publicly available.",
  "published": "2023-12-14",
  "updated": "2023-12-14",
  "year": "2023",
  "authors": [
   "Xiang Wang",
   "Shiwei Zhang",
   "Han Zhang",
   "Yu Liu",
   "Yingya Zhang",
   "Changxin Gao",
   "Nong Sang"
  ],
  "author_count": 7,
  "categories": [
   "cs.CV",
   "cs.AI"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 95,
  "influential_citations": 6,
  "tldr": "The VideoLCM framework is presented, which leverages the concept of consistency models from image generation to efficiently synthesize videos with minimal steps while maintaining high quality and computational efficiency.",
  "doi": "10.48550/arXiv.2312.09109",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xiang Wang",
    "id": "2265658953",
    "h_index": 15,
    "papers": 32
   },
   {
    "name": "Shiwei Zhang",
    "id": "2268468730",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Han Zhang",
    "id": "2274245518",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Yu Liu",
    "id": "2272142557",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Yingya Zhang",
    "id": "2244766555",
    "h_index": 17,
    "papers": 35
   },
   {
    "name": "Changxin Gao",
    "id": "2258335873",
    "h_index": 15,
    "papers": 76
   },
   {
    "name": "Nong Sang",
    "id": "2270665345",
    "h_index": 15,
    "papers": 81
   }
  ],
  "comment": "",
  "topics": [
   "data-teleop",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2312.09109v1",
  "pdf_url": "https://arxiv.org/pdf/2312.09109v1",
  "html_url": "https://arxiv.org/html/2312.09109v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.98
 },
 {
  "id": "2312.08344",
  "slug": "foundationpose-unified-6d-pose-estimation-and-tracking-of-novel-object",
  "title": "FoundationPose: Unified 6D Pose Estimation and Tracking of Novel Objects",
  "abstract": "We present FoundationPose, a unified foundation model for 6D object pose estimation and tracking, supporting both model-based and model-free setups. Our approach can be instantly applied at test-time to a novel object without fine-tuning, as long as its CAD model is given, or a small number of reference images are captured. We bridge the gap between these two setups with a neural implicit representation that allows for effective novel view synthesis, keeping the downstream pose estimation modules invariant under the same unified framework. Strong generalizability is achieved via large-scale synthetic training, aided by a large language model (LLM), a novel transformer-based architecture, and contrastive learning formulation. Extensive evaluation on multiple public datasets involving challenging scenarios and objects indicate our unified approach outperforms existing methods specialized for each task by a large margin. In addition, it even achieves comparable results to instance-level methods despite the reduced assumptions. Project page: https://nvlabs.github.io/FoundationPose/",
  "published": "2023-12-13",
  "updated": "2024-03-26",
  "year": "2023",
  "authors": [
   "Bowen Wen",
   "Wei Yang",
   "Jan Kautz",
   "Stan Birchfield"
  ],
  "author_count": 4,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 719,
  "influential_citations": 119,
  "tldr": "Examination on multiple public datasets involving challenging scenarios and objects indicate the unified approach outperforms existing methods specialized for each task by a large margin, and even achieves comparable results to instance-level methods despite the reduced assumptions.",
  "doi": "10.1109/CVPR52733.2024.01692",
  "oa_pdf": "https://arxiv.org/pdf/2312.08344",
  "s2_authors": [
   {
    "name": "Bowen Wen",
    "id": "101349262",
    "h_index": 17,
    "papers": 25
   },
   {
    "name": "Wei Yang",
    "id": "2374281462",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Jan Kautz",
    "id": "2376331447",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Stanley T. Birchfield",
    "id": "2257232566",
    "h_index": 9,
    "papers": 26
   }
  ],
  "comment": "",
  "topics": [
   "spatial-3d",
   "foundation-pretraining"
  ],
  "orgs": [
   "NVIDIA"
  ],
  "abs_url": "https://arxiv.org/abs/2312.08344v2",
  "pdf_url": "https://arxiv.org/pdf/2312.08344v2",
  "html_url": "https://arxiv.org/html/2312.08344v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.86
 },
 {
  "id": "2312.07533",
  "slug": "vila-on-pre-training-for-visual-language-models",
  "title": "VILA: On Pre-training for Visual Language Models",
  "abstract": "Visual language models (VLMs) rapidly progressed with the recent success of large language models. There have been growing efforts on visual instruction tuning to extend the LLM with visual inputs, but lacks an in-depth study of the visual language pre-training process, where the model learns to perform joint modeling on both modalities. In this work, we examine the design options for VLM pre-training by augmenting LLM towards VLM through step-by-step controllable comparisons. We introduce three main findings: (1) freezing LLMs during pre-training can achieve decent zero-shot performance, but lack in-context learning capability, which requires unfreezing the LLM; (2) interleaved pre-training data is beneficial whereas image-text pairs alone are not optimal; (3) re-blending text-only instruction data to image-text data during instruction fine-tuning not only remedies the degradation of text-only tasks, but also boosts VLM task accuracy. With an enhanced pre-training recipe we build VILA, a Visual Language model family that consistently outperforms the state-of-the-art models, e.g., LLaVA-1.5, across main benchmarks without bells and whistles. Multi-modal pre-training also helps unveil appealing properties of VILA, including multi-image reasoning, enhanced in-context learning, and better world knowledge.",
  "published": "2023-12-12",
  "updated": "2024-05-16",
  "year": "2023",
  "authors": [
   "Ji Lin",
   "Hongxu Yin",
   "Wei Ping",
   "Yao Lu",
   "Pavlo Molchanov",
   "Andrew Tao",
   "Huizi Mao",
   "Jan Kautz",
   "Mohammad Shoeybi",
   "Song Han"
  ],
  "author_count": 10,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 922,
  "influential_citations": 89,
  "tldr": "This work examines the design options for VLM pre-training by augmenting LLM towards VLM through step-by-step controllable comparisons, and builds VILA, a Visual Language model family that consistently outperforms the state-of-the-art models, e.g., LLaVA-1.5, across main benchmarks without bells and whistles.",
  "doi": "10.1109/CVPR52733.2024.02520",
  "oa_pdf": "https://arxiv.org/pdf/2312.07533",
  "s2_authors": [
   {
    "name": "Ji Lin",
    "id": "2273904689",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Hongxu Yin",
    "id": "1989015",
    "h_index": 40,
    "papers": 72
   },
   {
    "name": "Wei Ping",
    "id": "2253664013",
    "h_index": 20,
    "papers": 39
   },
   {
    "name": "Yao Lu",
    "id": "2274926912",
    "h_index": 15,
    "papers": 18
   },
   {
    "name": "Pavlo Molchanov",
    "id": "2824500",
    "h_index": 50,
    "papers": 147
   },
   {
    "name": "Andrew Tao",
    "id": "29955511",
    "h_index": 31,
    "papers": 67
   },
   {
    "name": "Huizi Mao",
    "id": "3123774",
    "h_index": 19,
    "papers": 27
   },
   {
    "name": "Jan Kautz",
    "id": "2273651410",
    "h_index": 44,
    "papers": 107
   },
   {
    "name": "M. Shoeybi",
    "id": "1911755",
    "h_index": 45,
    "papers": 99
   },
   {
    "name": "Song Han",
    "id": "2273855886",
    "h_index": 13,
    "papers": 21
   }
  ],
  "comment": "CVPR 2024",
  "topics": [
   "foundation-pretraining",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2312.07533v4",
  "pdf_url": "https://arxiv.org/pdf/2312.07533v4",
  "html_url": "https://arxiv.org/html/2312.07533v4",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.47
 },
 {
  "id": "2312.06662",
  "slug": "photorealistic-video-generation-with-diffusion-models",
  "title": "Photorealistic Video Generation with Diffusion Models",
  "abstract": "We present W.A.L.T, a transformer-based approach for photorealistic video generation via diffusion modeling. Our approach has two key design decisions. First, we use a causal encoder to jointly compress images and videos within a unified latent space, enabling training and generation across modalities. Second, for memory and training efficiency, we use a window attention architecture tailored for joint spatial and spatiotemporal generative modeling. Taken together these design decisions enable us to achieve state-of-the-art performance on established video (UCF-101 and Kinetics-600) and image (ImageNet) generation benchmarks without using classifier free guidance. Finally, we also train a cascade of three models for the task of text-to-video generation consisting of a base latent video diffusion model, and two video super-resolution diffusion models to generate videos of $512 \\times 896$ resolution at $8$ frames per second.",
  "published": "2023-12-11",
  "updated": "2023-12-11",
  "year": "2023",
  "authors": [
   "Agrim Gupta",
   "Lijun Yu",
   "Kihyuk Sohn",
   "Xiuye Gu",
   "Meera Hahn",
   "Li Fei-Fei",
   "Irfan Essa",
   "Lu Jiang",
   "Jos\u00e9 Lezama"
  ],
  "author_count": 9,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.CV",
  "venue": "ECCV",
  "venue_source": "semantic-scholar",
  "citations": 350,
  "influential_citations": 18,
  "tldr": "This work presents W.A.L.T, a transformer-based approach for photorealistic video generation via diffusion modeling that uses a causal encoder to jointly compress images and videos within a unified latent space, enabling training and generation across modalities.",
  "doi": "10.48550/arXiv.2312.06662",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Agrim Gupta",
    "id": "2265716291",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Lijun Yu",
    "id": "8547960",
    "h_index": 14,
    "papers": 24
   },
   {
    "name": "Kihyuk Sohn",
    "id": "2256996545",
    "h_index": 10,
    "papers": 20
   },
   {
    "name": "Xiuye Gu",
    "id": "2257336985",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Meera Hahn",
    "id": "2268398776",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Fei-Fei Li",
    "id": "2273586794",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Irfan Essa",
    "id": "145955800",
    "h_index": 22,
    "papers": 56
   },
   {
    "name": "Lu Jiang",
    "id": "39978626",
    "h_index": 48,
    "papers": 78
   },
   {
    "name": "Jos\u00e9 Lezama",
    "id": "2256999290",
    "h_index": 5,
    "papers": 13
   }
  ],
  "comment": "Project website https://walt-video-diffusion.github.io/",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2312.06662v1",
  "pdf_url": "https://arxiv.org/pdf/2312.06662v1",
  "html_url": "https://arxiv.org/html/2312.06662v1",
  "code_url": "https://walt-video-diffusion.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.05
 },
 {
  "id": "2312.06647",
  "slug": "4m-massively-multimodal-masked-modeling",
  "title": "4M: Massively Multimodal Masked Modeling",
  "abstract": "Current machine learning models for vision are often highly specialized and limited to a single modality and task. In contrast, recent large language models exhibit a wide range of capabilities, hinting at a possibility for similarly versatile models in computer vision. In this paper, we take a step in this direction and propose a multimodal training scheme called 4M. It consists of training a single unified Transformer encoder-decoder using a masked modeling objective across a wide range of input/output modalities - including text, images, geometric, and semantic modalities, as well as neural network feature maps. 4M achieves scalability by unifying the representation space of all modalities through mapping them into discrete tokens and performing multimodal masked modeling on a small randomized subset of tokens. 4M leads to models that exhibit several key capabilities: (1) they can perform a diverse set of vision tasks out of the box, (2) they excel when fine-tuned for unseen downstream tasks or new input modalities, and (3) they can function as a generative model that can be conditioned on arbitrary modalities, enabling a wide variety of expressive multimodal editing capabilities with remarkable flexibility. Through experimental analyses, we demonstrate the potential of 4M for training versatile and scalable foundation models for vision tasks, setting the stage for further exploration in multimodal learning for vision and other domains.",
  "published": "2023-12-11",
  "updated": "2023-12-11",
  "year": "2023",
  "authors": [
   "David Mizrahi",
   "Roman Bachmann",
   "O\u011fuzhan Fatih Kar",
   "Teresa Yeo",
   "Mingfei Gao",
   "Afshin Dehghan",
   "Amir Zamir"
  ],
  "author_count": 7,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.CV",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 143,
  "influential_citations": 13,
  "tldr": "The potential of 4M for training versatile and scalable foundation models for vision tasks is demonstrated, setting the stage for further exploration in multimodal learning for vision and other domains.",
  "doi": "10.48550/arXiv.2312.06647",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "David Mizrahi",
    "id": "2111623708",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Roman Bachmann",
    "id": "153825349",
    "h_index": 11,
    "papers": 15
   },
   {
    "name": "Ouguzhan Fatih Kar",
    "id": "2273474116",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Teresa Yeo",
    "id": "143895090",
    "h_index": 9,
    "papers": 17
   },
   {
    "name": "Mingfei Gao",
    "id": "2273661239",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Afshin Dehghan",
    "id": "2273361790",
    "h_index": 13,
    "papers": 23
   },
   {
    "name": "Amir Zamir",
    "id": "40029556",
    "h_index": 36,
    "papers": 63
   }
  ],
  "comment": "NeurIPS 2023 Spotlight. Project page at https://4m.epfl.ch/",
  "topics": [
   "navigation",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2312.06647v1",
  "pdf_url": "https://arxiv.org/pdf/2312.06647v1",
  "html_url": "https://arxiv.org/html/2312.06647v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.66
 },
 {
  "id": "2312.06639",
  "slug": "harmonic-mobile-manipulation",
  "title": "Harmonic Mobile Manipulation",
  "abstract": "Recent advancements in robotics have enabled robots to navigate complex scenes or manipulate diverse objects independently. However, robots are still impotent in many household tasks requiring coordinated behaviors such as opening doors. The factorization of navigation and manipulation, while effective for some tasks, fails in scenarios requiring coordinated actions. To address this challenge, we introduce, HarmonicMM, an end-to-end learning method that optimizes both navigation and manipulation, showing notable improvement over existing techniques in everyday tasks. This approach is validated in simulated and real-world environments and adapts to novel unseen settings without additional tuning. Our contributions include a new benchmark for mobile manipulation and the successful deployment with only RGB visual observation in a real unseen apartment, demonstrating the potential for practical indoor robot deployment in daily life. More results are on our project site: https://rchalyang.github.io/HarmonicMM/",
  "published": "2023-12-11",
  "updated": "2024-12-05",
  "year": "2023",
  "authors": [
   "Ruihan Yang",
   "Yejin Kim",
   "Rose Hendrix",
   "Aniruddha Kembhavi",
   "Xiaolong Wang",
   "Kiana Ehsani"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS",
  "venue_source": "semantic-scholar",
  "citations": 33,
  "influential_citations": 0,
  "tldr": "HarmonicMM is introduced, an end-to-end learning method that optimizes both navigation and manipulation, showing notable improvement over existing techniques in everyday tasks and demonstrating the potential for practical indoor robot deployment in daily life.",
  "doi": "10.1109/IROS58592.2024.10802201",
  "oa_pdf": "http://arxiv.org/pdf/2312.06639",
  "s2_authors": [
   {
    "name": "Ruihan Yang",
    "id": "143955842",
    "h_index": 18,
    "papers": 22
   },
   {
    "name": "Yejin Kim",
    "id": "2269749997",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Aniruddha Kembhavi",
    "id": "2684226",
    "h_index": 49,
    "papers": 119
   },
   {
    "name": "Xiaolong Wang",
    "id": "2253445806",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "Kiana Ehsani",
    "id": "2883417",
    "h_index": 26,
    "papers": 42
   }
  ],
  "comment": "More results are on our project site: https://rchalyang.github.io/HarmonicMM/",
  "topics": [
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2312.06639v3",
  "pdf_url": "https://arxiv.org/pdf/2312.06639v3",
  "html_url": "https://arxiv.org/html/2312.06639v3",
  "code_url": "https://rchalyang.github.io/HarmonicMM/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.03
 },
 {
  "id": "2312.05251",
  "slug": "reconstructing-hands-in-3d-with-transformers",
  "title": "Reconstructing Hands in 3D with Transformers",
  "abstract": "We present an approach that can reconstruct hands in 3D from monocular input. Our approach for Hand Mesh Recovery, HaMeR, follows a fully transformer-based architecture and can analyze hands with significantly increased accuracy and robustness compared to previous work. The key to HaMeR's success lies in scaling up both the data used for training and the capacity of the deep network for hand reconstruction. For training data, we combine multiple datasets that contain 2D or 3D hand annotations. For the deep model, we use a large scale Vision Transformer architecture. Our final model consistently outperforms the previous baselines on popular 3D hand pose benchmarks. To further evaluate the effect of our design in non-controlled settings, we annotate existing in-the-wild datasets with 2D hand keypoint annotations. On this newly collected dataset of annotations, HInt, we demonstrate significant improvements over existing baselines. We make our code, data and models available on the project website: https://geopavlakos.github.io/hamer/.",
  "published": "2023-12-08",
  "updated": "2023-12-08",
  "year": "2023",
  "authors": [
   "Georgios Pavlakos",
   "Dandan Shan",
   "Ilija Radosavovic",
   "Angjoo Kanazawa",
   "David Fouhey",
   "Jitendra Malik"
  ],
  "author_count": 6,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 408,
  "influential_citations": 77,
  "tldr": "This work follows a fully transformer-based architecture and can analyze hands with significantly increased accuracy and robustness compared to previous work, and demonstrates significant improvements over existing baselines on popular 3D hand pose benchmarks.",
  "doi": "10.1109/CVPR52733.2024.00938",
  "oa_pdf": "http://arxiv.org/pdf/2312.05251",
  "s2_authors": [
   {
    "name": "G. Pavlakos",
    "id": "2829330",
    "h_index": 28,
    "papers": 65
   },
   {
    "name": "Dandan Shan",
    "id": "2058873326",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Ilija Radosavovic",
    "id": "30407997",
    "h_index": 20,
    "papers": 22
   },
   {
    "name": "Angjoo Kanazawa",
    "id": "20615377",
    "h_index": 60,
    "papers": 126
   },
   {
    "name": "David F. Fouhey",
    "id": "1786435",
    "h_index": 30,
    "papers": 67
   },
   {
    "name": "Jitendra Malik",
    "id": "2257249601",
    "h_index": 11,
    "papers": 16
   }
  ],
  "comment": "",
  "topics": [
   "data-teleop",
   "video-generation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2312.05251v1",
  "pdf_url": "https://arxiv.org/pdf/2312.05251v1",
  "html_url": "https://arxiv.org/html/2312.05251v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.11
 },
 {
  "id": "2312.03913",
  "slug": "controllable-human-object-interaction-synthesis",
  "title": "Controllable Human-Object Interaction Synthesis",
  "abstract": "Synthesizing semantic-aware, long-horizon, human-object interaction is critical to simulate realistic human behaviors. In this work, we address the challenging problem of generating synchronized object motion and human motion guided by language descriptions in 3D scenes. We propose Controllable Human-Object Interaction Synthesis (CHOIS), an approach that generates object motion and human motion simultaneously using a conditional diffusion model given a language description, initial object and human states, and sparse object waypoints. Here, language descriptions inform style and intent, and waypoints, which can be effectively extracted from high-level planning, ground the motion in the scene. Naively applying a diffusion model fails to predict object motion aligned with the input waypoints; it also cannot ensure the realism of interactions that require precise hand-object and human-floor contact. To overcome these problems, we introduce an object geometry loss as additional supervision to improve the matching between generated object motion and input object waypoints; we also design guidance terms to enforce contact constraints during the sampling process of the trained diffusion model. We demonstrate that our learned interaction module can synthesize realistic human-object interactions, adhering to provided textual descriptions and sparse waypoint conditions. Additionally, our module seamlessly integrates with a path planning module, enabling the generation of long-term interactions in 3D environments.",
  "published": "2023-12-06",
  "updated": "2024-07-14",
  "year": "2023",
  "authors": [
   "Jiaman Li",
   "Alexander Clegg",
   "Roozbeh Mottaghi",
   "Jiajun Wu",
   "Xavier Puig",
   "C. Karen Liu"
  ],
  "author_count": 6,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "ECCV",
  "venue_source": "semantic-scholar",
  "citations": 123,
  "influential_citations": 26,
  "tldr": "This work proposes Controllable Human-Object Interaction Synthesis (CHOIS), an approach that generates object motion and human motion simultaneously using a conditional diffusion model given a language description, initial object and human states, and sparse object waypoints, and seamlessly integrates with a path planning module.",
  "doi": "10.48550/arXiv.2312.03913",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jiaman Li",
    "id": "22133106",
    "h_index": 16,
    "papers": 22
   },
   {
    "name": "Alexander Clegg",
    "id": "30933599",
    "h_index": 20,
    "papers": 24
   },
   {
    "name": "Roozbeh Mottaghi",
    "id": "3012475",
    "h_index": 46,
    "papers": 102
   },
   {
    "name": "Jiajun Wu",
    "id": "3045089",
    "h_index": 80,
    "papers": 228
   },
   {
    "name": "Xavi Puig",
    "id": "2260947618",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "C. K. Liu",
    "id": "2247934447",
    "h_index": 9,
    "papers": 11
   }
  ],
  "comment": "ECCV 2024, project webpage: https://lijiaman.github.io/projects/chois/",
  "topics": [
   "navigation",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2312.03913v2",
  "pdf_url": "https://arxiv.org/pdf/2312.03913v2",
  "html_url": "https://arxiv.org/html/2312.03913v2",
  "code_url": "https://lijiaman.github.io/projects/chois/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.59
 },
 {
  "id": "2312.02975",
  "slug": "dexterous-functional-grasping",
  "title": "Dexterous Functional Grasping",
  "abstract": "While there have been significant strides in dexterous manipulation, most of it is limited to benchmark tasks like in-hand reorientation which are of limited utility in the real world. The main benefit of dexterous hands over two-fingered ones is their ability to pickup tools and other objects (including thin ones) and grasp them firmly to apply force. However, this task requires both a complex understanding of functional affordances as well as precise low-level control. While prior work obtains affordances from human data this approach doesn't scale to low-level control. Similarly, simulation training cannot give the robot an understanding of real-world semantics. In this paper, we aim to combine the best of both worlds to accomplish functional grasping for in-the-wild objects. We use a modular approach. First, affordances are obtained by matching corresponding regions of different objects and then a low-level policy trained in sim is run to grasp it. We propose a novel application of eigengrasps to reduce the search space of RL using a small amount of human data and find that it leads to more stable and physically realistic motion. We find that eigengrasp action space beats baselines in simulation and outperforms hardcoded grasping in real and matches or outperforms a trained human teleoperator. Results visualizations and videos at https://dexfunc.github.io/",
  "published": "2023-12-05",
  "updated": "2023-12-05",
  "year": "2023",
  "authors": [
   "Ananye Agarwal",
   "Shagun Uppal",
   "Kenneth Shaw",
   "Deepak Pathak"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 65,
  "influential_citations": 4,
  "tldr": "This paper proposes a novel application of eigengrasps to reduce the search space of RL using a small amount of human data and finds that it leads to more stable and physically realistic motion.",
  "doi": "10.48550/arXiv.2312.02975",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ananye Agarwal",
    "id": "2107063491",
    "h_index": 9,
    "papers": 14
   },
   {
    "name": "Shagun Uppal",
    "id": "2269735302",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Kenneth Shaw",
    "id": "2263541750",
    "h_index": 10,
    "papers": 18
   },
   {
    "name": "Deepak Pathak",
    "id": "2269734979",
    "h_index": 11,
    "papers": 17
   }
  ],
  "comment": "In CoRL 2023. Website at https://dexfunc.github.io/",
  "topics": [
   "dexterous-manipulation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2312.02975v1",
  "pdf_url": "https://arxiv.org/pdf/2312.02975v1",
  "html_url": "https://arxiv.org/html/2312.02975v1",
  "code_url": "https://dexfunc.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.32
 },
 {
  "id": "2312.14937",
  "slug": "sc-gs-sparse-controlled-gaussian-splatting-for-editable-dynamic-scenes",
  "title": "SC-GS: Sparse-Controlled Gaussian Splatting for Editable Dynamic Scenes",
  "abstract": "Novel view synthesis for dynamic scenes is still a challenging problem in computer vision and graphics. Recently, Gaussian splatting has emerged as a robust technique to represent static scenes and enable high-quality and real-time novel view synthesis. Building upon this technique, we propose a new representation that explicitly decomposes the motion and appearance of dynamic scenes into sparse control points and dense Gaussians, respectively. Our key idea is to use sparse control points, significantly fewer in number than the Gaussians, to learn compact 6 DoF transformation bases, which can be locally interpolated through learned interpolation weights to yield the motion field of 3D Gaussians. We employ a deformation MLP to predict time-varying 6 DoF transformations for each control point, which reduces learning complexities, enhances learning abilities, and facilitates obtaining temporal and spatial coherent motion patterns. Then, we jointly learn the 3D Gaussians, the canonical space locations of control points, and the deformation MLP to reconstruct the appearance, geometry, and dynamics of 3D scenes. During learning, the location and number of control points are adaptively adjusted to accommodate varying motion complexities in different regions, and an ARAP loss following the principle of as rigid as possible is developed to enforce spatial continuity and local rigidity of learned motions. Finally, thanks to the explicit sparse motion representation and its decomposition from appearance, our method can enable user-controlled motion editing while retaining high-fidelity appearances. Extensive experiments demonstrate that our approach outperforms existing approaches on novel view synthesis with a high rendering speed and enables novel appearance-preserved motion editing applications. Project page: https://yihua7.github.io/SC-GS-web/",
  "published": "2023-12-04",
  "updated": "2024-03-31",
  "year": "2023",
  "authors": [
   "Yi-Hua Huang",
   "Yang-Tian Sun",
   "Ziyi Yang",
   "Xiaoyang Lyu",
   "Yan-Pei Cao",
   "Xiaojuan Qi"
  ],
  "author_count": 6,
  "categories": [
   "cs.CV",
   "cs.GR"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 428,
  "influential_citations": 66,
  "tldr": "This work proposes a new representation that explicitly decomposes the motion and appearance of dynamic scenes into sparse control points and dense Gaussians, respectively, and outperforms existing approaches on novel view synthesis with a high rendering speed and enables novel appearance-preserved motion editing applications.",
  "doi": "10.1109/CVPR52733.2024.00404",
  "oa_pdf": "https://arxiv.org/pdf/2312.14937",
  "s2_authors": [
   {
    "name": "Yi-Hua Huang",
    "id": "2276484450",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Yang-Tian Sun",
    "id": "2276552777",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "Ziyi Yang",
    "id": "2276448654",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Xiaoyang Lyu",
    "id": "2238952101",
    "h_index": 11,
    "papers": 21
   },
   {
    "name": "Yan-Pei Cao",
    "id": "2276490175",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Xiaojuan Qi",
    "id": "2264513980",
    "h_index": 8,
    "papers": 14
   }
  ],
  "comment": "Code link: https://github.com/yihua7/SC-GS",
  "topics": [
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2312.14937v3",
  "pdf_url": "https://arxiv.org/pdf/2312.14937v3",
  "html_url": "https://arxiv.org/html/2312.14937v3",
  "code_url": "https://github.com/yihua7/SC-GS",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.13
 },
 {
  "id": "2312.00849",
  "slug": "rlhf-v-towards-trustworthy-mllms-via-behavior-alignment-from-fine-grai",
  "title": "RLHF-V: Towards Trustworthy MLLMs via Behavior Alignment from Fine-grained Correctional Human Feedback",
  "abstract": "Multimodal Large Language Models (MLLMs) have recently demonstrated impressive capabilities in multimodal understanding, reasoning, and interaction. However, existing MLLMs prevalently suffer from serious hallucination problems, generating text that is not factually grounded in associated images. The problem makes existing MLLMs untrustworthy and thus impractical in real-world (especially high-stakes) applications. To address the challenge, we present RLHF-V, which enhances MLLM trustworthiness via behavior alignment from fine-grained correctional human feedback. Specifically, RLHF-V collects human preference in the form of segment-level corrections on hallucinations, and performs dense direct preference optimization over the human feedback. Comprehensive experiments on five benchmarks in both automatic and human evaluation show that, RLHF-V can enable substantially more trustworthy MLLM behaviors with promising data and computation efficiency. Remarkably, using 1.4k annotated data samples, RLHF-V significantly reduces the hallucination rate of the base MLLM by 34.8%, outperforming the concurrent LLaVA-RLHF trained on 10k annotated data. The final model achieves state-of-the-art performance in trustworthiness among open-source MLLMs, and shows better robustness than GPT-4V in preventing hallucinations aroused from over-generalization. We open-source our code, model, and data at https://github.com/RLHF-V/RLHF-V.",
  "published": "2023-12-01",
  "updated": "2024-03-08",
  "year": "2023",
  "authors": [
   "Tianyu Yu",
   "Yuan Yao",
   "Haoye Zhang",
   "Taiwen He",
   "Yifeng Han",
   "Ganqu Cui",
   "Jinyi Hu",
   "Zhiyuan Liu",
   "Hai-Tao Zheng",
   "Maosong Sun",
   "Tat-Seng Chua"
  ],
  "author_count": 11,
  "categories": [
   "cs.CL",
   "cs.CV"
  ],
  "primary_category": "cs.CL",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 486,
  "influential_citations": 59,
  "tldr": "RLHF-V is presented, which enhances MLLM trustworthiness via behavior alignment from fine-grained correctional human feedback, and achieves state-of-the-art performance in trustwor-thiness among open-source MLLMs.",
  "doi": "10.1109/CVPR52733.2024.01310",
  "oa_pdf": "https://arxiv.org/pdf/2312.00849",
  "s2_authors": [
   {
    "name": "Tianyu Yu",
    "id": "2117902355",
    "h_index": 10,
    "papers": 17
   },
   {
    "name": "Yuan Yao",
    "id": "1390925224",
    "h_index": 31,
    "papers": 43
   },
   {
    "name": "Haoye Zhang",
    "id": "2233320353",
    "h_index": 7,
    "papers": 9
   },
   {
    "name": "Taiwen He",
    "id": "2269469438",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Yifeng Han",
    "id": "1733075484",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Ganqu Cui",
    "id": "52297757",
    "h_index": 34,
    "papers": 84
   },
   {
    "name": "Jinyi Hu",
    "id": "92837695",
    "h_index": 15,
    "papers": 24
   },
   {
    "name": "Zhiyuan Liu",
    "id": "2141313179",
    "h_index": 42,
    "papers": 130
   },
   {
    "name": "Hai-Tao Zheng",
    "id": "2242734235",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Maosong Sun",
    "id": "1753344",
    "h_index": 101,
    "papers": 491
   },
   {
    "name": "Tat-Seng Chua",
    "id": "2257036129",
    "h_index": 31,
    "papers": 89
   }
  ],
  "comment": "Accepted by CVPR 2024",
  "topics": [
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2312.00849v2",
  "pdf_url": "https://arxiv.org/pdf/2312.00849v2",
  "html_url": "https://arxiv.org/html/2312.00849v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.19
 },
 {
  "id": "2312.00785",
  "slug": "sequential-modeling-enables-scalable-learning-for-large-vision-models",
  "title": "Sequential Modeling Enables Scalable Learning for Large Vision Models",
  "abstract": "We introduce a novel sequential modeling approach which enables learning a Large Vision Model (LVM) without making use of any linguistic data. To do this, we define a common format, \"visual sentences\", in which we can represent raw images and videos as well as annotated data sources such as semantic segmentations and depth reconstructions without needing any meta-knowledge beyond the pixels. Once this wide variety of visual data (comprising 420 billion tokens) is represented as sequences, the model can be trained to minimize a cross-entropy loss for next token prediction. By training across various scales of model architecture and data diversity, we provide empirical evidence that our models scale effectively. Many different vision tasks can be solved by designing suitable visual prompts at test time.",
  "published": "2023-12-01",
  "updated": "2023-12-01",
  "year": "2023",
  "authors": [
   "Yutong Bai",
   "Xinyang Geng",
   "Karttikeya Mangalam",
   "Amir Bar",
   "Alan Yuille",
   "Trevor Darrell",
   "Jitendra Malik",
   "Alexei A Efros"
  ],
  "author_count": 8,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 285,
  "influential_citations": 26,
  "tldr": "A novel sequential modeling approach which enables learning a Large Vision Model (LVM) without making use of any linguistic data, and defines a common format, \u201cvisual sentences \u201d, in which raw images and videos as well as annotated data sources can be represented as sequences.",
  "doi": "10.1109/CVPR52733.2024.02157",
  "oa_pdf": "https://arxiv.org/pdf/2312.00785",
  "s2_authors": [
   {
    "name": "Yutong Bai",
    "id": "48442730",
    "h_index": 20,
    "papers": 34
   },
   {
    "name": "Xinyang Geng",
    "id": "3468192",
    "h_index": 19,
    "papers": 33
   },
   {
    "name": "K. Mangalam",
    "id": "11379939",
    "h_index": 23,
    "papers": 45
   },
   {
    "name": "Amir Bar",
    "id": "2063958674",
    "h_index": 13,
    "papers": 23
   },
   {
    "name": "Alan Yuille",
    "id": "2253485890",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Trevor Darrell",
    "id": "2257173088",
    "h_index": 12,
    "papers": 18
   },
   {
    "name": "Jitendra Malik",
    "id": "2244770235",
    "h_index": 7,
    "papers": 9
   },
   {
    "name": "Alexei A. Efros",
    "id": "144495271",
    "h_index": 6,
    "papers": 22
   }
  ],
  "comment": "Website: https://yutongbai.com/lvm.html",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2312.00785v1",
  "pdf_url": "https://arxiv.org/pdf/2312.00785v1",
  "html_url": "https://arxiv.org/html/2312.00785v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.96
 },
 {
  "id": "2312.00775",
  "slug": "towards-generalizable-zero-shot-manipulation-via-translating-human-int",
  "title": "Towards Generalizable Zero-Shot Manipulation via Translating Human Interaction Plans",
  "abstract": "We pursue the goal of developing robots that can interact zero-shot with generic unseen objects via a diverse repertoire of manipulation skills and show how passive human videos can serve as a rich source of data for learning such generalist robots. Unlike typical robot learning approaches which directly learn how a robot should act from interaction data, we adopt a factorized approach that can leverage large-scale human videos to learn how a human would accomplish a desired task (a human plan), followed by translating this plan to the robots embodiment. Specifically, we learn a human plan predictor that, given a current image of a scene and a goal image, predicts the future hand and object configurations. We combine this with a translation module that learns a plan-conditioned robot manipulation policy, and allows following humans plans for generic manipulation tasks in a zero-shot manner with no deployment-time training. Importantly, while the plan predictor can leverage large-scale human videos for learning, the translation module only requires a small amount of in-domain data, and can generalize to tasks not seen during training. We show that our learned system can perform over 16 manipulation skills that generalize to 40 objects, encompassing 100 real-world tasks for table-top manipulation and diverse in-the-wild manipulation. https://homangab.github.io/hopman/",
  "published": "2023-12-01",
  "updated": "2023-12-01",
  "year": "2023",
  "authors": [
   "Homanga Bharadhwaj",
   "Abhinav Gupta",
   "Vikash Kumar",
   "Shubham Tulsiani"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA",
  "venue_source": "semantic-scholar",
  "citations": 77,
  "influential_citations": 2,
  "tldr": "A factorized approach that can leverage large-scale human videos to learn how a human would accomplish a desired task (a human \u2018plan\u2019), followed by \u2018translating\u2019 this plan to the robot\u2019s embodiment and allows following humans plans for generic manipulation tasks in a zero-shot manner with no deployment-time training.",
  "doi": "10.1109/ICRA57147.2024.10610288",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Homanga Bharadhwaj",
    "id": "51113848",
    "h_index": 23,
    "papers": 59
   },
   {
    "name": "Abhi Gupta",
    "id": "2117767136",
    "h_index": 16,
    "papers": 24
   },
   {
    "name": "Vikash Kumar",
    "id": "2269751524",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Shubham Tulsiani",
    "id": "2757335",
    "h_index": 45,
    "papers": 98
   }
  ],
  "comment": "Preprint. Under Review",
  "topics": [
   "egocentric-data"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2312.00775v1",
  "pdf_url": "https://arxiv.org/pdf/2312.00775v1",
  "html_url": "https://arxiv.org/html/2312.00775v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.39
 },
 {
  "id": "2312.00583",
  "slug": "deformgs-scene-flow-in-highly-deformable-scenes-for-deformable-object",
  "title": "DeformGS: Scene Flow in Highly Deformable Scenes for Deformable Object Manipulation",
  "abstract": "Teaching robots to fold, drape, or reposition deformable objects such as cloth will unlock a variety of automation applications. While remarkable progress has been made for rigid object manipulation, manipulating deformable objects poses unique challenges, including frequent occlusions, infinite-dimensional state spaces and complex dynamics. Just as object pose estimation and tracking have aided robots for rigid manipulation, dense 3D tracking (scene flow) of highly deformable objects will enable new applications in robotics while aiding existing approaches, such as imitation learning or creating digital twins with real2sim transfer. We propose DeformGS, an approach to recover scene flow in highly deformable scenes, using simultaneous video captures of a dynamic scene from multiple cameras. DeformGS builds on recent advances in Gaussian splatting, a method that learns the properties of a large number of Gaussians for state-of-the-art and fast novel-view synthesis. DeformGS learns a deformation function to project a set of Gaussians with canonical properties into world space. The deformation function uses a neural-voxel encoding and a multilayer perceptron (MLP) to infer Gaussian position, rotation, and a shadow scalar. We enforce physics-inspired regularization terms based on conservation of momentum and isometry, which leads to trajectories with smaller trajectory errors. We also leverage existing foundation models SAM and XMEM to produce noisy masks, and learn a per-Gaussian mask for better physics-inspired regularization. DeformGS achieves high-quality 3D tracking on highly deformable scenes with shadows and occlusions. In experiments, DeformGS improves 3D tracking by an average of 55.8% compared to the state-of-the-art. With sufficient texture, DeformGS achieves a median tracking error of 3.3 mm on a cloth of 1.5 x 1.5 m in area. Website: https://deformgs.github.io",
  "published": "2023-11-30",
  "updated": "2024-08-30",
  "year": "2023",
  "authors": [
   "Bardienus P. Duisterhof",
   "Zhao Mandi",
   "Yunchao Yao",
   "Jia-Wei Liu",
   "Jenny Seidenschwarz",
   "Mike Zheng Shou",
   "Deva Ramanan",
   "Shuran Song",
   "Stan Birchfield",
   "Bowen Wen",
   "Jeffrey Ichnowski"
  ],
  "author_count": 11,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 54,
  "influential_citations": 2,
  "tldr": "DeformGS, an approach to recover scene flow in highly deformable scenes, using simultaneous video captures of a dynamic scene from multiple cameras, builds on recent advances in Gaussian splatting, a method that learns the properties of a large number of Gaussians for state-of-the-art and fast novel-view synthesis.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "B. Duisterhof",
    "id": "146254448",
    "h_index": 10,
    "papers": 26
   },
   {
    "name": "Zhao Mandi",
    "id": "2126966292",
    "h_index": 6,
    "papers": 6
   },
   {
    "name": "Yunchao Yao",
    "id": "2269410520",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Jia-Wei Liu",
    "id": "2108267957",
    "h_index": 18,
    "papers": 32
   },
   {
    "name": "M. Shou",
    "id": "2047358650",
    "h_index": 49,
    "papers": 278
   },
   {
    "name": "Shuran Song",
    "id": "2297819190",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Jeffrey Ichnowski",
    "id": "2269146110",
    "h_index": 9,
    "papers": 29
   }
  ],
  "comment": "",
  "topics": [
   "imitation-diffusion",
   "spatial-3d",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2312.00583v2",
  "pdf_url": "https://arxiv.org/pdf/2312.00583v2",
  "html_url": "https://arxiv.org/html/2312.00583v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.74
 },
 {
  "id": "2311.18259",
  "slug": "ego-exo4d-understanding-skilled-human-activity-from-first-and-third-pe",
  "title": "Ego-Exo4D: Understanding Skilled Human Activity from First- and Third-Person Perspectives",
  "abstract": "We present Ego-Exo4D, a diverse, large-scale multimodal multiview video dataset and benchmark challenge. Ego-Exo4D centers around simultaneously-captured egocentric and exocentric video of skilled human activities (e.g., sports, music, dance, bike repair). 740 participants from 13 cities worldwide performed these activities in 123 different natural scene contexts, yielding long-form captures from 1 to 42 minutes each and 1,286 hours of video combined. The multimodal nature of the dataset is unprecedented: the video is accompanied by multichannel audio, eye gaze, 3D point clouds, camera poses, IMU, and multiple paired language descriptions -- including a novel \"expert commentary\" done by coaches and teachers and tailored to the skilled-activity domain. To push the frontier of first-person video understanding of skilled human activity, we also present a suite of benchmark tasks and their annotations, including fine-grained activity understanding, proficiency estimation, cross-view translation, and 3D hand/body pose. All resources are open sourced to fuel new research in the community. Project page: http://ego-exo4d-data.org/",
  "published": "2023-11-30",
  "updated": "2024-09-25",
  "year": "2023",
  "authors": [
   "Kristen Grauman",
   "Andrew Westbury",
   "Lorenzo Torresani",
   "Kris Kitani",
   "Jitendra Malik",
   "Triantafyllos Afouras",
   "Kumar Ashutosh",
   "Vijay Baiyya",
   "Siddhant Bansal",
   "Bikram Boote",
   "Eugene Byrne",
   "Zach Chavis",
   "Joya Chen",
   "Feng Cheng",
   "Fu-Jen Chu",
   "Sean Crane",
   "Avijit Dasgupta",
   "Jing Dong",
   "Maria Escobar",
   "Cristhian Forigua",
   "Abrham Gebreselasie",
   "Sanjay Haresh",
   "Jing Huang",
   "Md Mohaiminul Islam",
   "Suyog Jain",
   "Rawal Khirodkar",
   "Devansh Kukreja",
   "Kevin J Liang",
   "Jia-Wei Liu",
   "Sagnik Majumder",
   "Yongsen Mao",
   "Miguel Martin",
   "Effrosyni Mavroudi",
   "Tushar Nagarajan",
   "Francesco Ragusa",
   "Santhosh Kumar Ramakrishnan",
   "Luigi Seminara",
   "Arjun Somayazulu",
   "Yale Song",
   "Shan Su",
   "Zihui Xue",
   "Edward Zhang",
   "Jinxu Zhang",
   "Angela Castillo",
   "Changan Chen",
   "Xinzhu Fu",
   "Ryosuke Furuta",
   "Cristina Gonzalez",
   "Prince Gupta",
   "Jiabo Hu",
   "Yifei Huang",
   "Yiming Huang",
   "Weslie Khoo",
   "Anush Kumar",
   "Robert Kuo",
   "Sach Lakhavani",
   "Miao Liu",
   "Mi Luo",
   "Zhengyi Luo",
   "Brighid Meredith",
   "Austin Miller",
   "Oluwatumininu Oguntola",
   "Xiaqing Pan",
   "Penny Peng",
   "Shraman Pramanick",
   "Merey Ramazanova",
   "Fiona Ryan",
   "Wei Shan",
   "Kiran Somasundaram",
   "Chenan Song",
   "Audrey Southerland",
   "Masatoshi Tateno",
   "Huiyu Wang",
   "Yuchen Wang",
   "Takuma Yagi",
   "Mingfei Yan",
   "Xitong Yang",
   "Zecheng Yu",
   "Shengxin Cindy Zha",
   "Chen Zhao",
   "Ziwei Zhao",
   "Zhifan Zhu",
   "Jeff Zhuo",
   "Pablo Arbelaez",
   "Gedas Bertasius",
   "David Crandall",
   "Dima Damen",
   "Jakob Engel",
   "Giovanni Maria Farinella",
   "Antonino Furnari",
   "Bernard Ghanem",
   "Judy Hoffman",
   "C. V. Jawahar",
   "Richard Newcombe",
   "Hyun Soo Park",
   "James M. Rehg",
   "Yoichi Sato",
   "Manolis Savva",
   "Jianbo Shi",
   "Mike Zheng Shou",
   "Michael Wray"
  ],
  "author_count": 101,
  "categories": [
   "cs.CV",
   "cs.AI"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 563,
  "influential_citations": 74,
  "tldr": "To push the frontier of first-person video understanding of skilled human activity, a suite of benchmark tasks and their annotations are presented, including fine-grained activity understanding, proficiency estimation, cross-view translation, and 3D hand/body pose.",
  "doi": "10.1007/s11263-025-02557-6",
  "oa_pdf": "https://doi.org/10.1007/s11263-025-02557-6",
  "s2_authors": [
   {
    "name": "K. Grauman",
    "id": "1794409",
    "h_index": 99,
    "papers": 295
   },
   {
    "name": "Andrew Westbury",
    "id": "2127379149",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "L. Torresani",
    "id": "1732879",
    "h_index": 57,
    "papers": 143
   },
   {
    "name": "Kris Kitani",
    "id": "2256989593",
    "h_index": 10,
    "papers": 20
   },
   {
    "name": "Jitendra Malik",
    "id": "2244770235",
    "h_index": 7,
    "papers": 9
   },
   {
    "name": "Triantafyllos Afouras",
    "id": "2285516",
    "h_index": 27,
    "papers": 41
   },
   {
    "name": "Kumar Ashutosh",
    "id": "49133777",
    "h_index": 11,
    "papers": 27
   },
   {
    "name": "Vijay Baiyya",
    "id": "2268759525",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Siddhant Bansal",
    "id": "16936840",
    "h_index": 8,
    "papers": 17
   },
   {
    "name": "Bikram Boote",
    "id": "2023707382",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Eugene Byrne",
    "id": "2132203390",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Zachary Chavis",
    "id": "2312205852",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Joya Chen",
    "id": "2268796016",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Feng Cheng",
    "id": "2268761868",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Fu-Jen Chu",
    "id": "2268761245",
    "h_index": 7,
    "papers": 8
   },
   {
    "name": "Sean Crane",
    "id": "2268759283",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Avijit Dasgupta",
    "id": "2268761333",
    "h_index": 2,
    "papers": 8
   },
   {
    "name": "Jing Dong",
    "id": "2397713428",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Mar\u00eda Escobar",
    "id": "152479570",
    "h_index": 9,
    "papers": 17
   },
   {
    "name": "Cristhian Forigua",
    "id": "2185755102",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "A. Gebreselasie",
    "id": "150013809",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "S. Haresh",
    "id": "2330587187",
    "h_index": 9,
    "papers": 19
   },
   {
    "name": "Jing Huang",
    "id": "2220616407",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Md Mohaiminul Islam",
    "id": "2269101371",
    "h_index": 10,
    "papers": 11
   },
   {
    "name": "S. Jain",
    "id": "3347530",
    "h_index": 14,
    "papers": 20
   },
   {
    "name": "Rawal Khirodkar",
    "id": "51927417",
    "h_index": 13,
    "papers": 31
   },
   {
    "name": "Devansh Kukreja",
    "id": "2268759210",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Kevin J. Liang",
    "id": "2268760054",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Jia-Wei Liu",
    "id": "2108267957",
    "h_index": 18,
    "papers": 32
   },
   {
    "name": "Sagnik Majumder",
    "id": "47927907",
    "h_index": 12,
    "papers": 28
   },
   {
    "name": "Yongsen Mao",
    "id": "2160729619",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Miguel Martin",
    "id": "2268818182",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "E. Mavroudi",
    "id": "2065610871",
    "h_index": 8,
    "papers": 10
   },
   {
    "name": "Tushar Nagarajan",
    "id": "38661780",
    "h_index": 14,
    "papers": 33
   },
   {
    "name": "Francesco Ragusa",
    "id": "2276741850",
    "h_index": 13,
    "papers": 48
   },
   {
    "name": "Santhosh K. Ramakrishnan",
    "id": "21810992",
    "h_index": 18,
    "papers": 30
   },
   {
    "name": "Luigi Seminara",
    "id": "2268757871",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Arjun Somayazulu",
    "id": "2190280533",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Yale Song",
    "id": "2268777158",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Shan Su",
    "id": "2066630446",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Zihui Xue",
    "id": "2268760036",
    "h_index": 9,
    "papers": 19
   },
   {
    "name": "Edward Zhang",
    "id": "145214194",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Jinxu Zhang",
    "id": "2327911103",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "\u00c1ngela Castillo",
    "id": "2268761467",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Changan Chen",
    "id": "2268807177",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Xinzhu Fu",
    "id": "2393953198",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Ryosuke Furuta",
    "id": "2256987514",
    "h_index": 5,
    "papers": 20
   },
   {
    "name": "Cristina Gonz\u00e1lez",
    "id": "2269047429",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Prince Gupta",
    "id": "2119998894",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Jiabo Hu",
    "id": "2269020646",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Yifei Huang",
    "id": "48355651",
    "h_index": 24,
    "papers": 70
   },
   {
    "name": "Yiming Huang",
    "id": "2273119677",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "Weslie Khoo",
    "id": "2260652099",
    "h_index": 6,
    "papers": 18
   },
   {
    "name": "Anushk Kumar",
    "id": "2348792214",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Robert Kuo",
    "id": "2268759520",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Sach Lakhavani",
    "id": "2268756996",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Miao Liu",
    "id": "2108511234",
    "h_index": 16,
    "papers": 28
   },
   {
    "name": "Romy Mi Luo",
    "id": "2109495772",
    "h_index": 11,
    "papers": 16
   },
   {
    "name": "Zhengyi Luo",
    "id": "2566332",
    "h_index": 15,
    "papers": 20
   },
   {
    "name": "Brighid Meredith",
    "id": "2268759094",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Austin Miller",
    "id": "2405642162",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Oluwatumininu Oguntola",
    "id": "2268759070",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Xiaqing Pan",
    "id": "2268853555",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Penny Peng",
    "id": "13317717",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Shraman Pramanick",
    "id": "1564558163",
    "h_index": 12,
    "papers": 18
   },
   {
    "name": "Merey Ramazanova",
    "id": "4042496",
    "h_index": 9,
    "papers": 20
   },
   {
    "name": "Fiona Ryan",
    "id": "119797486",
    "h_index": 12,
    "papers": 17
   },
   {
    "name": "W. Shan",
    "id": "2356751361",
    "h_index": 5,
    "papers": 94
   },
   {
    "name": "Kiran Somasundaram",
    "id": "2249761206",
    "h_index": 6,
    "papers": 6
   },
   {
    "name": "Chenan Song",
    "id": "2269120657",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "A. Southerland",
    "id": "7824981",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "Masatoshi Tateno",
    "id": "2254899134",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Huiyu Wang",
    "id": "2268818266",
    "h_index": 9,
    "papers": 11
   },
   {
    "name": "Yuchen Wang",
    "id": "2268823415",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Takuma Yagi",
    "id": "2052544968",
    "h_index": 9,
    "papers": 21
   },
   {
    "name": "Mingfei Yan",
    "id": "2234026075",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Xitong Yang",
    "id": "2328261065",
    "h_index": 7,
    "papers": 7
   },
   {
    "name": "Ze Yu",
    "id": "2350049794",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Shengxin Zha",
    "id": "2815926",
    "h_index": 11,
    "papers": 21
   },
   {
    "name": "Chen Zhao",
    "id": "2268804964",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Ziwei Zhao",
    "id": "2186108324",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Zhifan Zhu",
    "id": "74123394",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "J. Zhuo",
    "id": "2268758124",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Pablo Arbel\u00e1ez",
    "id": "2251012614",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Gedas Bertasius",
    "id": "3313330",
    "h_index": 29,
    "papers": 76
   },
   {
    "name": "David J. Crandall",
    "id": "2260652429",
    "h_index": 6,
    "papers": 17
   },
   {
    "name": "D. Damen",
    "id": "145089978",
    "h_index": 45,
    "papers": 201
   },
   {
    "name": "J. Engel",
    "id": "2241357086",
    "h_index": 9,
    "papers": 13
   },
   {
    "name": "G. Farinella",
    "id": "1729739",
    "h_index": 37,
    "papers": 320
   },
   {
    "name": "Antonino Furnari",
    "id": "1792681",
    "h_index": 29,
    "papers": 152
   },
   {
    "name": "Bernard Ghanem",
    "id": "2268676398",
    "h_index": 8,
    "papers": 21
   },
   {
    "name": "Judy Hoffman",
    "id": "2268757045",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "C. V. Jawahar",
    "id": "2265550049",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Richard A. Newcombe",
    "id": "50366818",
    "h_index": 31,
    "papers": 57
   },
   {
    "name": "H. Park",
    "id": "2289827917",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "James M. Rehg",
    "id": "2091835870",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Yoichi Sato",
    "id": "9467266",
    "h_index": 61,
    "papers": 263
   },
   {
    "name": "M. Savva",
    "id": "2295141",
    "h_index": 47,
    "papers": 118
   },
   {
    "name": "Jianbo Shi",
    "id": "2242193127",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "M. Shou",
    "id": "2047358650",
    "h_index": 49,
    "papers": 278
   },
   {
    "name": "Michael Wray",
    "id": "145032628",
    "h_index": 15,
    "papers": 51
   }
  ],
  "comment": "Expanded manuscript (compared to arxiv v1 from Nov 2023 and CVPR 2024 paper from June 2024) for more comprehensive dataset and benchmark presentation, plus new results on v2 data release",
  "topics": [
   "egocentric-data",
   "spatial-3d",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2311.18259v4",
  "pdf_url": "https://arxiv.org/pdf/2311.18259v4",
  "html_url": "https://arxiv.org/html/2311.18259v4",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.25
 },
 {
  "id": "2311.17984",
  "slug": "4d-fy-text-to-4d-generation-using-hybrid-score-distillation-sampling",
  "title": "4D-fy: Text-to-4D Generation Using Hybrid Score Distillation Sampling",
  "abstract": "Recent breakthroughs in text-to-4D generation rely on pre-trained text-to-image and text-to-video models to generate dynamic 3D scenes. However, current text-to-4D methods face a three-way tradeoff between the quality of scene appearance, 3D structure, and motion. For example, text-to-image models and their 3D-aware variants are trained on internet-scale image datasets and can be used to produce scenes with realistic appearance and 3D structure -- but no motion. Text-to-video models are trained on relatively smaller video datasets and can produce scenes with motion, but poorer appearance and 3D structure. While these models have complementary strengths, they also have opposing weaknesses, making it difficult to combine them in a way that alleviates this three-way tradeoff. Here, we introduce hybrid score distillation sampling, an alternating optimization procedure that blends supervision signals from multiple pre-trained diffusion models and incorporates benefits of each for high-fidelity text-to-4D generation. Using hybrid SDS, we demonstrate synthesis of 4D scenes with compelling appearance, 3D structure, and motion.",
  "published": "2023-11-29",
  "updated": "2024-05-26",
  "year": "2023",
  "authors": [
   "Sherwin Bahmani",
   "Ivan Skorokhodov",
   "Victor Rong",
   "Gordon Wetzstein",
   "Leonidas Guibas",
   "Peter Wonka",
   "Sergey Tulyakov",
   "Jeong Joon Park",
   "Andrea Tagliasacchi",
   "David B. Lindell"
  ],
  "author_count": 10,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 214,
  "influential_citations": 24,
  "tldr": "This work introduces hybrid score distillation sampling, an alternating optimization procedure that blends supervision signals from multiple pre-trained diffusion models and incorporates benefits of each for high-fidelity text-to-4D generation and demonstrates synthesis of 4D scenes with compelling appearance, 3D structure, and motion.",
  "doi": "10.1109/CVPR52733.2024.00764",
  "oa_pdf": "https://arxiv.org/pdf/2311.17984",
  "s2_authors": [
   {
    "name": "Sherwin Bahmani",
    "id": "2142454988",
    "h_index": 11,
    "papers": 14
   },
   {
    "name": "Ivan Skorokhodov",
    "id": "51118864",
    "h_index": 22,
    "papers": 45
   },
   {
    "name": "Victor Rong",
    "id": "2268759553",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Gordon Wetzstein",
    "id": "2256985147",
    "h_index": 21,
    "papers": 43
   },
   {
    "name": "Leonidas J. Guibas",
    "id": "2231868572",
    "h_index": 12,
    "papers": 21
   },
   {
    "name": "Peter Wonka",
    "id": "2262444458",
    "h_index": 14,
    "papers": 82
   },
   {
    "name": "S. Tulyakov",
    "id": "145582202",
    "h_index": 48,
    "papers": 171
   },
   {
    "name": "J. Park",
    "id": "2148838020",
    "h_index": 14,
    "papers": 18
   },
   {
    "name": "Andrea Tagliasacchi",
    "id": "2237987366",
    "h_index": 13,
    "papers": 26
   },
   {
    "name": "David B. Lindell",
    "id": "2202838",
    "h_index": 28,
    "papers": 50
   }
  ],
  "comment": "CVPR 2024; Project page: https://sherwinbahmani.github.io/4dfy",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2311.17984v2",
  "pdf_url": "https://arxiv.org/pdf/2311.17984v2",
  "html_url": "https://arxiv.org/html/2311.17984v2",
  "code_url": "https://sherwinbahmani.github.io/4dfy",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.83
 },
 {
  "id": "2311.17982",
  "slug": "vbench-comprehensive-benchmark-suite-for-video-generative-models",
  "title": "VBench: Comprehensive Benchmark Suite for Video Generative Models",
  "abstract": "Video generation has witnessed significant advancements, yet evaluating these models remains a challenge. A comprehensive evaluation benchmark for video generation is indispensable for two reasons: 1) Existing metrics do not fully align with human perceptions; 2) An ideal evaluation system should provide insights to inform future developments of video generation. To this end, we present VBench, a comprehensive benchmark suite that dissects \"video generation quality\" into specific, hierarchical, and disentangled dimensions, each with tailored prompts and evaluation methods. VBench has three appealing properties: 1) Comprehensive Dimensions: VBench comprises 16 dimensions in video generation (e.g., subject identity inconsistency, motion smoothness, temporal flickering, and spatial relationship, etc). The evaluation metrics with fine-grained levels reveal individual models' strengths and weaknesses. 2) Human Alignment: We also provide a dataset of human preference annotations to validate our benchmarks' alignment with human perception, for each evaluation dimension respectively. 3) Valuable Insights: We look into current models' ability across various evaluation dimensions, and various content types. We also investigate the gaps between video and image generation models. We will open-source VBench, including all prompts, evaluation methods, generated videos, and human preference annotations, and also include more video generation models in VBench to drive forward the field of video generation.",
  "published": "2023-11-29",
  "updated": "2023-11-29",
  "year": "2023",
  "authors": [
   "Ziqi Huang",
   "Yinan He",
   "Jiashuo Yu",
   "Fan Zhang",
   "Chenyang Si",
   "Yuming Jiang",
   "Yuanhan Zhang",
   "Tianxing Wu",
   "Qingyang Jin",
   "Nattapol Chanpaisit",
   "Yaohui Wang",
   "Xinyuan Chen",
   "Limin Wang",
   "Dahua Lin",
   "Yu Qiao",
   "Ziwei Liu"
  ],
  "author_count": 16,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 1831,
  "influential_citations": 330,
  "tldr": "VBench is a comprehensive benchmark suite that dissects \u201cvideo generation quality\u201d into specific, hierarchical, and disentangled dimensions, each with tailored prompts and evaluation methods, and investi-gate the gaps between video and image generation models.",
  "doi": "10.1109/CVPR52733.2024.02060",
  "oa_pdf": "https://arxiv.org/pdf/2311.17982",
  "s2_authors": [
   {
    "name": "Ziqi Huang",
    "id": "2243375536",
    "h_index": 13,
    "papers": 28
   },
   {
    "name": "Yinan He",
    "id": "2118918324",
    "h_index": 28,
    "papers": 48
   },
   {
    "name": "Jiashuo Yu",
    "id": "2116034742",
    "h_index": 19,
    "papers": 33
   },
   {
    "name": "Fan Zhang",
    "id": "2268852317",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Chenyang Si",
    "id": "2243153986",
    "h_index": 14,
    "papers": 23
   },
   {
    "name": "Yuming Jiang",
    "id": "2441865976",
    "h_index": 20,
    "papers": 35
   },
   {
    "name": "Yuanhan Zhang",
    "id": "2145784327",
    "h_index": 22,
    "papers": 47
   },
   {
    "name": "Tianxing Wu",
    "id": "2116518381",
    "h_index": 11,
    "papers": 17
   },
   {
    "name": "Qin Jin",
    "id": "2290783013",
    "h_index": 6,
    "papers": 16
   },
   {
    "name": "Nattapol Chanpaisit",
    "id": "2268758475",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Yaohui Wang",
    "id": "2267905077",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Xinyuan Chen",
    "id": "2257124855",
    "h_index": 14,
    "papers": 38
   },
   {
    "name": "Limin Wang",
    "id": "2141353278",
    "h_index": 19,
    "papers": 35
   },
   {
    "name": "Dahua Lin",
    "id": "2258618427",
    "h_index": 18,
    "papers": 31
   },
   {
    "name": "Yu Qiao",
    "id": "2257027525",
    "h_index": 12,
    "papers": 22
   },
   {
    "name": "Ziwei Liu",
    "id": "2243875466",
    "h_index": 18,
    "papers": 29
   }
  ],
  "comment": "Equal contributions from first four authors. Project page: https://vchitect.github.io/VBench-project/ Code: https://github.com/Vchitect/VBench",
  "topics": [
   "data-teleop",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2311.17982v1",
  "pdf_url": "https://arxiv.org/pdf/2311.17982v1",
  "html_url": "https://arxiv.org/html/2311.17982v1",
  "code_url": "https://vchitect.github.io/VBench-project/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.5
 },
 {
  "id": "2311.17946",
  "slug": "dreamsync-aligning-text-to-image-generation-with-image-understanding-f",
  "title": "DreamSync: Aligning Text-to-Image Generation with Image Understanding Feedback",
  "abstract": "Despite their wide-spread success, Text-to-Image models (T2I) still struggle to produce images that are both aesthetically pleasing and faithful to the user's input text. We introduce DreamSync, a model-agnostic training algorithm by design that improves T2I models to be faithful to the text input. DreamSync builds off a recent insight from TIFA's evaluation framework -- that large vision-language models (VLMs) can effectively identify the fine-grained discrepancies between generated images and the text inputs. DreamSync uses this insight to train T2I models without any labeled data; it improves T2I models using its own generations. First, it prompts the model to generate several candidate images for a given input text. Then, it uses two VLMs to select the best generation: a Visual Question Answering model that measures the alignment of generated images to the text, and another that measures the generation's aesthetic quality. After selection, we use LoRA to iteratively finetune the T2I model to guide its generation towards the selected best generations. DreamSync does not need any additional human annotation. model architecture changes, or reinforcement learning. Despite its simplicity, DreamSync improves both the semantic alignment and aesthetic appeal of two diffusion-based T2I models, evidenced by multiple benchmarks (+1.7% on TIFA, +2.9% on DSG1K, +3.4% on VILA aesthetic) and human evaluation.",
  "published": "2023-11-29",
  "updated": "2023-11-29",
  "year": "2023",
  "authors": [
   "Jiao Sun",
   "Deqing Fu",
   "Yushi Hu",
   "Su Wang",
   "Royi Rassin",
   "Da-Cheng Juan",
   "Dana Alon",
   "Charles Herrmann",
   "Sjoerd van Steenkiste",
   "Ranjay Krishna",
   "Cyrus Rashtchian"
  ],
  "author_count": 11,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.CL"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 70,
  "influential_citations": 3,
  "tldr": "DreamSync is introduced, a model-agnostic training algorithm by design that improves T2I models to be faithful to the text input and improves both the semantic alignment and aesthetic appeal of two diffusion-based T1I models.",
  "doi": "10.48550/arXiv.2311.17946",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jiao Sun",
    "id": "2268780887",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Deqing Fu",
    "id": "2135593484",
    "h_index": 10,
    "papers": 25
   },
   {
    "name": "Yushi Hu",
    "id": "2112209725",
    "h_index": 14,
    "papers": 21
   },
   {
    "name": "Su Wang",
    "id": "2268798066",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Royi Rassin",
    "id": "2188241744",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Da-Cheng Juan",
    "id": "50270386",
    "h_index": 19,
    "papers": 46
   },
   {
    "name": "Dana Alon",
    "id": "13920082",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Charles Herrmann",
    "id": "2363348424",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Sjoerd van Steenkiste",
    "id": "3440930",
    "h_index": 22,
    "papers": 40
   },
   {
    "name": "Ranjay Krishna",
    "id": "2262217505",
    "h_index": 22,
    "papers": 61
   },
   {
    "name": "Cyrus Rashtchian",
    "id": "3125805",
    "h_index": 20,
    "papers": 56
   }
  ],
  "comment": "",
  "topics": [
   "rl-control",
   "data-teleop",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2311.17946v1",
  "pdf_url": "https://arxiv.org/pdf/2311.17946v1",
  "html_url": "https://arxiv.org/html/2311.17946v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.85
 },
 {
  "id": "2311.16098",
  "slug": "on-bringing-robots-home",
  "title": "On Bringing Robots Home",
  "abstract": "Throughout history, we have successfully integrated various machines into our homes. Dishwashers, laundry machines, stand mixers, and robot vacuums are a few recent examples. However, these machines excel at performing only a single task effectively. The concept of a \"generalist machine\" in homes - a domestic assistant that can adapt and learn from our needs, all while remaining cost-effective - has long been a goal in robotics that has been steadily pursued for decades. In this work, we initiate a large-scale effort towards this goal by introducing Dobb-E, an affordable yet versatile general-purpose system for learning robotic manipulation within household settings. Dobb-E can learn a new task with only five minutes of a user showing it how to do it, thanks to a demonstration collection tool (\"The Stick\") we built out of cheap parts and iPhones. We use the Stick to collect 13 hours of data in 22 homes of New York City, and train Home Pretrained Representations (HPR). Then, in a novel home environment, with five minutes of demonstrations and fifteen minutes of adapting the HPR model, we show that Dobb-E can reliably solve the task on the Stretch, a mobile robot readily available on the market. Across roughly 30 days of experimentation in homes of New York City and surrounding areas, we test our system in 10 homes, with a total of 109 tasks in different environments, and finally achieve a success rate of 81%. Beyond success percentages, our experiments reveal a plethora of unique challenges absent or ignored in lab robotics. These range from effects of strong shadows, to variable demonstration quality by non-expert users. With the hope of accelerating research on home robots, and eventually seeing robot butlers in every home, we open-source Dobb-E software stack and models, our data, and our hardware designs at https://dobb-e.com",
  "published": "2023-11-27",
  "updated": "2023-11-27",
  "year": "2023",
  "authors": [
   "Nur Muhammad Mahi Shafiullah",
   "Anant Rai",
   "Haritheja Etukuru",
   "Yiqian Liu",
   "Ishan Misra",
   "Soumith Chintala",
   "Lerrel Pinto"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 161,
  "influential_citations": 12,
  "tldr": "This work introduces Dobb-E, an affordable yet versatile general-purpose system for learning robotic manipulation within household settings, and opens-source the software stack and models, the data, and the hardware designs.",
  "doi": "10.48550/arXiv.2311.16098",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Nur Muhammad (Mahi) Shafiullah",
    "id": "84146411",
    "h_index": 9,
    "papers": 11
   },
   {
    "name": "Anant Rai",
    "id": "2253550685",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Haritheja Etukuru",
    "id": "2268398574",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Yiqian Liu",
    "id": "2373579181",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Ishan Misra",
    "id": "2267241285",
    "h_index": 13,
    "papers": 60
   },
   {
    "name": "Soumith Chintala",
    "id": "2127604",
    "h_index": 32,
    "papers": 49
   },
   {
    "name": "Lerrel Pinto",
    "id": "2253567347",
    "h_index": 15,
    "papers": 25
   }
  ],
  "comment": "Project website and videos are available at https://dobb-e.com, technical documentation for getting started is available at https://docs.dobb-e.com, and code is released at https://github.com/notmahi/dobb-e",
  "topics": [
   "data-teleop",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2311.16098v1",
  "pdf_url": "https://arxiv.org/pdf/2311.16098v1",
  "html_url": "https://arxiv.org/html/2311.16098v1",
  "code_url": "https://github.com/notmahi/dobb-e",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.21
 },
 {
  "id": "2311.15127",
  "slug": "stable-video-diffusion-scaling-latent-video-diffusion-models-to-large",
  "title": "Stable Video Diffusion: Scaling Latent Video Diffusion Models to Large Datasets",
  "abstract": "We present Stable Video Diffusion - a latent video diffusion model for high-resolution, state-of-the-art text-to-video and image-to-video generation. Recently, latent diffusion models trained for 2D image synthesis have been turned into generative video models by inserting temporal layers and finetuning them on small, high-quality video datasets. However, training methods in the literature vary widely, and the field has yet to agree on a unified strategy for curating video data. In this paper, we identify and evaluate three different stages for successful training of video LDMs: text-to-image pretraining, video pretraining, and high-quality video finetuning. Furthermore, we demonstrate the necessity of a well-curated pretraining dataset for generating high-quality videos and present a systematic curation process to train a strong base model, including captioning and filtering strategies. We then explore the impact of finetuning our base model on high-quality data and train a text-to-video model that is competitive with closed-source video generation. We also show that our base model provides a powerful motion representation for downstream tasks such as image-to-video generation and adaptability to camera motion-specific LoRA modules. Finally, we demonstrate that our model provides a strong multi-view 3D-prior and can serve as a base to finetune a multi-view diffusion model that jointly generates multiple views of objects in a feedforward fashion, outperforming image-based methods at a fraction of their compute budget. We release code and model weights at https://github.com/Stability-AI/generative-models .",
  "published": "2023-11-25",
  "updated": "2023-11-25",
  "year": "2023",
  "authors": [
   "Andreas Blattmann",
   "Tim Dockhorn",
   "Sumith Kulal",
   "Daniel Mendelevitch",
   "Maciej Kilian",
   "Dominik Lorenz",
   "Yam Levi",
   "Zion English",
   "Vikram Voleti",
   "Adam Letts",
   "Varun Jampani",
   "Robin Rombach"
  ],
  "author_count": 12,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 2849,
  "influential_citations": 330,
  "tldr": "This paper identifies and evaluates three different stages for successful training of video LDMs: text-to-image Pretraining, video pretraining, and high-quality video finetuning, and shows that the necessity of a well-curated pretraining dataset for generating high- quality videos and a systematic curation process to train a strong base model.",
  "doi": "10.48550/arXiv.2311.15127",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "A. Blattmann",
    "id": "119843260",
    "h_index": 17,
    "papers": 25
   },
   {
    "name": "Tim Dockhorn",
    "id": "102541178",
    "h_index": 13,
    "papers": 17
   },
   {
    "name": "Sumith Kulal",
    "id": "3411322",
    "h_index": 12,
    "papers": 19
   },
   {
    "name": "Daniel Mendelevitch",
    "id": "2267536530",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Maciej Kilian",
    "id": "2302771628",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Dominik Lorenz",
    "id": "2053482699",
    "h_index": 7,
    "papers": 11
   }
  ],
  "comment": "",
  "topics": [
   "foundation-pretraining",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2311.15127v1",
  "pdf_url": "https://arxiv.org/pdf/2311.15127v1",
  "html_url": "https://arxiv.org/html/2311.15127v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.0
 },
 {
  "id": "2311.13231",
  "slug": "using-human-feedback-to-fine-tune-diffusion-models-without-any-reward",
  "title": "Using Human Feedback to Fine-tune Diffusion Models without Any Reward Model",
  "abstract": "Using reinforcement learning with human feedback (RLHF) has shown significant promise in fine-tuning diffusion models. Previous methods start by training a reward model that aligns with human preferences, then leverage RL techniques to fine-tune the underlying models. However, crafting an efficient reward model demands extensive datasets, optimal architecture, and manual hyperparameter tuning, making the process both time and cost-intensive. The direct preference optimization (DPO) method, effective in fine-tuning large language models, eliminates the necessity for a reward model. However, the extensive GPU memory requirement of the diffusion model's denoising process hinders the direct application of the DPO method. To address this issue, we introduce the Direct Preference for Denoising Diffusion Policy Optimization (D3PO) method to directly fine-tune diffusion models. The theoretical analysis demonstrates that although D3PO omits training a reward model, it effectively functions as the optimal reward model trained using human feedback data to guide the learning process. This approach requires no training of a reward model, proving to be more direct, cost-effective, and minimizing computational overhead. In experiments, our method uses the relative scale of objectives as a proxy for human preference, delivering comparable results to methods using ground-truth rewards. Moreover, D3PO demonstrates the ability to reduce image distortion rates and generate safer images, overcoming challenges lacking robust reward models. Our code is publicly available at https://github.com/yk7333/D3PO.",
  "published": "2023-11-22",
  "updated": "2024-03-23",
  "year": "2023",
  "authors": [
   "Kai Yang",
   "Jian Tao",
   "Jiafei Lyu",
   "Chunjiang Ge",
   "Jiaxin Chen",
   "Qimai Li",
   "Weihan Shen",
   "Xiaolong Zhu",
   "Xiu Li"
  ],
  "author_count": 9,
  "categories": [
   "cs.LG",
   "cs.AI",
   "cs.CV"
  ],
  "primary_category": "cs.LG",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 283,
  "influential_citations": 29,
  "tldr": "The theoretical analysis demonstrates that although D3PO omits training a reward model, it effectively functions as the optimal re-ward model trained using human feedback data to guide the learning process, proving to be more direct, cost-effective, and minimizing computational overhead.",
  "doi": "10.1109/CVPR52733.2024.00854",
  "oa_pdf": "http://arxiv.org/pdf/2311.13231",
  "s2_authors": [
   {
    "name": "Kai Yang",
    "id": "2239414219",
    "h_index": 5,
    "papers": 18
   },
   {
    "name": "Jian Tao",
    "id": "2219923621",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Jiafei Lyu",
    "id": "2008151131",
    "h_index": 16,
    "papers": 55
   },
   {
    "name": "Chunjiang Ge",
    "id": "2333235506",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Jiaxing Chen",
    "id": "2218894812",
    "h_index": 10,
    "papers": 23
   },
   {
    "name": "Qimai Li",
    "id": "2265619322",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Weihan Shen",
    "id": "2199247868",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Xiaolong Zhu",
    "id": "2148597516",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Xiu Li",
    "id": "2199418209",
    "h_index": 3,
    "papers": 6
   }
  ],
  "comment": "CVPR 2024 accepted; huggingface daily paper",
  "topics": [
   "imitation-diffusion",
   "rl-control",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2311.13231v3",
  "pdf_url": "https://arxiv.org/pdf/2311.13231v3",
  "html_url": "https://arxiv.org/html/2311.13231v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.95
 },
 {
  "id": "2311.12908",
  "slug": "diffusion-model-alignment-using-direct-preference-optimization",
  "title": "Diffusion Model Alignment Using Direct Preference Optimization",
  "abstract": "Large language models (LLMs) are fine-tuned using human comparison data with Reinforcement Learning from Human Feedback (RLHF) methods to make them better aligned with users' preferences. In contrast to LLMs, human preference learning has not been widely explored in text-to-image diffusion models; the best existing approach is to fine-tune a pretrained model using carefully curated high quality images and captions to improve visual appeal and text alignment. We propose Diffusion-DPO, a method to align diffusion models to human preferences by directly optimizing on human comparison data. Diffusion-DPO is adapted from the recently developed Direct Preference Optimization (DPO), a simpler alternative to RLHF which directly optimizes a policy that best satisfies human preferences under a classification objective. We re-formulate DPO to account for a diffusion model notion of likelihood, utilizing the evidence lower bound to derive a differentiable objective. Using the Pick-a-Pic dataset of 851K crowdsourced pairwise preferences, we fine-tune the base model of the state-of-the-art Stable Diffusion XL (SDXL)-1.0 model with Diffusion-DPO. Our fine-tuned base model significantly outperforms both base SDXL-1.0 and the larger SDXL-1.0 model consisting of an additional refinement model in human evaluation, improving visual appeal and prompt alignment. We also develop a variant that uses AI feedback and has comparable performance to training on human preferences, opening the door for scaling of diffusion model alignment methods.",
  "published": "2023-11-21",
  "updated": "2023-11-21",
  "year": "2023",
  "authors": [
   "Bram Wallace",
   "Meihua Dang",
   "Rafael Rafailov",
   "Linqi Zhou",
   "Aaron Lou",
   "Senthil Purushwalkam",
   "Stefano Ermon",
   "Caiming Xiong",
   "Shafiq Joty",
   "Nikhil Naik"
  ],
  "author_count": 10,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.GR",
   "cs.LG"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 852,
  "influential_citations": 163,
  "tldr": "This work proposes Diffusion-DPO, a method to align diffusion models to human preferences by directly optimizing on human comparison data, and fine-tuned the base model of the state-of-the-art Stable Diffusion XL (SDXL)-1.0 model with Diffusion-DPO.",
  "doi": "10.1109/CVPR52733.2024.00786",
  "oa_pdf": "https://arxiv.org/pdf/2311.12908",
  "s2_authors": [
   {
    "name": "Bram Wallace",
    "id": "152823401",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Meihua Dang",
    "id": "2267728871",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Rafael Rafailov",
    "id": "102801230",
    "h_index": 25,
    "papers": 44
   },
   {
    "name": "Linqi Zhou",
    "id": "2249592181",
    "h_index": 9,
    "papers": 9
   },
   {
    "name": "Aaron Lou",
    "id": "2261494043",
    "h_index": 7,
    "papers": 8
   },
   {
    "name": "Senthil Purushwalkam",
    "id": "3234247",
    "h_index": 18,
    "papers": 36
   },
   {
    "name": "S. Ermon",
    "id": "2490652",
    "h_index": 104,
    "papers": 479
   },
   {
    "name": "Caiming Xiong",
    "id": "2267728986",
    "h_index": 15,
    "papers": 30
   },
   {
    "name": "Shafiq R. Joty",
    "id": "2708940",
    "h_index": 65,
    "papers": 296
   },
   {
    "name": "Nikhil Naik",
    "id": "2265756339",
    "h_index": 4,
    "papers": 4
   }
  ],
  "comment": "",
  "topics": [
   "rl-control",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2311.12908v1",
  "pdf_url": "https://arxiv.org/pdf/2311.12908v1",
  "html_url": "https://arxiv.org/html/2311.12908v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.43
 },
 {
  "id": "2311.12398",
  "slug": "rftrans-leveraging-refractive-flow-of-transparent-objects-for-surface",
  "title": "RFTrans: Leveraging Refractive Flow of Transparent Objects for Surface Normal Estimation and Manipulation",
  "abstract": "Transparent objects are widely used in our daily lives, making it important to teach robots to interact with them. However, it's not easy because the reflective and refractive effects can make depth cameras fail to give accurate geometry measurements. To solve this problem, this paper introduces RFTrans, an RGB-D-based method for surface normal estimation and manipulation of transparent objects. By leveraging refractive flow as an intermediate representation, the proposed method circumvents the drawbacks of directly predicting the geometry (e.g. surface normal) from images and helps bridge the sim-to-real gap. It integrates the RFNet, which predicts refractive flow, object mask, and boundaries, followed by the F2Net, which estimates surface normal from the refractive flow. To make manipulation possible, a global optimization module will take in the predictions, refine the raw depth, and construct the point cloud with normal. An off-the-shelf analytical grasp planning algorithm is followed to generate the grasp poses. We build a synthetic dataset with physically plausible ray-tracing rendering techniques to train the networks. Results show that the proposed method trained on the synthetic dataset can consistently outperform the baseline method in both synthetic and real-world benchmarks by a large margin. Finally, a real-world robot grasping task witnesses an 83% success rate, proving that refractive flow can help enable direct sim-to-real transfer. The code, data, and supplementary materials are available at https://rftrans.robotflow.ai.",
  "published": "2023-11-21",
  "updated": "2024-02-08",
  "year": "2023",
  "authors": [
   "Tutian Tang",
   "Jiyu Liu",
   "Jieyi Zhang",
   "Haoyuan Fu",
   "Wenqiang Xu",
   "Cewu Lu"
  ],
  "author_count": 6,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "RA-L",
  "venue_source": "semantic-scholar",
  "citations": 12,
  "influential_citations": 1,
  "tldr": "RFTrans, an RGB-D-based method for surface normal estimation and manipulation of transparent objects and a real-world robot grasping task witnesses an 83% success rate, proving that refractive flow can help enable direct sim-to-real transfer.",
  "doi": "10.1109/LRA.2024.3364837",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Tutian Tang",
    "id": "2030708645",
    "h_index": 10,
    "papers": 19
   },
   {
    "name": "Jiyu Liu",
    "id": "2358502925",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Jieyi Zhang",
    "id": "2267756111",
    "h_index": 7,
    "papers": 8
   },
   {
    "name": "Haoyuan Fu",
    "id": "2089776116",
    "h_index": 10,
    "papers": 12
   },
   {
    "name": "Wenqiang Xu",
    "id": "2000358050",
    "h_index": 20,
    "papers": 53
   },
   {
    "name": "Cewu Lu",
    "id": "2264977082",
    "h_index": 18,
    "papers": 53
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "sim2real",
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2311.12398v2",
  "pdf_url": "https://arxiv.org/pdf/2311.12398v2",
  "html_url": "https://arxiv.org/html/2311.12398v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.61
 },
 {
  "id": "2311.08400",
  "slug": "towards-open-ended-visual-recognition-with-large-language-model",
  "title": "Towards Open-Ended Visual Recognition with Large Language Model",
  "abstract": "Localizing and recognizing objects in the open-ended physical world poses a long-standing challenge within the domain of machine perception. Recent methods have endeavored to address the issue by employing a class-agnostic mask (or box) proposal model, complemented by an open-vocabulary classifier (e.g., CLIP) using pre-extracted text embeddings. However, it is worth noting that these open-vocabulary recognition models still exhibit limitations in practical applications. On one hand, they rely on the provision of class names during testing, where the recognition performance heavily depends on this predefined set of semantic classes by users. On the other hand, when training with multiple datasets, human intervention is required to alleviate the label definition conflict between them. In this paper, we introduce the OmniScient Model (OSM), a novel Large Language Model (LLM) based mask classifier, as a straightforward and effective solution to the aforementioned challenges. Specifically, OSM predicts class labels in a generative manner, thus removing the supply of class names during both training and testing. It also enables cross-dataset training without any human interference, exhibiting robust generalization capabilities due to the world knowledge acquired from the LLM. By combining OSM with an off-the-shelf mask proposal model, we present promising results on various benchmarks, and demonstrate its effectiveness in handling novel concepts. Code/model are available at https://github.com/bytedance/OmniScient-Model.",
  "published": "2023-11-14",
  "updated": "2023-11-14",
  "year": "2023",
  "authors": [
   "Qihang Yu",
   "Xiaohui Shen",
   "Liang-Chieh Chen"
  ],
  "author_count": 3,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 8,
  "influential_citations": 1,
  "tldr": "The OmniScient Model is introduced, a novel Large Language Model (LLM) based mask classifier that predicts class labels in a generative manner, thus removing the supply of class names during both training and testing, and enables cross-dataset training without any human interference.",
  "doi": "10.48550/arXiv.2311.08400",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Qihang Yu",
    "id": "2156559",
    "h_index": 23,
    "papers": 41
   },
   {
    "name": "Xiaohui Shen",
    "id": "2266472250",
    "h_index": 11,
    "papers": 18
   },
   {
    "name": "Liang-Chieh Chen",
    "id": "2266697544",
    "h_index": 13,
    "papers": 26
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [
   "ByteDance"
  ],
  "abs_url": "https://arxiv.org/abs/2311.08400v1",
  "pdf_url": "https://arxiv.org/pdf/2311.08400v1",
  "html_url": "https://arxiv.org/html/2311.08400v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.45
 },
 {
  "id": "2311.02848",
  "slug": "consistent4d-consistent-360-dynamic-object-generation-from-monocular-v",
  "title": "Consistent4D: Consistent 360\u00b0 Dynamic Object Generation from Monocular Video",
  "abstract": "In this paper, we present Consistent4D, a novel approach for generating 4D dynamic objects from uncalibrated monocular videos. Uniquely, we cast the 360-degree dynamic object reconstruction as a 4D generation problem, eliminating the need for tedious multi-view data collection and camera calibration. This is achieved by leveraging the object-level 3D-aware image diffusion model as the primary supervision signal for training Dynamic Neural Radiance Fields (DyNeRF). Specifically, we propose a Cascade DyNeRF to facilitate stable convergence and temporal continuity under the supervision signal which is discrete along the time axis. To achieve spatial and temporal consistency, we further introduce an Interpolation-driven Consistency Loss. It is optimized by minimizing the discrepancy between rendered frames from DyNeRF and interpolated frames from a pre-trained video interpolation model. Extensive experiments show that our Consistent4D can perform competitively to prior art alternatives, opening up new possibilities for 4D dynamic object generation from monocular videos, whilst also demonstrating advantage for conventional text-to-3D generation tasks. Our project page is https://consistent4d.github.io/.",
  "published": "2023-11-06",
  "updated": "2023-11-06",
  "year": "2023",
  "authors": [
   "Yanqin Jiang",
   "Li Zhang",
   "Jin Gao",
   "Weimin Hu",
   "Yao Yao"
  ],
  "author_count": 5,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 143,
  "influential_citations": 45,
  "tldr": "This paper cast the 360-degree dynamic object reconstruction as a 4D generation problem, eliminating the need for tedious multi-view data collection and camera calibration, and proposes a Cascade DyNeRF to facilitate stable convergence and temporal continuity under the supervision signal which is discrete along the time axis.",
  "doi": "10.48550/arXiv.2311.02848",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yanqin Jiang",
    "id": "2265542118",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Li Zhang",
    "id": "2265652138",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Jin Gao",
    "id": "2265517996",
    "h_index": 2,
    "papers": 9
   },
   {
    "name": "Weiming Hu",
    "id": "2240303745",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Yao Yao",
    "id": "2265522585",
    "h_index": 4,
    "papers": 6
   }
  ],
  "comment": "Technique report. Project page: https://consistent4d.github.io/",
  "topics": [
   "spatial-3d",
   "data-teleop",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2311.02848v1",
  "pdf_url": "https://arxiv.org/pdf/2311.02848v1",
  "html_url": "https://arxiv.org/html/2311.02848v1",
  "code_url": "https://consistent4d.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.16
 },
 {
  "id": "2311.01832",
  "slug": "on-hand-held-grippers-and-the-morphological-gap-in-human-manipulation",
  "title": "On Hand-Held Grippers and the Morphological Gap in Human Manipulation Demonstration",
  "abstract": "Collecting manipulation demonstrations with robotic hardware is tedious - and thus difficult to scale. Recording data on robot hardware ensures that it is in the appropriate format for Learning from Demonstrations (LfD) methods. By contrast, humans are proficient manipulators, and recording their actions would be easy to scale, but it is challenging to use that data format with LfD methods. The question we explore is whether there is a method to collect data in a format that can be used with LfD while retaining some of the attractive features of recording human manipulation. We propose equipping humans with hand-held, hand-actuated parallel grippers and a head-mounted camera to record demonstrations of manipulation tasks. Using customised and reproducible grippers, we collect an initial dataset of common manipulation tasks. We show that there are tasks that, against our initial intuition, can be performed using parallel grippers. Qualitative insights are obtained regarding the impact of the difference in morphology on LfD by comparing the strategies used to complete tasks with human hands and grippers. Our data collection method bridges the gap between robot- and human-native manipulation demonstration. By making the design of our gripper prototype available, we hope to reduce other researchers effort to collect manipulation data.",
  "published": "2023-11-03",
  "updated": "2023-11-03",
  "year": "2023",
  "authors": [
   "Kiran Doshi",
   "Yijiang Huang",
   "Stelian Coros"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 7,
  "influential_citations": 0,
  "tldr": "This work proposes equipping humans with hand-held, hand-actuated parallel grippers and a head-mounted camera to record demonstrations of manipulation tasks, and collects an initial dataset of common manipulation tasks.",
  "doi": "10.48550/arXiv.2311.01832",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Kiran Doshi",
    "id": "2265387014",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Yijiang Huang",
    "id": "2265459398",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Stelian Coros",
    "id": "1783776",
    "h_index": 45,
    "papers": 194
   }
  ],
  "comment": "",
  "topics": [
   "data-teleop",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2311.01832v1",
  "pdf_url": "https://arxiv.org/pdf/2311.01832v1",
  "html_url": "https://arxiv.org/html/2311.01832v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 0.9
 },
 {
  "id": "2311.00924",
  "slug": "the-power-of-the-senses-generalizable-manipulation-from-vision-and-tou",
  "title": "The Power of the Senses: Generalizable Manipulation from Vision and Touch through Masked Multimodal Learning",
  "abstract": "Humans rely on the synergy of their senses for most essential tasks. For tasks requiring object manipulation, we seamlessly and effectively exploit the complementarity of our senses of vision and touch. This paper draws inspiration from such capabilities and aims to find a systematic approach to fuse visual and tactile information in a reinforcement learning setting. We propose Masked Multimodal Learning (M3L), which jointly learns a policy and visual-tactile representations based on masked autoencoding. The representations jointly learned from vision and touch improve sample efficiency, and unlock generalization capabilities beyond those achievable through each of the senses separately. Remarkably, representations learned in a multimodal setting also benefit vision-only policies at test time. We evaluate M3L on three simulated environments with both visual and tactile observations: robotic insertion, door opening, and dexterous in-hand manipulation, demonstrating the benefits of learning a multimodal policy. Code and videos of the experiments are available at https://sferrazza.cc/m3l_site.",
  "published": "2023-11-02",
  "updated": "2023-11-02",
  "year": "2023",
  "authors": [
   "Carmelo Sferrazza",
   "Younggyo Seo",
   "Hao Liu",
   "Youngwoon Lee",
   "Pieter Abbeel"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS",
  "venue_source": "semantic-scholar",
  "citations": 41,
  "influential_citations": 4,
  "tldr": "This paper proposes Masked Multimodal Learning (M3L), which jointly learns a policy and visual-tactile representations based on masked autoencoding that improve sample efficiency, unlock generalization capabilities beyond those achievable through each of the senses separately.",
  "doi": "10.1109/IROS58592.2024.10802719",
  "oa_pdf": "https://arxiv.org/pdf/2311.00924",
  "s2_authors": [
   {
    "name": "Carmelo Sferrazza",
    "id": "47218071",
    "h_index": 21,
    "papers": 44
   },
   {
    "name": "Younggyo Seo",
    "id": "2067714176",
    "h_index": 20,
    "papers": 36
   },
   {
    "name": "Hao Liu",
    "id": "2256317240",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Youngwoon Lee",
    "id": "2264988120",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Pieter Abbeel",
    "id": "2253464956",
    "h_index": 10,
    "papers": 18
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2311.00924v1",
  "pdf_url": "https://arxiv.org/pdf/2311.00924v1",
  "html_url": "https://arxiv.org/html/2311.00924v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.12
 },
 {
  "id": "2310.19797",
  "slug": "deft-dexterous-fine-tuning-for-real-world-hand-policies",
  "title": "DEFT: Dexterous Fine-Tuning for Real-World Hand Policies",
  "abstract": "Dexterity is often seen as a cornerstone of complex manipulation. Humans are able to perform a host of skills with their hands, from making food to operating tools. In this paper, we investigate these challenges, especially in the case of soft, deformable objects as well as complex, relatively long-horizon tasks. However, learning such behaviors from scratch can be data inefficient. To circumvent this, we propose a novel approach, DEFT (DExterous Fine-Tuning for Hand Policies), that leverages human-driven priors, which are executed directly in the real world. In order to improve upon these priors, DEFT involves an efficient online optimization procedure. With the integration of human-based learning and online fine-tuning, coupled with a soft robotic hand, DEFT demonstrates success across various tasks, establishing a robust, data-efficient pathway toward general dexterous manipulation. Please see our website at https://dexterous-finetuning.github.io for video results.",
  "published": "2023-10-30",
  "updated": "2023-12-12",
  "year": "2023",
  "authors": [
   "Aditya Kannan",
   "Kenneth Shaw",
   "Shikhar Bahl",
   "Pragna Mannam",
   "Deepak Pathak"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL 2023",
  "venue_source": "arxiv-comment",
  "citations": 28,
  "influential_citations": 1,
  "tldr": "A novel approach, DEFT (DExterous Fine-Tuning for Hand Policies), that leverages human-driven priors, which are executed directly in the real world, which demonstrates success across various tasks, establishing a robust, data-efficient pathway toward general dexterous manipulation.",
  "doi": "10.48550/arXiv.2310.19797",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Aditya Kannan",
    "id": "2223325382",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Kenneth Shaw",
    "id": "2263541750",
    "h_index": 10,
    "papers": 18
   },
   {
    "name": "Shikhar Bahl",
    "id": "8527563",
    "h_index": 20,
    "papers": 38
   },
   {
    "name": "Pragna Mannam",
    "id": "4724804",
    "h_index": 8,
    "papers": 11
   },
   {
    "name": "Deepak Pathak",
    "id": "2256629847",
    "h_index": 9,
    "papers": 16
   }
  ],
  "comment": "In CoRL 2023. Website at https://dexterous-finetuning.github.io/",
  "topics": [
   "dexterous-manipulation",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2310.19797v2",
  "pdf_url": "https://arxiv.org/pdf/2310.19797v2",
  "html_url": "https://arxiv.org/html/2310.19797v2",
  "code_url": "https://dexterous-finetuning.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.96
 },
 {
  "id": "2310.17596",
  "slug": "mimicgen-a-data-generation-system-for-scalable-robot-learning-using-hu",
  "title": "MimicGen: A Data Generation System for Scalable Robot Learning using Human Demonstrations",
  "abstract": "Imitation learning from a large set of human demonstrations has proved to be an effective paradigm for building capable robot agents. However, the demonstrations can be extremely costly and time-consuming to collect. We introduce MimicGen, a system for automatically synthesizing large-scale, rich datasets from only a small number of human demonstrations by adapting them to new contexts. We use MimicGen to generate over 50K demonstrations across 18 tasks with diverse scene configurations, object instances, and robot arms from just ~200 human demonstrations. We show that robot agents can be effectively trained on this generated dataset by imitation learning to achieve strong performance in long-horizon and high-precision tasks, such as multi-part assembly and coffee preparation, across broad initial state distributions. We further demonstrate that the effectiveness and utility of MimicGen data compare favorably to collecting additional human demonstrations, making it a powerful and economical approach towards scaling up robot learning. Datasets, simulation environments, videos, and more at https://mimicgen.github.io .",
  "published": "2023-10-26",
  "updated": "2023-10-26",
  "year": "2023",
  "authors": [
   "Ajay Mandlekar",
   "Soroush Nasiriany",
   "Bowen Wen",
   "Iretiayo Akinola",
   "Yashraj Narang",
   "Linxi Fan",
   "Yuke Zhu",
   "Dieter Fox"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 421,
  "influential_citations": 41,
  "tldr": "It is demonstrated that robot agents can be effectively trained on this generated dataset by imitation learning to achieve strong performance in long-horizon and high-precision tasks, such as multi-part assembly and coffee preparation, across broad initial state distributions.",
  "doi": "10.48550/arXiv.2310.17596",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "A. Mandlekar",
    "id": "49686756",
    "h_index": 36,
    "papers": 67
   },
   {
    "name": "Soroush Nasiriany",
    "id": "3457048",
    "h_index": 18,
    "papers": 24
   },
   {
    "name": "Bowen Wen",
    "id": "2261740421",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Iretiayo Akinola",
    "id": "2856639",
    "h_index": 19,
    "papers": 38
   },
   {
    "name": "Yashraj S. Narang",
    "id": "5046361",
    "h_index": 20,
    "papers": 36
   },
   {
    "name": "L. Fan",
    "id": "2257381161",
    "h_index": 18,
    "papers": 25
   },
   {
    "name": "Yuke Zhu",
    "id": "2253507326",
    "h_index": 16,
    "papers": 20
   },
   {
    "name": "Dieter Fox",
    "id": "2258436157",
    "h_index": 12,
    "papers": 14
   }
  ],
  "comment": "Conference on Robot Learning (CoRL) 2023",
  "topics": [
   "egocentric-data",
   "imitation-diffusion",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2310.17596v1",
  "pdf_url": "https://arxiv.org/pdf/2310.17596v1",
  "html_url": "https://arxiv.org/html/2310.17596v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.13
 },
 {
  "id": "2310.16917",
  "slug": "mimictouch-leveraging-multi-modal-human-tactile-demonstrations-for-con",
  "title": "MimicTouch: Leveraging Multi-modal Human Tactile Demonstrations for Contact-rich Manipulation",
  "abstract": "Tactile sensing is critical to fine-grained, contact-rich manipulation tasks, such as insertion and assembly. Prior research has shown the possibility of learning tactile-guided policy from teleoperated demonstration data. However, to provide the demonstration, human users often rely on visual feedback to control the robot. This creates a gap between the sensing modality used for controlling the robot (visual) and the modality of interest (tactile). To bridge this gap, we introduce \"MimicTouch\", a novel framework for learning policies directly from demonstrations provided by human users with their hands. The key innovations are i) a human tactile data collection system which collects multi-modal tactile dataset for learning human's tactile-guided control strategy, ii) an imitation learning-based framework for learning human's tactile-guided control strategy through such data, and iii) an online residual RL framework to bridge the embodiment gap between the human hand and the robot gripper. Through comprehensive experiments, we highlight the efficacy of utilizing human's tactile-guided control strategy to resolve contact-rich manipulation tasks. The project website is at https://sites.google.com/view/MimicTouch.",
  "published": "2023-10-25",
  "updated": "2025-02-06",
  "year": "2023",
  "authors": [
   "Kelin Yu",
   "Yunhai Han",
   "Qixian Wang",
   "Vaibhav Saxena",
   "Danfei Xu",
   "Ye Zhao"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 73,
  "influential_citations": 2,
  "tldr": "This work introduces \"MimicTouch\", a novel framework for learning policies directly from demonstrations provided by human users with their hands, and highlights the efficacy of utilizing human's tactile-guided control strategy to resolve contact-rich manipulation tasks.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Kelin Yu",
    "id": "2261909068",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Yunhai Han",
    "id": "1995513527",
    "h_index": 8,
    "papers": 14
   },
   {
    "name": "Qixian Wang",
    "id": "2304894766",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Vaibhav Saxena",
    "id": "2311885269",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Danfei Xu",
    "id": "2264393671",
    "h_index": 4,
    "papers": 15
   },
   {
    "name": "Ye Zhao",
    "id": "2261889422",
    "h_index": 2,
    "papers": 3
   }
  ],
  "comment": "Accepted by CoRL 2024, Best Paper Award at NeurIPS 2023 Touch Processing Workshop",
  "topics": [
   "tactile",
   "imitation-diffusion",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2310.16917v4",
  "pdf_url": "https://arxiv.org/pdf/2310.16917v4",
  "html_url": "https://arxiv.org/html/2310.16917v4",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.37
 },
 {
  "id": "2310.16828",
  "slug": "td-mpc2-scalable-robust-world-models-for-continuous-control",
  "title": "TD-MPC2: Scalable, Robust World Models for Continuous Control",
  "abstract": "TD-MPC is a model-based reinforcement learning (RL) algorithm that performs local trajectory optimization in the latent space of a learned implicit (decoder-free) world model. In this work, we present TD-MPC2: a series of improvements upon the TD-MPC algorithm. We demonstrate that TD-MPC2 improves significantly over baselines across 104 online RL tasks spanning 4 diverse task domains, achieving consistently strong results with a single set of hyperparameters. We further show that agent capabilities increase with model and data size, and successfully train a single 317M parameter agent to perform 80 tasks across multiple task domains, embodiments, and action spaces. We conclude with an account of lessons, opportunities, and risks associated with large TD-MPC2 agents. Explore videos, models, data, code, and more at https://tdmpc2.com",
  "published": "2023-10-25",
  "updated": "2024-03-21",
  "year": "2023",
  "authors": [
   "Nicklas Hansen",
   "Hao Su",
   "Xiaolong Wang"
  ],
  "author_count": 3,
  "categories": [
   "cs.LG",
   "cs.AI",
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.LG",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 571,
  "influential_citations": 91,
  "tldr": "It is demonstrated that TD-MPC2 improves significantly over baselines across 104 online RL tasks spanning 4 diverse task domains, achieving consistently strong results with a single set of hyperparameters.",
  "doi": "10.48550/arXiv.2310.16828",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Nicklas Hansen",
    "id": "1491707104",
    "h_index": 19,
    "papers": 39
   },
   {
    "name": "Hao Su",
    "id": "2255041135",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Xiaolong Wang",
    "id": "2255251113",
    "h_index": 10,
    "papers": 17
   }
  ],
  "comment": "ICLR 2024. Explore videos, models, data, code, and more at https://tdmpc2.com",
  "topics": [
   "world-models",
   "rl-control",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2310.16828v2",
  "pdf_url": "https://arxiv.org/pdf/2310.16828v2",
  "html_url": "https://arxiv.org/html/2310.16828v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.26
 },
 {
  "id": "2310.14400",
  "slug": "a-pytorch-reproduction-of-masked-generative-image-transformer",
  "title": "A Pytorch Reproduction of Masked Generative Image Transformer",
  "abstract": "In this technical report, we present a reproduction of MaskGIT: Masked Generative Image Transformer, using PyTorch. The approach involves leveraging a masked bidirectional transformer architecture, enabling image generation with only few steps (8~16 steps) for 512 x 512 resolution images, i.e., ~64x faster than an auto-regressive approach. Through rigorous experimentation and optimization, we achieved results that closely align with the findings presented in the original paper. We match the reported FID of 7.32 with our replication and obtain 7.59 with similar hyperparameters on ImageNet at resolution 512 x 512. Moreover, we improve over the official implementation with some minor hyperparameter tweaking, achieving FID of 7.26. At the lower resolution of 256 x 256 pixels, our reimplementation scores 6.80, in comparison to the original paper's 6.18. To promote further research on Masked Generative Models and facilitate their reproducibility, we released our code and pre-trained weights openly at https://github.com/valeoai/MaskGIT-pytorch/",
  "published": "2023-10-22",
  "updated": "2023-10-22",
  "year": "2023",
  "authors": [
   "Victor Besnier",
   "Mickael Chen"
  ],
  "author_count": 2,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 29,
  "influential_citations": 3,
  "tldr": "A reproduction of MaskGIT: Masked Generative Image Transformer, using PyTorch is presented, enabling image generation with only few steps for 512 x 512 resolution images, i.e., ~64x faster than an auto-regressive approach.",
  "doi": "10.48550/arXiv.2310.14400",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Victor Besnier",
    "id": "1400349847",
    "h_index": 8,
    "papers": 11
   },
   {
    "name": "Mickael Chen",
    "id": "2335122090",
    "h_index": 4,
    "papers": 5
   }
  ],
  "comment": "",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2310.14400v1",
  "pdf_url": "https://arxiv.org/pdf/2310.14400v1",
  "html_url": "https://arxiv.org/html/2310.14400v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.48
 },
 {
  "id": "2310.14189",
  "slug": "improved-techniques-for-training-consistency-models",
  "title": "Improved Techniques for Training Consistency Models",
  "abstract": "Consistency models are a nascent family of generative models that can sample high quality data in one step without the need for adversarial training. Current consistency models achieve optimal sample quality by distilling from pre-trained diffusion models and employing learned metrics such as LPIPS. However, distillation limits the quality of consistency models to that of the pre-trained diffusion model, and LPIPS causes undesirable bias in evaluation. To tackle these challenges, we present improved techniques for consistency training, where consistency models learn directly from data without distillation. We delve into the theory behind consistency training and identify a previously overlooked flaw, which we address by eliminating Exponential Moving Average from the teacher consistency model. To replace learned metrics like LPIPS, we adopt Pseudo-Huber losses from robust statistics. Additionally, we introduce a lognormal noise schedule for the consistency training objective, and propose to double total discretization steps every set number of training iterations. Combined with better hyperparameter tuning, these modifications enable consistency models to achieve FID scores of 2.51 and 3.25 on CIFAR-10 and ImageNet $64\\times 64$ respectively in a single sampling step. These scores mark a 3.5$\\times$ and 4$\\times$ improvement compared to prior consistency training approaches. Through two-step sampling, we further reduce FID scores to 2.24 and 2.77 on these two datasets, surpassing those obtained via distillation in both one-step and two-step settings, while narrowing the gap between consistency models and other state-of-the-art generative models.",
  "published": "2023-10-22",
  "updated": "2023-10-22",
  "year": "2023",
  "authors": [
   "Yang Song",
   "Prafulla Dhariwal"
  ],
  "author_count": 2,
  "categories": [
   "cs.LG"
  ],
  "primary_category": "cs.LG",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 471,
  "influential_citations": 91,
  "tldr": "The theory behind consistency training is explored and a previously overlooked flaw is addressed by eliminating Exponential Moving Average from the teacher consistency model, and Pseudo-Huber losses from robust statistics are adopted to replace learned metrics like LPIPS.",
  "doi": "10.48550/arXiv.2310.14189",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yang Song",
    "id": "2299946245",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "Prafulla Dhariwal",
    "id": "6515819",
    "h_index": 20,
    "papers": 43
   }
  ],
  "comment": "",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2310.14189v1",
  "pdf_url": "https://arxiv.org/pdf/2310.14189v1",
  "html_url": "https://arxiv.org/html/2310.14189v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.17
 },
 {
  "id": "2310.12982",
  "slug": "putting-the-object-back-into-video-object-segmentation",
  "title": "Putting the Object Back into Video Object Segmentation",
  "abstract": "We present Cutie, a video object segmentation (VOS) network with object-level memory reading, which puts the object representation from memory back into the video object segmentation result. Recent works on VOS employ bottom-up pixel-level memory reading which struggles due to matching noise, especially in the presence of distractors, resulting in lower performance in more challenging data. In contrast, Cutie performs top-down object-level memory reading by adapting a small set of object queries. Via those, it interacts with the bottom-up pixel features iteratively with a query-based object transformer (qt, hence Cutie). The object queries act as a high-level summary of the target object, while high-resolution feature maps are retained for accurate segmentation. Together with foreground-background masked attention, Cutie cleanly separates the semantics of the foreground object from the background. On the challenging MOSE dataset, Cutie improves by 8.7 J&F over XMem with a similar running time and improves by 4.2 J&F over DeAOT while being three times faster. Code is available at: https://hkchengrex.github.io/Cutie",
  "published": "2023-10-19",
  "updated": "2024-04-11",
  "year": "2023",
  "authors": [
   "Ho Kei Cheng",
   "Seoung Wug Oh",
   "Brian Price",
   "Joon-Young Lee",
   "Alexander Schwing"
  ],
  "author_count": 5,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 263,
  "influential_citations": 43,
  "tldr": "",
  "doi": "10.1109/CVPR52733.2024.00304",
  "oa_pdf": "https://arxiv.org/pdf/2310.12982",
  "s2_authors": [
   {
    "name": "H. Cheng",
    "id": "2238270458",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Seoung Wug Oh",
    "id": "3451982",
    "h_index": 21,
    "papers": 52
   },
   {
    "name": "Brian L. Price",
    "id": "2260337537",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Joon-Young Lee",
    "id": "2260646673",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Alexander G. Schwing",
    "id": "2238205921",
    "h_index": 4,
    "papers": 9
   }
  ],
  "comment": "CVPR 2024 Highlight. Project page: https://hkchengrex.github.io/Cutie",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2310.12982v2",
  "pdf_url": "https://arxiv.org/pdf/2310.12982v2",
  "html_url": "https://arxiv.org/html/2310.12982v2",
  "code_url": "https://hkchengrex.github.io/Cutie",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.92
 },
 {
  "id": "2310.12190",
  "slug": "dynamicrafter-animating-open-domain-images-with-video-diffusion-priors",
  "title": "DynamiCrafter: Animating Open-domain Images with Video Diffusion Priors",
  "abstract": "Animating a still image offers an engaging visual experience. Traditional image animation techniques mainly focus on animating natural scenes with stochastic dynamics (e.g. clouds and fluid) or domain-specific motions (e.g. human hair or body motions), and thus limits their applicability to more general visual content. To overcome this limitation, we explore the synthesis of dynamic content for open-domain images, converting them into animated videos. The key idea is to utilize the motion prior of text-to-video diffusion models by incorporating the image into the generative process as guidance. Given an image, we first project it into a text-aligned rich context representation space using a query transformer, which facilitates the video model to digest the image content in a compatible fashion. However, some visual details still struggle to be preserved in the resultant videos. To supplement with more precise image information, we further feed the full image to the diffusion model by concatenating it with the initial noises. Experimental results show that our proposed method can produce visually convincing and more logical & natural motions, as well as higher conformity to the input image. Comparative evaluation demonstrates the notable superiority of our approach over existing competitors.",
  "published": "2023-10-18",
  "updated": "2023-11-27",
  "year": "2023",
  "authors": [
   "Jinbo Xing",
   "Menghan Xia",
   "Yong Zhang",
   "Haoxin Chen",
   "Wangbo Yu",
   "Hanyuan Liu",
   "Xintao Wang",
   "Tien-Tsin Wong",
   "Ying Shan"
  ],
  "author_count": 9,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "ECCV",
  "venue_source": "semantic-scholar",
  "citations": 521,
  "influential_citations": 70,
  "tldr": "Experimental results show that the proposed method can produce visually convincing and more logical&natural motions, as well as higher conformity to the input image, over existing competitors.",
  "doi": "10.48550/arXiv.2310.12190",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jinbo Xing",
    "id": "2087273800",
    "h_index": 16,
    "papers": 25
   },
   {
    "name": "Menghan Xia",
    "id": "2257035878",
    "h_index": 16,
    "papers": 33
   },
   {
    "name": "Yong Zhang",
    "id": "2257199953",
    "h_index": 17,
    "papers": 24
   },
   {
    "name": "Haoxin Chen",
    "id": "2149052351",
    "h_index": 14,
    "papers": 20
   },
   {
    "name": "Gongye Liu",
    "id": "2269171464",
    "h_index": 9,
    "papers": 18
   },
   {
    "name": "Xintao Wang",
    "id": "2253795356",
    "h_index": 26,
    "papers": 44
   },
   {
    "name": "Tien-Tsin Wong",
    "id": "2267727086",
    "h_index": 6,
    "papers": 16
   },
   {
    "name": "Ying Shan",
    "id": "2257019659",
    "h_index": 23,
    "papers": 36
   }
  ],
  "comment": "Project page: https://doubiiu.github.io/projects/DynamiCrafter",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2310.12190v2",
  "pdf_url": "https://arxiv.org/pdf/2310.12190v2",
  "html_url": "https://arxiv.org/html/2310.12190v2",
  "code_url": "https://doubiiu.github.io/projects/DynamiCrafter",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.22
 },
 {
  "id": "2310.11513",
  "slug": "geneval-an-object-focused-framework-for-evaluating-text-to-image-align",
  "title": "GenEval: An Object-Focused Framework for Evaluating Text-to-Image Alignment",
  "abstract": "Recent breakthroughs in diffusion models, multimodal pretraining, and efficient finetuning have led to an explosion of text-to-image generative models. Given human evaluation is expensive and difficult to scale, automated methods are critical for evaluating the increasingly large number of new models. However, most current automated evaluation metrics like FID or CLIPScore only offer a holistic measure of image quality or image-text alignment, and are unsuited for fine-grained or instance-level analysis. In this paper, we introduce GenEval, an object-focused framework to evaluate compositional image properties such as object co-occurrence, position, count, and color. We show that current object detection models can be leveraged to evaluate text-to-image models on a variety of generation tasks with strong human agreement, and that other discriminative vision models can be linked to this pipeline to further verify properties like object color. We then evaluate several open-source text-to-image models and analyze their relative generative capabilities on our benchmark. We find that recent models demonstrate significant improvement on these tasks, though they are still lacking in complex capabilities such as spatial relations and attribute binding. Finally, we demonstrate how GenEval might be used to help discover existing failure modes, in order to inform development of the next generation of text-to-image models. Our code to run the GenEval framework is publicly available at https://github.com/djghosh13/geneval.",
  "published": "2023-10-17",
  "updated": "2023-10-17",
  "year": "2023",
  "authors": [
   "Dhruba Ghosh",
   "Hanna Hajishirzi",
   "Ludwig Schmidt"
  ],
  "author_count": 3,
  "categories": [
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.CV",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 972,
  "influential_citations": 244,
  "tldr": "GenEval is introduced, an object-focused framework to evaluate compositional image properties such as object co-occurrence, position, count, and color, and it is shown that current object detection models can be leveraged to evaluate text-to-image models on a variety of generation tasks with strong human agreement.",
  "doi": "10.48550/arXiv.2310.11513",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Dhruba Ghosh",
    "id": "2143028576",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "H. Hajishirzi",
    "id": "2259915495",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Ludwig Schmidt",
    "id": "2259918151",
    "h_index": 3,
    "papers": 4
   }
  ],
  "comment": "",
  "topics": [
   "foundation-pretraining",
   "video-generation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2310.11513v1",
  "pdf_url": "https://arxiv.org/pdf/2310.11513v1",
  "html_url": "https://arxiv.org/html/2310.11513v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.49
 },
 {
  "id": "2310.10639",
  "slug": "zero-shot-robotic-manipulation-with-pretrained-image-editing-diffusion",
  "title": "Zero-Shot Robotic Manipulation with Pretrained Image-Editing Diffusion Models",
  "abstract": "If generalist robots are to operate in truly unstructured environments, they need to be able to recognize and reason about novel objects and scenarios. Such objects and scenarios might not be present in the robot's own training data. We propose SuSIE, a method that leverages an image-editing diffusion model to act as a high-level planner by proposing intermediate subgoals that a low-level controller can accomplish. Specifically, we finetune InstructPix2Pix on video data, consisting of both human videos and robot rollouts, such that it outputs hypothetical future \"subgoal\" observations given the robot's current observation and a language command. We also use the robot data to train a low-level goal-conditioned policy to act as the aforementioned low-level controller. We find that the high-level subgoal predictions can utilize Internet-scale pretraining and visual understanding to guide the low-level goal-conditioned policy, achieving significantly better generalization and precision than conventional language-conditioned policies. We achieve state-of-the-art results on the CALVIN benchmark, and also demonstrate robust generalization on real-world manipulation tasks, beating strong baselines that have access to privileged information or that utilize orders of magnitude more compute and training data. The project website can be found at http://rail-berkeley.github.io/susie .",
  "published": "2023-10-16",
  "updated": "2023-10-16",
  "year": "2023",
  "authors": [
   "Kevin Black",
   "Mitsuhiko Nakamoto",
   "Pranav Atreya",
   "Homer Walke",
   "Chelsea Finn",
   "Aviral Kumar",
   "Sergey Levine"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 331,
  "influential_citations": 35,
  "tldr": "SuSIE, a method that leverages an image-editing diffusion model to act as a high-level planner by proposing intermediate subgoals that a low-level controller can accomplish, is proposed, achieving significantly better generalization and precision than conventional language-conditioned policies.",
  "doi": "10.48550/arXiv.2310.10639",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Kevin Black",
    "id": "2258959388",
    "h_index": 13,
    "papers": 15
   },
   {
    "name": "Mitsuhiko Nakamoto",
    "id": "2211096149",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "P. Atreya",
    "id": "1643955309",
    "h_index": 9,
    "papers": 13
   },
   {
    "name": "H. Walke",
    "id": "2029241116",
    "h_index": 17,
    "papers": 23
   },
   {
    "name": "Chelsea Finn",
    "id": "2257346440",
    "h_index": 23,
    "papers": 32
   },
   {
    "name": "Aviral Kumar",
    "id": "1488785534",
    "h_index": 46,
    "papers": 83
   },
   {
    "name": "Sergey Levine",
    "id": "2249615151",
    "h_index": 32,
    "papers": 53
   }
  ],
  "comment": "22 pages, 8 figures",
  "topics": [
   "egocentric-data",
   "foundation-pretraining",
   "video-generation"
  ],
  "orgs": [
   "UC Berkeley"
  ],
  "abs_url": "https://arxiv.org/abs/2310.10639v1",
  "pdf_url": "https://arxiv.org/pdf/2310.10639v1",
  "html_url": "https://arxiv.org/html/2310.10639v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.52
 },
 {
  "id": "2310.09615",
  "slug": "storm-efficient-stochastic-transformer-based-world-models-for-reinforc",
  "title": "STORM: Efficient Stochastic Transformer based World Models for Reinforcement Learning",
  "abstract": "Recently, model-based reinforcement learning algorithms have demonstrated remarkable efficacy in visual input environments. These approaches begin by constructing a parameterized simulation world model of the real environment through self-supervised learning. By leveraging the imagination of the world model, the agent's policy is enhanced without the constraints of sampling from the real environment. The performance of these algorithms heavily relies on the sequence modeling and generation capabilities of the world model. However, constructing a perfectly accurate model of a complex unknown environment is nearly impossible. Discrepancies between the model and reality may cause the agent to pursue virtual goals, resulting in subpar performance in the real environment. Introducing random noise into model-based reinforcement learning has been proven beneficial. In this work, we introduce Stochastic Transformer-based wORld Model (STORM), an efficient world model architecture that combines the strong sequence modeling and generation capabilities of Transformers with the stochastic nature of variational autoencoders. STORM achieves a mean human performance of $126.7\\%$ on the Atari $100$k benchmark, setting a new record among state-of-the-art methods that do not employ lookahead search techniques. Moreover, training an agent with $1.85$ hours of real-time interaction experience on a single NVIDIA GeForce RTX 3090 graphics card requires only $4.3$ hours, showcasing improved efficiency compared to previous methodologies.",
  "published": "2023-10-14",
  "updated": "2023-10-14",
  "year": "2023",
  "authors": [
   "Weipu Zhang",
   "Gang Wang",
   "Jian Sun",
   "Yetian Yuan",
   "Gao Huang"
  ],
  "author_count": 5,
  "categories": [
   "cs.LG"
  ],
  "primary_category": "cs.LG",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 144,
  "influential_citations": 9,
  "tldr": "STORM is introduced, an efficient world model architecture that combines the strong sequence modeling and generation capabilities of Transformers with the stochastic nature of variational autoencoders and achieves a new record among state-of-the-art methods that do not employ lookahead search techniques.",
  "doi": "10.48550/arXiv.2310.09615",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Weipu Zhang",
    "id": "2258753215",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Gang Wang",
    "id": "2258688044",
    "h_index": 8,
    "papers": 53
   },
   {
    "name": "Jian Sun",
    "id": "2152148509",
    "h_index": 17,
    "papers": 84
   },
   {
    "name": "Yetian Yuan",
    "id": "2258795744",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Gao Huang",
    "id": "2258709443",
    "h_index": 3,
    "papers": 7
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "rl-control"
  ],
  "orgs": [
   "NVIDIA"
  ],
  "abs_url": "https://arxiv.org/abs/2310.09615v1",
  "pdf_url": "https://arxiv.org/pdf/2310.09615v1",
  "html_url": "https://arxiv.org/html/2310.09615v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.16
 },
 {
  "id": "2310.08864",
  "slug": "open-x-embodiment-robotic-learning-datasets-and-rt-x-models",
  "title": "Open X-Embodiment: Robotic Learning Datasets and RT-X Models",
  "abstract": "Large, high-capacity models trained on diverse datasets have shown remarkable successes on efficiently tackling downstream applications. In domains from NLP to Computer Vision, this has led to a consolidation of pretrained models, with general pretrained backbones serving as a starting point for many applications. Can such a consolidation happen in robotics? Conventionally, robotic learning methods train a separate model for every application, every robot, and even every environment. Can we instead train generalist X-robot policy that can be adapted efficiently to new robots, tasks, and environments? In this paper, we provide datasets in standardized data formats and models to make it possible to explore this possibility in the context of robotic manipulation, alongside experimental results that provide an example of effective X-robot policies. We assemble a dataset from 22 different robots collected through a collaboration between 21 institutions, demonstrating 527 skills (160266 tasks). We show that a high-capacity model trained on this data, which we call RT-X, exhibits positive transfer and improves the capabilities of multiple robots by leveraging experience from other platforms. More details can be found on the project website https://robotics-transformer-x.github.io.",
  "published": "2023-10-13",
  "updated": "2025-05-14",
  "year": "2023",
  "authors": [
   "Open X-Embodiment Collaboration",
   "Abby O'Neill",
   "Abdul Rehman",
   "Abhinav Gupta",
   "Abhiram Maddukuri",
   "Abhishek Gupta",
   "Abhishek Padalkar",
   "Abraham Lee",
   "Acorn Pooley",
   "Agrim Gupta",
   "Ajay Mandlekar",
   "Ajinkya Jain",
   "Albert Tung",
   "Alex Bewley",
   "Alex Herzog",
   "Alex Irpan",
   "Alexander Khazatsky",
   "Anant Rai",
   "Anchit Gupta",
   "Andrew Wang",
   "Andrey Kolobov",
   "Anikait Singh",
   "Animesh Garg",
   "Aniruddha Kembhavi",
   "Annie Xie",
   "Anthony Brohan",
   "Antonin Raffin",
   "Archit Sharma",
   "Arefeh Yavary",
   "Arhan Jain",
   "Ashwin Balakrishna",
   "Ayzaan Wahid",
   "Ben Burgess-Limerick",
   "Beomjoon Kim",
   "Bernhard Sch\u00f6lkopf",
   "Blake Wulfe",
   "Brian Ichter",
   "Cewu Lu",
   "Charles Xu",
   "Charlotte Le",
   "Chelsea Finn",
   "Chen Wang",
   "Chenfeng Xu",
   "Cheng Chi",
   "Chenguang Huang",
   "Christine Chan",
   "Christopher Agia",
   "Chuer Pan",
   "Chuyuan Fu",
   "Coline Devin",
   "Danfei Xu",
   "Daniel Morton",
   "Danny Driess",
   "Daphne Chen",
   "Deepak Pathak",
   "Dhruv Shah",
   "Dieter B\u00fcchler",
   "Dinesh Jayaraman",
   "Dmitry Kalashnikov",
   "Dorsa Sadigh",
   "Edward Johns",
   "Ethan Foster",
   "Fangchen Liu",
   "Federico Ceola",
   "Fei Xia",
   "Feiyu Zhao",
   "Felipe Vieira Frujeri",
   "Freek Stulp",
   "Gaoyue Zhou",
   "Gaurav S. Sukhatme",
   "Gautam Salhotra",
   "Ge Yan",
   "Gilbert Feng",
   "Giulio Schiavi",
   "Glen Berseth",
   "Gregory Kahn",
   "Guangwen Yang",
   "Guanzhi Wang",
   "Hao Su",
   "Hao-Shu Fang",
   "Haochen Shi",
   "Henghui Bao",
   "Heni Ben Amor",
   "Henrik I Christensen",
   "Hiroki Furuta",
   "Homanga Bharadhwaj",
   "Homer Walke",
   "Hongjie Fang",
   "Huy Ha",
   "Igor Mordatch",
   "Ilija Radosavovic",
   "Isabel Leal",
   "Jacky Liang",
   "Jad Abou-Chakra",
   "Jaehyung Kim",
   "Jaimyn Drake",
   "Jan Peters",
   "Jan Schneider",
   "Jasmine Hsu",
   "Jay Vakil",
   "Jeannette Bohg",
   "Jeffrey Bingham",
   "Jeffrey Wu",
   "Jensen Gao",
   "Jiaheng Hu",
   "Jiajun Wu",
   "Jialin Wu",
   "Jiankai Sun",
   "Jianlan Luo",
   "Jiayuan Gu",
   "Jie Tan",
   "Jihoon Oh",
   "Jimmy Wu",
   "Jingpei Lu",
   "Jingyun Yang",
   "Jitendra Malik",
   "Jo\u00e3o Silv\u00e9rio",
   "Joey Hejna",
   "Jonathan Booher",
   "Jonathan Tompson",
   "Jonathan Yang",
   "Jordi Salvador",
   "Joseph J. Lim",
   "Junhyek Han",
   "Kaiyuan Wang",
   "Kanishka Rao",
   "Karl Pertsch",
   "Karol Hausman",
   "Keegan Go",
   "Keerthana Gopalakrishnan",
   "Ken Goldberg",
   "Kendra Byrne",
   "Kenneth Oslund",
   "Kento Kawaharazuka",
   "Kevin Black",
   "Kevin Lin",
   "Kevin Zhang",
   "Kiana Ehsani",
   "Kiran Lekkala",
   "Kirsty Ellis",
   "Krishan Rana",
   "Krishnan Srinivasan",
   "Kuan Fang",
   "Kunal Pratap Singh",
   "Kuo-Hao Zeng",
   "Kyle Hatch",
   "Kyle Hsu",
   "Laurent Itti",
   "Lawrence Yunliang Chen",
   "Lerrel Pinto",
   "Li Fei-Fei",
   "Liam Tan",
   "Linxi \"Jim\" Fan",
   "Lionel Ott",
   "Lisa Lee",
   "Luca Weihs",
   "Magnum Chen",
   "Marion Lepert",
   "Marius Memmel",
   "Masayoshi Tomizuka",
   "Masha Itkina",
   "Mateo Guaman Castro",
   "Max Spero",
   "Maximilian Du",
   "Michael Ahn",
   "Michael C. Yip",
   "Mingtong Zhang",
   "Mingyu Ding",
   "Minho Heo",
   "Mohan Kumar Srirama",
   "Mohit Sharma",
   "Moo Jin Kim",
   "Muhammad Zubair Irshad",
   "Naoaki Kanazawa",
   "Nicklas Hansen",
   "Nicolas Heess",
   "Nikhil J Joshi",
   "Niko Suenderhauf",
   "Ning Liu",
   "Norman Di Palo",
   "Nur Muhammad Mahi Shafiullah",
   "Oier Mees",
   "Oliver Kroemer",
   "Osbert Bastani",
   "Pannag R Sanketi",
   "Patrick \"Tree\" Miller",
   "Patrick Yin",
   "Paul Wohlhart",
   "Peng Xu",
   "Peter David Fagan",
   "Peter Mitrano",
   "Pierre Sermanet",
   "Pieter Abbeel",
   "Priya Sundaresan",
   "Qiuyu Chen",
   "Quan Vuong",
   "Rafael Rafailov",
   "Ran Tian",
   "Ria Doshi",
   "Roberto Mart\u00edn-Mart\u00edn",
   "Rohan Baijal",
   "Rosario Scalise",
   "Rose Hendrix",
   "Roy Lin",
   "Runjia Qian",
   "Ruohan Zhang",
   "Russell Mendonca",
   "Rutav Shah",
   "Ryan Hoque",
   "Ryan Julian",
   "Samuel Bustamante",
   "Sean Kirmani",
   "Sergey Levine",
   "Shan Lin",
   "Sherry Moore",
   "Shikhar Bahl",
   "Shivin Dass",
   "Shubham Sonawani",
   "Shubham Tulsiani",
   "Shuran Song",
   "Sichun Xu",
   "Siddhant Haldar",
   "Siddharth Karamcheti",
   "Simeon Adebola",
   "Simon Guist",
   "Soroush Nasiriany",
   "Stefan Schaal",
   "Stefan Welker",
   "Stephen Tian",
   "Subramanian Ramamoorthy",
   "Sudeep Dasari",
   "Suneel Belkhale",
   "Sungjae Park",
   "Suraj Nair",
   "Suvir Mirchandani",
   "Takayuki Osa",
   "Tanmay Gupta",
   "Tatsuya Harada",
   "Tatsuya Matsushima",
   "Ted Xiao",
   "Thomas Kollar",
   "Tianhe Yu",
   "Tianli Ding",
   "Todor Davchev",
   "Tony Z. Zhao",
   "Travis Armstrong",
   "Trevor Darrell",
   "Trinity Chung",
   "Vidhi Jain",
   "Vikash Kumar",
   "Vincent Vanhoucke",
   "Vitor Guizilini",
   "Wei Zhan",
   "Wenxuan Zhou",
   "Wolfram Burgard",
   "Xi Chen",
   "Xiangyu Chen",
   "Xiaolong Wang",
   "Xinghao Zhu",
   "Xinyang Geng",
   "Xiyuan Liu",
   "Xu Liangwei",
   "Xuanlin Li",
   "Yansong Pang",
   "Yao Lu",
   "Yecheng Jason Ma",
   "Yejin Kim",
   "Yevgen Chebotar",
   "Yifan Zhou",
   "Yifeng Zhu",
   "Yilin Wu",
   "Ying Xu",
   "Yixuan Wang",
   "Yonatan Bisk",
   "Yongqiang Dou",
   "Yoonyoung Cho",
   "Youngwoon Lee",
   "Yuchen Cui",
   "Yue Cao",
   "Yueh-Hua Wu",
   "Yujin Tang",
   "Yuke Zhu",
   "Yunchu Zhang",
   "Yunfan Jiang",
   "Yunshuang Li",
   "Yunzhu Li",
   "Yusuke Iwasawa",
   "Yutaka Matsuo",
   "Zehan Ma",
   "Zhuo Xu",
   "Zichen Jeff Cui",
   "Zichen Zhang",
   "Zipeng Fu",
   "Zipeng Lin"
  ],
  "author_count": 294,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA",
  "venue_source": "semantic-scholar",
  "citations": 1153,
  "influential_citations": 81,
  "tldr": "",
  "doi": "10.1109/ICRA57147.2024.10611477",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "A. Padalkar",
    "id": "5836681",
    "h_index": 7,
    "papers": 16
   },
   {
    "name": "A. Pooley",
    "id": "2253748715",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Ajinkya Jain",
    "id": "2256951061",
    "h_index": 7,
    "papers": 7
   },
   {
    "name": "Alex Bewley",
    "id": "2238127835",
    "h_index": 11,
    "papers": 14
   },
   {
    "name": "Alex Herzog",
    "id": "2253642204",
    "h_index": 8,
    "papers": 14
   },
   {
    "name": "A. Irpan",
    "id": "17818078",
    "h_index": 22,
    "papers": 32
   },
   {
    "name": "Alexander Khazatsky",
    "id": "121873407",
    "h_index": 10,
    "papers": 11
   },
   {
    "name": "Anant Rai",
    "id": "2253550685",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Anikait Singh",
    "id": "2111007256",
    "h_index": 16,
    "papers": 26
   },
   {
    "name": "Anthony Brohan",
    "id": "118025075",
    "h_index": 12,
    "papers": 19
   },
   {
    "name": "A. Raffin",
    "id": "51247992",
    "h_index": 11,
    "papers": 33
   },
   {
    "name": "Ayzaan Wahid",
    "id": "88728227",
    "h_index": 21,
    "papers": 27
   },
   {
    "name": "Ben Burgess-Limerick",
    "id": "2375921719",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Beomjoon Kim",
    "id": "2238120623",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Bernhard Sch\u00f6lkopf",
    "id": "2237290631",
    "h_index": 7,
    "papers": 28
   },
   {
    "name": "Brian Ichter",
    "id": "2704814",
    "h_index": 37,
    "papers": 60
   },
   {
    "name": "Cewu Lu",
    "id": "2281998765",
    "h_index": 24,
    "papers": 45
   },
   {
    "name": "Charles Xu",
    "id": "2254150689",
    "h_index": 10,
    "papers": 14
   },
   {
    "name": "Chelsea Finn",
    "id": "46881670",
    "h_index": 94,
    "papers": 240
   },
   {
    "name": "Chenfeng Xu",
    "id": "1490695028",
    "h_index": 25,
    "papers": 46
   },
   {
    "name": "Cheng Chi",
    "id": "2253746565",
    "h_index": 8,
    "papers": 11
   },
   {
    "name": "Chenguang Huang",
    "id": "2256612775",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Christine Chan",
    "id": "2256938625",
    "h_index": 8,
    "papers": 8
   },
   {
    "name": "Chuer Pan",
    "id": "2253801737",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Chuyuan Fu",
    "id": "3430433",
    "h_index": 15,
    "papers": 21
   },
   {
    "name": "Coline Devin",
    "id": "144373380",
    "h_index": 24,
    "papers": 39
   },
   {
    "name": "Danny Driess",
    "id": "2283848260",
    "h_index": 27,
    "papers": 35
   },
   {
    "name": "Deepak Pathak",
    "id": "2004879394",
    "h_index": 24,
    "papers": 32
   },
   {
    "name": "Dhruv Shah",
    "id": "2322628540",
    "h_index": 29,
    "papers": 63
   },
   {
    "name": "Dieter B\u00fcchler",
    "id": "108281807",
    "h_index": 10,
    "papers": 24
   },
   {
    "name": "Dmitry Kalashnikov",
    "id": "48313860",
    "h_index": 21,
    "papers": 34
   },
   {
    "name": "Dorsa Sadigh",
    "id": "1779671",
    "h_index": 69,
    "papers": 226
   },
   {
    "name": "Edward Johns",
    "id": "2253752957",
    "h_index": 10,
    "papers": 14
   },
   {
    "name": "Federico Ceola",
    "id": "1443777921",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Fei Xia",
    "id": "2237987648",
    "h_index": 11,
    "papers": 15
   },
   {
    "name": "F. Stulp",
    "id": "50707365",
    "h_index": 36,
    "papers": 179
   },
   {
    "name": "Gaoyue Zhou",
    "id": "2257386929",
    "h_index": 9,
    "papers": 12
   },
   {
    "name": "G. Sukhatme",
    "id": "1732493",
    "h_index": 95,
    "papers": 782
   },
   {
    "name": "Gautam Salhotra",
    "id": "101987257",
    "h_index": 9,
    "papers": 21
   },
   {
    "name": "Ge Yan",
    "id": "2253536756",
    "h_index": 7,
    "papers": 9
   },
   {
    "name": "Giulio Schiavi",
    "id": "2184722575",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Hao Su",
    "id": "2255041135",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Haoshu Fang",
    "id": "122851212",
    "h_index": 34,
    "papers": 61
   },
   {
    "name": "Haochen Shi",
    "id": "2257452648",
    "h_index": 7,
    "papers": 9
   },
   {
    "name": "H. B. Amor",
    "id": "2207330",
    "h_index": 32,
    "papers": 163
   },
   {
    "name": "Henrik I Christensen",
    "id": "2243047488",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Hiroki Furuta",
    "id": "2052903664",
    "h_index": 16,
    "papers": 31
   },
   {
    "name": "H. Walke",
    "id": "2029241116",
    "h_index": 17,
    "papers": 23
   },
   {
    "name": "Hongjie Fang",
    "id": "2152115958",
    "h_index": 16,
    "papers": 31
   },
   {
    "name": "Igor Mordatch",
    "id": "2080746",
    "h_index": 34,
    "papers": 49
   },
   {
    "name": "Ilija Radosavovic",
    "id": "30407997",
    "h_index": 20,
    "papers": 22
   },
   {
    "name": "Isabel Leal",
    "id": "2057988112",
    "h_index": 12,
    "papers": 15
   },
   {
    "name": "Jacky Liang",
    "id": "6454541",
    "h_index": 23,
    "papers": 40
   },
   {
    "name": "Jaehyung Kim",
    "id": "2238088091",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Jan Schneider",
    "id": "2221192892",
    "h_index": 7,
    "papers": 14
   },
   {
    "name": "Jasmine Hsu",
    "id": "2726592",
    "h_index": 15,
    "papers": 20
   },
   {
    "name": "Jeannette Bohg",
    "id": "1775407",
    "h_index": 50,
    "papers": 161
   },
   {
    "name": "Jeff Bingham",
    "id": "1696556124",
    "h_index": 8,
    "papers": 9
   },
   {
    "name": "Jiajun Wu",
    "id": "2326444500",
    "h_index": 35,
    "papers": 214
   },
   {
    "name": "Jialin Wu",
    "id": "9095876",
    "h_index": 14,
    "papers": 29
   },
   {
    "name": "Jiankai Sun",
    "id": "2282025",
    "h_index": 22,
    "papers": 61
   },
   {
    "name": "Jianlan Luo",
    "id": "2238220544",
    "h_index": 13,
    "papers": 21
   },
   {
    "name": "Jiayuan Gu",
    "id": "2256468903",
    "h_index": 12,
    "papers": 15
   },
   {
    "name": "Jie Tan",
    "id": "2257005884",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Jihoon Oh",
    "id": "2256484322",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Jitendra Malik",
    "id": "2257249601",
    "h_index": 11,
    "papers": 16
   },
   {
    "name": "Jonathan Tompson",
    "id": "2704494",
    "h_index": 43,
    "papers": 71
   },
   {
    "name": "Jonathan Yang",
    "id": "2143067490",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Joseph J. Lim",
    "id": "2253826001",
    "h_index": 7,
    "papers": 9
   },
   {
    "name": "Jo\u00e3o Silv\u00e9rio",
    "id": "2184921450",
    "h_index": 6,
    "papers": 17
   },
   {
    "name": "Junhyek Han",
    "id": "2238178686",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Kanishka Rao",
    "id": "2256967164",
    "h_index": 9,
    "papers": 14
   },
   {
    "name": "Karl Pertsch",
    "id": "31719101",
    "h_index": 32,
    "papers": 56
   },
   {
    "name": "Karol Hausman",
    "id": "1944801",
    "h_index": 47,
    "papers": 122
   },
   {
    "name": "Keegan Go",
    "id": "2253750510",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "K. Gopalakrishnan",
    "id": "2161342233",
    "h_index": 17,
    "papers": 25
   },
   {
    "name": "Ken Goldberg",
    "id": "2253727638",
    "h_index": 8,
    "papers": 13
   },
   {
    "name": "Kendra Byrne",
    "id": "2253740486",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Kenneth Oslund",
    "id": "21095952",
    "h_index": 9,
    "papers": 10
   },
   {
    "name": "Kento Kawaharazuka",
    "id": "8308607",
    "h_index": 17,
    "papers": 222
   },
   {
    "name": "Kevin Zhang",
    "id": "2254189637",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "K. Majd",
    "id": "81905590",
    "h_index": 6,
    "papers": 16
   },
   {
    "name": "Krishan Rana",
    "id": "51047912",
    "h_index": 13,
    "papers": 24
   },
   {
    "name": "K. Srinivasan",
    "id": "2093939303",
    "h_index": 16,
    "papers": 29
   },
   {
    "name": "L. Chen",
    "id": "2143804724",
    "h_index": 14,
    "papers": 24
   },
   {
    "name": "Lerrel Pinto",
    "id": "2253567347",
    "h_index": 15,
    "papers": 25
   },
   {
    "name": "Liam Tan",
    "id": "2253731404",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Lionel Ott",
    "id": "2253741217",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Lisa Lee",
    "id": "2256481400",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Masayoshi Tomizuka",
    "id": "2245825283",
    "h_index": 24,
    "papers": 99
   },
   {
    "name": "Maximilian Du",
    "id": "117791840",
    "h_index": 10,
    "papers": 17
   },
   {
    "name": "Michael Ahn",
    "id": "2106194123",
    "h_index": 10,
    "papers": 15
   },
   {
    "name": "Mingtong Zhang",
    "id": "2247824858",
    "h_index": 11,
    "papers": 24
   },
   {
    "name": "Mingyu Ding",
    "id": "2253455336",
    "h_index": 15,
    "papers": 33
   },
   {
    "name": "M. K. Srirama",
    "id": "2193493900",
    "h_index": 12,
    "papers": 14
   },
   {
    "name": "Mohit Sharma",
    "id": "145467103",
    "h_index": 17,
    "papers": 35
   },
   {
    "name": "Moo Jin Kim",
    "id": "2159987907",
    "h_index": 10,
    "papers": 10
   },
   {
    "name": "Muhammad Zubair Irshad",
    "id": "147495445",
    "h_index": 17,
    "papers": 37
   },
   {
    "name": "Naoaki Kanazawa",
    "id": "2173085172",
    "h_index": 8,
    "papers": 29
   },
   {
    "name": "Nicklas Hansen",
    "id": "1491707104",
    "h_index": 19,
    "papers": 39
   },
   {
    "name": "N. Heess",
    "id": "2801204",
    "h_index": 73,
    "papers": 192
   },
   {
    "name": "Nikhil J. Joshi",
    "id": "2052368480",
    "h_index": 11,
    "papers": 14
   },
   {
    "name": "Niko Suenderhauf",
    "id": "2253748689",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Norman Di Palo",
    "id": "52218759",
    "h_index": 14,
    "papers": 20
   },
   {
    "name": "Nur Muhammad Shafiullah",
    "id": "2253752930",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Oier Mees",
    "id": "7264115",
    "h_index": 28,
    "papers": 45
   },
   {
    "name": "Oliver Kroemer",
    "id": "2250726149",
    "h_index": 9,
    "papers": 30
   },
   {
    "name": "Pannag R. Sanketi",
    "id": "2840758",
    "h_index": 22,
    "papers": 44
   },
   {
    "name": "Paul Wohlhart",
    "id": "3202367",
    "h_index": 24,
    "papers": 47
   },
   {
    "name": "Peng Xu",
    "id": "2153917744",
    "h_index": 17,
    "papers": 22
   },
   {
    "name": "P. Sermanet",
    "id": "3142556",
    "h_index": 39,
    "papers": 77
   },
   {
    "name": "Priya Sundaresan",
    "id": "123235030",
    "h_index": 19,
    "papers": 31
   },
   {
    "name": "Q. Vuong",
    "id": "144579461",
    "h_index": 23,
    "papers": 40
   },
   {
    "name": "Rafael Rafailov",
    "id": "102801230",
    "h_index": 25,
    "papers": 44
   },
   {
    "name": "Ran Tian",
    "id": "2253748783",
    "h_index": 7,
    "papers": 9
   },
   {
    "name": "Ria Doshi",
    "id": "2197078118",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "R. Mendonca",
    "id": "35509365",
    "h_index": 15,
    "papers": 24
   },
   {
    "name": "Rutav Shah",
    "id": "2117716975",
    "h_index": 12,
    "papers": 17
   },
   {
    "name": "Ryan Hoque",
    "id": "1387872831",
    "h_index": 16,
    "papers": 25
   },
   {
    "name": "Ryan C. Julian",
    "id": "144885996",
    "h_index": 19,
    "papers": 34
   },
   {
    "name": "Samuel Bustamante",
    "id": "2253588342",
    "h_index": 5,
    "papers": 15
   },
   {
    "name": "Sean Kirmani",
    "id": "51881277",
    "h_index": 21,
    "papers": 29
   },
   {
    "name": "Sergey Levine",
    "id": "2249615151",
    "h_index": 32,
    "papers": 53
   },
   {
    "name": "Sherry Moore",
    "id": "144375552",
    "h_index": 9,
    "papers": 12
   },
   {
    "name": "Shikhar Bahl",
    "id": "8527563",
    "h_index": 20,
    "papers": 38
   },
   {
    "name": "Shivin Dass",
    "id": "2193057311",
    "h_index": 9,
    "papers": 15
   },
   {
    "name": "Shuran Song",
    "id": "2254874914",
    "h_index": 12,
    "papers": 12
   },
   {
    "name": "Sichun Xu",
    "id": "3068504",
    "h_index": 11,
    "papers": 16
   },
   {
    "name": "Siddhant Haldar",
    "id": "51445278",
    "h_index": 13,
    "papers": 21
   },
   {
    "name": "S. Adebola",
    "id": "1722033888",
    "h_index": 8,
    "papers": 27
   },
   {
    "name": "Simon Guist",
    "id": "1631651129",
    "h_index": 8,
    "papers": 16
   },
   {
    "name": "Soroush Nasiriany",
    "id": "3457048",
    "h_index": 18,
    "papers": 24
   },
   {
    "name": "S. Schaal",
    "id": "1745219",
    "h_index": 97,
    "papers": 443
   },
   {
    "name": "Stefan Welker",
    "id": "69426588",
    "h_index": 13,
    "papers": 17
   },
   {
    "name": "Stephen Tian",
    "id": "71692259",
    "h_index": 13,
    "papers": 14
   },
   {
    "name": "S. Dasari",
    "id": "36076404",
    "h_index": 23,
    "papers": 39
   },
   {
    "name": "Suneel Belkhale",
    "id": "69879999",
    "h_index": 18,
    "papers": 25
   },
   {
    "name": "Takayuki Osa",
    "id": "2253748734",
    "h_index": 5,
    "papers": 22
   },
   {
    "name": "Tatsuya Harada",
    "id": "2253762492",
    "h_index": 4,
    "papers": 21
   },
   {
    "name": "T. Matsushima",
    "id": "145930468",
    "h_index": 10,
    "papers": 35
   },
   {
    "name": "Ted Xiao",
    "id": "9961095",
    "h_index": 33,
    "papers": 45
   },
   {
    "name": "Tianhe Yu",
    "id": "10909315",
    "h_index": 31,
    "papers": 48
   },
   {
    "name": "Tianli Ding",
    "id": "95691186",
    "h_index": 15,
    "papers": 24
   },
   {
    "name": "Todor Davchev",
    "id": "121676884",
    "h_index": 15,
    "papers": 30
   },
   {
    "name": "Tony Zhao",
    "id": "145914976",
    "h_index": 17,
    "papers": 19
   },
   {
    "name": "Travis Armstrong",
    "id": "2057103965",
    "h_index": 9,
    "papers": 11
   },
   {
    "name": "T. Darrell",
    "id": "1398038481",
    "h_index": 9,
    "papers": 13
   },
   {
    "name": "Vidhi Jain",
    "id": "2253472236",
    "h_index": 9,
    "papers": 17
   },
   {
    "name": "Vincent Vanhoucke",
    "id": "2657155",
    "h_index": 34,
    "papers": 60
   },
   {
    "name": "Wei Zhan",
    "id": "144267500",
    "h_index": 42,
    "papers": 164
   },
   {
    "name": "Wenxuan Zhou",
    "id": "2246658976",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Wolfram Burgard",
    "id": "2106871731",
    "h_index": 128,
    "papers": 809
   },
   {
    "name": "Xi Chen",
    "id": "2254208802",
    "h_index": 9,
    "papers": 15
   },
   {
    "name": "Xiaolong Wang",
    "id": "2255251113",
    "h_index": 10,
    "papers": 17
   },
   {
    "name": "Xinghao Zhu",
    "id": "8362363",
    "h_index": 14,
    "papers": 29
   },
   {
    "name": "Xuanlin Li",
    "id": "2108263986",
    "h_index": 17,
    "papers": 21
   },
   {
    "name": "Yao Lu",
    "id": "2161346119",
    "h_index": 16,
    "papers": 22
   },
   {
    "name": "Yevgen Chebotar",
    "id": "2527420",
    "h_index": 33,
    "papers": 57
   },
   {
    "name": "Yifan Zhou",
    "id": "2256290706",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Yifeng Zhu",
    "id": "2255171532",
    "h_index": 11,
    "papers": 11
   },
   {
    "name": "Ying Xu",
    "id": "2256016502",
    "h_index": 10,
    "papers": 25
   },
   {
    "name": "Yixuan Wang",
    "id": "2253810575",
    "h_index": 8,
    "papers": 10
   },
   {
    "name": "Yonatan Bisk",
    "id": "3312309",
    "h_index": 46,
    "papers": 144
   },
   {
    "name": "Yoonyoung Cho",
    "id": "2255540603",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Youngwoon Lee",
    "id": "46358230",
    "h_index": 18,
    "papers": 27
   },
   {
    "name": "Yuchen Cui",
    "id": "2238151901",
    "h_index": 9,
    "papers": 24
   },
   {
    "name": "Yueh-Hua Wu",
    "id": "2253837067",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Yujin Tang",
    "id": "2217908034",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Yuke Zhu",
    "id": "2253507326",
    "h_index": 16,
    "papers": 20
   },
   {
    "name": "Yunzhu Li",
    "id": "2294926592",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Yusuke Iwasawa",
    "id": "1715282",
    "h_index": 23,
    "papers": 172
   },
   {
    "name": "Yutaka Matsuo",
    "id": "2241471533",
    "h_index": 13,
    "papers": 81
   },
   {
    "name": "Zhuo Xu",
    "id": "152248240",
    "h_index": 10,
    "papers": 17
   },
   {
    "name": "Zichen Jeff Cui",
    "id": "2172045711",
    "h_index": 5,
    "papers": 8
   }
  ],
  "comment": "Project website: https://robotics-transformer-x.github.io",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2310.08864v9",
  "pdf_url": "https://arxiv.org/pdf/2310.08864v9",
  "html_url": "https://arxiv.org/html/2310.08864v9",
  "code_url": "https://robotics-transformer-x.github.io",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.5
 },
 {
  "id": "2310.08576",
  "slug": "learning-to-act-from-actionless-videos-through-dense-correspondences",
  "title": "Learning to Act from Actionless Videos through Dense Correspondences",
  "abstract": "In this work, we present an approach to construct a video-based robot policy capable of reliably executing diverse tasks across different robots and environments from few video demonstrations without using any action annotations. Our method leverages images as a task-agnostic representation, encoding both the state and action information, and text as a general representation for specifying robot goals. By synthesizing videos that ``hallucinate'' robot executing actions and in combination with dense correspondences between frames, our approach can infer the closed-formed action to execute to an environment without the need of any explicit action labels. This unique capability allows us to train the policy solely based on RGB videos and deploy learned policies to various robotic tasks. We demonstrate the efficacy of our approach in learning policies on table-top manipulation and navigation tasks. Additionally, we contribute an open-source framework for efficient video modeling, enabling the training of high-fidelity policy models with four GPUs within a single day.",
  "published": "2023-10-12",
  "updated": "2023-10-12",
  "year": "2023",
  "authors": [
   "Po-Chen Ko",
   "Jiayuan Mao",
   "Yilun Du",
   "Shao-Hua Sun",
   "Joshua B. Tenenbaum"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.CV",
   "cs.LG",
   "stat.ML"
  ],
  "primary_category": "cs.RO",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 218,
  "influential_citations": 42,
  "tldr": "",
  "doi": "10.48550/arXiv.2310.08576",
  "oa_pdf": "https://arxiv.org/pdf/2310.08576",
  "s2_authors": [
   {
    "name": "Po-Chen Ko",
    "id": "2257344090",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Jiayuan Mao",
    "id": "13589371",
    "h_index": 23,
    "papers": 58
   },
   {
    "name": "Yilun Du",
    "id": "15394275",
    "h_index": 48,
    "papers": 86
   },
   {
    "name": "Shao-Hua Sun",
    "id": "2109374418",
    "h_index": 15,
    "papers": 20
   },
   {
    "name": "Josh Tenenbaum",
    "id": "2243002911",
    "h_index": 10,
    "papers": 16
   }
  ],
  "comment": "Project page: https://flow-diffusion.github.io/",
  "topics": [
   "navigation",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2310.08576v1",
  "pdf_url": "https://arxiv.org/pdf/2310.08576v1",
  "html_url": "https://arxiv.org/html/2310.08576v1",
  "code_url": "https://flow-diffusion.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.84
 },
 {
  "id": "2310.08528",
  "slug": "4d-gaussian-splatting-for-real-time-dynamic-scene-rendering",
  "title": "4D Gaussian Splatting for Real-Time Dynamic Scene Rendering",
  "abstract": "Representing and rendering dynamic scenes has been an important but challenging task. Especially, to accurately model complex motions, high efficiency is usually hard to guarantee. To achieve real-time dynamic scene rendering while also enjoying high training and storage efficiency, we propose 4D Gaussian Splatting (4D-GS) as a holistic representation for dynamic scenes rather than applying 3D-GS for each individual frame. In 4D-GS, a novel explicit representation containing both 3D Gaussians and 4D neural voxels is proposed. A decomposed neural voxel encoding algorithm inspired by HexPlane is proposed to efficiently build Gaussian features from 4D neural voxels and then a lightweight MLP is applied to predict Gaussian deformations at novel timestamps. Our 4D-GS method achieves real-time rendering under high resolutions, 82 FPS at an 800$\\times$800 resolution on an RTX 3090 GPU while maintaining comparable or better quality than previous state-of-the-art methods. More demos and code are available at https://guanjunwu.github.io/4dgs/.",
  "published": "2023-10-12",
  "updated": "2024-07-15",
  "year": "2023",
  "authors": [
   "Guanjun Wu",
   "Taoran Yi",
   "Jiemin Fang",
   "Lingxi Xie",
   "Xiaopeng Zhang",
   "Wei Wei",
   "Wenyu Liu",
   "Qi Tian",
   "Xinggang Wang"
  ],
  "author_count": 9,
  "categories": [
   "cs.CV",
   "cs.GR"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 1570,
  "influential_citations": 241,
  "tldr": "This work proposes 4D Gaussian Splatting (4D-GS) as a holistic representation for dynamic scenes rather than applying 3D-GS for each individual frame, and achieves real-time rendering under high resolutions, 82 FPS at an 800x800 resolution on an RTX 3090 GPU while maintaining comparable or better quality than previous state- of-the-art methods.",
  "doi": "10.1109/CVPR52733.2024.01920",
  "oa_pdf": "https://arxiv.org/pdf/2310.08528",
  "s2_authors": [
   {
    "name": "Guanjun Wu",
    "id": "2257429038",
    "h_index": 9,
    "papers": 10
   },
   {
    "name": "Taoran Yi",
    "id": "2167029536",
    "h_index": 10,
    "papers": 14
   },
   {
    "name": "Jiemin Fang",
    "id": "50882585",
    "h_index": 24,
    "papers": 51
   },
   {
    "name": "Lingxi Xie",
    "id": "3041937",
    "h_index": 69,
    "papers": 220
   },
   {
    "name": "Xiaopeng Zhang",
    "id": "2257369199",
    "h_index": 11,
    "papers": 23
   },
   {
    "name": "Wei Wei",
    "id": "2257698239",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Wenyu Liu",
    "id": "2238154287",
    "h_index": 12,
    "papers": 19
   },
   {
    "name": "Qi Tian",
    "id": "2257417852",
    "h_index": 14,
    "papers": 30
   },
   {
    "name": "Xinggang Wang",
    "id": "2257846814",
    "h_index": 12,
    "papers": 22
   }
  ],
  "comment": "CVPR 2024. Project page: https://guanjunwu.github.io/4dgs/",
  "topics": [
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2310.08528v3",
  "pdf_url": "https://arxiv.org/pdf/2310.08528v3",
  "html_url": "https://arxiv.org/html/2310.08528v3",
  "code_url": "https://guanjunwu.github.io/4dgs/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.5
 },
 {
  "id": "2310.08235",
  "slug": "groot-learning-to-follow-instructions-by-watching-gameplay-videos",
  "title": "GROOT: Learning to Follow Instructions by Watching Gameplay Videos",
  "abstract": "We study the problem of building a controller that can follow open-ended instructions in open-world environments. We propose to follow reference videos as instructions, which offer expressive goal specifications while eliminating the need for expensive text-gameplay annotations. A new learning framework is derived to allow learning such instruction-following controllers from gameplay videos while producing a video instruction encoder that induces a structured goal space. We implement our agent GROOT in a simple yet effective encoder-decoder architecture based on causal transformers. We evaluate GROOT against open-world counterparts and human players on a proposed Minecraft SkillForge benchmark. The Elo ratings clearly show that GROOT is closing the human-machine gap as well as exhibiting a 70% winning rate over the best generalist agent baseline. Qualitative analysis of the induced goal space further demonstrates some interesting emergent properties, including the goal composition and complex gameplay behavior synthesis. The project page is available at https://craftjarvis-groot.github.io.",
  "published": "2023-10-12",
  "updated": "2023-11-29",
  "year": "2023",
  "authors": [
   "Shaofei Cai",
   "Bowei Zhang",
   "Zihao Wang",
   "Xiaojian Ma",
   "Anji Liu",
   "Yitao Liang"
  ],
  "author_count": 6,
  "categories": [
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.AI",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 47,
  "influential_citations": 8,
  "tldr": "This work proposes to follow reference videos as instructions, which offer expressive goal specifications while eliminating the need for expensive text-gameplay annotations, and implements the agent GROOT in a simple yet effective encoder-decoder architecture based on causal transformers.",
  "doi": "10.48550/arXiv.2310.08235",
  "oa_pdf": "https://arxiv.org/pdf/2310.08235",
  "s2_authors": [
   {
    "name": "Shaofei Cai",
    "id": "1993661033",
    "h_index": 13,
    "papers": 32
   },
   {
    "name": "Bowei Zhang",
    "id": "2257373058",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Zihao Wang",
    "id": "47196237",
    "h_index": 15,
    "papers": 26
   },
   {
    "name": "Xiaojian Ma",
    "id": "2257636629",
    "h_index": 8,
    "papers": 14
   },
   {
    "name": "Anji Liu",
    "id": "70097297",
    "h_index": 20,
    "papers": 49
   },
   {
    "name": "Yitao Liang",
    "id": "2257367774",
    "h_index": 12,
    "papers": 36
   }
  ],
  "comment": "",
  "topics": [
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2310.08235v2",
  "pdf_url": "https://arxiv.org/pdf/2310.08235v2",
  "html_url": "https://arxiv.org/html/2310.08235v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.18
 },
 {
  "id": "2310.06114",
  "slug": "learning-interactive-real-world-simulators",
  "title": "Learning Interactive Real-World Simulators",
  "abstract": "Generative models trained on internet data have revolutionized how text, image, and video content can be created. Perhaps the next milestone for generative models is to simulate realistic experience in response to actions taken by humans, robots, and other interactive agents. Applications of a real-world simulator range from controllable content creation in games and movies, to training embodied agents purely in simulation that can be directly deployed in the real world. We explore the possibility of learning a universal simulator (UniSim) of real-world interaction through generative modeling. We first make the important observation that natural datasets available for learning a real-world simulator are often rich along different dimensions (e.g., abundant objects in image data, densely sampled actions in robotics data, and diverse movements in navigation data). With careful orchestration of diverse datasets, each providing a different aspect of the overall experience, we can simulate the visual outcome of both high-level instructions such as \"open the drawer\" and low-level controls from otherwise static scenes and objects. We use the simulator to train both high-level vision-language policies and low-level reinforcement learning policies, each of which can be deployed in the real world in zero shot after training purely in simulation. We also show that other types of intelligence such as video captioning models can benefit from training with simulated experience, opening up even wider applications. Video demos can be found at https://universal-simulator.github.io.",
  "published": "2023-10-09",
  "updated": "2024-09-26",
  "year": "2023",
  "authors": [
   "Sherry Yang",
   "Yilun Du",
   "Kamyar Ghasemipour",
   "Jonathan Tompson",
   "Leslie Kaelbling",
   "Dale Schuurmans",
   "Pieter Abbeel"
  ],
  "author_count": 7,
  "categories": [
   "cs.AI"
  ],
  "primary_category": "cs.AI",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 497,
  "influential_citations": 24,
  "tldr": "This work uses the simulator to train both high-level vision-language policies and low-level reinforcement learning policies, each of which can be deployed in the real world in zero shot after training purely in simulation, and shows that other types of intelligence such as video captioning models can benefit from training with simulated experience, opening up even wider applications.",
  "doi": "10.48550/arXiv.2310.06114",
  "oa_pdf": "https://arxiv.org/pdf/2310.06114",
  "s2_authors": [
   {
    "name": "Mengjiao Yang",
    "id": "2257252407",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Yilun Du",
    "id": "15394275",
    "h_index": 48,
    "papers": 86
   },
   {
    "name": "Kamyar Ghasemipour",
    "id": "2256999030",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Jonathan Tompson",
    "id": "2704494",
    "h_index": 43,
    "papers": 71
   },
   {
    "name": "Dale Schuurmans",
    "id": "1714772",
    "h_index": 65,
    "papers": 264
   },
   {
    "name": "Pieter Abbeel",
    "id": "2257003229",
    "h_index": 11,
    "papers": 17
   }
  ],
  "comment": "https://universal-simulator.github.io",
  "topics": [
   "sim2real",
   "rl-control",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2310.06114v3",
  "pdf_url": "https://arxiv.org/pdf/2310.06114v3",
  "html_url": "https://arxiv.org/html/2310.06114v3",
  "code_url": "https://universal-simulator.github.io",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.2
 },
 {
  "id": "2310.05737",
  "slug": "language-model-beats-diffusion-tokenizer-is-key-to-visual-generation",
  "title": "Language Model Beats Diffusion -- Tokenizer is Key to Visual Generation",
  "abstract": "While Large Language Models (LLMs) are the dominant models for generative tasks in language, they do not perform as well as diffusion models on image and video generation. To effectively use LLMs for visual generation, one crucial component is the visual tokenizer that maps pixel-space inputs to discrete tokens appropriate for LLM learning. In this paper, we introduce MAGVIT-v2, a video tokenizer designed to generate concise and expressive tokens for both videos and images using a common token vocabulary. Equipped with this new tokenizer, we show that LLMs outperform diffusion models on standard image and video generation benchmarks including ImageNet and Kinetics. In addition, we demonstrate that our tokenizer surpasses the previously top-performing video tokenizer on two more tasks: (1) video compression comparable to the next-generation video codec (VCC) according to human evaluations, and (2) learning effective representations for action recognition tasks.",
  "published": "2023-10-09",
  "updated": "2024-03-29",
  "year": "2023",
  "authors": [
   "Lijun Yu",
   "Jos\u00e9 Lezama",
   "Nitesh B. Gundavarapu",
   "Luca Versari",
   "Kihyuk Sohn",
   "David Minnen",
   "Yong Cheng",
   "Vighnesh Birodkar",
   "Agrim Gupta",
   "Xiuye Gu",
   "Alexander G. Hauptmann",
   "Boqing Gong",
   "Ming-Hsuan Yang",
   "Irfan Essa",
   "David A. Ross",
   "Lu Jiang"
  ],
  "author_count": 16,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.MM"
  ],
  "primary_category": "cs.CV",
  "venue": "ICLR 2024",
  "venue_source": "arxiv-comment",
  "citations": 687,
  "influential_citations": 84,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Lijun Yu",
    "id": "8547960",
    "h_index": 14,
    "papers": 24
   },
   {
    "name": "Jos\u00e9 Lezama",
    "id": "2256999290",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "N. B. Gundavarapu",
    "id": "1387987945",
    "h_index": 8,
    "papers": 13
   },
   {
    "name": "Luca Versari",
    "id": "2256995349",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Kihyuk Sohn",
    "id": "2256996545",
    "h_index": 10,
    "papers": 20
   },
   {
    "name": "David C. Minnen",
    "id": "3144223",
    "h_index": 28,
    "papers": 49
   },
   {
    "name": "Yong Cheng",
    "id": "2198464317",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Agrim Gupta",
    "id": "2265716291",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Xiuye Gu",
    "id": "2257336985",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Alexander G. Hauptmann",
    "id": "2257000091",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Boqing Gong",
    "id": "2257000670",
    "h_index": 9,
    "papers": 16
   },
   {
    "name": "Ming-Hsuan Yang",
    "id": "2257132345",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Irfan Essa",
    "id": "145955800",
    "h_index": 22,
    "papers": 56
   },
   {
    "name": "David A. Ross",
    "id": "2257003564",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Lu Jiang",
    "id": "39978626",
    "h_index": 48,
    "papers": 78
   }
  ],
  "comment": "ICLR 2024",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2310.05737v3",
  "pdf_url": "https://arxiv.org/pdf/2310.05737v3",
  "html_url": "https://arxiv.org/html/2310.05737v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.34
 },
 {
  "id": "2310.05400",
  "slug": "efficient-vqgan-towards-high-resolution-image-generation-with-efficien",
  "title": "Efficient-VQGAN: Towards High-Resolution Image Generation with Efficient Vision Transformers",
  "abstract": "Vector-quantized image modeling has shown great potential in synthesizing high-quality images. However, generating high-resolution images remains a challenging task due to the quadratic computational overhead of the self-attention process. In this study, we seek to explore a more efficient two-stage framework for high-resolution image generation with improvements in the following three aspects. (1) Based on the observation that the first quantization stage has solid local property, we employ a local attention-based quantization model instead of the global attention mechanism used in previous methods, leading to better efficiency and reconstruction quality. (2) We emphasize the importance of multi-grained feature interaction during image generation and introduce an efficient attention mechanism that combines global attention (long-range semantic consistency within the whole image) and local attention (fined-grained details). This approach results in faster generation speed, higher generation fidelity, and improved resolution. (3) We propose a new generation pipeline incorporating autoencoding training and autoregressive generation strategy, demonstrating a better paradigm for image synthesis. Extensive experiments demonstrate the superiority of our approach in high-quality and high-resolution image reconstruction and generation.",
  "published": "2023-10-09",
  "updated": "2023-10-09",
  "year": "2023",
  "authors": [
   "Shiyue Cao",
   "Yueqin Yin",
   "Lianghua Huang",
   "Yu Liu",
   "Xin Zhao",
   "Deli Zhao",
   "Kaiqi Huang"
  ],
  "author_count": 7,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "ICCV",
  "venue_source": "semantic-scholar",
  "citations": 39,
  "influential_citations": 2,
  "tldr": "A local attention-based quantization model is employed instead of the global attention mechanism used in previous methods, leading to better efficiency and reconstruction quality and a new generation pipeline incorporating autoencoding training and autoregressive generation strategy is proposed, demonstrating a better paradigm for image synthesis.",
  "doi": "10.1109/ICCV51070.2023.00677",
  "oa_pdf": "https://arxiv.org/pdf/2310.05400",
  "s2_authors": [
   {
    "name": "Shiyue Cao",
    "id": "2256996726",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Yue Yin",
    "id": "2217961999",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Lianghua Huang",
    "id": "2109047017",
    "h_index": 14,
    "papers": 25
   },
   {
    "name": "Yu Liu",
    "id": "2192813197",
    "h_index": 11,
    "papers": 14
   },
   {
    "name": "Xin Zhao",
    "id": "2145733730",
    "h_index": 23,
    "papers": 56
   },
   {
    "name": "Deli Zhao",
    "id": "2258097324",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Kaiqi Huang",
    "id": "2257213157",
    "h_index": 10,
    "papers": 15
   }
  ],
  "comment": "This paper is accepted to ICCV2023",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2310.05400v1",
  "pdf_url": "https://arxiv.org/pdf/2310.05400v1",
  "html_url": "https://arxiv.org/html/2310.05400v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.1
 },
 {
  "id": "2310.03191",
  "slug": "sim-to-real-learning-for-humanoid-box-loco-manipulation",
  "title": "Sim-to-Real Learning for Humanoid Box Loco-Manipulation",
  "abstract": "In this work we propose a learning-based approach to box loco-manipulation for a humanoid robot. This is a particularly challenging problem due to the need for whole-body coordination in order to lift boxes of varying weight, position, and orientation while maintaining balance. To address this challenge, we present a sim-to-real reinforcement learning approach for training general box pickup and carrying skills for the bipedal robot Digit. Our reward functions are designed to produce the desired interactions with the box while also valuing balance and gait quality. We combine the learned skills into a full system for box loco-manipulation to achieve the task of moving boxes from one table to another with a variety of sizes, weights, and initial configurations. In addition to quantitative simulation results, we demonstrate successful sim-to-real transfer on the humanoid r",
  "published": "2023-10-04",
  "updated": "2023-10-04",
  "year": "2023",
  "authors": [
   "Jeremy Dao",
   "Helei Duan",
   "Alan Fern"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA",
  "venue_source": "semantic-scholar",
  "citations": 76,
  "influential_citations": 1,
  "tldr": "This work presents a sim-to-real reinforcement learning approach for training general box pickup and carrying skills for the bipedal robot Digit and demonstrates successful sim-to-real transfer on the humanoid robot Digit.",
  "doi": "10.1109/ICRA57147.2024.10610977",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jeremy Dao",
    "id": "7699850",
    "h_index": 15,
    "papers": 23
   },
   {
    "name": "Helei Duan",
    "id": "10865892",
    "h_index": 11,
    "papers": 15
   },
   {
    "name": "Alan Fern",
    "id": "2217085116",
    "h_index": 6,
    "papers": 9
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "sim2real",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2310.03191v1",
  "pdf_url": "https://arxiv.org/pdf/2310.03191v1",
  "html_url": "https://arxiv.org/html/2310.03191v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.39
 },
 {
  "id": "2310.01218",
  "slug": "making-llama-see-and-draw-with-seed-tokenizer",
  "title": "Making LLaMA SEE and Draw with SEED Tokenizer",
  "abstract": "The great success of Large Language Models (LLMs) has expanded the potential of multimodality, contributing to the gradual evolution of General Artificial Intelligence (AGI). A true AGI agent should not only possess the capability to perform predefined multi-tasks but also exhibit emergent abilities in an open-world context. However, despite the considerable advancements made by recent multimodal LLMs, they still fall short in effectively unifying comprehension and generation tasks, let alone open-world emergent abilities. We contend that the key to overcoming the present impasse lies in enabling text and images to be represented and processed interchangeably within a unified autoregressive Transformer. To this end, we introduce SEED, an elaborate image tokenizer that empowers LLMs with the ability to SEE and Draw at the same time. We identify two crucial design principles: (1) Image tokens should be independent of 2D physical patch positions and instead be produced with a 1D causal dependency, exhibiting intrinsic interdependence that aligns with the left-to-right autoregressive prediction mechanism in LLMs. (2) Image tokens should capture high-level semantics consistent with the degree of semantic abstraction in words, and be optimized for both discriminativeness and reconstruction during the tokenizer training phase. With SEED tokens, LLM is able to perform scalable multimodal autoregression under its original training recipe, i.e., next-word prediction. SEED-LLaMA is therefore produced by large-scale pretraining and instruction tuning on the interleaved textual and visual data, demonstrating impressive performance on a broad range of multimodal comprehension and generation tasks. More importantly, SEED-LLaMA has exhibited compositional emergent abilities such as multi-turn in-context multimodal generation, acting like your AI assistant.",
  "published": "2023-10-02",
  "updated": "2023-10-02",
  "year": "2023",
  "authors": [
   "Yuying Ge",
   "Sijie Zhao",
   "Ziyun Zeng",
   "Yixiao Ge",
   "Chen Li",
   "Xintao Wang",
   "Ying Shan"
  ],
  "author_count": 7,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 219,
  "influential_citations": 32,
  "tldr": "SEED is introduced, an elaborate image tokenizer that empowers LLMs with the ability to SEE and Draw at the same time, and with SEED tokens, LLM is able to perform scalable multimodal autoregression under its original training recipe, i.e., next-word prediction.",
  "doi": "10.48550/arXiv.2310.01218",
  "oa_pdf": "https://arxiv.org/pdf/2310.01218",
  "s2_authors": [
   {
    "name": "Yuying Ge",
    "id": "51123495",
    "h_index": 29,
    "papers": 59
   },
   {
    "name": "Sijie Zhao",
    "id": "2254048096",
    "h_index": 8,
    "papers": 9
   },
   {
    "name": "Ziyun Zeng",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Yixiao Ge",
    "id": "152988335",
    "h_index": 45,
    "papers": 97
   },
   {
    "name": "Chen Li",
    "id": "2256784925",
    "h_index": 9,
    "papers": 10
   },
   {
    "name": "Xintao Wang",
    "id": "2253795356",
    "h_index": 26,
    "papers": 44
   },
   {
    "name": "Ying Shan",
    "id": "2257019659",
    "h_index": 23,
    "papers": 36
   }
  ],
  "comment": "Project released at: https://github.com/AILab-CVC/SEED. arXiv admin note: substantial text overlap with arXiv:2307.08041",
  "topics": [
   "foundation-pretraining",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2310.01218v1",
  "pdf_url": "https://arxiv.org/pdf/2310.01218v1",
  "html_url": "https://arxiv.org/html/2310.01218v1",
  "code_url": "https://github.com/AILab-CVC/SEED.",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.84
 },
 {
  "id": "2309.17080",
  "slug": "gaia-1-a-generative-world-model-for-autonomous-driving",
  "title": "GAIA-1: A Generative World Model for Autonomous Driving",
  "abstract": "Autonomous driving promises transformative improvements to transportation, but building systems capable of safely navigating the unstructured complexity of real-world scenarios remains challenging. A critical problem lies in effectively predicting the various potential outcomes that may emerge in response to the vehicle's actions as the world evolves. To address this challenge, we introduce GAIA-1 ('Generative AI for Autonomy'), a generative world model that leverages video, text, and action inputs to generate realistic driving scenarios while offering fine-grained control over ego-vehicle behavior and scene features. Our approach casts world modeling as an unsupervised sequence modeling problem by mapping the inputs to discrete tokens, and predicting the next token in the sequence. Emerging properties from our model include learning high-level structures and scene dynamics, contextual awareness, generalization, and understanding of geometry. The power of GAIA-1's learned representation that captures expectations of future events, combined with its ability to generate realistic samples, provides new possibilities for innovation in the field of autonomy, enabling enhanced and accelerated training of autonomous driving technology.",
  "published": "2023-09-29",
  "updated": "2023-09-29",
  "year": "2023",
  "authors": [
   "Anthony Hu",
   "Lloyd Russell",
   "Hudson Yeo",
   "Zak Murez",
   "George Fedoseev",
   "Alex Kendall",
   "Jamie Shotton",
   "Gianluca Corrado"
  ],
  "author_count": 8,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 650,
  "influential_citations": 36,
  "tldr": "GAIA-1 ('Generative AI for Autonomy'), a generative world model that leverages video, text, and action inputs to generate realistic driving scenarios while offering fine-grained control over ego-vehicle behavior and scene features, is introduced.",
  "doi": "10.48550/arXiv.2309.17080",
  "oa_pdf": "https://arxiv.org/pdf/2309.17080",
  "s2_authors": [
   {
    "name": "Anthony Hu",
    "id": "1909776",
    "h_index": 8,
    "papers": 11
   },
   {
    "name": "Lloyd Russell",
    "id": "2249537838",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Hudson Yeo",
    "id": "2187874918",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Zak Murez",
    "id": "2375421",
    "h_index": 11,
    "papers": 16
   },
   {
    "name": "George Fedoseev",
    "id": "2249528986",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Alex Kendall",
    "id": "2249535147",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Jamie Shotton",
    "id": "2249528512",
    "h_index": 8,
    "papers": 9
   },
   {
    "name": "Gianluca Corrado",
    "id": "2249536755",
    "h_index": 4,
    "papers": 4
   }
  ],
  "comment": "Technical Report",
  "topics": [
   "world-models",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2309.17080v1",
  "pdf_url": "https://arxiv.org/pdf/2309.17080v1",
  "html_url": "https://arxiv.org/html/2309.17080v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.81
 },
 {
  "id": "2309.17024",
  "slug": "holoassist-an-egocentric-human-interaction-dataset-for-interactive-ai",
  "title": "HoloAssist: an Egocentric Human Interaction Dataset for Interactive AI Assistants in the Real World",
  "abstract": "Building an interactive AI assistant that can perceive, reason, and collaborate with humans in the real world has been a long-standing pursuit in the AI community. This work is part of a broader research effort to develop intelligent agents that can interactively guide humans through performing tasks in the physical world. As a first step in this direction, we introduce HoloAssist, a large-scale egocentric human interaction dataset, where two people collaboratively complete physical manipulation tasks. The task performer executes the task while wearing a mixed-reality headset that captures seven synchronized data streams. The task instructor watches the performer's egocentric video in real time and guides them verbally. By augmenting the data with action and conversational annotations and observing the rich behaviors of various participants, we present key insights into how human assistants correct mistakes, intervene in the task completion procedure, and ground their instructions to the environment. HoloAssist spans 166 hours of data captured by 350 unique instructor-performer pairs. Furthermore, we construct and present benchmarks on mistake detection, intervention type prediction, and hand forecasting, along with detailed analysis. We expect HoloAssist will provide an important resource for building AI assistants that can fluidly collaborate with humans in the real world. Data can be downloaded at https://holoassist.github.io/.",
  "published": "2023-09-29",
  "updated": "2023-09-29",
  "year": "2023",
  "authors": [
   "Xin Wang",
   "Taein Kwon",
   "Mahdi Rad",
   "Bowen Pan",
   "Ishani Chakraborty",
   "Sean Andrist",
   "Dan Bohus",
   "Ashley Feniello",
   "Bugra Tekin",
   "Felipe Vieira Frujeri",
   "Neel Joshi",
   "Marc Pollefeys"
  ],
  "author_count": 12,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "ICCV",
  "venue_source": "semantic-scholar",
  "citations": 192,
  "influential_citations": 24,
  "tldr": "HoloAssist is introduced, a large-scale egocentric human interaction dataset, where two people collaboratively complete physical manipulation tasks, and key insights into how human assistants correct mistakes, intervene in the task completion procedure, and ground their instructions to the environment are presented.",
  "doi": "10.1109/ICCV51070.2023.01854",
  "oa_pdf": "https://arxiv.org/pdf/2309.17024",
  "s2_authors": [
   {
    "name": "Xin Wang",
    "id": "2308010128",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Taein Kwon",
    "id": "8197167",
    "h_index": 6,
    "papers": 17
   },
   {
    "name": "Mahdi Rad",
    "id": "2243230113",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Bowen Pan",
    "id": "2249538421",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Ishani Chakraborty",
    "id": "2249585199",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Sean Andrist",
    "id": "2211183",
    "h_index": 23,
    "papers": 54
   },
   {
    "name": "D. Bohus",
    "id": "2314124",
    "h_index": 31,
    "papers": 98
   },
   {
    "name": "Ashley Feniello",
    "id": "2420223",
    "h_index": 7,
    "papers": 7
   },
   {
    "name": "Bugra Tekin",
    "id": "39307390",
    "h_index": 18,
    "papers": 35
   },
   {
    "name": "F. Frujeri",
    "id": "1844283112",
    "h_index": 12,
    "papers": 21
   },
   {
    "name": "Neel Joshi",
    "id": "2250480908",
    "h_index": 10,
    "papers": 17
   },
   {
    "name": "Marc Pollefeys",
    "id": "2243233287",
    "h_index": 6,
    "papers": 18
   }
  ],
  "comment": "ICCV 2023",
  "topics": [
   "egocentric-data",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2309.17024v1",
  "pdf_url": "https://arxiv.org/pdf/2309.17024v1",
  "html_url": "https://arxiv.org/html/2309.17024v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.79
 },
 {
  "id": "2309.16653",
  "slug": "dreamgaussian-generative-gaussian-splatting-for-efficient-3d-content-c",
  "title": "DreamGaussian: Generative Gaussian Splatting for Efficient 3D Content Creation",
  "abstract": "Recent advances in 3D content creation mostly leverage optimization-based 3D generation via score distillation sampling (SDS). Though promising results have been exhibited, these methods often suffer from slow per-sample optimization, limiting their practical usage. In this paper, we propose DreamGaussian, a novel 3D content generation framework that achieves both efficiency and quality simultaneously. Our key insight is to design a generative 3D Gaussian Splatting model with companioned mesh extraction and texture refinement in UV space. In contrast to the occupancy pruning used in Neural Radiance Fields, we demonstrate that the progressive densification of 3D Gaussians converges significantly faster for 3D generative tasks. To further enhance the texture quality and facilitate downstream applications, we introduce an efficient algorithm to convert 3D Gaussians into textured meshes and apply a fine-tuning stage to refine the details. Extensive experiments demonstrate the superior efficiency and competitive generation quality of our proposed approach. Notably, DreamGaussian produces high-quality textured meshes in just 2 minutes from a single-view image, achieving approximately 10 times acceleration compared to existing methods.",
  "published": "2023-09-28",
  "updated": "2024-03-29",
  "year": "2023",
  "authors": [
   "Jiaxiang Tang",
   "Jiawei Ren",
   "Hang Zhou",
   "Ziwei Liu",
   "Gang Zeng"
  ],
  "author_count": 5,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 1055,
  "influential_citations": 89,
  "tldr": "This paper proposes DreamGaussian, a novel 3D content generation framework that achieves both efficiency and quality simultaneously, and introduces an efficient algorithm to convert 3D Gaussians into textured meshes and apply a fine-tuning stage to refine the details.",
  "doi": "10.48550/arXiv.2309.16653",
  "oa_pdf": "https://arxiv.org/pdf/2309.16653",
  "s2_authors": [
   {
    "name": "Jiaxiang Tang",
    "id": "1397711601",
    "h_index": 20,
    "papers": 35
   },
   {
    "name": "Jiawei Ren",
    "id": "1820909323",
    "h_index": 20,
    "papers": 30
   },
   {
    "name": "Hang Zhou",
    "id": "2315725007",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Ziwei Liu",
    "id": "2249080787",
    "h_index": 9,
    "papers": 9
   },
   {
    "name": "Gang Zeng",
    "id": "2247995148",
    "h_index": 7,
    "papers": 10
   }
  ],
  "comment": "Camera-ready version. Project page: https://dreamgaussian.github.io/",
  "topics": [
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2309.16653v2",
  "pdf_url": "https://arxiv.org/pdf/2309.16653v2",
  "html_url": "https://arxiv.org/html/2309.16653v2",
  "code_url": "https://dreamgaussian.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.5
 },
 {
  "id": "2309.16588",
  "slug": "vision-transformers-need-registers",
  "title": "Vision Transformers Need Registers",
  "abstract": "Transformers have recently emerged as a powerful tool for learning visual representations. In this paper, we identify and characterize artifacts in feature maps of both supervised and self-supervised ViT networks. The artifacts correspond to high-norm tokens appearing during inference primarily in low-informative background areas of images, that are repurposed for internal computations. We propose a simple yet effective solution based on providing additional tokens to the input sequence of the Vision Transformer to fill that role. We show that this solution fixes that problem entirely for both supervised and self-supervised models, sets a new state of the art for self-supervised visual models on dense visual prediction tasks, enables object discovery methods with larger models, and most importantly leads to smoother feature maps and attention maps for downstream visual processing.",
  "published": "2023-09-28",
  "updated": "2024-04-12",
  "year": "2023",
  "authors": [
   "Timoth\u00e9e Darcet",
   "Maxime Oquab",
   "Julien Mairal",
   "Piotr Bojanowski"
  ],
  "author_count": 4,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 972,
  "influential_citations": 123,
  "tldr": "This paper identifies and characterize artifacts in feature maps of both supervised and self-supervised ViT networks, and proposes a simple yet effective solution based on providing additional tokens to the input sequence of the Vision Transformer to fill that role.",
  "doi": "10.48550/arXiv.2309.16588",
  "oa_pdf": "https://arxiv.org/pdf/2309.16588",
  "s2_authors": [
   {
    "name": "Timoth\u00e9e Darcet",
    "id": "2214523349",
    "h_index": 7,
    "papers": 7
   },
   {
    "name": "Maxime Oquab",
    "id": "2248163611",
    "h_index": 8,
    "papers": 13
   },
   {
    "name": "J. Mairal",
    "id": "2599292",
    "h_index": 58,
    "papers": 162
   },
   {
    "name": "Piotr Bojanowski",
    "id": "2329288",
    "h_index": 38,
    "papers": 77
   }
  ],
  "comment": "",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2309.16588v2",
  "pdf_url": "https://arxiv.org/pdf/2309.16588v2",
  "html_url": "https://arxiv.org/html/2309.16588v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.49
 },
 {
  "id": "2309.16237",
  "slug": "object-motion-guided-human-motion-synthesis",
  "title": "Object Motion Guided Human Motion Synthesis",
  "abstract": "Modeling human behaviors in contextual environments has a wide range of applications in character animation, embodied AI, VR/AR, and robotics. In real-world scenarios, humans frequently interact with the environment and manipulate various objects to complete daily tasks. In this work, we study the problem of full-body human motion synthesis for the manipulation of large-sized objects. We propose Object MOtion guided human MOtion synthesis (OMOMO), a conditional diffusion framework that can generate full-body manipulation behaviors from only the object motion. Since naively applying diffusion models fails to precisely enforce contact constraints between the hands and the object, OMOMO learns two separate denoising processes to first predict hand positions from object motion and subsequently synthesize full-body poses based on the predicted hand positions. By employing the hand positions as an intermediate representation between the two denoising processes, we can explicitly enforce contact constraints, resulting in more physically plausible manipulation motions. With the learned model, we develop a novel system that captures full-body human manipulation motions by simply attaching a smartphone to the object being manipulated. Through extensive experiments, we demonstrate the effectiveness of our proposed pipeline and its ability to generalize to unseen objects. Additionally, as high-quality human-object interaction datasets are scarce, we collect a large-scale dataset consisting of 3D object geometry, object motion, and human motion. Our dataset contains human-object interaction motion for 15 objects, with a total duration of approximately 10 hours.",
  "published": "2023-09-28",
  "updated": "2023-09-28",
  "year": "2023",
  "authors": [
   "Jiaman Li",
   "Jiajun Wu",
   "C. Karen Liu"
  ],
  "author_count": 3,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "SIGGRAPH 2023",
  "venue_source": "arxiv-comment",
  "citations": 251,
  "influential_citations": 46,
  "tldr": "This work proposes Object MOtion guided human MOtion synthesis (OMOMO), a conditional diffusion framework that can generate full- body manipulation behaviors from only the object motion, and develops a novel system that captures full-body human manipulation motions by simply attaching a smartphone to the object being manipulated.",
  "doi": "10.1145/3618333",
  "oa_pdf": "https://dl.acm.org/doi/pdf/10.1145/3618333",
  "s2_authors": [
   {
    "name": "Jiaman Li",
    "id": "22133106",
    "h_index": 16,
    "papers": 22
   },
   {
    "name": "Jiajun Wu",
    "id": "3045089",
    "h_index": 80,
    "papers": 228
   },
   {
    "name": "C. K. Liu",
    "id": "2247934447",
    "h_index": 9,
    "papers": 11
   }
  ],
  "comment": "SIGGRAPH Asia 2023",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2309.16237v1",
  "pdf_url": "https://arxiv.org/pdf/2309.16237v1",
  "html_url": "https://arxiv.org/html/2309.16237v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.9
 },
 {
  "id": "2309.15505",
  "slug": "finite-scalar-quantization-vq-vae-made-simple",
  "title": "Finite Scalar Quantization: VQ-VAE Made Simple",
  "abstract": "We propose to replace vector quantization (VQ) in the latent representation of VQ-VAEs with a simple scheme termed finite scalar quantization (FSQ), where we project the VAE representation down to a few dimensions (typically less than 10). Each dimension is quantized to a small set of fixed values, leading to an (implicit) codebook given by the product of these sets. By appropriately choosing the number of dimensions and values each dimension can take, we obtain the same codebook size as in VQ. On top of such discrete representations, we can train the same models that have been trained on VQ-VAE representations. For example, autoregressive and masked transformer models for image generation, multimodal generation, and dense prediction computer vision tasks. Concretely, we employ FSQ with MaskGIT for image generation, and with UViM for depth estimation, colorization, and panoptic segmentation. Despite the much simpler design of FSQ, we obtain competitive performance in all these tasks. We emphasize that FSQ does not suffer from codebook collapse and does not need the complex machinery employed in VQ (commitment losses, codebook reseeding, code splitting, entropy penalties, etc.) to learn expressive discrete representations.",
  "published": "2023-09-27",
  "updated": "2023-10-12",
  "year": "2023",
  "authors": [
   "Fabian Mentzer",
   "David Minnen",
   "Eirikur Agustsson",
   "Michael Tschannen"
  ],
  "author_count": 4,
  "categories": [
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.CV",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 562,
  "influential_citations": 75,
  "tldr": "Finite scalar quantization (FSQ) is proposed, where each dimension is quantized to a small set of fixed values, leading to an (implicit) codebook given by the product of these sets.",
  "doi": "10.48550/arXiv.2309.15505",
  "oa_pdf": "https://arxiv.org/pdf/2309.15505",
  "s2_authors": [
   {
    "name": "Fabian Mentzer",
    "id": "3468078",
    "h_index": 21,
    "papers": 27
   },
   {
    "name": "David C. Minnen",
    "id": "3144223",
    "h_index": 28,
    "papers": 49
   },
   {
    "name": "E. Agustsson",
    "id": "2794259",
    "h_index": 32,
    "papers": 49
   },
   {
    "name": "Michael Tschannen",
    "id": "143902495",
    "h_index": 37,
    "papers": 72
   }
  ],
  "comment": "Code: https://github.com/google-research/google-research/tree/master/fsq",
  "topics": [
   "spatial-3d",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2309.15505v2",
  "pdf_url": "https://arxiv.org/pdf/2309.15505v2",
  "html_url": "https://arxiv.org/html/2309.15505v2",
  "code_url": "https://github.com/google-research/google-research",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.25
 },
 {
  "id": "2309.14975",
  "slug": "airexo-low-cost-exoskeletons-for-learning-whole-arm-manipulation-in-th",
  "title": "AirExo: Low-Cost Exoskeletons for Learning Whole-Arm Manipulation in the Wild",
  "abstract": "While humans can use parts of their arms other than the hands for manipulations like gathering and supporting, whether robots can effectively learn and perform the same type of operations remains relatively unexplored. As these manipulations require joint-level control to regulate the complete poses of the robots, we develop AirExo, a low-cost, adaptable, and portable dual-arm exoskeleton, for teleoperation and demonstration collection. As collecting teleoperated data is expensive and time-consuming, we further leverage AirExo to collect cheap in-the-wild demonstrations at scale. Under our in-the-wild learning framework, we show that with only 3 minutes of the teleoperated demonstrations, augmented by diverse and extensive in-the-wild data collected by AirExo, robots can learn a policy that is comparable to or even better than one learned from teleoperated demonstrations lasting over 20 minutes. Experiments demonstrate that our approach enables the model to learn a more general and robust policy across the various stages of the task, enhancing the success rates in task completion even with the presence of disturbances. Project website: https://airexo.github.io/",
  "published": "2023-09-26",
  "updated": "2024-05-09",
  "year": "2023",
  "authors": [
   "Hongjie Fang",
   "Hao-Shu Fang",
   "Yiming Wang",
   "Jieji Ren",
   "Jingjing Chen",
   "Ruo Zhang",
   "Weiming Wang",
   "Cewu Lu"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA",
  "venue_source": "semantic-scholar",
  "citations": 93,
  "influential_citations": 1,
  "tldr": "Under this in-the-wild learning framework, robots can learn a policy that is comparable to or even better than one learned from teleoperated demonstrations lasting over 20 minutes, and experiments demonstrate that this approach enables the model to learn a more general and robust policy across the various stages of the task, enhancing the success rates in task completion even with the presence of disturbances.",
  "doi": "10.1109/ICRA57147.2024.10610799",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hongjie Fang",
    "id": "2152115958",
    "h_index": 16,
    "papers": 31
   },
   {
    "name": "Haoshu Fang",
    "id": "122851212",
    "h_index": 34,
    "papers": 61
   },
   {
    "name": "Yiming Wang",
    "id": "21595671",
    "h_index": 19,
    "papers": 29
   },
   {
    "name": "Jieji Ren",
    "id": "120532949",
    "h_index": 12,
    "papers": 34
   },
   {
    "name": "Jing Chen",
    "id": "47740650",
    "h_index": 131,
    "papers": 3633
   },
   {
    "name": "Ruo Zhang",
    "id": "2228276343",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Weiming Wang",
    "id": "39899748",
    "h_index": 15,
    "papers": 36
   },
   {
    "name": "Cewu Lu",
    "id": "2281998765",
    "h_index": 24,
    "papers": 45
   }
  ],
  "comment": "Project page: https://airexo.github.io/",
  "topics": [
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2309.14975v2",
  "pdf_url": "https://arxiv.org/pdf/2309.14975v2",
  "html_url": "https://arxiv.org/html/2309.14975v2",
  "code_url": "https://airexo.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.47
 },
 {
  "id": "2309.14860",
  "slug": "a-wearable-robotic-hand-for-hand-over-hand-imitation-learning",
  "title": "A Wearable Robotic Hand for Hand-over-Hand Imitation Learning",
  "abstract": "Dexterous manipulation through imitation learning has gained significant attention in robotics research. The collection of high-quality expert data holds paramount importance when using imitation learning. The existing approaches for acquiring expert data commonly involve utilizing a data glove to capture hand motion information. However, this method suffers from limitations as the collected information cannot be directly mapped to the robotic hand due to discrepancies in their degrees of freedom or structures. Furthermore,it fails to accurately capture force feedback information between the hand and objects during the demonstration process. To overcome these challenges, this paper presents a novel solution in the form of a wearable dexterous hand, namely Hand-over-hand Imitation learning wearable RObotic Hand (HIRO Hand),which integrates expert data collection and enables the implementation of dexterous operations. This HIRO Hand empowers the operator to utilize their own tactile feedback to determine appropriate force, position, and actions, resulting in more accurate imitation of the expert's actions. We develop both non-learning and visual behavior cloning based controllers allowing HIRO Hand successfully achieves grasping and in-hand manipulation ability.",
  "published": "2023-09-26",
  "updated": "2023-09-26",
  "year": "2023",
  "authors": [
   "Dehao Wei",
   "Huazhe Xu"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA",
  "venue_source": "semantic-scholar",
  "citations": 22,
  "influential_citations": 0,
  "tldr": "This paper presents a novel solution in the form of a wearable dexterous hand, namely Handover-hand Imitation learning wearable RObotic Hand (HIRO Hand), which integrates expert data collection and enables the implementation of dexterous operations.",
  "doi": "10.1109/ICRA57147.2024.10610516",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Deh-Chang Wei",
    "id": "98200020",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Huazhe Xu",
    "id": "2283877819",
    "h_index": 2,
    "papers": 6
   }
  ],
  "comment": "7 pages",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "imitation-diffusion",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2309.14860v1",
  "pdf_url": "https://arxiv.org/pdf/2309.14860v1",
  "html_url": "https://arxiv.org/html/2309.14860v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.86
 },
 {
  "id": "2309.14341",
  "slug": "extreme-parkour-with-legged-robots",
  "title": "Extreme Parkour with Legged Robots",
  "abstract": "Humans can perform parkour by traversing obstacles in a highly dynamic fashion requiring precise eye-muscle coordination and movement. Getting robots to do the same task requires overcoming similar challenges. Classically, this is done by independently engineering perception, actuation, and control systems to very low tolerances. This restricts them to tightly controlled settings such as a predetermined obstacle course in labs. In contrast, humans are able to learn parkour through practice without significantly changing their underlying biology. In this paper, we take a similar approach to developing robot parkour on a small low-cost robot with imprecise actuation and a single front-facing depth camera for perception which is low-frequency, jittery, and prone to artifacts. We show how a single neural net policy operating directly from a camera image, trained in simulation with large-scale RL, can overcome imprecise sensing and actuation to output highly precise control behavior end-to-end. We show our robot can perform a high jump on obstacles 2x its height, long jump across gaps 2x its length, do a handstand and run across tilted ramps, and generalize to novel obstacle courses with different physical properties. Parkour videos at https://extreme-parkour.github.io/",
  "published": "2023-09-25",
  "updated": "2023-09-25",
  "year": "2023",
  "authors": [
   "Xuxin Cheng",
   "Kexin Shi",
   "Ananye Agarwal",
   "Deepak Pathak"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA",
  "venue_source": "semantic-scholar",
  "citations": 372,
  "influential_citations": 33,
  "tldr": "This paper shows how a single neural net policy operating directly from a camera image, trained in simulation with large-scale RL, can overcome imprecise sensing and actuation to output highly precise control behavior end-to-end.",
  "doi": "10.1109/ICRA57147.2024.10610200",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xuxin Cheng",
    "id": "90080090",
    "h_index": 13,
    "papers": 13
   },
   {
    "name": "Kexin Shi",
    "id": "2151814686",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Ananye Agarwal",
    "id": "2107063491",
    "h_index": 9,
    "papers": 14
   },
   {
    "name": "Deepak Pathak",
    "id": "2004879394",
    "h_index": 24,
    "papers": 32
   }
  ],
  "comment": "Website and videos at https://extreme-parkour.github.io/",
  "topics": [
   "humanoids"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2309.14341v1",
  "pdf_url": "https://arxiv.org/pdf/2309.14341v1",
  "html_url": "https://arxiv.org/html/2309.14341v1",
  "code_url": "https://extreme-parkour.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.07
 },
 {
  "id": "2309.13037",
  "slug": "gello-a-general-low-cost-and-intuitive-teleoperation-framework-for-rob",
  "title": "GELLO: A General, Low-Cost, and Intuitive Teleoperation Framework for Robot Manipulators",
  "abstract": "Humans can teleoperate robots to accomplish complex manipulation tasks. Imitation learning has emerged as a powerful framework that leverages human teleoperated demonstrations to teach robots new skills. However, the performance of the learned policies is bottlenecked by the quality, scale, and variety of the demonstration data. In this paper, we aim to lower the barrier to collecting large and high-quality human demonstration data by proposing a GEneraL framework for building LOw-cost and intuitive teleoperation systems for robotic manipulation (GELLO). Given a target robot arm, we build a GELLO controller device that has the same kinematic structure as the target arm, leveraging 3D-printed parts and economical off-the-shelf motors. GELLO is easy to build and intuitive to use. Through an extensive user study, we show that GELLO enables more reliable and efficient demonstration collection compared to other cost efficient teleoperation devices commonly used in the imitation learning literature such as virtual reality controllers and 3D spacemouses. We further demonstrate the capabilities of GELLO for performing complex bi-manual and contact-rich manipulation tasks. To make GELLO accessible to everyone, we have designed and built GELLO systems for 3 commonly used robotic arms: Franka, UR5, and xArm. All software and hardware are open-sourced and can be found on our website: https://wuphilipp.github.io/gello/.",
  "published": "2023-09-22",
  "updated": "2024-07-18",
  "year": "2023",
  "authors": [
   "Philipp Wu",
   "Yide Shentu",
   "Zhongke Yi",
   "Xingyu Lin",
   "Pieter Abbeel"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS",
  "venue_source": "semantic-scholar",
  "citations": 364,
  "influential_citations": 18,
  "tldr": "A GEneraL framework for building LOw-cost and intuitive teleoperation systems for robotic manipulation (GELLO) and it is shown that GELLO enables more reliable and efficient demonstration collection compared to other cost efficient teleoperation devices commonly used in the imitation learning literature such as virtual reality controllers and 3D spacemouses.",
  "doi": "10.1109/IROS58592.2024.10801581",
  "oa_pdf": "http://arxiv.org/pdf/2309.13037",
  "s2_authors": [
   {
    "name": "Philipp Wu",
    "id": "2108864104",
    "h_index": 12,
    "papers": 18
   },
   {
    "name": "Yide Shentu",
    "id": "41019645",
    "h_index": 7,
    "papers": 9
   },
   {
    "name": "Zhongke Yi",
    "id": "2405522638",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Xingyu Lin",
    "id": "2115252655",
    "h_index": 15,
    "papers": 31
   },
   {
    "name": "P. Abbeel",
    "id": "1689992",
    "h_index": 165,
    "papers": 596
   }
  ],
  "comment": "",
  "topics": [
   "egocentric-data",
   "tactile",
   "imitation-diffusion",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2309.13037v2",
  "pdf_url": "https://arxiv.org/pdf/2309.13037v2",
  "html_url": "https://arxiv.org/html/2309.13037v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.06
 },
 {
  "id": "2309.12300",
  "slug": "see-to-touch-learning-tactile-dexterity-through-visual-incentives",
  "title": "See to Touch: Learning Tactile Dexterity through Visual Incentives",
  "abstract": "Equipping multi-fingered robots with tactile sensing is crucial for achieving the precise, contact-rich, and dexterous manipulation that humans excel at. However, relying solely on tactile sensing fails to provide adequate cues for reasoning about objects' spatial configurations, limiting the ability to correct errors and adapt to changing situations. In this paper, we present Tactile Adaptation from Visual Incentives (TAVI), a new framework that enhances tactile-based dexterity by optimizing dexterous policies using vision-based rewards. First, we use a contrastive-based objective to learn visual representations. Next, we construct a reward function using these visual representations through optimal-transport based matching on one human demonstration. Finally, we use online reinforcement learning on our robot to optimize tactile-based policies that maximize the visual reward. On six challenging tasks, such as peg pick-and-place, unstacking bowls, and flipping slender objects, TAVI achieves a success rate of 73% using our four-fingered Allegro robot hand. The increase in performance is 108% higher than policies using tactile and vision-based rewards and 135% higher than policies without tactile observational input. Robot videos are best viewed on our project website: https://see-to-touch.github.io/.",
  "published": "2023-09-21",
  "updated": "2023-09-21",
  "year": "2023",
  "authors": [
   "Irmak Guzey",
   "Yinlong Dai",
   "Ben Evans",
   "Soumith Chintala",
   "Lerrel Pinto"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA",
  "venue_source": "semantic-scholar",
  "citations": 60,
  "influential_citations": 2,
  "tldr": "TAVI is presented, a new framework that enhances tactile-based dexterity by optimizing dexterous policies using vision-based rewards using contrastive-based objective to learn visual representations and uses online reinforcement learning on the authors' robot to optimize tactile-based policies that maximize the visual reward.",
  "doi": "10.1109/ICRA57147.2024.10611407",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Irmak Guzey",
    "id": "2143167646",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Yinlong Dai",
    "id": "2243500130",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Ben Evans",
    "id": "2153473632",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Soumith Chintala",
    "id": "2127604",
    "h_index": 32,
    "papers": 49
   },
   {
    "name": "Lerrel Pinto",
    "id": "34026610",
    "h_index": 41,
    "papers": 70
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "egocentric-data",
   "tactile",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2309.12300v1",
  "pdf_url": "https://arxiv.org/pdf/2309.12300v1",
  "html_url": "https://arxiv.org/html/2309.12300v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.29
 },
 {
  "id": "2309.11419",
  "slug": "kosmos-2-5-a-multimodal-literate-model",
  "title": "KOSMOS-2.5: A Multimodal Literate Model",
  "abstract": "The automatic reading of text-intensive images represents a significant advancement toward achieving Artificial General Intelligence (AGI). In this paper we present KOSMOS-2.5, a multimodal literate model for machine reading of text-intensive images. Pre-trained on a large-scale corpus of text-intensive images, KOSMOS-2.5 excels in two distinct yet complementary transcription tasks: (1) generating spatially-aware text blocks, where each block of text is assigned spatial coordinates within the image, and (2) producing structured text output that captures both style and structure in markdown format. This unified multimodal literate capability is achieved through a shared decoder-only autoregressive Transformer architecture and task-specific prompts. Building on this foundation, we fine-tune KOSMOS-2.5 for document understanding tasks, resulting in a document understanding generalist named KOSMOS-2.5-CHAT. Additionally, a large corpus of 357.4 million document pages spanning diverse domains was curated for pre-training. We evaluate KOSMOS-2.5 on two newly proposed benchmarks, OCREval and MarkdownEval, for document-level text recognition and image-to-markdown generation, demonstrating impressive literate capabilities comparable to GPT-4o. KOSMOS-2.5-CHAT achieves performance comparable to other state-of-the-art generalists that are five times larger (1.3B vs. 7B) across nine text-rich visual question answering benchmarks. Models and code have been available at \\url{https://aka.ms/kosmos25}.",
  "published": "2023-09-20",
  "updated": "2024-08-21",
  "year": "2023",
  "authors": [
   "Tengchao Lv",
   "Yupan Huang",
   "Jingye Chen",
   "Yuzhong Zhao",
   "Yilin Jia",
   "Lei Cui",
   "Shuming Ma",
   "Yaoyao Chang",
   "Shaohan Huang",
   "Wenhui Wang",
   "Li Dong",
   "Weiyao Luo",
   "Shaoxiang Wu",
   "Guoxin Wang",
   "Cha Zhang",
   "Furu Wei"
  ],
  "author_count": 16,
  "categories": [
   "cs.CL",
   "cs.CV"
  ],
  "primary_category": "cs.CL",
  "venue": "",
  "venue_source": "",
  "citations": 118,
  "influential_citations": 8,
  "tldr": "This paper presents KOSMOS-2.5, a multimodal literate model for machine reading of text-intensive images, and fine-tunes KOSMOS-2.5 for document understanding tasks, resulting in a document understanding generalist named KOSMOS-2.5-CHAT.",
  "doi": "10.48550/arXiv.2309.11419",
  "oa_pdf": "https://arxiv.org/pdf/2309.11419",
  "s2_authors": [
   {
    "name": "Tengchao Lv",
    "id": "1379581011",
    "h_index": 16,
    "papers": 26
   },
   {
    "name": "Yupan Huang",
    "id": "102665943",
    "h_index": 10,
    "papers": 17
   },
   {
    "name": "Jingye Chen",
    "id": "2244136105",
    "h_index": 15,
    "papers": 27
   },
   {
    "name": "Lei Cui",
    "id": "2114843952",
    "h_index": 16,
    "papers": 37
   },
   {
    "name": "Shuming Ma",
    "id": "2118866998",
    "h_index": 29,
    "papers": 77
   },
   {
    "name": "Ya-Chi Chang",
    "id": "2157845614",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Shaohan Huang",
    "id": "3110003",
    "h_index": 45,
    "papers": 115
   },
   {
    "name": "Wenhui Wang",
    "id": "51456429",
    "h_index": 24,
    "papers": 40
   },
   {
    "name": "Li Dong",
    "id": "145307652",
    "h_index": 63,
    "papers": 118
   },
   {
    "name": "Weiyao Luo",
    "id": "2242906251",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Shaoxiang Wu",
    "id": "2243250139",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Guoxin Wang",
    "id": "2107947686",
    "h_index": 12,
    "papers": 17
   },
   {
    "name": "Cha Zhang",
    "id": "2256775919",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Furu Wei",
    "id": "49807919",
    "h_index": 105,
    "papers": 326
   }
  ],
  "comment": "",
  "topics": [
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2309.11419v2",
  "pdf_url": "https://arxiv.org/pdf/2309.11419v2",
  "html_url": "https://arxiv.org/html/2309.11419v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.08
 },
 {
  "id": "2309.08587",
  "slug": "compositional-foundation-models-for-hierarchical-planning",
  "title": "Compositional Foundation Models for Hierarchical Planning",
  "abstract": "To make effective decisions in novel environments with long-horizon goals, it is crucial to engage in hierarchical reasoning across spatial and temporal scales. This entails planning abstract subgoal sequences, visually reasoning about the underlying plans, and executing actions in accordance with the devised plan through visual-motor control. We propose Compositional Foundation Models for Hierarchical Planning (HiP), a foundation model which leverages multiple expert foundation model trained on language, vision and action data individually jointly together to solve long-horizon tasks. We use a large language model to construct symbolic plans that are grounded in the environment through a large video diffusion model. Generated video plans are then grounded to visual-motor control, through an inverse dynamics model that infers actions from generated videos. To enable effective reasoning within this hierarchy, we enforce consistency between the models via iterative refinement. We illustrate the efficacy and adaptability of our approach in three different long-horizon table-top manipulation tasks.",
  "published": "2023-09-15",
  "updated": "2023-09-21",
  "year": "2023",
  "authors": [
   "Anurag Ajay",
   "Seungwook Han",
   "Yilun Du",
   "Shuang Li",
   "Abhi Gupta",
   "Tommi Jaakkola",
   "Josh Tenenbaum",
   "Leslie Kaelbling",
   "Akash Srivastava",
   "Pulkit Agrawal"
  ],
  "author_count": 10,
  "categories": [
   "cs.LG",
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.LG",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 131,
  "influential_citations": 19,
  "tldr": "A foundation model which leverages multiple expert foundation model trained on language, vision and action data individually jointly together to solve long-horizon tasks and enforce consistency between the models via iterative refinement is proposed.",
  "doi": "10.48550/arXiv.2309.08587",
  "oa_pdf": "https://arxiv.org/pdf/2309.08587",
  "s2_authors": [
   {
    "name": "A. Ajay",
    "id": "150004828",
    "h_index": 13,
    "papers": 22
   },
   {
    "name": "Seung-Jun Han",
    "id": "2197109",
    "h_index": 10,
    "papers": 25
   },
   {
    "name": "Yilun Du",
    "id": "15394275",
    "h_index": 48,
    "papers": 86
   },
   {
    "name": "Shaun Li",
    "id": "2389424910",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Abhishek Gupta",
    "id": "144150274",
    "h_index": 33,
    "papers": 287
   },
   {
    "name": "T. Jaakkola",
    "id": "35132120",
    "h_index": 112,
    "papers": 427
   },
   {
    "name": "Josh Tenenbaum",
    "id": "2243002911",
    "h_index": 10,
    "papers": 16
   },
   {
    "name": "L. Kaelbling",
    "id": "1709512",
    "h_index": 77,
    "papers": 429
   },
   {
    "name": "Akash Srivastava",
    "id": "2243025154",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Pulkit Agrawal",
    "id": "33932184",
    "h_index": 41,
    "papers": 114
   }
  ],
  "comment": "Website: https://hierarchical-planning-foundation-model.github.io/",
  "topics": [
   "foundation-pretraining",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2309.08587v2",
  "pdf_url": "https://arxiv.org/pdf/2309.08587v2",
  "html_url": "https://arxiv.org/html/2309.08587v2",
  "code_url": "https://hierarchical-planning-foundation-model.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.62
 },
 {
  "id": "2309.06440",
  "slug": "leap-hand-low-cost-efficient-and-anthropomorphic-hand-for-robot-learni",
  "title": "LEAP Hand: Low-Cost, Efficient, and Anthropomorphic Hand for Robot Learning",
  "abstract": "Dexterous manipulation has been a long-standing challenge in robotics. While machine learning techniques have shown some promise, results have largely been currently limited to simulation. This can be mostly attributed to the lack of suitable hardware. In this paper, we present LEAP Hand, a low-cost dexterous and anthropomorphic hand for machine learning research. In contrast to previous hands, LEAP Hand has a novel kinematic structure that allows maximal dexterity regardless of finger pose. LEAP Hand is low-cost and can be assembled in 4 hours at a cost of 2000 USD from readily available parts. It is capable of consistently exerting large torques over long durations of time. We show that LEAP Hand can be used to perform several manipulation tasks in the real world -- from visual teleoperation to learning from passive video data and sim2real. LEAP Hand significantly outperforms its closest competitor Allegro Hand in all our experiments while being 1/8th of the cost. We release detailed assembly instructions, the Sim2Real pipeline and a development platform with useful APIs on our website at https://leap-hand.github.io/",
  "published": "2023-09-12",
  "updated": "2023-09-12",
  "year": "2023",
  "authors": [
   "Kenneth Shaw",
   "Ananye Agarwal",
   "Deepak Pathak"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "RSS",
  "venue_source": "semantic-scholar",
  "citations": 237,
  "influential_citations": 19,
  "tldr": "LEAP Hand is a low-cost dexterous and anthropomorphic hand for machine learning research that has a novel kinematic structure that allows maximal dexterity regardless of finger pose and significantly outperforms its closest competitor Allegro Hand while being 1/8th of the cost.",
  "doi": "10.15607/RSS.2023.XIX.089",
  "oa_pdf": "https://doi.org/10.15607/rss.2023.xix.089",
  "s2_authors": [
   {
    "name": "Kenneth Shaw",
    "id": "2072761493",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Ananye Agarwal",
    "id": "2107063491",
    "h_index": 9,
    "papers": 14
   },
   {
    "name": "Deepak Pathak",
    "id": "2004879394",
    "h_index": 24,
    "papers": 32
   }
  ],
  "comment": "Website at https://leap-hand.github.io/",
  "topics": [
   "dexterous-manipulation",
   "sim2real",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2309.06440v1",
  "pdf_url": "https://arxiv.org/pdf/2309.06440v1",
  "html_url": "https://arxiv.org/html/2309.06440v1",
  "code_url": "https://leap-hand.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.88
 },
 {
  "id": "2309.05655",
  "slug": "dynamic-handover-throw-and-catch-with-bimanual-hands",
  "title": "Dynamic Handover: Throw and Catch with Bimanual Hands",
  "abstract": "Humans throw and catch objects all the time. However, such a seemingly common skill introduces a lot of challenges for robots to achieve: The robots need to operate such dynamic actions at high-speed, collaborate precisely, and interact with diverse objects. In this paper, we design a system with two multi-finger hands attached to robot arms to solve this problem. We train our system using Multi-Agent Reinforcement Learning in simulation and perform Sim2Real transfer to deploy on the real robots. To overcome the Sim2Real gap, we provide multiple novel algorithm designs including learning a trajectory prediction model for the object. Such a model can help the robot catcher has a real-time estimation of where the object will be heading, and then react accordingly. We conduct our experiments with multiple objects in the real-world system, and show significant improvements over multiple baselines. Our project page is available at \\url{https://binghao-huang.github.io/dynamic_handover/}.",
  "published": "2023-09-11",
  "updated": "2023-09-11",
  "year": "2023",
  "authors": [
   "Binghao Huang",
   "Yuanpei Chen",
   "Tianyu Wang",
   "Yuzhe Qin",
   "Yaodong Yang",
   "Nikolay Atanasov",
   "Xiaolong Wang"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 71,
  "influential_citations": 4,
  "tldr": "This paper designs a system with two multi-finger hands attached to robot arms to solve the Sim2Real gap, and provides multiple novel algorithm designs including learning a trajectory prediction model for the object.",
  "doi": "10.48550/arXiv.2309.05655",
  "oa_pdf": "https://arxiv.org/pdf/2309.05655",
  "s2_authors": [
   {
    "name": "Binghao Huang",
    "id": "2175672218",
    "h_index": 8,
    "papers": 8
   },
   {
    "name": "Yuanpei Chen",
    "id": "2261728034",
    "h_index": 13,
    "papers": 33
   },
   {
    "name": "Tianyu Wang",
    "id": "2286252382",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Yuzhe Qin",
    "id": "12701031",
    "h_index": 24,
    "papers": 34
   },
   {
    "name": "Yaodong Yang",
    "id": "2239164053",
    "h_index": 8,
    "papers": 21
   },
   {
    "name": "Nikolay Atanasov",
    "id": "2285801279",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Xiaolong Wang",
    "id": "2239141122",
    "h_index": 12,
    "papers": 13
   }
  ],
  "comment": "Accepted at CoRL 2023. https://binghao-huang.github.io/dynamic_handover/",
  "topics": [
   "dexterous-manipulation",
   "sim2real",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2309.05655v1",
  "pdf_url": "https://arxiv.org/pdf/2309.05655v1",
  "html_url": "https://arxiv.org/html/2309.05655v1",
  "code_url": "https://binghao-huang.github.io/dynamic_handover/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.36
 },
 {
  "id": "2309.05310",
  "slug": "imitationnet-unsupervised-human-to-robot-motion-retargeting-via-shared",
  "title": "ImitationNet: Unsupervised Human-to-Robot Motion Retargeting via Shared Latent Space",
  "abstract": "This paper introduces a novel deep-learning approach for human-to-robot motion retargeting, enabling robots to mimic human poses accurately. Contrary to prior deep-learning-based works, our method does not require paired human-to-robot data, which facilitates its translation to new robots. First, we construct a shared latent space between humans and robots via adaptive contrastive learning that takes advantage of a proposed cross-domain similarity metric between the human and robot poses. Additionally, we propose a consistency term to build a common latent space that captures the similarity of the poses with precision while allowing direct robot motion control from the latent space. For instance, we can generate in-between motion through simple linear interpolation between two projected human poses. We conduct a comprehensive evaluation of robot control from diverse modalities (i.e., texts, RGB videos, and key poses), which facilitates robot control for non-expert users. Our model outperforms existing works regarding human-to-robot retargeting in terms of efficiency and precision. Finally, we implemented our method in a real robot with self-collision avoidance through a whole-body controller to showcase the effectiveness of our approach. More information on our website https://evm7.github.io/UnsH2R/",
  "published": "2023-09-11",
  "updated": "2024-04-08",
  "year": "2023",
  "authors": [
   "Yashuai Yan",
   "Esteve Valls Mascaro",
   "Dongheui Lee"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "Humanoids",
  "venue_source": "semantic-scholar",
  "citations": 38,
  "influential_citations": 2,
  "tldr": "A novel deep-learning approach for human-to-robot motion retargeting, enabling robots to mimic human poses accurately and outperforms existing works regarding human-to-robot retargeting in terms of efficiency and precision.",
  "doi": "10.1109/Humanoids57100.2023.10375150",
  "oa_pdf": "https://arxiv.org/pdf/2309.05310",
  "s2_authors": [
   {
    "name": "Yashuai Yan",
    "id": "2239060301",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Esteve Valls Mascaro",
    "id": "2179106150",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "Dongheui Lee",
    "id": "2239064250",
    "h_index": 5,
    "papers": 18
   }
  ],
  "comment": "Accepted to Humanoids 2023. Website: https://evm7.github.io/UnsH2R/",
  "topics": [
   "humanoids"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2309.05310v3",
  "pdf_url": "https://arxiv.org/pdf/2309.05310v3",
  "html_url": "https://arxiv.org/html/2309.05310v3",
  "code_url": "https://evm7.github.io/UnsH2R/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.09
 },
 {
  "id": "2309.03897",
  "slug": "propainter-improving-propagation-and-transformer-for-video-inpainting",
  "title": "ProPainter: Improving Propagation and Transformer for Video Inpainting",
  "abstract": "Flow-based propagation and spatiotemporal Transformer are two mainstream mechanisms in video inpainting (VI). Despite the effectiveness of these components, they still suffer from some limitations that affect their performance. Previous propagation-based approaches are performed separately either in the image or feature domain. Global image propagation isolated from learning may cause spatial misalignment due to inaccurate optical flow. Moreover, memory or computational constraints limit the temporal range of feature propagation and video Transformer, preventing exploration of correspondence information from distant frames. To address these issues, we propose an improved framework, called ProPainter, which involves enhanced ProPagation and an efficient Transformer. Specifically, we introduce dual-domain propagation that combines the advantages of image and feature warping, exploiting global correspondences reliably. We also propose a mask-guided sparse video Transformer, which achieves high efficiency by discarding unnecessary and redundant tokens. With these components, ProPainter outperforms prior arts by a large margin of 1.46 dB in PSNR while maintaining appealing efficiency.",
  "published": "2023-09-07",
  "updated": "2023-09-07",
  "year": "2023",
  "authors": [
   "Shangchen Zhou",
   "Chongyi Li",
   "Kelvin C. K. Chan",
   "Chen Change Loy"
  ],
  "author_count": 4,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "ICCV",
  "venue_source": "semantic-scholar",
  "citations": 250,
  "influential_citations": 46,
  "tldr": "This work introduces dual-domain propagation that combines the advantages of image and feature warping, exploiting global correspondences reliably, and proposes a mask-guided sparse video Transformer, which achieves high efficiency by discarding unnecessary and redundant tokens.",
  "doi": "10.1109/ICCV51070.2023.00961",
  "oa_pdf": "http://arxiv.org/pdf/2309.03897",
  "s2_authors": [
   {
    "name": "Shangchen Zhou",
    "id": "7523259",
    "h_index": 31,
    "papers": 58
   },
   {
    "name": "Chongyi Li",
    "id": "2238391366",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Kelvin C. K. Chan",
    "id": "12009218",
    "h_index": 21,
    "papers": 25
   },
   {
    "name": "Chen Change Loy",
    "id": "1717179",
    "h_index": 131,
    "papers": 341
   }
  ],
  "comment": "Accepted by ICCV 2023. Code: https://github.com/sczhou/ProPainter",
  "topics": [
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2309.03897v1",
  "pdf_url": "https://arxiv.org/pdf/2309.03897v1",
  "html_url": "https://arxiv.org/html/2309.03897v1",
  "code_url": "https://github.com/sczhou/ProPainter",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.9
 },
 {
  "id": "2309.02591",
  "slug": "scaling-autoregressive-multi-modal-models-pretraining-and-instruction",
  "title": "Scaling Autoregressive Multi-Modal Models: Pretraining and Instruction Tuning",
  "abstract": "We present CM3Leon (pronounced \"Chameleon\"), a retrieval-augmented, token-based, decoder-only multi-modal language model capable of generating and infilling both text and images. CM3Leon uses the CM3 multi-modal architecture but additionally shows the extreme benefits of scaling up and tuning on more diverse instruction-style data. It is the first multi-modal model trained with a recipe adapted from text-only language models, including a large-scale retrieval-augmented pre-training stage and a second multi-task supervised fine-tuning (SFT) stage. It is also a general-purpose model that can do both text-to-image and image-to-text generation, allowing us to introduce self-contained contrastive decoding methods that produce high-quality outputs. Extensive experiments demonstrate that this recipe is highly effective for multi-modal models. CM3Leon achieves state-of-the-art performance in text-to-image generation with 5x less training compute than comparable methods (zero-shot MS-COCO FID of 4.88). After SFT, CM3Leon can also demonstrate unprecedented levels of controllability in tasks ranging from language-guided image editing to image-controlled generation and segmentation.",
  "published": "2023-09-05",
  "updated": "2023-09-05",
  "year": "2023",
  "authors": [
   "Lili Yu",
   "Bowen Shi",
   "Ramakanth Pasunuru",
   "Benjamin Muller",
   "Olga Golovneva",
   "Tianlu Wang",
   "Arun Babu",
   "Binh Tang",
   "Brian Karrer",
   "Shelly Sheynin",
   "Candace Ross",
   "Adam Polyak",
   "Russell Howes",
   "Vasu Sharma",
   "Puxin Xu",
   "Hovhannes Tamoyan",
   "Oron Ashual",
   "Uriel Singer",
   "Shang-Wen Li",
   "Susan Zhang",
   "Richard James",
   "Gargi Ghosh",
   "Yaniv Taigman",
   "Maryam Fazel-Zarandi",
   "Asli Celikyilmaz",
   "Luke Zettlemoyer",
   "Armen Aghajanyan"
  ],
  "author_count": 27,
  "categories": [
   "cs.LG",
   "cs.CL",
   "cs.CV"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 180,
  "influential_citations": 3,
  "tldr": "CM3Leon is a retrieval-augmented, token-based, decoder-only multi-modal language model capable of generating and infilling both text and images and introduces self-contained contrastive decoding methods that produce high-quality outputs.",
  "doi": "10.48550/arXiv.2309.02591",
  "oa_pdf": "https://arxiv.org/pdf/2309.02591",
  "s2_authors": [
   {
    "name": "L. Yu",
    "id": "49297123",
    "h_index": 15,
    "papers": 31
   },
   {
    "name": "Bowen Shi",
    "id": "2261676061",
    "h_index": 12,
    "papers": 31
   },
   {
    "name": "Ramakanth Pasunuru",
    "id": "10721120",
    "h_index": 30,
    "papers": 51
   },
   {
    "name": "Benjamin Muller",
    "id": "2335443692",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "O. Yu. Golovneva",
    "id": "100664938",
    "h_index": 6,
    "papers": 13
   },
   {
    "name": "Tianlu Wang",
    "id": "2238056517",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "A. Babu",
    "id": "2237983657",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Binh Tang",
    "id": "2237987675",
    "h_index": 9,
    "papers": 61
   },
   {
    "name": "Brian Karrer",
    "id": "2253591308",
    "h_index": 12,
    "papers": 22
   },
   {
    "name": "Shelly Sheynin",
    "id": "2086827528",
    "h_index": 12,
    "papers": 14
   },
   {
    "name": "Candace Ross",
    "id": "51519704",
    "h_index": 10,
    "papers": 14
   },
   {
    "name": "Adam Polyak",
    "id": "33964593",
    "h_index": 28,
    "papers": 38
   },
   {
    "name": "Russell Howes",
    "id": "2237983588",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Vasu Sharma",
    "id": "2237990986",
    "h_index": 7,
    "papers": 12
   },
   {
    "name": "Puxin Xu",
    "id": "2214843767",
    "h_index": 11,
    "papers": 58
   },
   {
    "name": "Hovhannes Tamoyan",
    "id": "2040866961",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Oron Ashual",
    "id": "1388005058",
    "h_index": 8,
    "papers": 10
   },
   {
    "name": "Uriel Singer",
    "id": "88622696",
    "h_index": 13,
    "papers": 27
   },
   {
    "name": "Shang-Wen Li",
    "id": "2530311",
    "h_index": 32,
    "papers": 90
   },
   {
    "name": "Susan Zhang",
    "id": "2238121623",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Rich James",
    "id": "2191899140",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Gargi Ghosh",
    "id": "134007132",
    "h_index": 17,
    "papers": 20
   },
   {
    "name": "Yaniv Taigman",
    "id": "2188620",
    "h_index": 30,
    "papers": 41
   },
   {
    "name": "Maryam Fazel-Zarandi",
    "id": "1399159921",
    "h_index": 18,
    "papers": 37
   },
   {
    "name": "Asli Celikyilmaz",
    "id": "1709797",
    "h_index": 48,
    "papers": 182
   },
   {
    "name": "Luke Zettlemoyer",
    "id": "1982950",
    "h_index": 118,
    "papers": 278
   },
   {
    "name": "Armen Aghajanyan",
    "id": "2201435",
    "h_index": 21,
    "papers": 40
   }
  ],
  "comment": "",
  "topics": [
   "foundation-pretraining",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2309.02591v1",
  "pdf_url": "https://arxiv.org/pdf/2309.02591v1",
  "html_url": "https://arxiv.org/html/2309.02591v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.26
 },
 {
  "id": "2309.01918",
  "slug": "roboagent-generalization-and-efficiency-in-robot-manipulation-via-sema",
  "title": "RoboAgent: Generalization and Efficiency in Robot Manipulation via Semantic Augmentations and Action Chunking",
  "abstract": "The grand aim of having a single robot that can manipulate arbitrary objects in diverse settings is at odds with the paucity of robotics datasets. Acquiring and growing such datasets is strenuous due to manual efforts, operational costs, and safety challenges. A path toward such an universal agent would require a structured framework capable of wide generalization but trained within a reasonable data budget. In this paper, we develop an efficient system (RoboAgent) for training universal agents capable of multi-task manipulation skills using (a) semantic augmentations that can rapidly multiply existing datasets and (b) action representations that can extract performant policies with small yet diverse multi-modal datasets without overfitting. In addition, reliable task conditioning and an expressive policy architecture enable our agent to exhibit a diverse repertoire of skills in novel situations specified using language commands. Using merely 7500 demonstrations, we are able to train a single agent capable of 12 unique skills, and demonstrate its generalization over 38 tasks spread across common daily activities in diverse kitchen scenes. On average, RoboAgent outperforms prior methods by over 40% in unseen situations while being more sample efficient and being amenable to capability improvements and extensions through fine-tuning. Videos at https://robopen.github.io/",
  "published": "2023-09-05",
  "updated": "2023-09-05",
  "year": "2023",
  "authors": [
   "Homanga Bharadhwaj",
   "Jay Vakil",
   "Mohit Sharma",
   "Abhinav Gupta",
   "Shubham Tulsiani",
   "Vikash Kumar"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA",
  "venue_source": "semantic-scholar",
  "citations": 263,
  "influential_citations": 13,
  "tldr": "An efficient framework for training universal agents capable of multi-task manipulation skills using semantic augmentations that can rapidly multiply existing datasets and action representations that can extract performant policies with small yet diverse multi-modal datasets without overfitting is developed.",
  "doi": "10.1109/ICRA57147.2024.10611293",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Homanga Bharadhwaj",
    "id": "51113848",
    "h_index": 23,
    "papers": 59
   },
   {
    "name": "Jay Vakil",
    "id": "2215216078",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Mohit Sharma",
    "id": "145467103",
    "h_index": 17,
    "papers": 35
   },
   {
    "name": "Abhi Gupta",
    "id": "2117767136",
    "h_index": 16,
    "papers": 24
   },
   {
    "name": "Shubham Tulsiani",
    "id": "2757335",
    "h_index": 45,
    "papers": 98
   },
   {
    "name": "Vikash Kumar",
    "id": "2238455355",
    "h_index": 7,
    "papers": 10
   }
  ],
  "comment": "",
  "topics": [
   "imitation-diffusion",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2309.01918v1",
  "pdf_url": "https://arxiv.org/pdf/2309.01918v1",
  "html_url": "https://arxiv.org/html/2309.01918v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.92
 },
 {
  "id": "2308.16891",
  "slug": "gnfactor-multi-task-real-robot-learning-with-generalizable-neural-feat",
  "title": "GNFactor: Multi-Task Real Robot Learning with Generalizable Neural Feature Fields",
  "abstract": "It is a long-standing problem in robotics to develop agents capable of executing diverse manipulation tasks from visual observations in unstructured real-world environments. To achieve this goal, the robot needs to have a comprehensive understanding of the 3D structure and semantics of the scene. In this work, we present $\\textbf{GNFactor}$, a visual behavior cloning agent for multi-task robotic manipulation with $\\textbf{G}$eneralizable $\\textbf{N}$eural feature $\\textbf{F}$ields. GNFactor jointly optimizes a generalizable neural field (GNF) as a reconstruction module and a Perceiver Transformer as a decision-making module, leveraging a shared deep 3D voxel representation. To incorporate semantics in 3D, the reconstruction module utilizes a vision-language foundation model ($\\textit{e.g.}$, Stable Diffusion) to distill rich semantic information into the deep 3D voxel. We evaluate GNFactor on 3 real robot tasks and perform detailed ablations on 10 RLBench tasks with a limited number of demonstrations. We observe a substantial improvement of GNFactor over current state-of-the-art methods in seen and unseen tasks, demonstrating the strong generalization ability of GNFactor. Our project website is https://yanjieze.com/GNFactor/ .",
  "published": "2023-08-31",
  "updated": "2024-07-28",
  "year": "2023",
  "authors": [
   "Yanjie Ze",
   "Ge Yan",
   "Yueh-Hua Wu",
   "Annabella Macaluso",
   "Yuying Ge",
   "Jianglong Ye",
   "Nicklas Hansen",
   "Li Erran Li",
   "Xiaolong Wang"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 161,
  "influential_citations": 12,
  "tldr": "A substantial improvement of GNFactor over current state-of-the-art methods in seen and unseen tasks is observed, demonstrating the strong generalization ability of GN factor.",
  "doi": "10.48550/arXiv.2308.16891",
  "oa_pdf": "https://arxiv.org/pdf/2308.16891",
  "s2_authors": [
   {
    "name": "Yanjie Ze",
    "id": "2151089356",
    "h_index": 12,
    "papers": 16
   },
   {
    "name": "Ge Yan",
    "id": "2181341253",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Yueh-Hua Wu",
    "id": "31609618",
    "h_index": 10,
    "papers": 13
   },
   {
    "name": "Annabella Macaluso",
    "id": "2196937970",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Yuying Ge",
    "id": "51123495",
    "h_index": 29,
    "papers": 59
   },
   {
    "name": "Jianglong Ye",
    "id": "2153258399",
    "h_index": 9,
    "papers": 17
   },
   {
    "name": "Nicklas Hansen",
    "id": "1491707104",
    "h_index": 19,
    "papers": 39
   },
   {
    "name": "L. Li",
    "id": "2156057522",
    "h_index": 9,
    "papers": 14
   },
   {
    "name": "X. Wang",
    "id": "39849136",
    "h_index": 45,
    "papers": 236
   }
  ],
  "comment": "CoRL 2023 Oral. Website: https://yanjieze.com/GNFactor/",
  "topics": [
   "imitation-diffusion",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2308.16891v3",
  "pdf_url": "https://arxiv.org/pdf/2308.16891v3",
  "html_url": "https://arxiv.org/html/2308.16891v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.71
 },
 {
  "id": "2308.13561",
  "slug": "project-aria-a-new-tool-for-egocentric-multi-modal-ai-research",
  "title": "Project Aria: A New Tool for Egocentric Multi-Modal AI Research",
  "abstract": "Egocentric, multi-modal data as available on future augmented reality (AR) devices provides unique challenges and opportunities for machine perception. These future devices will need to be all-day wearable in a socially acceptable form-factor to support always available, context-aware and personalized AI applications. Our team at Meta Reality Labs Research built the Aria device, an egocentric, multi-modal data recording and streaming device with the goal to foster and accelerate research in this area. In this paper, we describe the Aria device hardware including its sensor configuration and the corresponding software tools that enable recording and processing of such data.",
  "published": "2023-08-24",
  "updated": "2023-10-01",
  "year": "2023",
  "authors": [
   "Jakob Engel",
   "Kiran Somasundaram",
   "Michael Goesele",
   "Albert Sun",
   "Alexander Gamino",
   "Andrew Turner",
   "Arjang Talattof",
   "Arnie Yuan",
   "Bilal Souti",
   "Brighid Meredith",
   "Cheng Peng",
   "Chris Sweeney",
   "Cole Wilson",
   "Dan Barnes",
   "Daniel DeTone",
   "David Caruso",
   "Derek Valleroy",
   "Dinesh Ginjupalli",
   "Duncan Frost",
   "Edward Miller",
   "Elias Mueggler",
   "Evgeniy Oleinik",
   "Fan Zhang",
   "Guruprasad Somasundaram",
   "Gustavo Solaira",
   "Harry Lanaras",
   "Henry Howard-Jenkins",
   "Huixuan Tang",
   "Hyo Jin Kim",
   "Jaime Rivera",
   "Ji Luo",
   "Jing Dong",
   "Julian Straub",
   "Kevin Bailey",
   "Kevin Eckenhoff",
   "Lingni Ma",
   "Luis Pesqueira",
   "Mark Schwesinger",
   "Maurizio Monge",
   "Nan Yang",
   "Nick Charron",
   "Nikhil Raina",
   "Omkar Parkhi",
   "Peter Borschowa",
   "Pierre Moulon",
   "Prince Gupta",
   "Raul Mur-Artal",
   "Robbie Pennington",
   "Sachin Kulkarni",
   "Sagar Miglani",
   "Santosh Gondi",
   "Saransh Solanki",
   "Sean Diener",
   "Shangyi Cheng",
   "Simon Green",
   "Steve Saarinen",
   "Suvam Patra",
   "Tassos Mourikis",
   "Thomas Whelan",
   "Tripti Singh",
   "Vasileios Balntas",
   "Vijay Baiyya",
   "Wilson Dreewes",
   "Xiaqing Pan",
   "Yang Lou",
   "Yipu Zhao",
   "Yusuf Mansour",
   "Yuyang Zou",
   "Zhaoyang Lv",
   "Zijian Wang",
   "Mingfei Yan",
   "Carl Ren",
   "Renzo De Nardi",
   "Richard Newcombe"
  ],
  "author_count": 74,
  "categories": [
   "cs.HC",
   "cs.CV"
  ],
  "primary_category": "cs.HC",
  "venue": "",
  "venue_source": "",
  "citations": 272,
  "influential_citations": 34,
  "tldr": "The Aria device is described, an egocentric, multi-modal data recording and streaming device that will need to be all-day wearable in a socially acceptable form-factor to support always available, context-aware and personalized AI applications.",
  "doi": "10.48550/arXiv.2308.13561",
  "oa_pdf": "https://arxiv.org/pdf/2308.13561",
  "s2_authors": [
   {
    "name": "K. Somasundaram",
    "id": "31604945",
    "h_index": 13,
    "papers": 78
   },
   {
    "name": "Jing Dong",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Huixuan Tang",
    "id": "2494119",
    "h_index": 9,
    "papers": 21
   },
   {
    "name": "Julian Straub",
    "id": "20128275",
    "h_index": 20,
    "papers": 40
   },
   {
    "name": "Mingfei Yan",
    "id": "2234026075",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "M. Goesele",
    "id": "1689293",
    "h_index": 36,
    "papers": 159
   },
   {
    "name": "Jakob J. Engel",
    "id": "35152266",
    "h_index": 17,
    "papers": 22
   },
   {
    "name": "R. D. Nardi",
    "id": "1769365",
    "h_index": 15,
    "papers": 24
   },
   {
    "name": "Richard A. Newcombe",
    "id": "50366818",
    "h_index": 31,
    "papers": 57
   }
  ],
  "comment": "",
  "topics": [
   "egocentric-data"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2308.13561v3",
  "pdf_url": "https://arxiv.org/pdf/2308.13561v3",
  "html_url": "https://arxiv.org/html/2308.13561v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.44
 },
 {
  "id": "2308.12952",
  "slug": "bridgedata-v2-a-dataset-for-robot-learning-at-scale",
  "title": "BridgeData V2: A Dataset for Robot Learning at Scale",
  "abstract": "We introduce BridgeData V2, a large and diverse dataset of robotic manipulation behaviors designed to facilitate research on scalable robot learning. BridgeData V2 contains 60,096 trajectories collected across 24 environments on a publicly available low-cost robot. BridgeData V2 provides extensive task and environment variability, leading to skills that can generalize across environments, domains, and institutions, making the dataset a useful resource for a broad range of researchers. Additionally, the dataset is compatible with a wide variety of open-vocabulary, multi-task learning methods conditioned on goal images or natural language instructions. In our experiments, we train 6 state-of-the-art imitation learning and offline reinforcement learning methods on our dataset, and find that they succeed on a suite of tasks requiring varying amounts of generalization. We also demonstrate that the performance of these methods improves with more data and higher capacity models, and that training on a greater variety of skills leads to improved generalization. By publicly sharing BridgeData V2 and our pre-trained models, we aim to accelerate research in scalable robot learning methods. Project page at https://rail-berkeley.github.io/bridgedata",
  "published": "2023-08-24",
  "updated": "2024-01-17",
  "year": "2023",
  "authors": [
   "Homer Walke",
   "Kevin Black",
   "Abraham Lee",
   "Moo Jin Kim",
   "Max Du",
   "Chongyi Zheng",
   "Tony Zhao",
   "Philippe Hansen-Estruch",
   "Quan Vuong",
   "Andre He",
   "Vivek Myers",
   "Kuan Fang",
   "Chelsea Finn",
   "Sergey Levine"
  ],
  "author_count": 14,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 816,
  "influential_citations": 90,
  "tldr": "This work introduces BridgeData V2, a large and diverse dataset of robotic manipulation behaviors designed to facilitate research on scalable robot learning, and demonstrates that the performance of these methods improves with more data and higher capacity models, and that training on a greater variety of skills leads to improved generalization.",
  "doi": "10.48550/arXiv.2308.12952",
  "oa_pdf": "https://arxiv.org/pdf/2308.12952",
  "s2_authors": [
   {
    "name": "H. Walke",
    "id": "2029241116",
    "h_index": 17,
    "papers": 23
   },
   {
    "name": "Kevin Black",
    "id": "2069483822",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Abraham Lee",
    "id": "2233425761",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Moo Jin Kim",
    "id": "2159987907",
    "h_index": 10,
    "papers": 10
   },
   {
    "name": "Maximilian Du",
    "id": "117791840",
    "h_index": 10,
    "papers": 17
   },
   {
    "name": "Chongyi Zheng",
    "id": "1382713388",
    "h_index": 9,
    "papers": 16
   },
   {
    "name": "Tony Zhao",
    "id": "145914976",
    "h_index": 17,
    "papers": 19
   },
   {
    "name": "Philippe Hansen-Estruch",
    "id": "2163582335",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "Q. Vuong",
    "id": "144579461",
    "h_index": 23,
    "papers": 40
   },
   {
    "name": "Andre Wang He",
    "id": "2162736405",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Vivek Myers",
    "id": "1823943196",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Kuan Fang",
    "id": "145213709",
    "h_index": 15,
    "papers": 21
   },
   {
    "name": "Chelsea Finn",
    "id": "46881670",
    "h_index": 94,
    "papers": 240
   },
   {
    "name": "S. Levine",
    "id": "1736651",
    "h_index": 167,
    "papers": 551
   }
  ],
  "comment": "9 pages",
  "topics": [
   "imitation-diffusion",
   "rl-control"
  ],
  "orgs": [
   "UC Berkeley"
  ],
  "abs_url": "https://arxiv.org/abs/2308.12952v3",
  "pdf_url": "https://arxiv.org/pdf/2308.12952v3",
  "html_url": "https://arxiv.org/html/2308.12952v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.91
 },
 {
  "id": "2308.10901",
  "slug": "structured-world-models-from-human-videos",
  "title": "Structured World Models from Human Videos",
  "abstract": "We tackle the problem of learning complex, general behaviors directly in the real world. We propose an approach for robots to efficiently learn manipulation skills using only a handful of real-world interaction trajectories from many different settings. Inspired by the success of learning from large-scale datasets in the fields of computer vision and natural language, our belief is that in order to efficiently learn, a robot must be able to leverage internet-scale, human video data. Humans interact with the world in many interesting ways, which can allow a robot to not only build an understanding of useful actions and affordances but also how these actions affect the world for manipulation. Our approach builds a structured, human-centric action space grounded in visual affordances learned from human videos. Further, we train a world model on human videos and fine-tune on a small amount of robot interaction data without any task supervision. We show that this approach of affordance-space world models enables different robots to learn various manipulation skills in complex settings, in under 30 minutes of interaction. Videos can be found at https://human-world-model.github.io",
  "published": "2023-08-21",
  "updated": "2023-08-21",
  "year": "2023",
  "authors": [
   "Russell Mendonca",
   "Shikhar Bahl",
   "Deepak Pathak"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG",
   "cs.NE"
  ],
  "primary_category": "cs.RO",
  "venue": "RSS",
  "venue_source": "semantic-scholar",
  "citations": 181,
  "influential_citations": 16,
  "tldr": "This work proposes an approach for robots to efficiently learn manipulation skills using only a handful of real-world interaction trajectories from many different settings, and builds a structured, human-centric action space grounded in visual affordances learned from human videos.",
  "doi": "10.15607/RSS.2023.XIX.012",
  "oa_pdf": "https://doi.org/10.15607/rss.2023.xix.012",
  "s2_authors": [
   {
    "name": "R. Mendonca",
    "id": "35509365",
    "h_index": 15,
    "papers": 24
   },
   {
    "name": "Shikhar Bahl",
    "id": "8527563",
    "h_index": 20,
    "papers": 38
   },
   {
    "name": "Deepak Pathak",
    "id": "38236002",
    "h_index": 34,
    "papers": 78
   }
  ],
  "comment": "RSS 2023. Website at https://human-world-model.github.io",
  "topics": [
   "world-models",
   "egocentric-data"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2308.10901v1",
  "pdf_url": "https://arxiv.org/pdf/2308.10901v1",
  "html_url": "https://arxiv.org/html/2308.10901v1",
  "code_url": "https://human-world-model.github.io",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.76
 },
 {
  "id": "2308.07903",
  "slug": "relightable-and-animatable-neural-avatar-from-sparse-view-video",
  "title": "Relightable and Animatable Neural Avatar from Sparse-View Video",
  "abstract": "This paper tackles the challenge of creating relightable and animatable neural avatars from sparse-view (or even monocular) videos of dynamic humans under unknown illumination. Compared to studio environments, this setting is more practical and accessible but poses an extremely challenging ill-posed problem. Previous neural human reconstruction methods are able to reconstruct animatable avatars from sparse views using deformed Signed Distance Fields (SDF) but cannot recover material parameters for relighting. While differentiable inverse rendering-based methods have succeeded in material recovery of static objects, it is not straightforward to extend them to dynamic humans as it is computationally intensive to compute pixel-surface intersection and light visibility on deformed SDFs for inverse rendering. To solve this challenge, we propose a Hierarchical Distance Query (HDQ) algorithm to approximate the world space distances under arbitrary human poses. Specifically, we estimate coarse distances based on a parametric human model and compute fine distances by exploiting the local deformation invariance of SDF. Based on the HDQ algorithm, we leverage sphere tracing to efficiently estimate the surface intersection and light visibility. This allows us to develop the first system to recover animatable and relightable neural avatars from sparse view (or monocular) inputs. Experiments demonstrate that our approach is able to produce superior results compared to state-of-the-art methods. Our code will be released for reproducibility.",
  "published": "2023-08-15",
  "updated": "2023-08-17",
  "year": "2023",
  "authors": [
   "Zhen Xu",
   "Sida Peng",
   "Chen Geng",
   "Linzhan Mou",
   "Zihan Yan",
   "Jiaming Sun",
   "Hujun Bao",
   "Xiaowei Zhou"
  ],
  "author_count": 8,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.GR"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 42,
  "influential_citations": 8,
  "tldr": "A Hierarchical Distance Query (HDQ) algorithm is proposed to approximate the world space SDF under arbitrary human poses and is developed as the first system to recover relightable and animatable neural avatars from sparse or monocular inputs.",
  "doi": "10.1109/CVPR52733.2024.00100",
  "oa_pdf": "https://arxiv.org/pdf/2308.07903",
  "s2_authors": [
   {
    "name": "Zhen Xu",
    "id": "121832519",
    "h_index": 53,
    "papers": 623
   },
   {
    "name": "Sida Peng",
    "id": "2072712025",
    "h_index": 41,
    "papers": 95
   },
   {
    "name": "Chen Geng",
    "id": "2158857804",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Linzhan Mou",
    "id": "2205658464",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "Zihan Yan",
    "id": "2170737498",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Jiaming Sun",
    "id": "153552118",
    "h_index": 17,
    "papers": 25
   },
   {
    "name": "H. Bao",
    "id": "1679542",
    "h_index": 71,
    "papers": 405
   },
   {
    "name": "Xiaowei Zhou",
    "id": "145453113",
    "h_index": 51,
    "papers": 95
   }
  ],
  "comment": "Project page: https://zju3dv.github.io/relightable_avatar",
  "topics": [
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2308.07903v2",
  "pdf_url": "https://arxiv.org/pdf/2308.07903v2",
  "html_url": "https://arxiv.org/html/2308.07903v2",
  "code_url": "https://zju3dv.github.io/relightable_avatar",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.13
 },
 {
  "id": "2308.04079",
  "slug": "3d-gaussian-splatting-for-real-time-radiance-field-rendering",
  "title": "3D Gaussian Splatting for Real-Time Radiance Field Rendering",
  "abstract": "Radiance Field methods have recently revolutionized novel-view synthesis of scenes captured with multiple photos or videos. However, achieving high visual quality still requires neural networks that are costly to train and render, while recent faster methods inevitably trade off speed for quality. For unbounded and complete scenes (rather than isolated objects) and 1080p resolution rendering, no current method can achieve real-time display rates. We introduce three key elements that allow us to achieve state-of-the-art visual quality while maintaining competitive training times and importantly allow high-quality real-time (>= 30 fps) novel-view synthesis at 1080p resolution. First, starting from sparse points produced during camera calibration, we represent the scene with 3D Gaussians that preserve desirable properties of continuous volumetric radiance fields for scene optimization while avoiding unnecessary computation in empty space; Second, we perform interleaved optimization/density control of the 3D Gaussians, notably optimizing anisotropic covariance to achieve an accurate representation of the scene; Third, we develop a fast visibility-aware rendering algorithm that supports anisotropic splatting and both accelerates training and allows realtime rendering. We demonstrate state-of-the-art visual quality and real-time rendering on several established datasets.",
  "published": "2023-08-08",
  "updated": "2023-08-08",
  "year": "2023",
  "authors": [
   "Bernhard Kerbl",
   "Georgios Kopanas",
   "Thomas Leimk\u00fchler",
   "George Drettakis"
  ],
  "author_count": 4,
  "categories": [
   "cs.GR",
   "cs.CV"
  ],
  "primary_category": "cs.GR",
  "venue": "",
  "venue_source": "",
  "citations": 9994,
  "influential_citations": 2673,
  "tldr": "This work develops a fast visibility-aware rendering algorithm that supports anisotropic splatting and both accelerates training and allows realtime rendering and demonstrates state-of-the-art visual quality and real-time rendering on several established datasets.",
  "doi": "10.1145/3592433",
  "oa_pdf": "https://inria.hal.science/hal-04088161",
  "s2_authors": [
   {
    "name": "Bernhard Kerbl",
    "id": "2454128",
    "h_index": 14,
    "papers": 29
   },
   {
    "name": "Georgios Kopanas",
    "id": "72209802",
    "h_index": 9,
    "papers": 13
   },
   {
    "name": "Thomas Leimkuehler",
    "id": "41015873",
    "h_index": 12,
    "papers": 37
   },
   {
    "name": "G. Drettakis",
    "id": "1721779",
    "h_index": 58,
    "papers": 253
   }
  ],
  "comment": "https://repo-sam.inria.fr/fungraph/3d-gaussian-splatting/",
  "topics": [
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2308.04079v1",
  "pdf_url": "https://arxiv.org/pdf/2308.04079v1",
  "html_url": "https://arxiv.org/html/2308.04079v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.0
 },
 {
  "id": "2308.03624",
  "slug": "moma-force-visual-force-imitation-for-real-world-mobile-manipulation",
  "title": "MOMA-Force: Visual-Force Imitation for Real-World Mobile Manipulation",
  "abstract": "In this paper, we present a novel method for mobile manipulators to perform multiple contact-rich manipulation tasks. While learning-based methods have the potential to generate actions in an end-to-end manner, they often suffer from insufficient action accuracy and robustness against noise. On the other hand, classical control-based methods can enhance system robustness, but at the cost of extensive parameter tuning. To address these challenges, we present MOMA-Force, a visual-force imitation method that seamlessly combines representation learning for perception, imitation learning for complex motion generation, and admittance whole-body control for system robustness and controllability. MOMA-Force enables a mobile manipulator to learn multiple complex contact-rich tasks with high success rates and small contact forces. In a real household setting, our method outperforms baseline methods in terms of task success rates. Moreover, our method achieves smaller contact forces and smaller force variances compared to baseline methods without force imitation. Overall, we offer a promising approach for efficient and robust mobile manipulation in the real world. Videos and more details can be found on \\url{https://visual-force-imitation.github.io}",
  "published": "2023-08-07",
  "updated": "2023-08-07",
  "year": "2023",
  "authors": [
   "Taozheng Yang",
   "Ya Jing",
   "Hongtao Wu",
   "Jiafeng Xu",
   "Kuankuan Sima",
   "Guangzeng Chen",
   "Qie Sima",
   "Tao Kong"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS",
  "venue_source": "semantic-scholar",
  "citations": 31,
  "influential_citations": 1,
  "tldr": "MOMA-Force is presented, a visual-force imitation method that seamlessly combines representation learning for perception, imitation learning for complex motion generation, and admittance whole-body control for system robustness and controllability for efficient and robust mobile manipulation in the real world.",
  "doi": "10.1109/IROS55552.2023.10342371",
  "oa_pdf": "http://arxiv.org/pdf/2308.03624",
  "s2_authors": [
   {
    "name": "Taozheng Yang",
    "id": "2149225232",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Yaxing Jing",
    "id": "2190866731",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Hongtao Wu",
    "id": "2120430896",
    "h_index": 8,
    "papers": 14
   },
   {
    "name": "Jiafeng Xu",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Kuankuan Sima",
    "id": "2199959843",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Guangzeng Chen",
    "id": "9230409",
    "h_index": 7,
    "papers": 20
   },
   {
    "name": "Qie Sima",
    "id": "2300691797",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Tao Kong",
    "id": "145868988",
    "h_index": 30,
    "papers": 54
   }
  ],
  "comment": "IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS), 2023",
  "topics": [
   "humanoids",
   "tactile",
   "imitation-diffusion",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2308.03624v1",
  "pdf_url": "https://arxiv.org/pdf/2308.03624v1",
  "html_url": "https://arxiv.org/html/2308.03624v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.01
 },
 {
  "id": "2308.02487",
  "slug": "convolutions-die-hard-open-vocabulary-segmentation-with-single-frozen",
  "title": "Convolutions Die Hard: Open-Vocabulary Segmentation with Single Frozen Convolutional CLIP",
  "abstract": "Open-vocabulary segmentation is a challenging task requiring segmenting and recognizing objects from an open set of categories. One way to address this challenge is to leverage multi-modal models, such as CLIP, to provide image and text features in a shared embedding space, which bridges the gap between closed-vocabulary and open-vocabulary recognition. Hence, existing methods often adopt a two-stage framework to tackle the problem, where the inputs first go through a mask generator and then through the CLIP model along with the predicted masks. This process involves extracting features from images multiple times, which can be ineffective and inefficient. By contrast, we propose to build everything into a single-stage framework using a shared Frozen Convolutional CLIP backbone, which not only significantly simplifies the current two-stage pipeline, but also remarkably yields a better accuracy-cost trade-off. The proposed FC-CLIP, benefits from the following observations: the frozen CLIP backbone maintains the ability of open-vocabulary classification and can also serve as a strong mask generator, and the convolutional CLIP generalizes well to a larger input resolution than the one used during contrastive image-text pretraining. When training on COCO panoptic data only and testing in a zero-shot manner, FC-CLIP achieve 26.8 PQ, 16.8 AP, and 34.1 mIoU on ADE20K, 18.2 PQ, 27.9 mIoU on Mapillary Vistas, 44.0 PQ, 26.8 AP, 56.2 mIoU on Cityscapes, outperforming the prior art by +4.2 PQ, +2.4 AP, +4.2 mIoU on ADE20K, +4.0 PQ on Mapillary Vistas and +20.1 PQ on Cityscapes, respectively. Additionally, the training and testing time of FC-CLIP is 7.5x and 6.6x significantly faster than the same prior art, while using 5.9x fewer parameters. FC-CLIP also sets a new state-of-the-art performance across various open-vocabulary semantic segmentation datasets. Code at https://github.com/bytedance/fc-clip",
  "published": "2023-08-04",
  "updated": "2023-11-14",
  "year": "2023",
  "authors": [
   "Qihang Yu",
   "Ju He",
   "Xueqing Deng",
   "Xiaohui Shen",
   "Liang-Chieh Chen"
  ],
  "author_count": 5,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 278,
  "influential_citations": 53,
  "tldr": "This work proposes to build everything into a single-stage framework using a shared Frozen Convolutional CLIP backbone, which not only significantly simplifies the current two-stage pipeline, but also remarkably yields a better accuracy-cost trade-off.",
  "doi": "10.48550/arXiv.2308.02487",
  "oa_pdf": "https://arxiv.org/pdf/2308.02487",
  "s2_authors": [
   {
    "name": "Qihang Yu",
    "id": "2156559",
    "h_index": 23,
    "papers": 41
   },
   {
    "name": "Ju He",
    "id": "153146760",
    "h_index": 18,
    "papers": 35
   },
   {
    "name": "Xueqing Deng",
    "id": "50588067",
    "h_index": 12,
    "papers": 19
   },
   {
    "name": "Xiaohui Shen",
    "id": "1720987",
    "h_index": 59,
    "papers": 116
   },
   {
    "name": "Liang-Chieh Chen",
    "id": "34192119",
    "h_index": 41,
    "papers": 57
   }
  ],
  "comment": "NeurIPS 2023 camera ready. code and model available at https://github.com/bytedance/fc-clip",
  "topics": [
   "foundation-pretraining"
  ],
  "orgs": [
   "ByteDance"
  ],
  "abs_url": "https://arxiv.org/abs/2308.02487v2",
  "pdf_url": "https://arxiv.org/pdf/2308.02487v2",
  "html_url": "https://arxiv.org/html/2308.02487v2",
  "code_url": "https://github.com/bytedance/fc-clip",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.45
 },
 {
  "id": "2308.01898",
  "slug": "unisim-a-neural-closed-loop-sensor-simulator",
  "title": "UniSim: A Neural Closed-Loop Sensor Simulator",
  "abstract": "Rigorously testing autonomy systems is essential for making safe self-driving vehicles (SDV) a reality. It requires one to generate safety critical scenarios beyond what can be collected safely in the world, as many scenarios happen rarely on public roads. To accurately evaluate performance, we need to test the SDV on these scenarios in closed-loop, where the SDV and other actors interact with each other at each timestep. Previously recorded driving logs provide a rich resource to build these new scenarios from, but for closed loop evaluation, we need to modify the sensor data based on the new scene configuration and the SDV's decisions, as actors might be added or removed and the trajectories of existing actors and the SDV will differ from the original log. In this paper, we present UniSim, a neural sensor simulator that takes a single recorded log captured by a sensor-equipped vehicle and converts it into a realistic closed-loop multi-sensor simulation. UniSim builds neural feature grids to reconstruct both the static background and dynamic actors in the scene, and composites them together to simulate LiDAR and camera data at new viewpoints, with actors added or removed and at new placements. To better handle extrapolated views, we incorporate learnable priors for dynamic objects, and leverage a convolutional network to complete unseen regions. Our experiments show UniSim can simulate realistic sensor data with small domain gap on downstream tasks. With UniSim, we demonstrate closed-loop evaluation of an autonomy system on safety-critical scenarios as if it were in the real world.",
  "published": "2023-08-03",
  "updated": "2023-08-03",
  "year": "2023",
  "authors": [
   "Ze Yang",
   "Yun Chen",
   "Jingkang Wang",
   "Sivabalan Manivasagam",
   "Wei-Chiu Ma",
   "Anqi Joyce Yang",
   "Raquel Urtasun"
  ],
  "author_count": 7,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 385,
  "influential_citations": 30,
  "tldr": "UniSim is presented, a neural sensor simulator that takes a single recorded log captured by a sensor-equipped vehicle and converts it into a realistic closed-loop multi-sensor simulation, and demonstrates, for the first time, closed- loop evaluation of an autonomy system on safety-critical scenarios as if it were in the real world.",
  "doi": "10.1109/CVPR52729.2023.00140",
  "oa_pdf": "https://arxiv.org/pdf/2308.01898",
  "s2_authors": [
   {
    "name": "Ze Yang",
    "id": "2155451016",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Yun Chen",
    "id": "2144861649",
    "h_index": 12,
    "papers": 20
   },
   {
    "name": "Jingkang Wang",
    "id": "71563016",
    "h_index": 14,
    "papers": 34
   },
   {
    "name": "Sivabalan Manivasagam",
    "id": "39981216",
    "h_index": 19,
    "papers": 34
   },
   {
    "name": "Wei-Chiu Ma",
    "id": "2650832",
    "h_index": 26,
    "papers": 47
   },
   {
    "name": "A. Yang",
    "id": "150036428",
    "h_index": 13,
    "papers": 19
   },
   {
    "name": "R. Urtasun",
    "id": "2422559",
    "h_index": 116,
    "papers": 393
   }
  ],
  "comment": "CVPR 2023 Highlight. Project page: https://waabi.ai/research/unisim/",
  "topics": [
   "sim2real",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2308.01898v1",
  "pdf_url": "https://arxiv.org/pdf/2308.01898v1",
  "html_url": "https://arxiv.org/html/2308.01898v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.09
 },
 {
  "id": "2308.01399",
  "slug": "learning-to-model-the-world-with-language",
  "title": "Learning to Model the World with Language",
  "abstract": "To interact with humans and act in the world, agents need to understand the range of language that people use and relate it to the visual world. While current agents can learn to execute simple language instructions, we aim to build agents that leverage diverse language -- language like \"this button turns on the TV\" or \"I put the bowls away\" -- that conveys general knowledge, describes the state of the world, provides interactive feedback, and more. Our key idea is that agents should interpret such diverse language as a signal that helps them predict the future: what they will observe, how the world will behave, and which situations will be rewarded. This perspective unifies language understanding with future prediction as a powerful self-supervised learning objective. We instantiate this in Dynalang, an agent that learns a multimodal world model to predict future text and image representations, and learns to act from imagined model rollouts. While current methods that learn language-conditioned policies degrade in performance with more diverse types of language, we show that Dynalang learns to leverage environment descriptions, game rules, and instructions to excel on tasks ranging from game-playing to navigating photorealistic home scans. Finally, we show that our method enables additional capabilities due to learning a generative model: Dynalang can be pretrained on text-only data, enabling learning from offline datasets, and generate language grounded in an environment.",
  "published": "2023-07-31",
  "updated": "2024-05-31",
  "year": "2023",
  "authors": [
   "Jessy Lin",
   "Yuqing Du",
   "Olivia Watkins",
   "Danijar Hafner",
   "Pieter Abbeel",
   "Dan Klein",
   "Anca Dragan"
  ],
  "author_count": 7,
  "categories": [
   "cs.CL",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.CL",
  "venue": "ICML",
  "venue_source": "semantic-scholar",
  "citations": 88,
  "influential_citations": 8,
  "tldr": "Dynalang is an agent that learns a multimodal world model to predict future text and image representations, and learns to act from imagined model rollouts, and can be pretrained on text-only data, enabling learning from offline datasets, and generate language grounded in an environment.",
  "doi": "10.48550/arXiv.2308.01399",
  "oa_pdf": "https://arxiv.org/pdf/2308.01399",
  "s2_authors": [
   {
    "name": "Jessy Lin",
    "id": "32815692",
    "h_index": 8,
    "papers": 10
   },
   {
    "name": "Yuqing Du",
    "id": "144894286",
    "h_index": 16,
    "papers": 27
   },
   {
    "name": "Olivia Watkins",
    "id": "145695607",
    "h_index": 9,
    "papers": 12
   },
   {
    "name": "Danijar Hafner",
    "id": "35006479",
    "h_index": 25,
    "papers": 47
   },
   {
    "name": "P. Abbeel",
    "id": "1689992",
    "h_index": 165,
    "papers": 596
   },
   {
    "name": "D. Klein",
    "id": "38666915",
    "h_index": 86,
    "papers": 245
   },
   {
    "name": "A. Dragan",
    "id": "2745001",
    "h_index": 61,
    "papers": 190
   }
  ],
  "comment": "ICML 2024. Website: https://dynalang.github.io/",
  "topics": [
   "world-models"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2308.01399v2",
  "pdf_url": "https://arxiv.org/pdf/2308.01399v2",
  "html_url": "https://arxiv.org/html/2308.01399v2",
  "code_url": "https://dynalang.github.io/",
  "presented": [
   {
    "club": "saturday-robotics",
    "club_name": "Saturday Robotics",
    "session_number": 7,
    "session_title": "Robotics & World Models Reading Club 07: Learning to Dream: World Models, Imagination, Path to Foundation Models for Control \u2014 Los Altos",
    "date_text": "Saturday, May 9, 2026 | 2:00 PM \u2013 5:00 PM",
    "city": "",
    "url": "https://lu.ma/srhe0vuo",
    "listed_as": "Dynalang (2023)"
   }
  ],
  "club_note": "Adds language conditioning to world models",
  "featured": true,
  "signal": 7.45
 },
 {
  "id": "2307.15818",
  "slug": "rt-2-vision-language-action-models-transfer-web-knowledge-to-robotic-c",
  "title": "RT-2: Vision-Language-Action Models Transfer Web Knowledge to Robotic Control",
  "abstract": "We study how vision-language models trained on Internet-scale data can be incorporated directly into end-to-end robotic control to boost generalization and enable emergent semantic reasoning. Our goal is to enable a single end-to-end trained model to both learn to map robot observations to actions and enjoy the benefits of large-scale pretraining on language and vision-language data from the web. To this end, we propose to co-fine-tune state-of-the-art vision-language models on both robotic trajectory data and Internet-scale vision-language tasks, such as visual question answering. In contrast to other approaches, we propose a simple, general recipe to achieve this goal: in order to fit both natural language responses and robotic actions into the same format, we express the actions as text tokens and incorporate them directly into the training set of the model in the same way as natural language tokens. We refer to such category of models as vision-language-action models (VLA) and instantiate an example of such a model, which we call RT-2. Our extensive evaluation (6k evaluation trials) shows that our approach leads to performant robotic policies and enables RT-2 to obtain a range of emergent capabilities from Internet-scale training. This includes significantly improved generalization to novel objects, the ability to interpret commands not present in the robot training data (such as placing an object onto a particular number or icon), and the ability to perform rudimentary reasoning in response to user commands (such as picking up the smallest or largest object, or the one closest to another object). We further show that incorporating chain of thought reasoning allows RT-2 to perform multi-stage semantic reasoning, for example figuring out which object to pick up for use as an improvised hammer (a rock), or which type of drink is best suited for someone who is tired (an energy drink).",
  "published": "2023-07-28",
  "updated": "2023-07-28",
  "year": "2023",
  "authors": [
   "Anthony Brohan",
   "Noah Brown",
   "Justice Carbajal",
   "Yevgen Chebotar",
   "Xi Chen",
   "Krzysztof Choromanski",
   "Tianli Ding",
   "Danny Driess",
   "Avinava Dubey",
   "Chelsea Finn",
   "Pete Florence",
   "Chuyuan Fu",
   "Montse Gonzalez Arenas",
   "Keerthana Gopalakrishnan",
   "Kehang Han",
   "Karol Hausman",
   "Alexander Herzog",
   "Jasmine Hsu",
   "Brian Ichter",
   "Alex Irpan",
   "Nikhil Joshi",
   "Ryan Julian",
   "Dmitry Kalashnikov",
   "Yuheng Kuang",
   "Isabel Leal",
   "Lisa Lee",
   "Tsang-Wei Edward Lee",
   "Sergey Levine",
   "Yao Lu",
   "Henryk Michalewski",
   "Igor Mordatch",
   "Karl Pertsch",
   "Kanishka Rao",
   "Krista Reymann",
   "Michael Ryoo",
   "Grecia Salazar",
   "Pannag Sanketi",
   "Pierre Sermanet",
   "Jaspiar Singh",
   "Anikait Singh",
   "Radu Soricut",
   "Huong Tran",
   "Vincent Vanhoucke",
   "Quan Vuong",
   "Ayzaan Wahid",
   "Stefan Welker",
   "Paul Wohlhart",
   "Jialin Wu",
   "Fei Xia",
   "Ted Xiao",
   "Peng Xu",
   "Sichun Xu",
   "Tianhe Yu",
   "Brianna Zitkovich"
  ],
  "author_count": 54,
  "categories": [
   "cs.RO",
   "cs.CL",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 3982,
  "influential_citations": 225,
  "tldr": "This work proposes a simple, general recipe to enable a single end-to-end trained model to both learn to map robot observations to actions and enjoy the benefits of large-scale pretraining on language and vision-language data from the web.",
  "doi": "10.48550/arXiv.2307.15818",
  "oa_pdf": "https://arxiv.org/pdf/2307.15818",
  "s2_authors": [
   {
    "name": "Anthony Brohan",
    "id": "118025075",
    "h_index": 12,
    "papers": 19
   },
   {
    "name": "Noah Brown",
    "id": "2161343011",
    "h_index": 10,
    "papers": 16
   },
   {
    "name": "Justice Carbajal",
    "id": "2196517336",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Yevgen Chebotar",
    "id": "2527420",
    "h_index": 33,
    "papers": 57
   },
   {
    "name": "K. Choromanski",
    "id": "1805203",
    "h_index": 34,
    "papers": 132
   },
   {
    "name": "Tianli Ding",
    "id": "95691186",
    "h_index": 15,
    "papers": 24
   },
   {
    "name": "Danny Driess",
    "id": "2283848260",
    "h_index": 27,
    "papers": 35
   },
   {
    "name": "Kumar Avinava Dubey",
    "id": "89890133",
    "h_index": 23,
    "papers": 80
   },
   {
    "name": "Chelsea Finn",
    "id": "46881670",
    "h_index": 94,
    "papers": 240
   },
   {
    "name": "Peter R. Florence",
    "id": "47686265",
    "h_index": 30,
    "papers": 36
   },
   {
    "name": "Chuyuan Fu",
    "id": "3430433",
    "h_index": 15,
    "papers": 21
   },
   {
    "name": "Montse Gonzalez Arenas",
    "id": "153134021",
    "h_index": 13,
    "papers": 17
   },
   {
    "name": "K. Gopalakrishnan",
    "id": "2161342233",
    "h_index": 17,
    "papers": 25
   },
   {
    "name": "Kehang Han",
    "id": "2273880591",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Karol Hausman",
    "id": "1944801",
    "h_index": 47,
    "papers": 122
   },
   {
    "name": "Alexander Herzog",
    "id": "1505793452",
    "h_index": 22,
    "papers": 35
   },
   {
    "name": "Jasmine Hsu",
    "id": "2726592",
    "h_index": 15,
    "papers": 20
   },
   {
    "name": "Brian Ichter",
    "id": "2704814",
    "h_index": 37,
    "papers": 60
   },
   {
    "name": "A. Irpan",
    "id": "17818078",
    "h_index": 22,
    "papers": 32
   },
   {
    "name": "Nikhil J. Joshi",
    "id": "2052368480",
    "h_index": 11,
    "papers": 14
   },
   {
    "name": "Ryan C. Julian",
    "id": "144885996",
    "h_index": 19,
    "papers": 34
   },
   {
    "name": "Dmitry Kalashnikov",
    "id": "48313860",
    "h_index": 21,
    "papers": 34
   },
   {
    "name": "Yuheng Kuang",
    "id": "2161342687",
    "h_index": 9,
    "papers": 14
   },
   {
    "name": "Isabel Leal",
    "id": "2057988112",
    "h_index": 12,
    "papers": 15
   },
   {
    "name": "S. Levine",
    "id": "1736651",
    "h_index": 167,
    "papers": 551
   },
   {
    "name": "H. Michalewski",
    "id": "47407464",
    "h_index": 25,
    "papers": 108
   },
   {
    "name": "Igor Mordatch",
    "id": "2080746",
    "h_index": 34,
    "papers": 49
   },
   {
    "name": "Karl Pertsch",
    "id": "31719101",
    "h_index": 32,
    "papers": 56
   },
   {
    "name": "Kanishka Rao",
    "id": "2251957",
    "h_index": 31,
    "papers": 42
   },
   {
    "name": "Krista Reymann",
    "id": "2163522073",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "M. Ryoo",
    "id": "1766489",
    "h_index": 50,
    "papers": 164
   },
   {
    "name": "Grecia Salazar",
    "id": "2196524735",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Pannag R. Sanketi",
    "id": "2840758",
    "h_index": 22,
    "papers": 44
   },
   {
    "name": "P. Sermanet",
    "id": "3142556",
    "h_index": 39,
    "papers": 77
   },
   {
    "name": "Jaspiar Singh",
    "id": "2196040785",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Anikait Singh",
    "id": "2111007256",
    "h_index": 16,
    "papers": 26
   },
   {
    "name": "Radu Soricut",
    "id": "1737285",
    "h_index": 40,
    "papers": 104
   },
   {
    "name": "Huong Tran",
    "id": "2195355151",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Vincent Vanhoucke",
    "id": "2657155",
    "h_index": 34,
    "papers": 60
   },
   {
    "name": "Q. Vuong",
    "id": "144579461",
    "h_index": 23,
    "papers": 40
   },
   {
    "name": "Ayzaan Wahid",
    "id": "88728227",
    "h_index": 21,
    "papers": 27
   },
   {
    "name": "Stefan Welker",
    "id": "69426588",
    "h_index": 13,
    "papers": 17
   },
   {
    "name": "Paul Wohlhart",
    "id": "3202367",
    "h_index": 24,
    "papers": 47
   },
   {
    "name": "Ted Xiao",
    "id": "9961095",
    "h_index": 33,
    "papers": 45
   },
   {
    "name": "Tianhe Yu",
    "id": "10909315",
    "h_index": 31,
    "papers": 48
   },
   {
    "name": "Brianna Zitkovich",
    "id": "2196524598",
    "h_index": 4,
    "papers": 5
   }
  ],
  "comment": "Website: https://robotics-transformer.github.io/",
  "topics": [
   "vla",
   "foundation-pretraining",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2307.15818v1",
  "pdf_url": "https://arxiv.org/pdf/2307.15818v1",
  "html_url": "https://arxiv.org/html/2307.15818v1",
  "code_url": "https://robotics-transformer.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.5
 },
 {
  "id": "2307.09955",
  "slug": "xskill-cross-embodiment-skill-discovery",
  "title": "XSkill: Cross Embodiment Skill Discovery",
  "abstract": "Human demonstration videos are a widely available data source for robot learning and an intuitive user interface for expressing desired behavior. However, directly extracting reusable robot manipulation skills from unstructured human videos is challenging due to the big embodiment difference and unobserved action parameters. To bridge this embodiment gap, this paper introduces XSkill, an imitation learning framework that 1) discovers a cross-embodiment representation called skill prototypes purely from unlabeled human and robot manipulation videos, 2) transfers the skill representation to robot actions using conditional diffusion policy, and finally, 3) composes the learned skill to accomplish unseen tasks specified by a human prompt video. Our experiments in simulation and real-world environments show that the discovered skill prototypes facilitate both skill transfer and composition for unseen tasks, resulting in a more general and scalable imitation learning framework. The benchmark, code, and qualitative results are on https://xskill.cs.columbia.edu/",
  "published": "2023-07-19",
  "updated": "2023-09-28",
  "year": "2023",
  "authors": [
   "Mengda Xu",
   "Zhenjia Xu",
   "Cheng Chi",
   "Manuela Veloso",
   "Shuran Song"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 125,
  "influential_citations": 7,
  "tldr": "XSkill is introduced, an imitation learning framework that discovers a cross-embodiment representation called skill prototypes purely from unlabeled human and robot manipulation videos, transfers the skill representation to robot actions using conditional diffusion policy, and composes the learned skill to accomplish unseen tasks specified by a human prompt video.",
  "doi": "10.48550/arXiv.2307.09955",
  "oa_pdf": "https://arxiv.org/pdf/2307.09955",
  "s2_authors": [
   {
    "name": "Mengda Xu",
    "id": "2110683030",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "Zhenjia Xu",
    "id": "74498275",
    "h_index": 15,
    "papers": 22
   },
   {
    "name": "Cheng Chi",
    "id": "46859937",
    "h_index": 17,
    "papers": 83
   },
   {
    "name": "M. Veloso",
    "id": "1956361",
    "h_index": 81,
    "papers": 776
   },
   {
    "name": "Shuran Song",
    "id": "3340170",
    "h_index": 59,
    "papers": 90
   }
  ],
  "comment": "",
  "topics": [
   "egocentric-data",
   "imitation-diffusion",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2307.09955v2",
  "pdf_url": "https://arxiv.org/pdf/2307.09955v2",
  "html_url": "https://arxiv.org/html/2307.09955v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.6
 },
 {
  "id": "2307.09288",
  "slug": "llama-2-open-foundation-and-fine-tuned-chat-models",
  "title": "Llama 2: Open Foundation and Fine-Tuned Chat Models",
  "abstract": "In this work, we develop and release Llama 2, a collection of pretrained and fine-tuned large language models (LLMs) ranging in scale from 7 billion to 70 billion parameters. Our fine-tuned LLMs, called Llama 2-Chat, are optimized for dialogue use cases. Our models outperform open-source chat models on most benchmarks we tested, and based on our human evaluations for helpfulness and safety, may be a suitable substitute for closed-source models. We provide a detailed description of our approach to fine-tuning and safety improvements of Llama 2-Chat in order to enable the community to build on our work and contribute to the responsible development of LLMs.",
  "published": "2023-07-18",
  "updated": "2023-07-19",
  "year": "2023",
  "authors": [
   "Hugo Touvron",
   "Louis Martin",
   "Kevin Stone",
   "Peter Albert",
   "Amjad Almahairi",
   "Yasmine Babaei",
   "Nikolay Bashlykov",
   "Soumya Batra",
   "Prajjwal Bhargava",
   "Shruti Bhosale",
   "Dan Bikel",
   "Lukas Blecher",
   "Cristian Canton Ferrer",
   "Moya Chen",
   "Guillem Cucurull",
   "David Esiobu",
   "Jude Fernandes",
   "Jeremy Fu",
   "Wenyin Fu",
   "Brian Fuller",
   "Cynthia Gao",
   "Vedanuj Goswami",
   "Naman Goyal",
   "Anthony Hartshorn",
   "Saghar Hosseini",
   "Rui Hou",
   "Hakan Inan",
   "Marcin Kardas",
   "Viktor Kerkez",
   "Madian Khabsa",
   "Isabel Kloumann",
   "Artem Korenev",
   "Punit Singh Koura",
   "Marie-Anne Lachaux",
   "Thibaut Lavril",
   "Jenya Lee",
   "Diana Liskovich",
   "Yinghai Lu",
   "Yuning Mao",
   "Xavier Martinet",
   "Todor Mihaylov",
   "Pushkar Mishra",
   "Igor Molybog",
   "Yixin Nie",
   "Andrew Poulton",
   "Jeremy Reizenstein",
   "Rashi Rungta",
   "Kalyan Saladi",
   "Alan Schelten",
   "Ruan Silva",
   "Eric Michael Smith",
   "Ranjan Subramanian",
   "Xiaoqing Ellen Tan",
   "Binh Tang",
   "Ross Taylor",
   "Adina Williams",
   "Jian Xiang Kuan",
   "Puxin Xu",
   "Zheng Yan",
   "Iliyan Zarov",
   "Yuchen Zhang",
   "Angela Fan",
   "Melanie Kambadur",
   "Sharan Narang",
   "Aurelien Rodriguez",
   "Robert Stojnic",
   "Sergey Edunov",
   "Thomas Scialom"
  ],
  "author_count": 68,
  "categories": [
   "cs.CL",
   "cs.AI"
  ],
  "primary_category": "cs.CL",
  "venue": "",
  "venue_source": "",
  "citations": 18023,
  "influential_citations": 2246,
  "tldr": "This work develops and releases Llama 2, a collection of pretrained and fine-tuned large language models (LLMs) ranging in scale from 7 billion to 70 billion parameters, which may be a suitable substitute for closed-source models.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hugo Touvron",
    "id": "2113243762",
    "h_index": 17,
    "papers": 32
   },
   {
    "name": "Louis Martin",
    "id": "143792623",
    "h_index": 11,
    "papers": 26
   },
   {
    "name": "Kevin R. Stone",
    "id": "2059203763",
    "h_index": 8,
    "papers": 11
   },
   {
    "name": "Peter Albert",
    "id": "2214809450",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Amjad Almahairi",
    "id": "2634674",
    "h_index": 21,
    "papers": 56
   },
   {
    "name": "Yasmine Babaei",
    "id": "2223764353",
    "h_index": 10,
    "papers": 50
   },
   {
    "name": "Nikolay Bash-lykov",
    "id": "2223756247",
    "h_index": 11,
    "papers": 67
   },
   {
    "name": "Soumya Batra",
    "id": "47505161",
    "h_index": 12,
    "papers": 61
   },
   {
    "name": "Prajjwal Bhargava",
    "id": "51229603",
    "h_index": 13,
    "papers": 78
   },
   {
    "name": "Shruti Bhosale",
    "id": "2116473",
    "h_index": 22,
    "papers": 86
   },
   {
    "name": "D. Bikel",
    "id": "2023469",
    "h_index": 20,
    "papers": 69
   },
   {
    "name": "Lukas Blecher",
    "id": "2040305955",
    "h_index": 11,
    "papers": 63
   },
   {
    "name": "Cristian Canton Ferrer",
    "id": "66286536",
    "h_index": 15,
    "papers": 72
   },
   {
    "name": "Moya Chen",
    "id": "2108267192",
    "h_index": 11,
    "papers": 31
   },
   {
    "name": "Guillem Cucurull",
    "id": "7153363",
    "h_index": 15,
    "papers": 41
   },
   {
    "name": "David Esiobu",
    "id": "71039937",
    "h_index": 12,
    "papers": 66
   },
   {
    "name": "Jude Fernandes",
    "id": "2166312768",
    "h_index": 7,
    "papers": 25
   },
   {
    "name": "Jeremy Fu",
    "id": "2430465950",
    "h_index": 10,
    "papers": 65
   },
   {
    "name": "Wenyin Fu",
    "id": "2223742000",
    "h_index": 9,
    "papers": 59
   },
   {
    "name": "Brian Fuller",
    "id": "2223748737",
    "h_index": 8,
    "papers": 31
   },
   {
    "name": "Cynthia Gao",
    "id": "2107063269",
    "h_index": 16,
    "papers": 24
   },
   {
    "name": "Vedanuj Goswami",
    "id": "28554843",
    "h_index": 22,
    "papers": 88
   },
   {
    "name": "Naman Goyal",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "A. Hartshorn",
    "id": "2325255500",
    "h_index": 26,
    "papers": 101
   },
   {
    "name": "Saghar Hosseini",
    "id": "2195458",
    "h_index": 17,
    "papers": 32
   },
   {
    "name": "Rui Hou",
    "id": "2132302721",
    "h_index": 8,
    "papers": 13
   },
   {
    "name": "Hakan Inan",
    "id": "2065277797",
    "h_index": 9,
    "papers": 29
   },
   {
    "name": "Marcin Kardas",
    "id": "2059886128",
    "h_index": 10,
    "papers": 67
   },
   {
    "name": "Viktor Kerkez",
    "id": "2190957318",
    "h_index": 9,
    "papers": 60
   },
   {
    "name": "Madian Khabsa",
    "id": "2072010",
    "h_index": 30,
    "papers": 137
   },
   {
    "name": "Isabel M. Kloumann",
    "id": "2207049",
    "h_index": 18,
    "papers": 56
   },
   {
    "name": "A. Korenev",
    "id": "2294453195",
    "h_index": 9,
    "papers": 51
   },
   {
    "name": "Punit Singh Koura",
    "id": "2146367061",
    "h_index": 10,
    "papers": 14
   },
   {
    "name": "M. Lachaux",
    "id": "114952298",
    "h_index": 16,
    "papers": 60
   },
   {
    "name": "Thibaut Lavril",
    "id": "46183616",
    "h_index": 20,
    "papers": 69
   },
   {
    "name": "Jenya Lee",
    "id": "2223749565",
    "h_index": 9,
    "papers": 61
   },
   {
    "name": "Diana Liskovich",
    "id": "2145259939",
    "h_index": 9,
    "papers": 36
   },
   {
    "name": "Yinghai Lu",
    "id": "1768032",
    "h_index": 9,
    "papers": 17
   },
   {
    "name": "Yuning Mao",
    "id": "3375249",
    "h_index": 22,
    "papers": 40
   },
   {
    "name": "X. Martinet",
    "id": "1490887583",
    "h_index": 10,
    "papers": 45
   },
   {
    "name": "Todor Mihaylov",
    "id": "39980906",
    "h_index": 26,
    "papers": 87
   },
   {
    "name": "Pushkar Mishra",
    "id": "3047561",
    "h_index": 19,
    "papers": 38
   },
   {
    "name": "Igor Molybog",
    "id": "2322981055",
    "h_index": 8,
    "papers": 51
   },
   {
    "name": "Yixin Nie",
    "id": "40383658",
    "h_index": 16,
    "papers": 24
   },
   {
    "name": "Andrew Poulton",
    "id": "38579672",
    "h_index": 10,
    "papers": 20
   },
   {
    "name": "J. Reizenstein",
    "id": "39906022",
    "h_index": 14,
    "papers": 60
   },
   {
    "name": "Rashi Rungta",
    "id": "150282885",
    "h_index": 8,
    "papers": 33
   },
   {
    "name": "Kalyan Saladi",
    "id": "1859294",
    "h_index": 7,
    "papers": 26
   },
   {
    "name": "A. Schelten",
    "id": "14279694",
    "h_index": 11,
    "papers": 59
   },
   {
    "name": "Ruan Silva",
    "id": "2214818043",
    "h_index": 9,
    "papers": 54
   },
   {
    "name": "Eric Michael Smith",
    "id": "51324296",
    "h_index": 18,
    "papers": 25
   },
   {
    "name": "R. Subramanian",
    "id": "2066074360",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Xia Tan",
    "id": "2112782199",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Binh Tang",
    "id": "71292072",
    "h_index": 8,
    "papers": 11
   },
   {
    "name": "Ross Taylor",
    "id": "2110697298",
    "h_index": 9,
    "papers": 51
   },
   {
    "name": "Adina Williams",
    "id": "2110032535",
    "h_index": 15,
    "papers": 19
   },
   {
    "name": "Jian Xiang Kuan",
    "id": "2223770369",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Puxin Xu",
    "id": "2214843767",
    "h_index": 11,
    "papers": 58
   },
   {
    "name": "Zhengxu Yan",
    "id": "14701107",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Iliyan Zarov",
    "id": "121929334",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Yuchen Zhang",
    "id": "2108473229",
    "h_index": 9,
    "papers": 41
   },
   {
    "name": "Angela Fan",
    "id": "144270981",
    "h_index": 37,
    "papers": 64
   },
   {
    "name": "M. Kambadur",
    "id": "2165660870",
    "h_index": 12,
    "papers": 41
   },
   {
    "name": "Sharan Narang",
    "id": "46617804",
    "h_index": 31,
    "papers": 105
   },
   {
    "name": "Aur'elien Rodriguez",
    "id": "2166043087",
    "h_index": 9,
    "papers": 25
   },
   {
    "name": "Robert Stojnic",
    "id": "1962768",
    "h_index": 13,
    "papers": 61
   },
   {
    "name": "Sergey Edunov",
    "id": "2068070",
    "h_index": 19,
    "papers": 48
   },
   {
    "name": "Thomas Scialom",
    "id": "2073456043",
    "h_index": 17,
    "papers": 56
   }
  ],
  "comment": "",
  "topics": [
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2307.09288v2",
  "pdf_url": "https://arxiv.org/pdf/2307.09288v2",
  "html_url": "https://arxiv.org/html/2307.09288v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.0
 },
 {
  "id": "2307.07635",
  "slug": "cotracker-it-is-better-to-track-together",
  "title": "CoTracker: It is Better to Track Together",
  "abstract": "We introduce CoTracker, a transformer-based model that tracks a large number of 2D points in long video sequences. Differently from most existing approaches that track points independently, CoTracker tracks them jointly, accounting for their dependencies. We show that joint tracking significantly improves tracking accuracy and robustness, and allows CoTracker to track occluded points and points outside of the camera view. We also introduce several innovations for this class of trackers, including using token proxies that significantly improve memory efficiency and allow CoTracker to track 70k points jointly and simultaneously at inference on a single GPU. CoTracker is an online algorithm that operates causally on short windows. However, it is trained utilizing unrolled windows as a recurrent network, maintaining tracks for long periods of time even when points are occluded or leave the field of view. Quantitatively, CoTracker substantially outperforms prior trackers on standard point-tracking benchmarks.",
  "published": "2023-07-14",
  "updated": "2024-10-01",
  "year": "2023",
  "authors": [
   "Nikita Karaev",
   "Ignacio Rocco",
   "Benjamin Graham",
   "Natalia Neverova",
   "Andrea Vedaldi",
   "Christian Rupprecht"
  ],
  "author_count": 6,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "ECCV",
  "venue_source": "semantic-scholar",
  "citations": 642,
  "influential_citations": 98,
  "tldr": "It is shown that joint tracking significantly improves tracking accuracy and robustness, and allows CoTracker to track occluded points and points outside of the camera view, and substantially outperforms prior trackers on standard point-tracking benchmarks.",
  "doi": "10.48550/arXiv.2307.07635",
  "oa_pdf": "https://arxiv.org/pdf/2307.07635",
  "s2_authors": [
   {
    "name": "Nikita Karaev",
    "id": "16643610",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "Ignacio Rocco",
    "id": "2065198709",
    "h_index": 8,
    "papers": 11
   },
   {
    "name": "Benjamin Graham",
    "id": "143853801",
    "h_index": 19,
    "papers": 35
   },
   {
    "name": "N. Neverova",
    "id": "2759569",
    "h_index": 29,
    "papers": 56
   },
   {
    "name": "A. Vedaldi",
    "id": "1687524",
    "h_index": 108,
    "papers": 290
   },
   {
    "name": "C. Rupprecht",
    "id": "49359942",
    "h_index": 39,
    "papers": 76
   }
  ],
  "comment": "Code and model weights are available at: https://co-tracker.github.io/",
  "topics": [
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2307.07635v3",
  "pdf_url": "https://arxiv.org/pdf/2307.07635v3",
  "html_url": "https://arxiv.org/html/2307.07635v3",
  "code_url": "https://co-tracker.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.31
 },
 {
  "id": "2307.07358",
  "slug": "learn-from-incomplete-tactile-data-tactile-representation-learning-wit",
  "title": "Learn from Incomplete Tactile Data: Tactile Representation Learning with Masked Autoencoders",
  "abstract": "The missing signal caused by the objects being occluded or an unstable sensor is a common challenge during data collection. Such missing signals will adversely affect the results obtained from the data, and this issue is observed more frequently in robotic tactile perception. In tactile perception, due to the limited working space and the dynamic environment, the contact between the tactile sensor and the object is frequently insufficient and unstable, which causes the partial loss of signals, thus leading to incomplete tactile data. The tactile data will therefore contain fewer tactile cues with low information density. In this paper, we propose a tactile representation learning method, named TacMAE, based on Masked Autoencoder to address the problem of incomplete tactile data in tactile perception. In our framework, a portion of the tactile image is masked out to simulate the missing contact region. By reconstructing the missing signals in the tactile image, the trained model can achieve a high-level understanding of surface geometry and tactile properties from limited tactile cues. The experimental results of tactile texture recognition show that our proposed TacMAE can achieve a high recognition accuracy of 71.4% in the zero-shot transfer and 85.8% after fine-tuning, which are 15.2% and 8.2% higher than the results without using masked modeling. The extensive experiments on YCB objects demonstrate the knowledge transferability of our proposed method and the potential to improve efficiency in tactile exploration.",
  "published": "2023-07-14",
  "updated": "2023-07-14",
  "year": "2023",
  "authors": [
   "Guanqun Cao",
   "Jiaqi Jiang",
   "Danushka Bollegala",
   "Shan Luo"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS",
  "venue_source": "semantic-scholar",
  "citations": 19,
  "influential_citations": 0,
  "tldr": "This paper proposes a tactile representation learning method, named TacMAE, based on Masked Autoencoder to address the problem of incomplete tactile data in tactile perception, and demonstrates the knowledge transferability of the proposed method and the potential to improve efficiency in tactile exploration.",
  "doi": "10.1109/IROS55552.2023.10341788",
  "oa_pdf": "https://arxiv.org/pdf/2307.07358",
  "s2_authors": [
   {
    "name": "Guanqun Cao",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Jiaqi Jiang",
    "id": "2149531489",
    "h_index": 11,
    "papers": 18
   },
   {
    "name": "D. Bollegala",
    "id": "2075356592",
    "h_index": 15,
    "papers": 78
   },
   {
    "name": "Shan Luo",
    "id": "145524951",
    "h_index": 26,
    "papers": 65
   }
  ],
  "comment": "This paper is accepted at IROS 2023",
  "topics": [
   "tactile",
   "navigation",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2307.07358v1",
  "pdf_url": "https://arxiv.org/pdf/2307.07358v1",
  "html_url": "https://arxiv.org/html/2307.07358v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.8
 },
 {
  "id": "2307.05973",
  "slug": "voxposer-composable-3d-value-maps-for-robotic-manipulation-with-langua",
  "title": "VoxPoser: Composable 3D Value Maps for Robotic Manipulation with Language Models",
  "abstract": "Large language models (LLMs) are shown to possess a wealth of actionable knowledge that can be extracted for robot manipulation in the form of reasoning and planning. Despite the progress, most still rely on pre-defined motion primitives to carry out the physical interactions with the environment, which remains a major bottleneck. In this work, we aim to synthesize robot trajectories, i.e., a dense sequence of 6-DoF end-effector waypoints, for a large variety of manipulation tasks given an open-set of instructions and an open-set of objects. We achieve this by first observing that LLMs excel at inferring affordances and constraints given a free-form language instruction. More importantly, by leveraging their code-writing capabilities, they can interact with a vision-language model (VLM) to compose 3D value maps to ground the knowledge into the observation space of the agent. The composed value maps are then used in a model-based planning framework to zero-shot synthesize closed-loop robot trajectories with robustness to dynamic perturbations. We further demonstrate how the proposed framework can benefit from online experiences by efficiently learning a dynamics model for scenes that involve contact-rich interactions. We present a large-scale study of the proposed method in both simulated and real-robot environments, showcasing the ability to perform a large variety of everyday manipulation tasks specified in free-form natural language. Videos and code at https://voxposer.github.io",
  "published": "2023-07-12",
  "updated": "2023-11-02",
  "year": "2023",
  "authors": [
   "Wenlong Huang",
   "Chen Wang",
   "Ruohan Zhang",
   "Yunzhu Li",
   "Jiajun Wu",
   "Li Fei-Fei"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CL",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 1070,
  "influential_citations": 78,
  "tldr": "A large-scale study of the proposed framework to synthesize closed-loop robot trajectories with robustness to dynamic perturbations is presented, showcasing the ability to perform a large variety of everyday manipulation tasks specified in free-form natural language.",
  "doi": "10.48550/arXiv.2307.05973",
  "oa_pdf": "https://arxiv.org/pdf/2307.05973",
  "s2_authors": [
   {
    "name": "Wenlong Huang",
    "id": "2158105356",
    "h_index": 10,
    "papers": 11
   },
   {
    "name": "Chen Wang",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Ruohan Zhang",
    "id": "2657185",
    "h_index": 16,
    "papers": 46
   },
   {
    "name": "Yunzhu Li",
    "id": "3422021",
    "h_index": 25,
    "papers": 63
   },
   {
    "name": "Jiajun Wu",
    "id": "3045089",
    "h_index": 80,
    "papers": 228
   },
   {
    "name": "Li Fei-Fei",
    "id": "48004138",
    "h_index": 143,
    "papers": 606
   }
  ],
  "comment": "",
  "topics": [
   "tactile",
   "video-generation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2307.05973v2",
  "pdf_url": "https://arxiv.org/pdf/2307.05973v2",
  "html_url": "https://arxiv.org/html/2307.05973v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "2307.05663",
  "slug": "objaverse-xl-a-universe-of-10m-3d-objects",
  "title": "Objaverse-XL: A Universe of 10M+ 3D Objects",
  "abstract": "Natural language processing and 2D vision models have attained remarkable proficiency on many tasks primarily by escalating the scale of training data. However, 3D vision tasks have not seen the same progress, in part due to the challenges of acquiring high-quality 3D data. In this work, we present Objaverse-XL, a dataset of over 10 million 3D objects. Our dataset comprises deduplicated 3D objects from a diverse set of sources, including manually designed objects, photogrammetry scans of landmarks and everyday items, and professional scans of historic and antique artifacts. Representing the largest scale and diversity in the realm of 3D datasets, Objaverse-XL enables significant new possibilities for 3D vision. Our experiments demonstrate the improvements enabled with the scale provided by Objaverse-XL. We show that by training Zero123 on novel view synthesis, utilizing over 100 million multi-view rendered images, we achieve strong zero-shot generalization abilities. We hope that releasing Objaverse-XL will enable further innovations in the field of 3D vision at scale.",
  "published": "2023-07-11",
  "updated": "2023-07-11",
  "year": "2023",
  "authors": [
   "Matt Deitke",
   "Ruoshi Liu",
   "Matthew Wallingford",
   "Huong Ngo",
   "Oscar Michel",
   "Aditya Kusupati",
   "Alan Fan",
   "Christian Laforte",
   "Vikram Voleti",
   "Samir Yitzhak Gadre",
   "Eli VanderBilt",
   "Aniruddha Kembhavi",
   "Carl Vondrick",
   "Georgia Gkioxari",
   "Kiana Ehsani",
   "Ludwig Schmidt",
   "Ali Farhadi"
  ],
  "author_count": 17,
  "categories": [
   "cs.CV",
   "cs.AI"
  ],
  "primary_category": "cs.CV",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 896,
  "influential_citations": 75,
  "tldr": "This work presents Objaverse-XL, a dataset of over 10 million 3D objects that represents the largest scale and diversity in the realm of 3D datasets, and achieves strong zero-shot generalization abilities by training Zero123 on novel view synthesis.",
  "doi": "10.48550/arXiv.2307.05663",
  "oa_pdf": "https://arxiv.org/pdf/2307.05663",
  "s2_authors": [
   {
    "name": "Matt Deitke",
    "id": "1632916259",
    "h_index": 11,
    "papers": 14
   },
   {
    "name": "Ruoshi Liu",
    "id": "2143183492",
    "h_index": 14,
    "papers": 24
   },
   {
    "name": "Matthew Wallingford",
    "id": "1632957174",
    "h_index": 9,
    "papers": 21
   },
   {
    "name": "Huong Ngo",
    "id": "2223118889",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Oscar Michel",
    "id": "2196005933",
    "h_index": 9,
    "papers": 13
   },
   {
    "name": "Aditya Kusupati",
    "id": "52207562",
    "h_index": 14,
    "papers": 34
   },
   {
    "name": "Alan Fan",
    "id": "2219305108",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Christian Laforte",
    "id": "2223133147",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Vikram S. Voleti",
    "id": "2961618",
    "h_index": 17,
    "papers": 45
   },
   {
    "name": "S. Gadre",
    "id": "1387466862",
    "h_index": 15,
    "papers": 22
   },
   {
    "name": "Eli VanderBilt",
    "id": "1632920625",
    "h_index": 13,
    "papers": 19
   },
   {
    "name": "Aniruddha Kembhavi",
    "id": "2684226",
    "h_index": 49,
    "papers": 119
   },
   {
    "name": "Carl Vondrick",
    "id": "1856025",
    "h_index": 47,
    "papers": 116
   },
   {
    "name": "Georgia Gkioxari",
    "id": "2397488692",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Kiana Ehsani",
    "id": "2883417",
    "h_index": 26,
    "papers": 42
   },
   {
    "name": "Ludwig Schmidt",
    "id": "152772922",
    "h_index": 48,
    "papers": 84
   },
   {
    "name": "Ali Farhadi",
    "id": "143787583",
    "h_index": 78,
    "papers": 208
   }
  ],
  "comment": "",
  "topics": [
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2307.05663v1",
  "pdf_url": "https://arxiv.org/pdf/2307.05663v1",
  "html_url": "https://arxiv.org/html/2307.05663v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.45
 },
 {
  "id": "2307.05222",
  "slug": "emu-generative-pretraining-in-multimodality",
  "title": "Emu: Generative Pretraining in Multimodality",
  "abstract": "We present Emu, a Transformer-based multimodal foundation model, which can seamlessly generate images and texts in multimodal context. This omnivore model can take in any single-modality or multimodal data input indiscriminately (e.g., interleaved image, text and video) through a one-model-for-all autoregressive training process. First, visual signals are encoded into embeddings, and together with text tokens form an interleaved input sequence. Emu is then end-to-end trained with a unified objective of classifying the next text token or regressing the next visual embedding in the multimodal sequence. This versatile multimodality empowers the exploration of diverse pretraining data sources at scale, such as videos with interleaved frames and text, webpages with interleaved images and text, as well as web-scale image-text pairs and video-text pairs. Emu can serve as a generalist multimodal interface for both image-to-text and text-to-image tasks, and supports in-context image and text generation. Across a broad range of zero-shot/few-shot tasks including image captioning, visual question answering, video question answering and text-to-image generation, Emu demonstrates superb performance compared to state-of-the-art large multimodal models. Extended capabilities such as multimodal assistants via instruction tuning are also demonstrated with impressive performance.",
  "published": "2023-07-11",
  "updated": "2024-05-08",
  "year": "2023",
  "authors": [
   "Quan Sun",
   "Qiying Yu",
   "Yufeng Cui",
   "Fan Zhang",
   "Xiaosong Zhang",
   "Yueze Wang",
   "Hongcheng Gao",
   "Jingjing Liu",
   "Tiejun Huang",
   "Xinlong Wang"
  ],
  "author_count": 10,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 167,
  "influential_citations": 21,
  "tldr": "Emu, a Transformer-based multimodal foundation model, is presented, which can seamlessly generate images and texts in multi-modality context through a one-model-for-all autoregressive training process and demonstrates superb performance compared to state-of-the-art large multimodAL models.",
  "doi": "10.48550/arXiv.2307.05222",
  "oa_pdf": "https://arxiv.org/pdf/2307.05222",
  "s2_authors": [
   {
    "name": "Quan Sun",
    "id": "2112490007",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Qiying Yu",
    "id": "23716915",
    "h_index": 10,
    "papers": 10
   },
   {
    "name": "Yufeng Cui",
    "id": "2149499588",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Fan Zhang",
    "id": "2162659995",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Xiaosong Zhang",
    "id": "2108056766",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Yueze Wang",
    "id": "2217456303",
    "h_index": 16,
    "papers": 22
   },
   {
    "name": "Hongcheng Gao",
    "id": "2162081759",
    "h_index": 20,
    "papers": 28
   },
   {
    "name": "Jingjing Liu",
    "id": "2222717281",
    "h_index": 6,
    "papers": 6
   },
   {
    "name": "Tiejun Huang",
    "id": "34097174",
    "h_index": 61,
    "papers": 343
   },
   {
    "name": "Xinlong Wang",
    "id": "51316629",
    "h_index": 23,
    "papers": 42
   }
  ],
  "comment": "Accepted to ICLR 2024. Code and Models: https://github.com/baaivision/Emu",
  "topics": [
   "navigation",
   "foundation-pretraining",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2307.05222v2",
  "pdf_url": "https://arxiv.org/pdf/2307.05222v2",
  "html_url": "https://arxiv.org/html/2307.05222v2",
  "code_url": "https://github.com/baaivision/Emu",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.73
 },
 {
  "id": "2307.04577",
  "slug": "anyteleop-a-general-vision-based-dexterous-robot-arm-hand-teleoperatio",
  "title": "AnyTeleop: A General Vision-Based Dexterous Robot Arm-Hand Teleoperation System",
  "abstract": "Vision-based teleoperation offers the possibility to endow robots with human-level intelligence to physically interact with the environment, while only requiring low-cost camera sensors. However, current vision-based teleoperation systems are designed and engineered towards a particular robot model and deploy environment, which scales poorly as the pool of the robot models expands and the variety of the operating environment increases. In this paper, we propose AnyTeleop, a unified and general teleoperation system to support multiple different arms, hands, realities, and camera configurations within a single system. Although being designed to provide great flexibility to the choice of simulators and real hardware, our system can still achieve great performance. For real-world experiments, AnyTeleop can outperform a previous system that was designed for a specific robot hardware with a higher success rate, using the same robot. For teleoperation in simulation, AnyTeleop leads to better imitation learning performance, compared with a previous system that is particularly designed for that simulator. Project page: https://yzqin.github.io/anyteleop/.",
  "published": "2023-07-10",
  "updated": "2024-05-16",
  "year": "2023",
  "authors": [
   "Yuzhe Qin",
   "Wei Yang",
   "Binghao Huang",
   "Karl Van Wyk",
   "Hao Su",
   "Xiaolong Wang",
   "Yu-Wei Chao",
   "Dieter Fox"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "RSS",
  "venue_source": "semantic-scholar",
  "citations": 309,
  "influential_citations": 26,
  "tldr": "AnyTeleop is proposed, a unified and general teleoperation system to support multiple different arms, hands, realities, and camera configurations within a single system to support multiple different arms, hands, realities, and camera configurations within a single system.",
  "doi": "10.15607/RSS.2023.XIX.015",
  "oa_pdf": "https://doi.org/10.15607/rss.2023.xix.015",
  "s2_authors": [
   {
    "name": "Yuzhe Qin",
    "id": "12701031",
    "h_index": 24,
    "papers": 34
   },
   {
    "name": "Wei Yang",
    "id": "2150080732",
    "h_index": 24,
    "papers": 36
   },
   {
    "name": "Binghao Huang",
    "id": "2175672218",
    "h_index": 8,
    "papers": 8
   },
   {
    "name": "Karl Van Wyk",
    "id": "2423933",
    "h_index": 20,
    "papers": 54
   },
   {
    "name": "Hao Su",
    "id": "2087042750",
    "h_index": 21,
    "papers": 26
   },
   {
    "name": "Xiaolong Wang",
    "id": "2145748143",
    "h_index": 7,
    "papers": 8
   },
   {
    "name": "Yu-Wei Chao",
    "id": "2820136",
    "h_index": 25,
    "papers": 38
   },
   {
    "name": "D. Fox",
    "id": "145197953",
    "h_index": 133,
    "papers": 428
   }
  ],
  "comment": "http://anyteleop.com/ Robotics: Science and Systems 2023",
  "topics": [
   "dexterous-manipulation",
   "sim2real",
   "imitation-diffusion",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2307.04577v3",
  "pdf_url": "https://arxiv.org/pdf/2307.04577v3",
  "html_url": "https://arxiv.org/html/2307.04577v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.99
 },
 {
  "id": "2307.04011",
  "slug": "robust-learning-based-incipient-slip-detection-using-the-papillarray-o",
  "title": "Robust Learning-Based Incipient Slip Detection using the PapillArray Optical Tactile Sensor for Improved Robotic Gripping",
  "abstract": "The ability to detect slip, particularly incipient slip, enables robotic systems to take corrective measures to prevent a grasped object from being dropped. Therefore, slip detection can enhance the overall security of robotic gripping. However, accurately detecting incipient slip remains a significant challenge. In this paper, we propose a novel learning-based approach to detect incipient slip using the PapillArray (Contactile, Australia) tactile sensor. The resulting model is highly effective in identifying patterns associated with incipient slip, achieving a detection success rate of 95.6% when tested with an offline dataset. Furthermore, we introduce several data augmentation methods to enhance the robustness of our model. When transferring the trained model to a robotic gripping environment distinct from where the training data was collected, our model maintained robust performance, with a success rate of 96.8%, providing timely feedback for stabilizing several practical gripping tasks. Our project website: https://sites.google.com/view/incipient-slip-detection.",
  "published": "2023-07-08",
  "updated": "2023-07-08",
  "year": "2023",
  "authors": [
   "Qiang Wang",
   "Pablo Martinez Ulloa",
   "Robert Burke",
   "David Cordova Bulens",
   "Stephen J. Redmond"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "RA-L",
  "venue_source": "semantic-scholar",
  "citations": 10,
  "influential_citations": 0,
  "tldr": "A novel learning-based approach to detect incipient slip using the PapillArray (Contactile, Australia) tactile sensor is proposed, which is highly effective in identifying patterns associated with incipient slip, achieving a detection success rate of 95.6% when tested with an offline dataset.",
  "doi": "10.1109/LRA.2023.3347141",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Qiang Wang",
    "id": "2224668562",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Pablo Martinez Ulloa",
    "id": "2146114165",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "R. Burke",
    "id": "2056064553",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "David C\u00f3rdova Bulens",
    "id": "8399970",
    "h_index": 7,
    "papers": 19
   },
   {
    "name": "S. Redmond",
    "id": "144869332",
    "h_index": 39,
    "papers": 181
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2307.04011v1",
  "pdf_url": "https://arxiv.org/pdf/2307.04011v1",
  "html_url": "https://arxiv.org/html/2307.04011v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.54
 },
 {
  "id": "2307.03719",
  "slug": "polybot-training-one-policy-across-robots-while-embracing-variability",
  "title": "Polybot: Training One Policy Across Robots While Embracing Variability",
  "abstract": "Reusing large datasets is crucial to scale vision-based robotic manipulators to everyday scenarios due to the high cost of collecting robotic datasets. However, robotic platforms possess varying control schemes, camera viewpoints, kinematic configurations, and end-effector morphologies, posing significant challenges when transferring manipulation skills from one platform to another. To tackle this problem, we propose a set of key design decisions to train a single policy for deployment on multiple robotic platforms. Our framework first aligns the observation and action spaces of our policy across embodiments via utilizing wrist cameras and a unified, but modular codebase. To bridge the remaining domain shift, we align our policy's internal representations across embodiments through contrastive learning. We evaluate our method on a dataset collected over 60 hours spanning 6 tasks and 3 robots with varying joint configurations and sizes: the WidowX 250S, the Franka Emika Panda, and the Sawyer. Our results demonstrate significant improvements in success rate and sample efficiency for our policy when using new task data collected on a different robot, validating our proposed design decisions. More details and videos can be found on our anonymized project website: https://sites.google.com/view/polybot-multirobot",
  "published": "2023-07-07",
  "updated": "2023-07-07",
  "year": "2023",
  "authors": [
   "Jonathan Yang",
   "Dorsa Sadigh",
   "Chelsea Finn"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 53,
  "influential_citations": 0,
  "tldr": "This work proposes a set of key design decisions to train a single policy for deployment on multiple robotic platforms and aligns the observation and action spaces of the policy across embodiments via utilizing wrist cameras and a unified, but modular codebase.",
  "doi": "10.48550/arXiv.2307.03719",
  "oa_pdf": "https://arxiv.org/pdf/2307.03719",
  "s2_authors": [
   {
    "name": "Jonathan Yang",
    "id": "2143067490",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Dorsa Sadigh",
    "id": "1779671",
    "h_index": 69,
    "papers": 226
   },
   {
    "name": "Chelsea Finn",
    "id": "46881670",
    "h_index": 94,
    "papers": 240
   }
  ],
  "comment": "17 pages, 11 figures",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2307.03719v1",
  "pdf_url": "https://arxiv.org/pdf/2307.03719v1",
  "html_url": "https://arxiv.org/html/2307.03719v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.23
 },
 {
  "id": "2307.03659",
  "slug": "decomposing-the-generalization-gap-in-imitation-learning-for-visual-ro",
  "title": "Decomposing the Generalization Gap in Imitation Learning for Visual Robotic Manipulation",
  "abstract": "What makes generalization hard for imitation learning in visual robotic manipulation? This question is difficult to approach at face value, but the environment from the perspective of a robot can often be decomposed into enumerable factors of variation, such as the lighting conditions or the placement of the camera. Empirically, generalization to some of these factors have presented a greater obstacle than others, but existing work sheds little light on precisely how much each factor contributes to the generalization gap. Towards an answer to this question, we study imitation learning policies in simulation and on a real robot language-conditioned manipulation task to quantify the difficulty of generalization to different (sets of) factors. We also design a new simulated benchmark of 19 tasks with 11 factors of variation to facilitate more controlled evaluations of generalization. From our study, we determine an ordering of factors based on generalization difficulty, that is consistent across simulation and our real robot setup.",
  "published": "2023-07-07",
  "updated": "2023-07-07",
  "year": "2023",
  "authors": [
   "Annie Xie",
   "Lisa Lee",
   "Ted Xiao",
   "Chelsea Finn"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA",
  "venue_source": "semantic-scholar",
  "citations": 119,
  "influential_citations": 5,
  "tldr": "This work studies imitation learning policies in simulation and on a real robot language-conditioned manipulation task to quantify the difficulty of generalization to different (sets of) factors, and determines an ordering of factors based on generalization difficulty.",
  "doi": "10.1109/ICRA57147.2024.10611331",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Annie Xie",
    "id": "14484808",
    "h_index": 21,
    "papers": 23
   },
   {
    "name": "Lisa Lee",
    "id": "87068304",
    "h_index": 15,
    "papers": 25
   },
   {
    "name": "Ted Xiao",
    "id": "9961095",
    "h_index": 33,
    "papers": 45
   },
   {
    "name": "Chelsea Finn",
    "id": "46881670",
    "h_index": 94,
    "papers": 240
   }
  ],
  "comment": "Project webpage at https://sites.google.com/view/generalization-gap",
  "topics": [
   "vla",
   "imitation-diffusion"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2307.03659v1",
  "pdf_url": "https://arxiv.org/pdf/2307.03659v1",
  "html_url": "https://arxiv.org/html/2307.03659v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.58
 },
 {
  "id": "2307.01952",
  "slug": "sdxl-improving-latent-diffusion-models-for-high-resolution-image-synth",
  "title": "SDXL: Improving Latent Diffusion Models for High-Resolution Image Synthesis",
  "abstract": "We present SDXL, a latent diffusion model for text-to-image synthesis. Compared to previous versions of Stable Diffusion, SDXL leverages a three times larger UNet backbone: The increase of model parameters is mainly due to more attention blocks and a larger cross-attention context as SDXL uses a second text encoder. We design multiple novel conditioning schemes and train SDXL on multiple aspect ratios. We also introduce a refinement model which is used to improve the visual fidelity of samples generated by SDXL using a post-hoc image-to-image technique. We demonstrate that SDXL shows drastically improved performance compared the previous versions of Stable Diffusion and achieves results competitive with those of black-box state-of-the-art image generators. In the spirit of promoting open research and fostering transparency in large model training and evaluation, we provide access to code and model weights at https://github.com/Stability-AI/generative-models",
  "published": "2023-07-04",
  "updated": "2023-07-04",
  "year": "2023",
  "authors": [
   "Dustin Podell",
   "Zion English",
   "Kyle Lacey",
   "Andreas Blattmann",
   "Tim Dockhorn",
   "Jonas M\u00fcller",
   "Joe Penna",
   "Robin Rombach"
  ],
  "author_count": 8,
  "categories": [
   "cs.CV",
   "cs.AI"
  ],
  "primary_category": "cs.CV",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 5250,
  "influential_citations": 950,
  "tldr": "It is demonstrated that SDXL shows drastically improved performance compared the previous versions of Stable Diffusion and achieves results competitive with those of black-box state-of-the-art image generators.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Dustin Podell",
    "id": "2221125727",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Zion English",
    "id": "2221127565",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Kyle Lacey",
    "id": "2221126982",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "A. Blattmann",
    "id": "119843260",
    "h_index": 17,
    "papers": 25
   },
   {
    "name": "Tim Dockhorn",
    "id": "102541178",
    "h_index": 13,
    "papers": 17
   },
   {
    "name": "Jonas Muller",
    "id": "2188737195",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Joe Penna",
    "id": "2215904682",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Robin Rombach",
    "id": "1660819540",
    "h_index": 22,
    "papers": 29
   }
  ],
  "comment": "",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2307.01952v1",
  "pdf_url": "https://arxiv.org/pdf/2307.01952v1",
  "html_url": "https://arxiv.org/html/2307.01952v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "2307.00595",
  "slug": "rh20t-a-comprehensive-robotic-dataset-for-learning-diverse-skills-in-o",
  "title": "RH20T: A Comprehensive Robotic Dataset for Learning Diverse Skills in One-Shot",
  "abstract": "A key challenge in robotic manipulation in open domains is how to acquire diverse and generalizable skills for robots. Recent research in one-shot imitation learning has shown promise in transferring trained policies to new tasks based on demonstrations. This feature is attractive for enabling robots to acquire new skills and improving task and motion planning. However, due to limitations in the training dataset, the current focus of the community has mainly been on simple cases, such as push or pick-place tasks, relying solely on visual guidance. In reality, there are many complex skills, some of which may even require both visual and tactile perception to solve. This paper aims to unlock the potential for an agent to generalize to hundreds of real-world skills with multi-modal perception. To achieve this, we have collected a dataset comprising over 110,000 contact-rich robot manipulation sequences across diverse skills, contexts, robots, and camera viewpoints, all collected in the real world. Each sequence in the dataset includes visual, force, audio, and action information. Moreover, we also provide a corresponding human demonstration video and a language description for each robot sequence. We have invested significant efforts in calibrating all the sensors and ensuring a high-quality dataset. The dataset is made publicly available at rh20t.github.io",
  "published": "2023-07-02",
  "updated": "2023-09-26",
  "year": "2023",
  "authors": [
   "Hao-Shu Fang",
   "Hongjie Fang",
   "Zhenyu Tang",
   "Jirong Liu",
   "Chenxi Wang",
   "Junbo Wang",
   "Haoyi Zhu",
   "Cewu Lu"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA",
  "venue_source": "semantic-scholar",
  "citations": 247,
  "influential_citations": 10,
  "tldr": "This paper aims to unlock the potential for an agent to generalize to hundreds of real-world skills with multi-modal perception with a dataset comprising over 110,000 contact-rich robot manipulation sequences across diverse skills, contexts, robots, and camera viewpoints, all collected in the real world.",
  "doi": "10.1109/ICRA57147.2024.10611615",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Haoshu Fang",
    "id": "122851212",
    "h_index": 34,
    "papers": 61
   },
   {
    "name": "Hongjie Fang",
    "id": "2152115958",
    "h_index": 16,
    "papers": 31
   },
   {
    "name": "Zhenyu Tang",
    "id": "2087321693",
    "h_index": 11,
    "papers": 30
   },
   {
    "name": "Jirong Liu",
    "id": "2135259062",
    "h_index": 7,
    "papers": 12
   },
   {
    "name": "Junbo Wang",
    "id": "2220799617",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Haoyi Zhu",
    "id": "2171155650",
    "h_index": 9,
    "papers": 11
   },
   {
    "name": "Cewu Lu",
    "id": "2281998765",
    "h_index": 24,
    "papers": 45
   }
  ],
  "comment": "RSS 2023 workshop on LTAMP. The project page is at rh20t.github.io",
  "topics": [
   "egocentric-data",
   "tactile",
   "imitation-diffusion"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2307.00595v2",
  "pdf_url": "https://arxiv.org/pdf/2307.00595v2",
  "html_url": "https://arxiv.org/html/2307.00595v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.89
 },
 {
  "id": "2306.17817",
  "slug": "act3d-3d-feature-field-transformers-for-multi-task-robotic-manipulatio",
  "title": "Act3D: 3D Feature Field Transformers for Multi-Task Robotic Manipulation",
  "abstract": "3D perceptual representations are well suited for robot manipulation as they easily encode occlusions and simplify spatial reasoning. Many manipulation tasks require high spatial precision in end-effector pose prediction, which typically demands high-resolution 3D feature grids that are computationally expensive to process. As a result, most manipulation policies operate directly in 2D, foregoing 3D inductive biases. In this paper, we introduce Act3D, a manipulation policy transformer that represents the robot's workspace using a 3D feature field with adaptive resolutions dependent on the task at hand. The model lifts 2D pre-trained features to 3D using sensed depth, and attends to them to compute features for sampled 3D points. It samples 3D point grids in a coarse to fine manner, featurizes them using relative-position attention, and selects where to focus the next round of point sampling. In this way, it efficiently computes 3D action maps of high spatial resolution. Act3D sets a new state-of-the-art in RL-Bench, an established manipulation benchmark, where it achieves 10% absolute improvement over the previous SOTA 2D multi-view policy on 74 RLBench tasks and 22% absolute improvement with 3x less compute over the previous SOTA 3D policy. We quantify the importance of relative spatial attention, large-scale vision-language pre-trained 2D backbones, and weight tying across coarse-to-fine attentions in ablative experiments. Code and videos are available on our project website: https://act3d.github.io/.",
  "published": "2023-06-30",
  "updated": "2023-10-19",
  "year": "2023",
  "authors": [
   "Theophile Gervet",
   "Zhou Xian",
   "Nikolaos Gkanatsios",
   "Katerina Fragkiadaki"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 188,
  "influential_citations": 19,
  "tldr": "Act3D is introduced, a manipulation policy transformer that represents the robot's workspace using a 3D feature field with adaptive resolutions dependent on the task at hand, and sets a new state-of-the-art in RL-Bench, an established manipulation benchmark.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Th\u00e9ophile Gervet",
    "id": "81588783",
    "h_index": 15,
    "papers": 19
   },
   {
    "name": "Zhou Xian",
    "id": "2060151696",
    "h_index": 11,
    "papers": 17
   },
   {
    "name": "Nikolaos Gkanatsios",
    "id": "3070188",
    "h_index": 10,
    "papers": 16
   },
   {
    "name": "Katerina Fragkiadaki",
    "id": "1705557",
    "h_index": 32,
    "papers": 82
   }
  ],
  "comment": "",
  "topics": [
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2306.17817v2",
  "pdf_url": "https://arxiv.org/pdf/2306.17817v2",
  "html_url": "https://arxiv.org/html/2306.17817v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.78
 },
 {
  "id": "2306.14874",
  "slug": "anymal-parkour-learning-agile-navigation-for-quadrupedal-robots",
  "title": "ANYmal Parkour: Learning Agile Navigation for Quadrupedal Robots",
  "abstract": "Performing agile navigation with four-legged robots is a challenging task due to the highly dynamic motions, contacts with various parts of the robot, and the limited field of view of the perception sensors. In this paper, we propose a fully-learned approach to train such robots and conquer scenarios that are reminiscent of parkour challenges. The method involves training advanced locomotion skills for several types of obstacles, such as walking, jumping, climbing, and crouching, and then using a high-level policy to select and control those skills across the terrain. Thanks to our hierarchical formulation, the navigation policy is aware of the capabilities of each skill, and it will adapt its behavior depending on the scenario at hand. Additionally, a perception module is trained to reconstruct obstacles from highly occluded and noisy sensory data and endows the pipeline with scene understanding. Compared to previous attempts, our method can plan a path for challenging scenarios without expert demonstration, offline computation, a priori knowledge of the environment, or taking contacts explicitly into account. While these modules are trained from simulated data only, our real-world experiments demonstrate successful transfer on hardware, where the robot navigates and crosses consecutive challenging obstacles with speeds of up to two meters per second. The supplementary video can be found on the project website: https://sites.google.com/leggedrobotics.com/agile-navigation",
  "published": "2023-06-26",
  "updated": "2023-06-26",
  "year": "2023",
  "authors": [
   "David Hoeller",
   "Nikita Rudin",
   "Dhionis Sako",
   "Marco Hutter"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "Science Robotics",
  "venue_source": "semantic-scholar",
  "citations": 438,
  "influential_citations": 17,
  "tldr": "A fully learned approach to training a quadrupedal robot with locomotion skills, such as jumping, climbing, crouching, and walking, for rapid navigation around an obstacle parkour course, and shows potential for robot navigation on unstructured terrain where time is vital, such as in search and rescue.",
  "doi": "10.1126/scirobotics.adi7566",
  "oa_pdf": "https://arxiv.org/pdf/2306.14874",
  "s2_authors": [
   {
    "name": "David Hoeller",
    "id": "71054073",
    "h_index": 18,
    "papers": 19
   },
   {
    "name": "N. Rudin",
    "id": "2113243810",
    "h_index": 17,
    "papers": 17
   },
   {
    "name": "Dhionis V. Sako",
    "id": "37306990",
    "h_index": 6,
    "papers": 6
   },
   {
    "name": "Marco Hutter",
    "id": "14349870",
    "h_index": 80,
    "papers": 279
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2306.14874v1",
  "pdf_url": "https://arxiv.org/pdf/2306.14874v1",
  "html_url": "https://arxiv.org/html/2306.14874v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.14
 },
 {
  "id": "2306.12422",
  "slug": "dreamtime-an-improved-optimization-strategy-for-diffusion-guided-3d-ge",
  "title": "DreamTime: An Improved Optimization Strategy for Diffusion-Guided 3D Generation",
  "abstract": "Text-to-image diffusion models pre-trained on billions of image-text pairs have recently enabled 3D content creation by optimizing a randomly initialized differentiable 3D representation with score distillation. However, the optimization process suffers slow convergence and the resultant 3D models often exhibit two limitations: (a) quality concerns such as missing attributes and distorted shape and texture; (b) extremely low diversity comparing to text-guided image synthesis. In this paper, we show that the conflict between the 3D optimization process and uniform timestep sampling in score distillation is the main reason for these limitations. To resolve this conflict, we propose to prioritize timestep sampling with monotonically non-increasing functions, which aligns the 3D optimization process with the sampling process of diffusion model. Extensive experiments show that our simple redesign significantly improves 3D content creation with faster convergence, better quality and diversity.",
  "published": "2023-06-21",
  "updated": "2024-05-06",
  "year": "2023",
  "authors": [
   "Yukun Huang",
   "Jianan Wang",
   "Yukai Shi",
   "Boshi Tang",
   "Xianbiao Qi",
   "Lei Zhang"
  ],
  "author_count": 6,
  "categories": [
   "cs.CV",
   "cs.GR",
   "cs.LG"
  ],
  "primary_category": "cs.CV",
  "venue": "ICLR 2024",
  "venue_source": "arxiv-comment",
  "citations": 85,
  "influential_citations": 9,
  "tldr": "This paper proposes to prioritize timestep sampling with monotonically non-increasing functions, which aligns the 3D optimization process with the sampling process of diffusion model, and significantly improves 3D content creation with faster convergence, better quality and diversity.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yukun Huang",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Jianan Wang",
    "id": "2109495782",
    "h_index": 10,
    "papers": 14
   },
   {
    "name": "Yukai Shi",
    "id": "3435583",
    "h_index": 21,
    "papers": 60
   },
   {
    "name": "Xianbiao Qi",
    "id": "2689287",
    "h_index": 24,
    "papers": 55
   },
   {
    "name": "Zhengjun Zha",
    "id": "143962510",
    "h_index": 79,
    "papers": 412
   },
   {
    "name": "Lei Zhang",
    "id": "39089563",
    "h_index": 83,
    "papers": 599
   }
  ],
  "comment": "ICLR 2024",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2306.12422v2",
  "pdf_url": "https://arxiv.org/pdf/2306.12422v2",
  "html_url": "https://arxiv.org/html/2306.12422v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.43
 },
 {
  "id": "2306.04784",
  "slug": "designing-anthropomorphic-soft-hands-through-interaction",
  "title": "Designing Anthropomorphic Soft Hands through Interaction",
  "abstract": "Modeling and simulating soft robot hands can aid in design iteration for complex and high degree-of-freedom (DoF) morphologies. This can be further supplemented by iterating on the design based on its performance in real world manipulation tasks. However, iterating in the real world requires an approach that allows us to test new designs quickly at low costs. In this paper, we leverage rapid prototyping of the hand using 3D-printing, and utilize teleoperation to evaluate the hand in real world manipulation tasks. Using this method, we design a 3D-printed 16-DoF dexterous anthropomorphic soft hand (DASH) and iteratively improve its design over five iterations. Rapid prototyping techniques such as 3D-printing allow us to directly evaluate the fabricated hand without modeling it in simulation. We show that the design improves over five design iterations through evaluating the hand's performance in 30 real-world teleoperated manipulation tasks. Testing over 900 demonstrations shows that our final version of DASH can solve 19 of the 30 tasks compared to Allegro, a popular rigid hand in the market, which can only solve 7 tasks. We open-source our CAD models as well as the teleoperated dataset for further study.",
  "published": "2023-06-07",
  "updated": "2024-03-15",
  "year": "2023",
  "authors": [
   "Pragna Mannam",
   "Kenneth Shaw",
   "Dominik Bauer",
   "Jean Oh",
   "Deepak Pathak",
   "Nancy Pollard"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "Humanoids",
  "venue_source": "semantic-scholar",
  "citations": 14,
  "influential_citations": 2,
  "tldr": "This paper designs a 3D-printed 16-DoF dexterous anthropomorphic soft hand (DASH) and iteratively improves its design over five iterations and shows that the design improves over five design iterations through evaluating the hand's performance in 30 real-world teleoperated manipulation tasks.",
  "doi": "10.1109/Humanoids57100.2023.10375195",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Pragna Mannam",
    "id": "4724804",
    "h_index": 8,
    "papers": 11
   },
   {
    "name": "Kenneth Shaw",
    "id": "2072761493",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Dominik Bauer",
    "id": "1739101265",
    "h_index": 9,
    "papers": 21
   },
   {
    "name": "Jean Oh",
    "id": "143904954",
    "h_index": 26,
    "papers": 116
   },
   {
    "name": "Deepak Pathak",
    "id": "2004879394",
    "h_index": 24,
    "papers": 32
   },
   {
    "name": "N. Pollard",
    "id": "1735665",
    "h_index": 42,
    "papers": 136
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "data-teleop",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2306.04784v3",
  "pdf_url": "https://arxiv.org/pdf/2306.04784v3",
  "html_url": "https://arxiv.org/html/2306.04784v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.68
 },
 {
  "id": "2306.03881",
  "slug": "emergent-correspondence-from-image-diffusion",
  "title": "Emergent Correspondence from Image Diffusion",
  "abstract": "Finding correspondences between images is a fundamental problem in computer vision. In this paper, we show that correspondence emerges in image diffusion models without any explicit supervision. We propose a simple strategy to extract this implicit knowledge out of diffusion networks as image features, namely DIffusion FeaTures (DIFT), and use them to establish correspondences between real images. Without any additional fine-tuning or supervision on the task-specific data or annotations, DIFT is able to outperform both weakly-supervised methods and competitive off-the-shelf features in identifying semantic, geometric, and temporal correspondences. Particularly for semantic correspondence, DIFT from Stable Diffusion is able to outperform DINO and OpenCLIP by 19 and 14 accuracy points respectively on the challenging SPair-71k benchmark. It even outperforms the state-of-the-art supervised methods on 9 out of 18 categories while remaining on par for the overall performance. Project page: https://diffusionfeatures.github.io",
  "published": "2023-06-06",
  "updated": "2023-12-06",
  "year": "2023",
  "authors": [
   "Luming Tang",
   "Menglin Jia",
   "Qianqian Wang",
   "Cheng Perng Phoo",
   "Bharath Hariharan"
  ],
  "author_count": 5,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 526,
  "influential_citations": 73,
  "tldr": "This paper shows that correspondence emerges in image diffusion models without any explicit supervision, and proposes a simple strategy to extract this implicit knowledge out of diffusion networks as image features, namely DIffusion FeaTures (DIFT), and use them to establish correspondences between real images.",
  "doi": "10.48550/arXiv.2306.03881",
  "oa_pdf": "http://arxiv.org/pdf/2306.03881",
  "s2_authors": [
   {
    "name": "Luming Tang",
    "id": "34689393",
    "h_index": 11,
    "papers": 14
   },
   {
    "name": "Menglin Jia",
    "id": "51502783",
    "h_index": 17,
    "papers": 56
   },
   {
    "name": "Qianqian Wang",
    "id": "2144298123",
    "h_index": 18,
    "papers": 23
   },
   {
    "name": "Cheng Perng Phoo",
    "id": "51257044",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "Bharath Hariharan",
    "id": "1790580",
    "h_index": 38,
    "papers": 67
   }
  ],
  "comment": "NeurIPS 2023. Project page: https://diffusionfeatures.github.io",
  "topics": [
   "data-teleop",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2306.03881v2",
  "pdf_url": "https://arxiv.org/pdf/2306.03881v2",
  "html_url": "https://arxiv.org/html/2306.03881v2",
  "code_url": "https://diffusionfeatures.github.io",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.22
 },
 {
  "id": "2306.03310",
  "slug": "libero-benchmarking-knowledge-transfer-for-lifelong-robot-learning",
  "title": "LIBERO: Benchmarking Knowledge Transfer for Lifelong Robot Learning",
  "abstract": "Lifelong learning offers a promising paradigm of building a generalist agent that learns and adapts over its lifespan. Unlike traditional lifelong learning problems in image and text domains, which primarily involve the transfer of declarative knowledge of entities and concepts, lifelong learning in decision-making (LLDM) also necessitates the transfer of procedural knowledge, such as actions and behaviors. To advance research in LLDM, we introduce LIBERO, a novel benchmark of lifelong learning for robot manipulation. Specifically, LIBERO highlights five key research topics in LLDM: 1) how to efficiently transfer declarative knowledge, procedural knowledge, or the mixture of both; 2) how to design effective policy architectures and 3) effective algorithms for LLDM; 4) the robustness of a lifelong learner with respect to task ordering; and 5) the effect of model pretraining for LLDM. We develop an extendible procedural generation pipeline that can in principle generate infinitely many tasks. For benchmarking purpose, we create four task suites (130 tasks in total) that we use to investigate the above-mentioned research topics. To support sample-efficient learning, we provide high-quality human-teleoperated demonstration data for all tasks. Our extensive experiments present several insightful or even unexpected discoveries: sequential finetuning outperforms existing lifelong learning methods in forward transfer, no single visual encoder architecture excels at all types of knowledge transfer, and naive supervised pretraining can hinder agents' performance in the subsequent LLDM. Check the website at https://libero-project.github.io for the code and the datasets.",
  "published": "2023-06-05",
  "updated": "2023-10-14",
  "year": "2023",
  "authors": [
   "Bo Liu",
   "Yifeng Zhu",
   "Chongkai Gao",
   "Yihao Feng",
   "Qiang Liu",
   "Yuke Zhu",
   "Peter Stone"
  ],
  "author_count": 7,
  "categories": [
   "cs.AI"
  ],
  "primary_category": "cs.AI",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 1418,
  "influential_citations": 370,
  "tldr": "An extendible procedural generation pipeline that can in principle generate infinitely many tasks and highlight five key research topics in LLDM, including how to efficiently transfer declarative knowledge, procedural knowledge, or the mixture of both.",
  "doi": "10.48550/arXiv.2306.03310",
  "oa_pdf": "http://arxiv.org/pdf/2306.03310",
  "s2_authors": [
   {
    "name": "Bo Liu",
    "id": "1720831208",
    "h_index": 17,
    "papers": 37
   },
   {
    "name": "Yifeng Zhu",
    "id": "1557295600",
    "h_index": 14,
    "papers": 23
   },
   {
    "name": "Chongkai Gao",
    "id": "2114088654",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Yihao Feng",
    "id": "22758695",
    "h_index": 20,
    "papers": 39
   },
   {
    "name": "Qian Liu",
    "id": "2155193246",
    "h_index": 12,
    "papers": 16
   },
   {
    "name": "Yuke Zhu",
    "id": "2117748",
    "h_index": 57,
    "papers": 130
   },
   {
    "name": "Peter Stone",
    "id": "2113909888",
    "h_index": 11,
    "papers": 25
   }
  ],
  "comment": "",
  "topics": [
   "foundation-pretraining",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2306.03310v2",
  "pdf_url": "https://arxiv.org/pdf/2306.03310v2",
  "html_url": "https://arxiv.org/html/2306.03310v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "2306.00937",
  "slug": "steve-1-a-generative-model-for-text-to-behavior-in-minecraft",
  "title": "STEVE-1: A Generative Model for Text-to-Behavior in Minecraft",
  "abstract": "Constructing AI models that respond to text instructions is challenging, especially for sequential decision-making tasks. This work introduces a methodology, inspired by unCLIP, for instruction-tuning generative models of behavior without relying on a large dataset of instruction-labeled trajectories. Using this methodology, we create an instruction-tuned Video Pretraining (VPT) model called STEVE-1, which can follow short-horizon open-ended text and visual instructions in Minecraft. STEVE-1 is trained in two steps: adapting the pretrained VPT model to follow commands in MineCLIP's latent space, then training a prior to predict latent codes from text. This allows us to finetune VPT through self-supervised behavioral cloning and hindsight relabeling, reducing the need for costly human text annotations, and all for only $60 of compute. By leveraging pretrained models like VPT and MineCLIP and employing best practices from text-conditioned image generation, STEVE-1 sets a new bar for open-ended instruction-following in Minecraft with low-level controls (mouse and keyboard) and raw pixel inputs, far outperforming previous baselines and robustly completing 12 of 13 tasks in our early-game evaluation suite. We provide experimental evidence highlighting key factors for downstream performance, including pretraining, classifier-free guidance, and data scaling. All resources, including our model weights, training scripts, and evaluation tools are made available for further research.",
  "published": "2023-06-01",
  "updated": "2024-02-04",
  "year": "2023",
  "authors": [
   "Shalev Lifshitz",
   "Keiran Paster",
   "Harris Chan",
   "Jimmy Ba",
   "Sheila McIlraith"
  ],
  "author_count": 5,
  "categories": [
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.AI",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 119,
  "influential_citations": 24,
  "tldr": "This work introduces a methodology, inspired by unCLIP, for instruction-tuning generative models of behavior without relying on a large dataset of instruction-labeled trajectories, and creates an instruction-tuned Video Pretraining (VPT) model called STEVE-1, which can follow short-horizon open-ended text and visual instructions in Minecraft.",
  "doi": "10.48550/arXiv.2306.00937",
  "oa_pdf": "http://arxiv.org/pdf/2306.00937",
  "s2_authors": [
   {
    "name": "Shalev Lifshitz",
    "id": "2093408777",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Keiran Paster",
    "id": "73775191",
    "h_index": 10,
    "papers": 18
   },
   {
    "name": "Harris Chan",
    "id": "41228532",
    "h_index": 15,
    "papers": 39
   },
   {
    "name": "Jimmy Ba",
    "id": "2503659",
    "h_index": 46,
    "papers": 84
   },
   {
    "name": "Sheila A. McIlraith",
    "id": "1683896",
    "h_index": 60,
    "papers": 285
   }
  ],
  "comment": "",
  "topics": [
   "foundation-pretraining",
   "data-teleop",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2306.00937v3",
  "pdf_url": "https://arxiv.org/pdf/2306.00937v3",
  "html_url": "https://arxiv.org/html/2306.00937v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.58
 },
 {
  "id": "2305.18499",
  "slug": "pre-training-contextualized-world-models-with-in-the-wild-videos-for-r",
  "title": "Pre-training Contextualized World Models with In-the-wild Videos for Reinforcement Learning",
  "abstract": "Unsupervised pre-training methods utilizing large and diverse datasets have achieved tremendous success across a range of domains. Recent work has investigated such unsupervised pre-training methods for model-based reinforcement learning (MBRL) but is limited to domain-specific or simulated data. In this paper, we study the problem of pre-training world models with abundant in-the-wild videos for efficient learning of downstream visual control tasks. However, in-the-wild videos are complicated with various contextual factors, such as intricate backgrounds and textured appearance, which precludes a world model from extracting shared world knowledge to generalize better. To tackle this issue, we introduce Contextualized World Models (ContextWM) that explicitly separate context and dynamics modeling to overcome the complexity and diversity of in-the-wild videos and facilitate knowledge transfer between distinct scenes. Specifically, a contextualized extension of the latent dynamics model is elaborately realized by incorporating a context encoder to retain contextual information and empower the image decoder, which encourages the latent dynamics model to concentrate on essential temporal variations. Our experiments show that in-the-wild video pre-training equipped with ContextWM can significantly improve the sample efficiency of MBRL in various domains, including robotic manipulation, locomotion, and autonomous driving. Code is available at this repository: https://github.com/thuml/ContextWM.",
  "published": "2023-05-29",
  "updated": "2023-10-27",
  "year": "2023",
  "authors": [
   "Jialong Wu",
   "Haoyu Ma",
   "Chaoyi Deng",
   "Mingsheng Long"
  ],
  "author_count": 4,
  "categories": [
   "cs.CV",
   "cs.LG",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 54,
  "influential_citations": 4,
  "tldr": "Experiments show that in-the-wild video pre-training equipped with ContextWM can significantly improve the sample efficiency of MBRL in various domains, including robotic manipulation, locomotion, and autonomous driving.",
  "doi": "10.48550/arXiv.2305.18499",
  "oa_pdf": "http://arxiv.org/pdf/2305.18499",
  "s2_authors": [
   {
    "name": "Jialong Wu",
    "id": "2154707054",
    "h_index": 10,
    "papers": 21
   },
   {
    "name": "Haoyu Ma",
    "id": "2110816316",
    "h_index": 14,
    "papers": 22
   },
   {
    "name": "Chao Deng",
    "id": "2057945548",
    "h_index": 7,
    "papers": 9
   },
   {
    "name": "Mingsheng Long",
    "id": "2054275000",
    "h_index": 35,
    "papers": 88
   }
  ],
  "comment": "NeurIPS 2023. Code is available at https://github.com/thuml/ContextWM",
  "topics": [
   "world-models",
   "humanoids",
   "rl-control",
   "navigation",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2305.18499v2",
  "pdf_url": "https://arxiv.org/pdf/2305.18499v2",
  "html_url": "https://arxiv.org/html/2305.18499v2",
  "code_url": "https://github.com/thuml/ContextWM",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.24
 },
 {
  "id": "2305.18290",
  "slug": "direct-preference-optimization-your-language-model-is-secretly-a-rewar",
  "title": "Direct Preference Optimization: Your Language Model is Secretly a Reward Model",
  "abstract": "While large-scale unsupervised language models (LMs) learn broad world knowledge and some reasoning skills, achieving precise control of their behavior is difficult due to the completely unsupervised nature of their training. Existing methods for gaining such steerability collect human labels of the relative quality of model generations and fine-tune the unsupervised LM to align with these preferences, often with reinforcement learning from human feedback (RLHF). However, RLHF is a complex and often unstable procedure, first fitting a reward model that reflects the human preferences, and then fine-tuning the large unsupervised LM using reinforcement learning to maximize this estimated reward without drifting too far from the original model. In this paper we introduce a new parameterization of the reward model in RLHF that enables extraction of the corresponding optimal policy in closed form, allowing us to solve the standard RLHF problem with only a simple classification loss. The resulting algorithm, which we call Direct Preference Optimization (DPO), is stable, performant, and computationally lightweight, eliminating the need for sampling from the LM during fine-tuning or performing significant hyperparameter tuning. Our experiments show that DPO can fine-tune LMs to align with human preferences as well as or better than existing methods. Notably, fine-tuning with DPO exceeds PPO-based RLHF in ability to control sentiment of generations, and matches or improves response quality in summarization and single-turn dialogue while being substantially simpler to implement and train.",
  "published": "2023-05-29",
  "updated": "2024-07-29",
  "year": "2023",
  "authors": [
   "Rafael Rafailov",
   "Archit Sharma",
   "Eric Mitchell",
   "Stefano Ermon",
   "Christopher D. Manning",
   "Chelsea Finn"
  ],
  "author_count": 6,
  "categories": [
   "cs.LG",
   "cs.AI",
   "cs.CL"
  ],
  "primary_category": "cs.LG",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 10236,
  "influential_citations": 2123,
  "tldr": "A new parameterization of the reward model in RLHF that enables extraction of the corresponding optimal policy in closed form is introduced, allowing us to solve the standard RLHF problem with only a simple classification loss.",
  "doi": "10.52202/075280-2338",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Rafael Rafailov",
    "id": "102801230",
    "h_index": 25,
    "papers": 44
   },
   {
    "name": "Archit Sharma",
    "id": "50465276",
    "h_index": 22,
    "papers": 36
   },
   {
    "name": "E. Mitchell",
    "id": "49688913",
    "h_index": 22,
    "papers": 36
   },
   {
    "name": "S. Ermon",
    "id": "2490652",
    "h_index": 104,
    "papers": 479
   },
   {
    "name": "Christopher D. Manning",
    "id": "144783904",
    "h_index": 149,
    "papers": 490
   },
   {
    "name": "Chelsea Finn",
    "id": "46881670",
    "h_index": 94,
    "papers": 240
   }
  ],
  "comment": "",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2305.18290v3",
  "pdf_url": "https://arxiv.org/pdf/2305.18290v3",
  "html_url": "https://arxiv.org/html/2305.18290v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "2305.16381",
  "slug": "dpok-reinforcement-learning-for-fine-tuning-text-to-image-diffusion-mo",
  "title": "DPOK: Reinforcement Learning for Fine-tuning Text-to-Image Diffusion Models",
  "abstract": "Learning from human feedback has been shown to improve text-to-image models. These techniques first learn a reward function that captures what humans care about in the task and then improve the models based on the learned reward function. Even though relatively simple approaches (e.g., rejection sampling based on reward scores) have been investigated, fine-tuning text-to-image models with the reward function remains challenging. In this work, we propose using online reinforcement learning (RL) to fine-tune text-to-image models. We focus on diffusion models, defining the fine-tuning task as an RL problem, and updating the pre-trained text-to-image diffusion models using policy gradient to maximize the feedback-trained reward. Our approach, coined DPOK, integrates policy optimization with KL regularization. We conduct an analysis of KL regularization for both RL fine-tuning and supervised fine-tuning. In our experiments, we show that DPOK is generally superior to supervised fine-tuning with respect to both image-text alignment and image quality. Our code is available at https://github.com/google-research/google-research/tree/master/dpok.",
  "published": "2023-05-25",
  "updated": "2023-11-01",
  "year": "2023",
  "authors": [
   "Ying Fan",
   "Olivia Watkins",
   "Yuqing Du",
   "Hao Liu",
   "Moonkyung Ryu",
   "Craig Boutilier",
   "Pieter Abbeel",
   "Mohammad Ghavamzadeh",
   "Kangwook Lee",
   "Kimin Lee"
  ],
  "author_count": 10,
  "categories": [
   "cs.LG",
   "cs.CV"
  ],
  "primary_category": "cs.LG",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 475,
  "influential_citations": 62,
  "tldr": "This work proposes using online reinforcement learning (RL) to fine-tune text-to-image models, and integrates policy optimization with KL regularization, and shows that DPOK is generally superior to supervised fine-tuning with respect to both image-text alignment and image quality.",
  "doi": "10.48550/arXiv.2305.16381",
  "oa_pdf": "http://arxiv.org/pdf/2305.16381",
  "s2_authors": [
   {
    "name": "Ying Fan",
    "id": "2344807639",
    "h_index": 9,
    "papers": 16
   },
   {
    "name": "Olivia Watkins",
    "id": "145695607",
    "h_index": 9,
    "papers": 12
   },
   {
    "name": "Yuqing Du",
    "id": "144894286",
    "h_index": 16,
    "papers": 27
   },
   {
    "name": "Hao Liu",
    "id": "2143855835",
    "h_index": 21,
    "papers": 27
   },
   {
    "name": "M. Ryu",
    "id": "2060597867",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "Craig Boutilier",
    "id": "145646162",
    "h_index": 74,
    "papers": 280
   },
   {
    "name": "P. Abbeel",
    "id": "1689992",
    "h_index": 165,
    "papers": 596
   },
   {
    "name": "M. Ghavamzadeh",
    "id": "1678622",
    "h_index": 54,
    "papers": 187
   },
   {
    "name": "Kangwook Lee",
    "id": "2115495251",
    "h_index": 21,
    "papers": 47
   },
   {
    "name": "Kimin Lee",
    "id": "3436470",
    "h_index": 37,
    "papers": 69
   }
  ],
  "comment": "NeurIPS 2023",
  "topics": [
   "rl-control",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2305.16381v3",
  "pdf_url": "https://arxiv.org/pdf/2305.16381v3",
  "html_url": "https://arxiv.org/html/2305.16381v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.18
 },
 {
  "id": "2305.16291",
  "slug": "voyager-an-open-ended-embodied-agent-with-large-language-models",
  "title": "Voyager: An Open-Ended Embodied Agent with Large Language Models",
  "abstract": "We introduce Voyager, the first LLM-powered embodied lifelong learning agent in Minecraft that continuously explores the world, acquires diverse skills, and makes novel discoveries without human intervention. Voyager consists of three key components: 1) an automatic curriculum that maximizes exploration, 2) an ever-growing skill library of executable code for storing and retrieving complex behaviors, and 3) a new iterative prompting mechanism that incorporates environment feedback, execution errors, and self-verification for program improvement. Voyager interacts with GPT-4 via blackbox queries, which bypasses the need for model parameter fine-tuning. The skills developed by Voyager are temporally extended, interpretable, and compositional, which compounds the agent's abilities rapidly and alleviates catastrophic forgetting. Empirically, Voyager shows strong in-context lifelong learning capability and exhibits exceptional proficiency in playing Minecraft. It obtains 3.3x more unique items, travels 2.3x longer distances, and unlocks key tech tree milestones up to 15.3x faster than prior SOTA. Voyager is able to utilize the learned skill library in a new Minecraft world to solve novel tasks from scratch, while other techniques struggle to generalize. We open-source our full codebase and prompts at https://voyager.minedojo.org/.",
  "published": "2023-05-25",
  "updated": "2023-10-19",
  "year": "2023",
  "authors": [
   "Guanzhi Wang",
   "Yuqi Xie",
   "Yunfan Jiang",
   "Ajay Mandlekar",
   "Chaowei Xiao",
   "Yuke Zhu",
   "Linxi Fan",
   "Anima Anandkumar"
  ],
  "author_count": 8,
  "categories": [
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.AI",
  "venue": "Trans. Mach. Learn. Res.",
  "venue_source": "semantic-scholar",
  "citations": 2186,
  "influential_citations": 160,
  "tldr": "",
  "doi": "10.48550/arXiv.2305.16291",
  "oa_pdf": "http://arxiv.org/pdf/2305.16291",
  "s2_authors": [
   {
    "name": "Guanzhi Wang",
    "id": "96374437",
    "h_index": 16,
    "papers": 22
   },
   {
    "name": "Yuqi Xie",
    "id": "2218866691",
    "h_index": 11,
    "papers": 17
   },
   {
    "name": "Yunfan Jiang",
    "id": "2171112793",
    "h_index": 11,
    "papers": 16
   },
   {
    "name": "A. Mandlekar",
    "id": "49686756",
    "h_index": 36,
    "papers": 67
   },
   {
    "name": "Chaowei Xiao",
    "id": "2723309",
    "h_index": 46,
    "papers": 87
   },
   {
    "name": "Yuke Zhu",
    "id": "2117748",
    "h_index": 57,
    "papers": 130
   },
   {
    "name": "Linxi (Jim) Fan",
    "id": "3275727",
    "h_index": 18,
    "papers": 22
   },
   {
    "name": "Anima Anandkumar",
    "id": "47627049",
    "h_index": 41,
    "papers": 119
   }
  ],
  "comment": "Project website and open-source codebase: https://voyager.minedojo.org/",
  "topics": [
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2305.16291v2",
  "pdf_url": "https://arxiv.org/pdf/2305.16291v2",
  "html_url": "https://arxiv.org/html/2305.16291v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "2305.13567",
  "slug": "m-ember-tackling-long-horizon-mobile-manipulation-via-factorized-domai",
  "title": "M-EMBER: Tackling Long-Horizon Mobile Manipulation via Factorized Domain Transfer",
  "abstract": "In this paper, we propose a method to create visuomotor mobile manipulation solutions for long-horizon activities. We propose to leverage the recent advances in simulation to train visual solutions for mobile manipulation. While previous works have shown success applying this procedure to autonomous visual navigation and stationary manipulation, applying it to long-horizon visuomotor mobile manipulation is still an open challenge that demands both perceptual and compositional generalization of multiple skills. In this work, we develop Mobile-EMBER, or M-EMBER, a factorized method that decomposes a long-horizon mobile manipulation activity into a repertoire of primitive visual skills, reinforcement-learns each skill, and composes these skills to a long-horizon mobile manipulation activity. On a mobile manipulation robot, we find that M-EMBER completes a long-horizon mobile manipulation activity, cleaning_kitchen, achieving a 53% success rate. This requires successfully planning and executing five factorized, learned visual skills.",
  "published": "2023-05-23",
  "updated": "2023-05-23",
  "year": "2023",
  "authors": [
   "Bohan Wu",
   "Roberto Martin-Martin",
   "Li Fei-Fei"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA",
  "venue_source": "semantic-scholar",
  "citations": 16,
  "influential_citations": 1,
  "tldr": "This work develops Mobile-EMBER, a factorized method that decomposes a long-horizon mobile manipulation activity into a repertoire of primitive visual skills, reinforcement-learns each skill in simulation, and composes these skills to aLong-HorizonMobile manipulation activity.",
  "doi": "10.1109/ICRA48891.2023.10160934",
  "oa_pdf": "https://arxiv.org/pdf/2305.13567",
  "s2_authors": [
   {
    "name": "Bohan Wu",
    "id": "95528042",
    "h_index": 10,
    "papers": 13
   },
   {
    "name": "Roberto Mart\u00edn-Mart\u00edn",
    "id": "2334889156",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Li Fei-Fei",
    "id": "48004138",
    "h_index": 143,
    "papers": 606
   }
  ],
  "comment": "",
  "topics": [
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2305.13567v1",
  "pdf_url": "https://arxiv.org/pdf/2305.13567v1",
  "html_url": "https://arxiv.org/html/2305.13567v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.73
 },
 {
  "id": "2305.13301",
  "slug": "training-diffusion-models-with-reinforcement-learning",
  "title": "Training Diffusion Models with Reinforcement Learning",
  "abstract": "Diffusion models are a class of flexible generative models trained with an approximation to the log-likelihood objective. However, most use cases of diffusion models are not concerned with likelihoods, but instead with downstream objectives such as human-perceived image quality or drug effectiveness. In this paper, we investigate reinforcement learning methods for directly optimizing diffusion models for such objectives. We describe how posing denoising as a multi-step decision-making problem enables a class of policy gradient algorithms, which we refer to as denoising diffusion policy optimization (DDPO), that are more effective than alternative reward-weighted likelihood approaches. Empirically, DDPO is able to adapt text-to-image diffusion models to objectives that are difficult to express via prompting, such as image compressibility, and those derived from human feedback, such as aesthetic quality. Finally, we show that DDPO can improve prompt-image alignment using feedback from a vision-language model without the need for additional data collection or human annotation. The project's website can be found at http://rl-diffusion.github.io .",
  "published": "2023-05-22",
  "updated": "2024-01-04",
  "year": "2023",
  "authors": [
   "Kevin Black",
   "Michael Janner",
   "Yilun Du",
   "Ilya Kostrikov",
   "Sergey Levine"
  ],
  "author_count": 5,
  "categories": [
   "cs.LG",
   "cs.AI",
   "cs.CV"
  ],
  "primary_category": "cs.LG",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 1012,
  "influential_citations": 166,
  "tldr": "It is described how posing denoising as a multi-step decision-making problem enables a class of policy gradient algorithms, which are referred to as Denoising diffusion policy optimization (DDPO), that are more effective than alternative reward-weighted likelihood approaches.",
  "doi": "10.48550/arXiv.2305.13301",
  "oa_pdf": "https://arxiv.org/pdf/2305.13301",
  "s2_authors": [
   {
    "name": "Kevin Black",
    "id": "2069483822",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Michael Janner",
    "id": "35163402",
    "h_index": 16,
    "papers": 22
   },
   {
    "name": "Yilun Du",
    "id": "15394275",
    "h_index": 48,
    "papers": 86
   },
   {
    "name": "Ilya Kostrikov",
    "id": "2000906",
    "h_index": 29,
    "papers": 43
   },
   {
    "name": "S. Levine",
    "id": "1736651",
    "h_index": 167,
    "papers": 551
   }
  ],
  "comment": "23 pages, 16 figures",
  "topics": [
   "imitation-diffusion",
   "rl-control",
   "data-teleop",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2305.13301v4",
  "pdf_url": "https://arxiv.org/pdf/2305.13301v4",
  "html_url": "https://arxiv.org/html/2305.13301v4",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "2305.13245",
  "slug": "gqa-training-generalized-multi-query-transformer-models-from-multi-hea",
  "title": "GQA: Training Generalized Multi-Query Transformer Models from Multi-Head Checkpoints",
  "abstract": "Multi-query attention (MQA), which only uses a single key-value head, drastically speeds up decoder inference. However, MQA can lead to quality degradation, and moreover it may not be desirable to train a separate model just for faster inference. We (1) propose a recipe for uptraining existing multi-head language model checkpoints into models with MQA using 5% of original pre-training compute, and (2) introduce grouped-query attention (GQA), a generalization of multi-query attention which uses an intermediate (more than one, less than number of query heads) number of key-value heads. We show that uptrained GQA achieves quality close to multi-head attention with comparable speed to MQA.",
  "published": "2023-05-22",
  "updated": "2023-12-23",
  "year": "2023",
  "authors": [
   "Joshua Ainslie",
   "James Lee-Thorp",
   "Michiel de Jong",
   "Yury Zemlyanskiy",
   "Federico Lebr\u00f3n",
   "Sumit Sanghai"
  ],
  "author_count": 6,
  "categories": [
   "cs.CL",
   "cs.LG"
  ],
  "primary_category": "cs.CL",
  "venue": "",
  "venue_source": "",
  "citations": 1699,
  "influential_citations": 123,
  "tldr": "This work proposes a recipe for uptraining existing multi-head language model checkpoints into models with MQA using 5% of original pre-training compute, and introduces grouped-query attention (GQA), a generalization of multi- query attention which uses an intermediate number of query heads.",
  "doi": "10.48550/arXiv.2305.13245",
  "oa_pdf": "http://arxiv.org/pdf/2305.13245",
  "s2_authors": [
   {
    "name": "J. Ainslie",
    "id": "1643737606",
    "h_index": 19,
    "papers": 37
   },
   {
    "name": "J. Lee-Thorp",
    "id": "1405626394",
    "h_index": 14,
    "papers": 21
   },
   {
    "name": "Michiel de Jong",
    "id": "21379393",
    "h_index": 10,
    "papers": 15
   },
   {
    "name": "Yury Zemlyanskiy",
    "id": "51199981",
    "h_index": 11,
    "papers": 12
   },
   {
    "name": "Federico Lebr'on",
    "id": "2218311992",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Sumit K. Sanghai",
    "id": "144074891",
    "h_index": 17,
    "papers": 53
   }
  ],
  "comment": "Accepted at EMNLP 2023. Added to related work",
  "topics": [
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2305.13245v3",
  "pdf_url": "https://arxiv.org/pdf/2305.13245v3",
  "html_url": "https://arxiv.org/html/2305.13245v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.0
 },
 {
  "id": "2305.12821",
  "slug": "furniturebench-reproducible-real-world-benchmark-for-long-horizon-comp",
  "title": "FurnitureBench: Reproducible Real-World Benchmark for Long-Horizon Complex Manipulation",
  "abstract": "Reinforcement learning (RL), imitation learning (IL), and task and motion planning (TAMP) have demonstrated impressive performance across various robotic manipulation tasks. However, these approaches have been limited to learning simple behaviors in current real-world manipulation benchmarks, such as pushing or pick-and-place. To enable more complex, long-horizon behaviors of an autonomous robot, we propose to focus on real-world furniture assembly, a complex, long-horizon robot manipulation task that requires addressing many current robotic manipulation challenges to solve. We present FurnitureBench, a reproducible real-world furniture assembly benchmark aimed at providing a low barrier for entry and being easily reproducible, so that researchers across the world can reliably test their algorithms and compare them against prior work. For ease of use, we provide 200+ hours of pre-collected data (5000+ demonstrations), 3D printable furniture models, a robotic environment setup guide, and systematic task initialization. Furthermore, we provide FurnitureSim, a fast and realistic simulator of FurnitureBench. We benchmark the performance of offline RL and IL algorithms on our assembly tasks and demonstrate the need to improve such algorithms to be able to solve our tasks in the real world, providing ample opportunities for future research.",
  "published": "2023-05-22",
  "updated": "2023-05-22",
  "year": "2023",
  "authors": [
   "Minho Heo",
   "Youngwoon Lee",
   "Doohyun Lee",
   "Joseph J. Lim"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "RSS",
  "venue_source": "semantic-scholar",
  "citations": 204,
  "influential_citations": 31,
  "tldr": "FurnitureBench is presented, a reproducible real-world furniture assembly benchmark aimed at providing a low barrier for entry and being easily reproducible, so that researchers across the world can reliably test their algorithms and compare them against prior work.",
  "doi": "10.1177/02783649241304789",
  "oa_pdf": "http://arxiv.org/pdf/2305.12821",
  "s2_authors": [
   {
    "name": "Minho Heo",
    "id": "2218050062",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Youngwoon Lee",
    "id": "46358230",
    "h_index": 18,
    "papers": 27
   },
   {
    "name": "Doohyun Lee",
    "id": "2218134064",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Joseph J. Lim",
    "id": "35198686",
    "h_index": 28,
    "papers": 55
   }
  ],
  "comment": "Robotics: Science and Systems (RSS) 2023. Website: https://clvrai.com/furniture-bench",
  "topics": [
   "sim2real",
   "imitation-diffusion",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2305.12821v1",
  "pdf_url": "https://arxiv.org/pdf/2305.12821v1",
  "html_url": "https://arxiv.org/html/2305.12821v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.81
 },
 {
  "id": "2305.09765",
  "slug": "openvr-teleoperation-for-manipulation",
  "title": "OpenVR: Teleoperation for Manipulation",
  "abstract": "Across the robotics field, quality demonstrations are an integral part of many control pipelines. However, collecting high-quality demonstration trajectories remains time-consuming and difficult, often resulting in the number of demonstrations being the performance bottleneck. To address this issue, we present a method of Virtual Reality (VR) Teleoperation that uses an Oculus VR headset to teleoperate a Franka Emika Panda robot. Although other VR teleoperation methods exist, our code is open source, designed for readily available consumer hardware, easy to modify, agnostic to experimental setup, and simple to use.",
  "published": "2023-05-16",
  "updated": "2023-05-16",
  "year": "2023",
  "authors": [
   "Abraham George",
   "Alison Bartsch",
   "Amir Barati Farimani"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.HC",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "SoftwareX",
  "venue_source": "semantic-scholar",
  "citations": 27,
  "influential_citations": 0,
  "tldr": "A method of Virtual Reality (VR) Teleoperation that uses an Oculus VR headset to teleoperate a Franka Emika Panda robot and is designed for readily available consumer hardware, easy to modify, agnostic to experimental setup, and simple to use.",
  "doi": "10.48550/arXiv.2305.09765",
  "oa_pdf": "http://arxiv.org/pdf/2305.09765",
  "s2_authors": [
   {
    "name": "A. George",
    "id": "2057471145",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Alison Bartsch",
    "id": "2185508465",
    "h_index": 7,
    "papers": 14
   },
   {
    "name": "A. Farimani",
    "id": "3614493",
    "h_index": 46,
    "papers": 226
   }
  ],
  "comment": "8 pages, 8 figures, GitHub: https://github.com/Abraham190137/TeleoperationUnity",
  "topics": [
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2305.09765v1",
  "pdf_url": "https://arxiv.org/pdf/2305.09765v1",
  "html_url": "https://arxiv.org/html/2305.09765v1",
  "code_url": "https://github.com/Abraham190137/TeleoperationUnity",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.95
 },
 {
  "id": "2305.07759",
  "slug": "tinystories-how-small-can-language-models-be-and-still-speak-coherent",
  "title": "TinyStories: How Small Can Language Models Be and Still Speak Coherent English?",
  "abstract": "Language models (LMs) are powerful tools for natural language processing, but they often struggle to produce coherent and fluent text when they are small. Models with around 125M parameters such as GPT-Neo (small) or GPT-2 (small) can rarely generate coherent and consistent English text beyond a few words even after extensive training. This raises the question of whether the emergence of the ability to produce coherent English text only occurs at larger scales (with hundreds of millions of parameters or more) and complex architectures (with many layers of global attention). In this work, we introduce TinyStories, a synthetic dataset of short stories that only contain words that a typical 3 to 4-year-olds usually understand, generated by GPT-3.5 and GPT-4. We show that TinyStories can be used to train and evaluate LMs that are much smaller than the state-of-the-art models (below 10 million total parameters), or have much simpler architectures (with only one transformer block), yet still produce fluent and consistent stories with several paragraphs that are diverse and have almost perfect grammar, and demonstrate reasoning capabilities. We also introduce a new paradigm for the evaluation of language models: We suggest a framework which uses GPT-4 to grade the content generated by these models as if those were stories written by students and graded by a (human) teacher. This new paradigm overcomes the flaws of standard benchmarks which often requires the model's output to be very structures, and moreover provides a multidimensional score for the model, providing scores for different capabilities such as grammar, creativity and consistency. We hope that TinyStories can facilitate the development, analysis and research of LMs, especially for low-resource or specialized domains, and shed light on the emergence of language capabilities in LMs.",
  "published": "2023-05-12",
  "updated": "2023-05-24",
  "year": "2023",
  "authors": [
   "Ronen Eldan",
   "Yuanzhi Li"
  ],
  "author_count": 2,
  "categories": [
   "cs.CL",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.CL",
  "venue": "",
  "venue_source": "",
  "citations": 515,
  "influential_citations": 71,
  "tldr": "TinyStories, a synthetic dataset of short stories that only contain words that a typical 3 to 4-year-olds usually understand, is introduced and a new paradigm for the evaluation of language models is introduced, which uses GPT-4 to grade the content generated by these models as if those were stories written by students and graded by a (human) teacher.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ronen Eldan",
    "id": "2315830",
    "h_index": 37,
    "papers": 101
   },
   {
    "name": "Yuan-Fang Li",
    "id": "152244300",
    "h_index": 28,
    "papers": 88
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2305.07759v2",
  "pdf_url": "https://arxiv.org/pdf/2305.07759v2",
  "html_url": "https://arxiv.org/html/2305.07759v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.71
 },
 {
  "id": "2305.05658",
  "slug": "tidybot-personalized-robot-assistance-with-large-language-models",
  "title": "TidyBot: Personalized Robot Assistance with Large Language Models",
  "abstract": "For a robot to personalize physical assistance effectively, it must learn user preferences that can be generally reapplied to future scenarios. In this work, we investigate personalization of household cleanup with robots that can tidy up rooms by picking up objects and putting them away. A key challenge is determining the proper place to put each object, as people's preferences can vary greatly depending on personal taste or cultural background. For instance, one person may prefer storing shirts in the drawer, while another may prefer them on the shelf. We aim to build systems that can learn such preferences from just a handful of examples via prior interactions with a particular person. We show that robots can combine language-based planning and perception with the few-shot summarization capabilities of large language models (LLMs) to infer generalized user preferences that are broadly applicable to future interactions. This approach enables fast adaptation and achieves 91.2% accuracy on unseen objects in our benchmark dataset. We also demonstrate our approach on a real-world mobile manipulator called TidyBot, which successfully puts away 85.0% of objects in real-world test scenarios.",
  "published": "2023-05-09",
  "updated": "2023-10-11",
  "year": "2023",
  "authors": [
   "Jimmy Wu",
   "Rika Antonova",
   "Adam Kan",
   "Marion Lepert",
   "Andy Zeng",
   "Shuran Song",
   "Jeannette Bohg",
   "Szymon Rusinkiewicz",
   "Thomas Funkhouser"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CL",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS",
  "venue_source": "semantic-scholar",
  "citations": 462,
  "influential_citations": 21,
  "tldr": "It is shown that robots can combine language-based planning and perception with the few-shot summarization capabilities of large language models (LLMs) to infer generalized user preferences that are broadly applicable to future interactions.",
  "doi": "10.1007/s10514-023-10139-z",
  "oa_pdf": "https://arxiv.org/pdf/2305.05658",
  "s2_authors": [
   {
    "name": "Jimmy Wu",
    "id": "2155142153",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Rika Antonova",
    "id": "39534622",
    "h_index": 19,
    "papers": 40
   },
   {
    "name": "Adam Kan",
    "id": "2216606824",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Marion Lepert",
    "id": "10710717",
    "h_index": 8,
    "papers": 11
   },
   {
    "name": "Andy Zeng",
    "id": "38591293",
    "h_index": 34,
    "papers": 50
   },
   {
    "name": "Shuran Song",
    "id": "3340170",
    "h_index": 59,
    "papers": 90
   },
   {
    "name": "Jeannette Bohg",
    "id": "1775407",
    "h_index": 50,
    "papers": 161
   },
   {
    "name": "S. Rusinkiewicz",
    "id": "7723706",
    "h_index": 64,
    "papers": 155
   },
   {
    "name": "T. Funkhouser",
    "id": "1807080",
    "h_index": 89,
    "papers": 208
   }
  ],
  "comment": "Accepted to Autonomous Robots (AuRo) - Special Issue: Large Language Models in Robotics, 2023 and IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS), 2023. Project page: https://tidybot.cs.princeton.edu",
  "topics": [
   "hri"
  ],
  "orgs": [
   "Princeton"
  ],
  "abs_url": "https://arxiv.org/abs/2305.05658v2",
  "pdf_url": "https://arxiv.org/pdf/2305.05658v2",
  "html_url": "https://arxiv.org/html/2305.05658v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.67
 },
 {
  "id": "2305.04866",
  "slug": "causal-policy-gradient-for-whole-body-mobile-manipulation",
  "title": "Causal Policy Gradient for Whole-Body Mobile Manipulation",
  "abstract": "Developing the next generation of household robot helpers requires combining locomotion and interaction capabilities, which is generally referred to as mobile manipulation (MoMa). MoMa tasks are difficult due to the large action space of the robot and the common multi-objective nature of the task, e.g., efficiently reaching a goal while avoiding obstacles. Current approaches often segregate tasks into navigation without manipulation and stationary manipulation without locomotion by manually matching parts of the action space to MoMa sub-objectives (e.g. learning base actions for locomotion objectives and learning arm actions for manipulation). This solution prevents simultaneous combinations of locomotion and interaction degrees of freedom and requires human domain knowledge for both partitioning the action space and matching the action parts to the sub-objectives. In this paper, we introduce Causal MoMa, a new reinforcement learning framework to train policies for typical MoMa tasks that makes use of the most favorable subspace of the robot's action space to address each sub-objective. Causal MoMa automatically discovers the causal dependencies between actions and terms of the reward function and exploits these dependencies through causal policy gradient that reduces gradient variance compared to previous state-of-the-art reinforcement learning algorithms, improving convergence and results. We evaluate the performance of Causal MoMa on three types of simulated robots across different MoMa tasks and demonstrate success in transferring the policies trained in simulation directly to a real robot, where our agent is able to follow moving goals and react to dynamic obstacles while simultaneously and synergistically controlling the whole-body: base, arm, and head. More information at https://sites.google.com/view/causal-moma.",
  "published": "2023-05-04",
  "updated": "2023-09-28",
  "year": "2023",
  "authors": [
   "Jiaheng Hu",
   "Peter Stone",
   "Roberto Mart\u00edn-Mart\u00edn"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "RSS",
  "venue_source": "semantic-scholar",
  "citations": 37,
  "influential_citations": 2,
  "tldr": "Causal MoMa is introduced, a new reinforcement learning framework to train policies for typical MoMa tasks that makes use of the most favorable subspace of the robot's action space to address each sub-objective.",
  "doi": "10.48550/arXiv.2305.04866",
  "oa_pdf": "https://arxiv.org/pdf/2305.04866",
  "s2_authors": [
   {
    "name": "Jiaheng Hu",
    "id": "81703072",
    "h_index": 13,
    "papers": 29
   },
   {
    "name": "P. Stone",
    "id": "144848112",
    "h_index": 95,
    "papers": 704
   },
   {
    "name": "Roberto Mart\u00edn-Mart\u00edn",
    "id": "2316638007",
    "h_index": 6,
    "papers": 12
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "rl-control",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2305.04866v4",
  "pdf_url": "https://arxiv.org/pdf/2305.04866v4",
  "html_url": "https://arxiv.org/html/2305.04866v4",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.08
 },
 {
  "id": "2305.01569",
  "slug": "pick-a-pic-an-open-dataset-of-user-preferences-for-text-to-image-gener",
  "title": "Pick-a-Pic: An Open Dataset of User Preferences for Text-to-Image Generation",
  "abstract": "The ability to collect a large dataset of human preferences from text-to-image users is usually limited to companies, making such datasets inaccessible to the public. To address this issue, we create a web app that enables text-to-image users to generate images and specify their preferences. Using this web app we build Pick-a-Pic, a large, open dataset of text-to-image prompts and real users' preferences over generated images. We leverage this dataset to train a CLIP-based scoring function, PickScore, which exhibits superhuman performance on the task of predicting human preferences. Then, we test PickScore's ability to perform model evaluation and observe that it correlates better with human rankings than other automatic evaluation metrics. Therefore, we recommend using PickScore for evaluating future text-to-image generation models, and using Pick-a-Pic prompts as a more relevant dataset than MS-COCO. Finally, we demonstrate how PickScore can enhance existing text-to-image models via ranking.",
  "published": "2023-05-02",
  "updated": "2023-11-23",
  "year": "2023",
  "authors": [
   "Yuval Kirstain",
   "Adam Polyak",
   "Uriel Singer",
   "Shahbuland Matiana",
   "Joe Penna",
   "Omer Levy"
  ],
  "author_count": 6,
  "categories": [
   "cs.CV",
   "cs.AI"
  ],
  "primary_category": "cs.CV",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 1103,
  "influential_citations": 241,
  "tldr": "This work builds Pick-a-Pic, a large, open dataset of text-to-image prompts and real users' preferences over generated images, and uses it to train a CLIP-based scoring function, PickScore, which exhibits superhuman performance on the task of predicting human preferences.",
  "doi": "10.48550/arXiv.2305.01569",
  "oa_pdf": "http://arxiv.org/pdf/2305.01569",
  "s2_authors": [
   {
    "name": "Yuval Kirstain",
    "id": "2044194129",
    "h_index": 9,
    "papers": 10
   },
   {
    "name": "Adam Polyak",
    "id": "33964593",
    "h_index": 28,
    "papers": 38
   },
   {
    "name": "Uriel Singer",
    "id": "88622696",
    "h_index": 13,
    "papers": 27
   },
   {
    "name": "Shahbuland Matiana",
    "id": "2131098364",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Joe Penna",
    "id": "2215904682",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Omer Levy",
    "id": "39455775",
    "h_index": 53,
    "papers": 94
   }
  ],
  "comment": "",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2305.01569v2",
  "pdf_url": "https://arxiv.org/pdf/2305.01569v2",
  "html_url": "https://arxiv.org/html/2305.01569v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "2304.13705",
  "slug": "learning-fine-grained-bimanual-manipulation-with-low-cost-hardware",
  "title": "Learning Fine-Grained Bimanual Manipulation with Low-Cost Hardware",
  "abstract": "Fine manipulation tasks, such as threading cable ties or slotting a battery, are notoriously difficult for robots because they require precision, careful coordination of contact forces, and closed-loop visual feedback. Performing these tasks typically requires high-end robots, accurate sensors, or careful calibration, which can be expensive and difficult to set up. Can learning enable low-cost and imprecise hardware to perform these fine manipulation tasks? We present a low-cost system that performs end-to-end imitation learning directly from real demonstrations, collected with a custom teleoperation interface. Imitation learning, however, presents its own challenges, particularly in high-precision domains: errors in the policy can compound over time, and human demonstrations can be non-stationary. To address these challenges, we develop a simple yet novel algorithm, Action Chunking with Transformers (ACT), which learns a generative model over action sequences. ACT allows the robot to learn 6 difficult tasks in the real world, such as opening a translucent condiment cup and slotting a battery with 80-90% success, with only 10 minutes worth of demonstrations. Project website: https://tonyzhaozh.github.io/aloha/",
  "published": "2023-04-23",
  "updated": "2023-04-23",
  "year": "2023",
  "authors": [
   "Tony Z. Zhao",
   "Vikash Kumar",
   "Sergey Levine",
   "Chelsea Finn"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "RSS",
  "venue_source": "semantic-scholar",
  "citations": 2214,
  "influential_citations": 361,
  "tldr": "A low-cost system that performs end-to-end imitation learning directly from real demonstrations, collected with a custom teleoperation interface, and develops a simple yet novel algorithm, Action Chunking with Transformers (ACT), which learns a generative model over action sequences.",
  "doi": "10.48550/arXiv.2304.13705",
  "oa_pdf": "http://arxiv.org/pdf/2304.13705",
  "s2_authors": [
   {
    "name": "Tony Zhao",
    "id": "145914976",
    "h_index": 17,
    "papers": 19
   },
   {
    "name": "Vikash Kumar",
    "id": "2109446216",
    "h_index": 38,
    "papers": 76
   },
   {
    "name": "S. Levine",
    "id": "1736651",
    "h_index": 167,
    "papers": 551
   },
   {
    "name": "Chelsea Finn",
    "id": "46881670",
    "h_index": 94,
    "papers": 240
   }
  ],
  "comment": "",
  "topics": [
   "egocentric-data",
   "imitation-diffusion",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2304.13705v1",
  "pdf_url": "https://arxiv.org/pdf/2304.13705v1",
  "html_url": "https://arxiv.org/html/2304.13705v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "2304.11277",
  "slug": "pytorch-fsdp-experiences-on-scaling-fully-sharded-data-parallel",
  "title": "PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel",
  "abstract": "It is widely acknowledged that large models have the potential to deliver superior performance across a broad range of domains. Despite the remarkable progress made in the field of machine learning systems research, which has enabled the development and exploration of large models, such abilities remain confined to a small group of advanced users and industry leaders, resulting in an implicit technical barrier for the wider community to access and leverage these technologies. In this paper, we introduce PyTorch Fully Sharded Data Parallel (FSDP) as an industry-grade solution for large model training. FSDP has been closely co-designed with several key PyTorch core components including Tensor implementation, dispatcher system, and CUDA memory caching allocator, to provide non-intrusive user experiences and high training efficiency. Additionally, FSDP natively incorporates a range of techniques and settings to optimize resource utilization across a variety of hardware configurations. The experimental results demonstrate that FSDP is capable of achieving comparable performance to Distributed Data Parallel while providing support for significantly larger models with near-linear scalability in terms of TFLOPS.",
  "published": "2023-04-21",
  "updated": "2023-09-12",
  "year": "2023",
  "authors": [
   "Yanli Zhao",
   "Andrew Gu",
   "Rohan Varma",
   "Liang Luo",
   "Chien-Chin Huang",
   "Min Xu",
   "Less Wright",
   "Hamid Shojanazeri",
   "Myle Ott",
   "Sam Shleifer",
   "Alban Desmaison",
   "Can Balioglu",
   "Pritam Damania",
   "Bernard Nguyen",
   "Geeta Chauhan",
   "Yuchen Hao",
   "Ajit Mathews",
   "Shen Li"
  ],
  "author_count": 18,
  "categories": [
   "cs.DC",
   "cs.AI",
   "cs.LG",
   "cs.PF"
  ],
  "primary_category": "cs.DC",
  "venue": "",
  "venue_source": "",
  "citations": 843,
  "influential_citations": 64,
  "tldr": "PyTorch Fully Sharded Data Parallel is introduced as an industry-grade solution for large model training and is capable of achieving comparable performance to Distributed Data Parallel while providing support for significantly larger models with near-linear scalability in terms of TFLOPS.",
  "doi": "10.48550/arXiv.2304.11277",
  "oa_pdf": "https://arxiv.org/pdf/2304.11277",
  "s2_authors": [
   {
    "name": "Yanli Zhao",
    "id": "2109766944",
    "h_index": 6,
    "papers": 6
   },
   {
    "name": "A. Gu",
    "id": "152359468",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "R. Varma",
    "id": "39961043",
    "h_index": 12,
    "papers": 31
   },
   {
    "name": "Liangchen Luo",
    "id": "51225788",
    "h_index": 12,
    "papers": 23
   },
   {
    "name": "Chien-chin Huang",
    "id": "153622443",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "Min Xu",
    "id": "2041286121",
    "h_index": 9,
    "papers": 17
   },
   {
    "name": "Less Wright",
    "id": "2115145357",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Hamid Shojanazeri",
    "id": "2343773236",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Myle Ott",
    "id": "40511414",
    "h_index": 38,
    "papers": 135
   },
   {
    "name": "Sam Shleifer",
    "id": "88728159",
    "h_index": 16,
    "papers": 50
   },
   {
    "name": "Alban Desmaison",
    "id": "3050846",
    "h_index": 12,
    "papers": 21
   },
   {
    "name": "Can Balioglu",
    "id": "2280922389",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Bernard Nguyen",
    "id": "2215213203",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Geeta Chauhan",
    "id": "1896604752",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Yuchen Hao",
    "id": "8362396",
    "h_index": 12,
    "papers": 20
   },
   {
    "name": "Shen Li",
    "id": "2153698839",
    "h_index": 5,
    "papers": 6
   }
  ],
  "comment": "",
  "topics": [
   "navigation",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2304.11277v2",
  "pdf_url": "https://arxiv.org/pdf/2304.11277v2",
  "html_url": "https://arxiv.org/html/2304.11277v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.93
 },
 {
  "id": "2304.08488",
  "slug": "affordances-from-human-videos-as-a-versatile-representation-for-roboti",
  "title": "Affordances from Human Videos as a Versatile Representation for Robotics",
  "abstract": "Building a robot that can understand and learn to interact by watching humans has inspired several vision problems. However, despite some successful results on static datasets, it remains unclear how current models can be used on a robot directly. In this paper, we aim to bridge this gap by leveraging videos of human interactions in an environment centric manner. Utilizing internet videos of human behavior, we train a visual affordance model that estimates where and how in the scene a human is likely to interact. The structure of these behavioral affordances directly enables the robot to perform many complex tasks. We show how to seamlessly integrate our affordance model with four robot learning paradigms including offline imitation learning, exploration, goal-conditioned learning, and action parameterization for reinforcement learning. We show the efficacy of our approach, which we call VRB, across 4 real world environments, over 10 different tasks, and 2 robotic platforms operating in the wild. Results, visualizations and videos at https://robo-affordances.github.io/",
  "published": "2023-04-17",
  "updated": "2023-04-17",
  "year": "2023",
  "authors": [
   "Shikhar Bahl",
   "Russell Mendonca",
   "Lili Chen",
   "Unnat Jain",
   "Deepak Pathak"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG",
   "cs.NE"
  ],
  "primary_category": "cs.RO",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 351,
  "influential_citations": 25,
  "tldr": "This paper uses internet videos of human behavior to train a visual affordance model that estimates where and how in the scene a human is likely to interact and shows how to seamlessly integrate this model with four robot learning paradigms including offline imitation learning, exploration, goal-conditioned learning, and action parameterization for reinforcement learning.",
  "doi": "10.1109/CVPR52729.2023.01324",
  "oa_pdf": "https://arxiv.org/pdf/2304.08488",
  "s2_authors": [
   {
    "name": "Shikhar Bahl",
    "id": "8527563",
    "h_index": 20,
    "papers": 38
   },
   {
    "name": "R. Mendonca",
    "id": "35509365",
    "h_index": 15,
    "papers": 24
   },
   {
    "name": "Lili Chen",
    "id": "2108435457",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Unnat Jain",
    "id": "10680632",
    "h_index": 23,
    "papers": 33
   },
   {
    "name": "Deepak Pathak",
    "id": "2004879394",
    "h_index": 24,
    "papers": 32
   }
  ],
  "comment": "Accepted at CVPR 2023. Website at https://robo-affordances.github.io/",
  "topics": [
   "egocentric-data",
   "imitation-diffusion",
   "rl-control",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2304.08488v1",
  "pdf_url": "https://arxiv.org/pdf/2304.08488v1",
  "html_url": "https://arxiv.org/html/2304.08488v1",
  "code_url": "https://robo-affordances.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.05
 },
 {
  "id": "2304.08485",
  "slug": "visual-instruction-tuning",
  "title": "Visual Instruction Tuning",
  "abstract": "Instruction tuning large language models (LLMs) using machine-generated instruction-following data has improved zero-shot capabilities on new tasks, but the idea is less explored in the multimodal field. In this paper, we present the first attempt to use language-only GPT-4 to generate multimodal language-image instruction-following data. By instruction tuning on such generated data, we introduce LLaVA: Large Language and Vision Assistant, an end-to-end trained large multimodal model that connects a vision encoder and LLM for general-purpose visual and language understanding.Our early experiments show that LLaVA demonstrates impressive multimodel chat abilities, sometimes exhibiting the behaviors of multimodal GPT-4 on unseen images/instructions, and yields a 85.1% relative score compared with GPT-4 on a synthetic multimodal instruction-following dataset. When fine-tuned on Science QA, the synergy of LLaVA and GPT-4 achieves a new state-of-the-art accuracy of 92.53%. We make GPT-4 generated visual instruction tuning data, our model and code base publicly available.",
  "published": "2023-04-17",
  "updated": "2023-12-11",
  "year": "2023",
  "authors": [
   "Haotian Liu",
   "Chunyuan Li",
   "Qingyang Wu",
   "Yong Jae Lee"
  ],
  "author_count": 4,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.CL",
   "cs.LG"
  ],
  "primary_category": "cs.CV",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 10854,
  "influential_citations": 1677,
  "tldr": "This paper presents LLaVA: Large Language and Vision Assistant, an end-to-end trained large multimodal model that connects a vision encoder and LLM for general-purpose visual and language understanding and introduces GPT-4 generated visual instruction tuning data, the model and code base publicly available.",
  "doi": "10.48550/arXiv.2304.08485",
  "oa_pdf": "http://arxiv.org/pdf/2304.08485",
  "s2_authors": [
   {
    "name": "Haotian Liu",
    "id": "2143856368",
    "h_index": 15,
    "papers": 20
   },
   {
    "name": "Chunyuan Li",
    "id": "2109737569",
    "h_index": 49,
    "papers": 95
   },
   {
    "name": "Qingyang Wu",
    "id": "31060482",
    "h_index": 11,
    "papers": 22
   },
   {
    "name": "Yong Jae Lee",
    "id": "144756076",
    "h_index": 47,
    "papers": 96
   }
  ],
  "comment": "NeurIPS 2023 Oral; project page: https://llava-vl.github.io/",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2304.08485v2",
  "pdf_url": "https://arxiv.org/pdf/2304.08485v2",
  "html_url": "https://arxiv.org/html/2304.08485v2",
  "code_url": "https://llava-vl.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.5
 },
 {
  "id": "2304.07193",
  "slug": "dinov2-learning-robust-visual-features-without-supervision",
  "title": "DINOv2: Learning Robust Visual Features without Supervision",
  "abstract": "The recent breakthroughs in natural language processing for model pretraining on large quantities of data have opened the way for similar foundation models in computer vision. These models could greatly simplify the use of images in any system by producing all-purpose visual features, i.e., features that work across image distributions and tasks without finetuning. This work shows that existing pretraining methods, especially self-supervised methods, can produce such features if trained on enough curated data from diverse sources. We revisit existing approaches and combine different techniques to scale our pretraining in terms of data and model size. Most of the technical contributions aim at accelerating and stabilizing the training at scale. In terms of data, we propose an automatic pipeline to build a dedicated, diverse, and curated image dataset instead of uncurated data, as typically done in the self-supervised literature. In terms of models, we train a ViT model (Dosovitskiy et al., 2020) with 1B parameters and distill it into a series of smaller models that surpass the best available all-purpose features, OpenCLIP (Ilharco et al., 2021) on most of the benchmarks at image and pixel levels.",
  "published": "2023-04-14",
  "updated": "2024-02-02",
  "year": "2023",
  "authors": [
   "Maxime Oquab",
   "Timoth\u00e9e Darcet",
   "Th\u00e9o Moutakanni",
   "Huy Vo",
   "Marc Szafraniec",
   "Vasil Khalidov",
   "Pierre Fernandez",
   "Daniel Haziza",
   "Francisco Massa",
   "Alaaeldin El-Nouby",
   "Mahmoud Assran",
   "Nicolas Ballas",
   "Wojciech Galuba",
   "Russell Howes",
   "Po-Yao Huang",
   "Shang-Wen Li",
   "Ishan Misra",
   "Michael Rabbat",
   "Vasu Sharma",
   "Gabriel Synnaeve",
   "Hu Xu",
   "Herv\u00e9 Jegou",
   "Julien Mairal",
   "Patrick Labatut",
   "Armand Joulin",
   "Piotr Bojanowski"
  ],
  "author_count": 26,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "Trans. Mach. Learn. Res.",
  "venue_source": "semantic-scholar",
  "citations": 10015,
  "influential_citations": 1451,
  "tldr": "This work revisits existing approaches and combines different techniques to scale the pretraining in terms of data and model size, and proposes an automatic pipeline to build a dedicated, diverse, and curated image dataset instead of uncurated data, as typically done in the self-supervised literature.",
  "doi": "10.48550/arXiv.2304.07193",
  "oa_pdf": "http://arxiv.org/pdf/2304.07193",
  "s2_authors": [
   {
    "name": "M. Oquab",
    "id": "2093491",
    "h_index": 12,
    "papers": 27
   },
   {
    "name": "Timoth\u00e9e Darcet",
    "id": "2214523349",
    "h_index": 7,
    "papers": 7
   },
   {
    "name": "Th\u00e9o Moutakanni",
    "id": "1752699898",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "Huy V. Vo",
    "id": "3323377",
    "h_index": 14,
    "papers": 31
   },
   {
    "name": "Marc Szafraniec",
    "id": "23994377",
    "h_index": 11,
    "papers": 17
   },
   {
    "name": "Vasil Khalidov",
    "id": "2182694",
    "h_index": 16,
    "papers": 28
   },
   {
    "name": "Pierre Fernandez",
    "id": "2147013351",
    "h_index": 12,
    "papers": 15
   },
   {
    "name": "Daniel Haziza",
    "id": "40864100",
    "h_index": 11,
    "papers": 24
   },
   {
    "name": "Francisco Massa",
    "id": "1403239967",
    "h_index": 11,
    "papers": 14
   },
   {
    "name": "Alaaeldin El-Nouby",
    "id": "1388811741",
    "h_index": 20,
    "papers": 26
   },
   {
    "name": "Mahmoud Assran",
    "id": "38698856",
    "h_index": 15,
    "papers": 21
   },
   {
    "name": "Nicolas Ballas",
    "id": "2482072",
    "h_index": 30,
    "papers": 69
   },
   {
    "name": "Wojciech Galuba",
    "id": "2247475926",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Russ Howes",
    "id": "1410913697",
    "h_index": 12,
    "papers": 40
   },
   {
    "name": "Po-Yao (Bernie) Huang",
    "id": "2319973",
    "h_index": 27,
    "papers": 47
   },
   {
    "name": "Shang-Wen Li",
    "id": "2530311",
    "h_index": 32,
    "papers": 90
   },
   {
    "name": "Ishan Misra",
    "id": "1806773",
    "h_index": 44,
    "papers": 73
   },
   {
    "name": "Michael G. Rabbat",
    "id": "2066127975",
    "h_index": 25,
    "papers": 42
   },
   {
    "name": "Vasu Sharma",
    "id": "144582538",
    "h_index": 9,
    "papers": 21
   },
   {
    "name": "Gabriel Synnaeve",
    "id": "2282478",
    "h_index": 58,
    "papers": 209
   },
   {
    "name": "Hu Xu",
    "id": "2356046304",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Herv\u00e9 J\u00e9gou",
    "id": "1681054",
    "h_index": 52,
    "papers": 149
   },
   {
    "name": "J. Mairal",
    "id": "2599292",
    "h_index": 58,
    "papers": 162
   },
   {
    "name": "Patrick Labatut",
    "id": "1744868",
    "h_index": 20,
    "papers": 38
   },
   {
    "name": "Armand Joulin",
    "id": "2319608",
    "h_index": 72,
    "papers": 151
   },
   {
    "name": "Piotr Bojanowski",
    "id": "2329288",
    "h_index": 38,
    "papers": 77
   }
  ],
  "comment": "",
  "topics": [
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2304.07193v2",
  "pdf_url": "https://arxiv.org/pdf/2304.07193v2",
  "html_url": "https://arxiv.org/html/2304.07193v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "2304.03047",
  "slug": "etpnav-evolving-topological-planning-for-vision-language-navigation-in",
  "title": "ETPNav: Evolving Topological Planning for Vision-Language Navigation in Continuous Environments",
  "abstract": "Vision-language navigation is a task that requires an agent to follow instructions to navigate in environments. It becomes increasingly crucial in the field of embodied AI, with potential applications in autonomous navigation, search and rescue, and human-robot interaction. In this paper, we propose to address a more practical yet challenging counterpart setting - vision-language navigation in continuous environments (VLN-CE). To develop a robust VLN-CE agent, we propose a new navigation framework, ETPNav, which focuses on two critical skills: 1) the capability to abstract environments and generate long-range navigation plans, and 2) the ability of obstacle-avoiding control in continuous environments. ETPNav performs online topological mapping of environments by self-organizing predicted waypoints along a traversed path, without prior environmental experience. It privileges the agent to break down the navigation procedure into high-level planning and low-level control. Concurrently, ETPNav utilizes a transformer-based cross-modal planner to generate navigation plans based on topological maps and instructions. The plan is then performed through an obstacle-avoiding controller that leverages a trial-and-error heuristic to prevent navigation from getting stuck in obstacles. Experimental results demonstrate the effectiveness of the proposed method. ETPNav yields more than 10% and 20% improvements over prior state-of-the-art on R2R-CE and RxR-CE datasets, respectively. Our code is available at https://github.com/MarSaKi/ETPNav.",
  "published": "2023-04-06",
  "updated": "2024-01-22",
  "year": "2023",
  "authors": [
   "Dong An",
   "Hanqing Wang",
   "Wenguan Wang",
   "Zun Wang",
   "Yan Huang",
   "Keji He",
   "Liang Wang"
  ],
  "author_count": 7,
  "categories": [
   "cs.CV",
   "cs.CL",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 253,
  "influential_citations": 41,
  "tldr": "A new navigation framework, ETPNav, which focuses on two critical skills: the capability to abstract environments and generate long-range navigation plans, and the ability of obstacle-avoiding control in continuous environments.",
  "doi": "10.1109/TPAMI.2024.3386695",
  "oa_pdf": "http://arxiv.org/pdf/2304.03047",
  "s2_authors": [
   {
    "name": "Dongyan An",
    "id": "65990035",
    "h_index": 8,
    "papers": 14
   },
   {
    "name": "H. Wang",
    "id": "3228896",
    "h_index": 37,
    "papers": 102
   },
   {
    "name": "Wenguan Wang",
    "id": "2693875",
    "h_index": 78,
    "papers": 139
   },
   {
    "name": "Zun Wang",
    "id": "2382933980",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Yan Huang",
    "id": "144368930",
    "h_index": 44,
    "papers": 427
   },
   {
    "name": "Keji He",
    "id": "51054943",
    "h_index": 7,
    "papers": 17
   },
   {
    "name": "Liang Wang",
    "id": "2144696372",
    "h_index": 2,
    "papers": 3
   }
  ],
  "comment": "Project page: https://github.com/MarSaKi/ETPNav",
  "topics": [
   "navigation",
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2304.03047v3",
  "pdf_url": "https://arxiv.org/pdf/2304.03047v3",
  "html_url": "https://arxiv.org/html/2304.03047v3",
  "code_url": "https://github.com/MarSaKi/ETPNav",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.4
 },
 {
  "id": "2304.00410",
  "slug": "asc-adaptive-skill-coordination-for-robotic-mobile-manipulation",
  "title": "ASC: Adaptive Skill Coordination for Robotic Mobile Manipulation",
  "abstract": "We present Adaptive Skill Coordination (ASC) -- an approach for accomplishing long-horizon tasks like mobile pick-and-place (i.e., navigating to an object, picking it, navigating to another location, and placing it). ASC consists of three components -- (1) a library of basic visuomotor skills (navigation, pick, place), (2) a skill coordination policy that chooses which skill to use when, and (3) a corrective policy that adapts pre-trained skills in out-of-distribution states. All components of ASC rely only on onboard visual and proprioceptive sensing, without requiring detailed maps with obstacle layouts or precise object locations, easing real-world deployment. We train ASC in simulated indoor environments, and deploy it zero-shot (without any real-world experience or fine-tuning) on the Boston Dynamics Spot robot in eight novel real-world environments (one apartment, one lab, two microkitchens, two lounges, one office space, one outdoor courtyard). In rigorous quantitative comparisons in two environments, ASC achieves near-perfect performance (59/60 episodes, or 98%), while sequentially executing skills succeeds in only 44/60 (73%) episodes. Extensive perturbation experiments show that ASC is robust to hand-off errors, changes in the environment layout, dynamic obstacles (e.g., people), and unexpected disturbances. Supplementary videos at adaptiveskillcoordination.github.io.",
  "published": "2023-04-01",
  "updated": "2023-11-19",
  "year": "2023",
  "authors": [
   "Naoki Yokoyama",
   "Alex Clegg",
   "Joanne Truong",
   "Eric Undersander",
   "Tsung-Yen Yang",
   "Sergio Arnaud",
   "Sehoon Ha",
   "Dhruv Batra",
   "Akshara Rai"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "RA-L",
  "venue_source": "semantic-scholar",
  "citations": 90,
  "influential_citations": 4,
  "tldr": "Adaptive Skill Coordination is presented \u2013 an approach for accomplishing long-horizon tasks like mobile pick-and-place (i.e., navigating to an object, picking it, navigating to another location, and placing it) that is robust to hand-off errors, changes in the environment layout, dynamic obstacles, and unexpected disturbances.",
  "doi": "10.1109/LRA.2023.3336109",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Naoki Yokoyama",
    "id": "49327690",
    "h_index": 12,
    "papers": 19
   },
   {
    "name": "Alexander Clegg",
    "id": "30933599",
    "h_index": 20,
    "papers": 24
   },
   {
    "name": "Eric Undersander",
    "id": "29994440",
    "h_index": 10,
    "papers": 12
   },
   {
    "name": "Sergio Arnaud",
    "id": "2213251873",
    "h_index": 9,
    "papers": 15
   },
   {
    "name": "Sehoon Ha",
    "id": "2248552",
    "h_index": 26,
    "papers": 69
   },
   {
    "name": "Dhruv Batra",
    "id": "1746610",
    "h_index": 87,
    "papers": 327
   },
   {
    "name": "Akshara Rai",
    "id": "2762463",
    "h_index": 26,
    "papers": 45
   }
  ],
  "comment": "",
  "topics": [
   "navigation",
   "safety-eval"
  ],
  "orgs": [
   "Boston Dynamics"
  ],
  "abs_url": "https://arxiv.org/abs/2304.00410v5",
  "pdf_url": "https://arxiv.org/pdf/2304.00410v5",
  "html_url": "https://arxiv.org/html/2304.00410v5",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.96
 },
 {
  "id": "2304.00341",
  "slug": "jacobinerf-nerf-shaping-with-mutual-information-gradients",
  "title": "JacobiNeRF: NeRF Shaping with Mutual Information Gradients",
  "abstract": "We propose a method that trains a neural radiance field (NeRF) to encode not only the appearance of the scene but also semantic correlations between scene points, regions, or entities -- aiming to capture their mutual co-variation patterns. In contrast to the traditional first-order photometric reconstruction objective, our method explicitly regularizes the learning dynamics to align the Jacobians of highly-correlated entities, which proves to maximize the mutual information between them under random scene perturbations. By paying attention to this second-order information, we can shape a NeRF to express semantically meaningful synergies when the network weights are changed by a delta along the gradient of a single entity, region, or even a point. To demonstrate the merit of this mutual information modeling, we leverage the coordinated behavior of scene entities that emerges from our shaping to perform label propagation for semantic and instance segmentation. Our experiments show that a JacobiNeRF is more efficient in propagating annotations among 2D pixels and 3D points compared to NeRFs without mutual information shaping, especially in extremely sparse label regimes -- thus reducing annotation burden. The same machinery can further be used for entity selection or scene modifications.",
  "published": "2023-04-01",
  "updated": "2023-04-01",
  "year": "2023",
  "authors": [
   "Xiaomeng Xu",
   "Yanchao Yang",
   "Kaichun Mo",
   "Boxiao Pan",
   "Li Yi",
   "Leonidas Guibas"
  ],
  "author_count": 6,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 18,
  "influential_citations": 3,
  "tldr": "",
  "doi": "10.1109/CVPR52729.2023.01583",
  "oa_pdf": "https://arxiv.org/pdf/2304.00341",
  "s2_authors": [
   {
    "name": "Xiaomeng Xu",
    "id": "2158827128",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Yanchao Yang",
    "id": "2305637601",
    "h_index": 7,
    "papers": 16
   },
   {
    "name": "Kaichun Mo",
    "id": "2216377",
    "h_index": 23,
    "papers": 39
   },
   {
    "name": "Boxiao Pan",
    "id": "52170427",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "L. Yi",
    "id": "47782132",
    "h_index": 30,
    "papers": 109
   },
   {
    "name": "L. Guibas",
    "id": "51352814",
    "h_index": 76,
    "papers": 202
   }
  ],
  "comment": "",
  "topics": [
   "spatial-3d",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2304.00341v1",
  "pdf_url": "https://arxiv.org/pdf/2304.00341v1",
  "html_url": "https://arxiv.org/html/2304.00341v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.78
 },
 {
  "id": "2303.18240",
  "slug": "where-are-we-in-the-search-for-an-artificial-visual-cortex-for-embodie",
  "title": "Where are we in the search for an Artificial Visual Cortex for Embodied Intelligence?",
  "abstract": "We present the largest and most comprehensive empirical study of pre-trained visual representations (PVRs) or visual 'foundation models' for Embodied AI. First, we curate CortexBench, consisting of 17 different tasks spanning locomotion, navigation, dexterous, and mobile manipulation. Next, we systematically evaluate existing PVRs and find that none are universally dominant. To study the effect of pre-training data size and diversity, we combine over 4,000 hours of egocentric videos from 7 different sources (over 4.3M images) and ImageNet to train different-sized vision transformers using Masked Auto-Encoding (MAE) on slices of this data. Contrary to inferences from prior work, we find that scaling dataset size and diversity does not improve performance universally (but does so on average). Our largest model, named VC-1, outperforms all prior PVRs on average but does not universally dominate either. Next, we show that task- or domain-specific adaptation of VC-1 leads to substantial gains, with VC-1 (adapted) achieving competitive or superior performance than the best known results on all of the benchmarks in CortexBench. Finally, we present real-world hardware experiments, in which VC-1 and VC-1 (adapted) outperform the strongest pre-existing PVR. Overall, this paper presents no new techniques but a rigorous systematic evaluation, a broad set of findings about PVRs (that in some cases, refute those made in narrow domains in prior work), and open-sourced code and models (that required over 10,000 GPU-hours to train) for the benefit of the research community.",
  "published": "2023-03-31",
  "updated": "2024-02-01",
  "year": "2023",
  "authors": [
   "Arjun Majumdar",
   "Karmesh Yadav",
   "Sergio Arnaud",
   "Yecheng Jason Ma",
   "Claire Chen",
   "Sneha Silwal",
   "Aryan Jain",
   "Vincent-Pierre Berges",
   "Pieter Abbeel",
   "Jitendra Malik",
   "Dhruv Batra",
   "Yixin Lin",
   "Oleksandr Maksymets",
   "Aravind Rajeswaran",
   "Franziska Meier"
  ],
  "author_count": 15,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.LG",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 315,
  "influential_citations": 38,
  "tldr": "The largest and most comprehensive empirical study of pre-trained visual representations (PVRs) or visual 'foundation models' for Embodied AI is presented, and the largest model, named VC-1, outperforms all prior PVRs on average but does not universally dominate either.",
  "doi": "10.48550/arXiv.2303.18240",
  "oa_pdf": "http://arxiv.org/pdf/2303.18240",
  "s2_authors": [
   {
    "name": "Arjun Majumdar",
    "id": "2905057",
    "h_index": 15,
    "papers": 28
   },
   {
    "name": "K. Yadav",
    "id": "1838683872",
    "h_index": 10,
    "papers": 21
   },
   {
    "name": "Sergio Arnaud",
    "id": "2213251873",
    "h_index": 9,
    "papers": 15
   },
   {
    "name": "Y. Ma",
    "id": "2130215451",
    "h_index": 16,
    "papers": 23
   },
   {
    "name": "C. Chen",
    "id": "2143886539",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "S. Silwal",
    "id": "2159712500",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Aryan Jain",
    "id": "2148306679",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Vincent-Pierre Berges",
    "id": "51266557",
    "h_index": 12,
    "papers": 19
   },
   {
    "name": "P. Abbeel",
    "id": "1689992",
    "h_index": 165,
    "papers": 596
   },
   {
    "name": "J. Malik",
    "id": "153652147",
    "h_index": 64,
    "papers": 99
   },
   {
    "name": "Dhruv Batra",
    "id": "1746610",
    "h_index": 87,
    "papers": 327
   },
   {
    "name": "Yixin Lin",
    "id": "2404160135",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Oleksandr Maksymets",
    "id": "90536527",
    "h_index": 14,
    "papers": 20
   },
   {
    "name": "A. Rajeswaran",
    "id": "19275599",
    "h_index": 34,
    "papers": 58
   },
   {
    "name": "Franziska Meier",
    "id": "153145615",
    "h_index": 31,
    "papers": 77
   }
  ],
  "comment": "Project website: https://eai-vc.github.io",
  "topics": [
   "dexterous-manipulation",
   "humanoids",
   "egocentric-data",
   "navigation",
   "foundation-pretraining",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2303.18240v2",
  "pdf_url": "https://arxiv.org/pdf/2303.18240v2",
  "html_url": "https://arxiv.org/html/2303.18240v2",
  "code_url": "https://eai-vc.github.io",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.0
 },
 {
  "id": "2303.14389",
  "slug": "mdtv2-masked-diffusion-transformer-is-a-strong-image-synthesizer",
  "title": "MDTv2: Masked Diffusion Transformer is a Strong Image Synthesizer",
  "abstract": "Despite its success in image synthesis, we observe that diffusion probabilistic models (DPMs) often lack contextual reasoning ability to learn the relations among object parts in an image, leading to a slow learning process. To solve this issue, we propose a Masked Diffusion Transformer (MDT) that introduces a mask latent modeling scheme to explicitly enhance the DPMs' ability to contextual relation learning among object semantic parts in an image. During training, MDT operates in the latent space to mask certain tokens. Then, an asymmetric diffusion transformer is designed to predict masked tokens from unmasked ones while maintaining the diffusion generation process. Our MDT can reconstruct the full information of an image from its incomplete contextual input, thus enabling it to learn the associated relations among image tokens. We further improve MDT with a more efficient macro network structure and training strategy, named MDTv2. Experimental results show that MDTv2 achieves superior image synthesis performance, e.g., a new SOTA FID score of 1.58 on the ImageNet dataset, and has more than 10x faster learning speed than the previous SOTA DiT. The source code is released at https://github.com/sail-sg/MDT.",
  "published": "2023-03-25",
  "updated": "2024-02-21",
  "year": "2023",
  "authors": [
   "Shanghua Gao",
   "Pan Zhou",
   "Ming-Ming Cheng",
   "Shuicheng Yan"
  ],
  "author_count": 4,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "ICCV",
  "venue_source": "semantic-scholar",
  "citations": 317,
  "influential_citations": 43,
  "tldr": "A Masked Diffusion Transformer (MDT) is proposed that introduces a mask latent modeling scheme to explicitly enhance the DPMs\u2019 ability to contextual relation learning among object semantic parts in an image.",
  "doi": "10.1109/ICCV51070.2023.02117",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Shanghua Gao",
    "id": "3180489",
    "h_index": 18,
    "papers": 41
   },
   {
    "name": "Pan Zhou",
    "id": "2153245275",
    "h_index": 20,
    "papers": 31
   },
   {
    "name": "Mingg-Ming Cheng",
    "id": "1557350184",
    "h_index": 22,
    "papers": 36
   },
   {
    "name": "Shuicheng Yan",
    "id": "2111618103",
    "h_index": 14,
    "papers": 32
   }
  ],
  "comment": "Extension of ICCV 2023 work, source code: https://github.com/sail-sg/MDT",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2303.14389v2",
  "pdf_url": "https://arxiv.org/pdf/2303.14389v2",
  "html_url": "https://arxiv.org/html/2303.14389v2",
  "code_url": "https://github.com/sail-sg/MDT",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.0
 },
 {
  "id": "2303.14158",
  "slug": "bundlesdf-neural-6-dof-tracking-and-3d-reconstruction-of-unknown-objec",
  "title": "BundleSDF: Neural 6-DoF Tracking and 3D Reconstruction of Unknown Objects",
  "abstract": "We present a near real-time method for 6-DoF tracking of an unknown object from a monocular RGBD video sequence, while simultaneously performing neural 3D reconstruction of the object. Our method works for arbitrary rigid objects, even when visual texture is largely absent. The object is assumed to be segmented in the first frame only. No additional information is required, and no assumption is made about the interaction agent. Key to our method is a Neural Object Field that is learned concurrently with a pose graph optimization process in order to robustly accumulate information into a consistent 3D representation capturing both geometry and appearance. A dynamic pool of posed memory frames is automatically maintained to facilitate communication between these threads. Our approach handles challenging sequences with large pose changes, partial and full occlusion, untextured surfaces, and specular highlights. We show results on HO3D, YCBInEOAT, and BEHAVE datasets, demonstrating that our method significantly outperforms existing approaches. Project page: https://bundlesdf.github.io",
  "published": "2023-03-24",
  "updated": "2023-03-24",
  "year": "2023",
  "authors": [
   "Bowen Wen",
   "Jonathan Tremblay",
   "Valts Blukis",
   "Stephen Tyree",
   "Thomas Muller",
   "Alex Evans",
   "Dieter Fox",
   "Jan Kautz",
   "Stan Birchfield"
  ],
  "author_count": 9,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.GR",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 246,
  "influential_citations": 24,
  "tldr": "",
  "doi": "10.1109/CVPR52729.2023.00066",
  "oa_pdf": "https://arxiv.org/pdf/2303.14158",
  "s2_authors": [
   {
    "name": "Bowen Wen",
    "id": "101349262",
    "h_index": 17,
    "papers": 25
   },
   {
    "name": "Jonathan Tremblay",
    "id": "31943350",
    "h_index": 27,
    "papers": 56
   },
   {
    "name": "Valts Blukis",
    "id": "32481910",
    "h_index": 13,
    "papers": 20
   },
   {
    "name": "Stephen Tyree",
    "id": "2342481",
    "h_index": 27,
    "papers": 41
   },
   {
    "name": "T. Muller",
    "id": "2072669753",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Alex Evans",
    "id": "2114739960",
    "h_index": 11,
    "papers": 16
   },
   {
    "name": "D. Fox",
    "id": "145197953",
    "h_index": 133,
    "papers": 428
   },
   {
    "name": "Jan Kautz",
    "id": "2376331447",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Stan Birchfield",
    "id": "2238841",
    "h_index": 52,
    "papers": 152
   }
  ],
  "comment": "CVPR 2023",
  "topics": [
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2303.14158v1",
  "pdf_url": "https://arxiv.org/pdf/2303.14158v1",
  "html_url": "https://arxiv.org/html/2303.14158v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.89
 },
 {
  "id": "2303.13446",
  "slug": "on-the-utility-of-koopman-operator-theory-in-learning-dexterous-manipu",
  "title": "On the Utility of Koopman Operator Theory in Learning Dexterous Manipulation Skills",
  "abstract": "Despite impressive dexterous manipulation capabilities enabled by learning-based approaches, we are yet to witness widespread adoption beyond well-resourced laboratories. This is likely due to practical limitations, such as significant computational burden, inscrutable learned behaviors, sensitivity to initialization, and the considerable technical expertise required for implementation. In this work, we investigate the utility of Koopman operator theory in alleviating these limitations. Koopman operators are simple yet powerful control-theoretic structures to represent complex nonlinear dynamics as linear systems in higher dimensions. Motivated by the fact that complex nonlinear dynamics underlie dexterous manipulation, we develop a Koopman operator-based imitation learning framework to learn the desired motions of both the robotic hand and the object simultaneously. We show that Koopman operators are surprisingly effective for dexterous manipulation and offer a number of unique benefits. Notably, policies can be learned analytically, drastically reducing computation burden and eliminating sensitivity to initialization and the need for painstaking hyperparameter optimization. Our experiments reveal that a Koopman operator-based approach can perform comparably to state-of-the-art imitation learning algorithms in terms of success rate and sample efficiency, while being an order of magnitude faster. Policy videos can be viewed at https://sites.google.com/view/kodex-corl.",
  "published": "2023-03-23",
  "updated": "2023-08-31",
  "year": "2023",
  "authors": [
   "Yunhai Han",
   "Mandy Xie",
   "Ye Zhao",
   "Harish Ravichandar"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 30,
  "influential_citations": 1,
  "tldr": "It is shown that a Koopman operator-based approach can perform comparably to state-of-the-art imitation learning algorithms in terms of success rate and sample efficiency, while being an order of magnitude faster.",
  "doi": "10.48550/arXiv.2303.13446",
  "oa_pdf": "https://arxiv.org/pdf/2303.13446",
  "s2_authors": [
   {
    "name": "Yunhai Han",
    "id": "1995513527",
    "h_index": 8,
    "papers": 14
   },
   {
    "name": "Mandy Xie",
    "id": "2089797867",
    "h_index": 8,
    "papers": 13
   },
   {
    "name": "Ye Zhao",
    "id": "97522088",
    "h_index": 19,
    "papers": 77
   },
   {
    "name": "H. Ravichandar",
    "id": "2138933",
    "h_index": 17,
    "papers": 74
   }
  ],
  "comment": "This work has been accepted for an oral presentation at CORL 2023",
  "topics": [
   "dexterous-manipulation",
   "imitation-diffusion"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2303.13446v4",
  "pdf_url": "https://arxiv.org/pdf/2303.13446v4",
  "html_url": "https://arxiv.org/html/2303.13446v4",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.99
 },
 {
  "id": "2303.12076",
  "slug": "dexterity-from-touch-self-supervised-pre-training-of-tactile-represent",
  "title": "Dexterity from Touch: Self-Supervised Pre-Training of Tactile Representations with Robotic Play",
  "abstract": "Teaching dexterity to multi-fingered robots has been a longstanding challenge in robotics. Most prominent work in this area focuses on learning controllers or policies that either operate on visual observations or state estimates derived from vision. However, such methods perform poorly on fine-grained manipulation tasks that require reasoning about contact forces or about objects occluded by the hand itself. In this work, we present T-Dex, a new approach for tactile-based dexterity, that operates in two phases. In the first phase, we collect 2.5 hours of play data, which is used to train self-supervised tactile encoders. This is necessary to bring high-dimensional tactile readings to a lower-dimensional embedding. In the second phase, given a handful of demonstrations for a dexterous task, we learn non-parametric policies that combine the tactile observations with visual ones. Across five challenging dexterous tasks, we show that our tactile-based dexterity models outperform purely vision and torque-based models by an average of 1.7X. Finally, we provide a detailed analysis on factors critical to T-Dex including the importance of play data, architectures, and representation learning.",
  "published": "2023-03-21",
  "updated": "2023-03-21",
  "year": "2023",
  "authors": [
   "Irmak Guzey",
   "Ben Evans",
   "Soumith Chintala",
   "Lerrel Pinto"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 108,
  "influential_citations": 7,
  "tldr": "This work presents T-Dex, a new approach for tactile-based dexterity, that operates in two phases, and provides a detailed analysis on factors critical to T- Dex including the importance of play data, architectures, and representation learning.",
  "doi": "10.48550/arXiv.2303.12076",
  "oa_pdf": "http://arxiv.org/pdf/2303.12076",
  "s2_authors": [
   {
    "name": "Irmak G\u00fczey",
    "id": "2212471096",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Ben Evans",
    "id": "2153473632",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Soumith Chintala",
    "id": "2127604",
    "h_index": 32,
    "papers": 49
   },
   {
    "name": "Lerrel Pinto",
    "id": "34026610",
    "h_index": 41,
    "papers": 70
   }
  ],
  "comment": "Video and code can be accessed here: https://tactile-dexterity.github.io/",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2303.12076v1",
  "pdf_url": "https://arxiv.org/pdf/2303.12076v1",
  "html_url": "https://arxiv.org/html/2303.12076v1",
  "code_url": "https://tactile-dexterity.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.54
 },
 {
  "id": "2303.11897",
  "slug": "tifa-accurate-and-interpretable-text-to-image-faithfulness-evaluation",
  "title": "TIFA: Accurate and Interpretable Text-to-Image Faithfulness Evaluation with Question Answering",
  "abstract": "Despite thousands of researchers, engineers, and artists actively working on improving text-to-image generation models, systems often fail to produce images that accurately align with the text inputs. We introduce TIFA (Text-to-Image Faithfulness evaluation with question Answering), an automatic evaluation metric that measures the faithfulness of a generated image to its text input via visual question answering (VQA). Specifically, given a text input, we automatically generate several question-answer pairs using a language model. We calculate image faithfulness by checking whether existing VQA models can answer these questions using the generated image. TIFA is a reference-free metric that allows for fine-grained and interpretable evaluations of generated images. TIFA also has better correlations with human judgments than existing metrics. Based on this approach, we introduce TIFA v1.0, a benchmark consisting of 4K diverse text inputs and 25K questions across 12 categories (object, counting, etc.). We present a comprehensive evaluation of existing text-to-image models using TIFA v1.0 and highlight the limitations and challenges of current models. For instance, we find that current text-to-image models, despite doing well on color and material, still struggle in counting, spatial relations, and composing multiple objects. We hope our benchmark will help carefully measure the research progress in text-to-image synthesis and provide valuable insights for further research.",
  "published": "2023-03-21",
  "updated": "2023-08-17",
  "year": "2023",
  "authors": [
   "Yushi Hu",
   "Benlin Liu",
   "Jungo Kasai",
   "Yizhong Wang",
   "Mari Ostendorf",
   "Ranjay Krishna",
   "Noah A Smith"
  ],
  "author_count": 7,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "ICCV",
  "venue_source": "semantic-scholar",
  "citations": 465,
  "influential_citations": 72,
  "tldr": "The introduction of TIFA (Text-to-Image Faithfulness evaluation with question Answering), an automatic evaluation metric that measures the faithfulness of a generated image to its text input via visual question answering (VQA), and a comprehensive evaluation of existing text-to-image models using TIFA v1.0.",
  "doi": "10.1109/ICCV51070.2023.01866",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yushi Hu",
    "id": "2112209725",
    "h_index": 14,
    "papers": 21
   },
   {
    "name": "Benlin Liu",
    "id": "67215934",
    "h_index": 9,
    "papers": 11
   },
   {
    "name": "Jungo Kasai",
    "id": "11348687",
    "h_index": 34,
    "papers": 85
   },
   {
    "name": "Yizhong Wang",
    "id": "1705260",
    "h_index": 28,
    "papers": 36
   },
   {
    "name": "Mari Ostendorf",
    "id": "144339506",
    "h_index": 63,
    "papers": 397
   },
   {
    "name": "Ranjay Krishna",
    "id": "145237361",
    "h_index": 37,
    "papers": 70
   },
   {
    "name": "Noah A. Smith",
    "id": "144365875",
    "h_index": 115,
    "papers": 425
   }
  ],
  "comment": "Accepted to ICCV 2023",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2303.11897v3",
  "pdf_url": "https://arxiv.org/pdf/2303.11897v3",
  "html_url": "https://arxiv.org/html/2303.11897v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.17
 },
 {
  "id": "2303.10880",
  "slug": "rotating-without-seeing-towards-in-hand-dexterity-through-touch",
  "title": "Rotating without Seeing: Towards In-hand Dexterity through Touch",
  "abstract": "Tactile information plays a critical role in human dexterity. It reveals useful contact information that may not be inferred directly from vision. In fact, humans can even perform in-hand dexterous manipulation without using vision. Can we enable the same ability for the multi-finger robot hand? In this paper, we present Touch Dexterity, a new system that can perform in-hand object rotation using only touching without seeing the object. Instead of relying on precise tactile sensing in a small region, we introduce a new system design using dense binary force sensors (touch or no touch) overlaying one side of the whole robot hand (palm, finger links, fingertips). Such a design is low-cost, giving a larger coverage of the object, and minimizing the Sim2Real gap at the same time. We train an in-hand rotation policy using Reinforcement Learning on diverse objects in simulation. Relying on touch-only sensing, we can directly deploy the policy in a real robot hand and rotate novel objects that are not presented in training. Extensive ablations are performed on how tactile information help in-hand manipulation.Our project is available at https://touchdexterity.github.io.",
  "published": "2023-03-20",
  "updated": "2023-03-27",
  "year": "2023",
  "authors": [
   "Zhao-Heng Yin",
   "Binghao Huang",
   "Yuzhe Qin",
   "Qifeng Chen",
   "Xiaolong Wang"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "RSS",
  "venue_source": "semantic-scholar",
  "citations": 161,
  "influential_citations": 6,
  "tldr": "A new system that can perform in-hand object rotation using only touching without seeing the object, using dense binary force sensors overlaying one side of the whole robot hand (palm, finger links, fingertips).",
  "doi": "10.48550/arXiv.2303.10880",
  "oa_pdf": "http://arxiv.org/pdf/2303.10880",
  "s2_authors": [
   {
    "name": "Zhao-Heng Yin",
    "id": "1693997019",
    "h_index": 7,
    "papers": 9
   },
   {
    "name": "Binghao Huang",
    "id": "2175672218",
    "h_index": 8,
    "papers": 8
   },
   {
    "name": "Yuzhe Qin",
    "id": "12701031",
    "h_index": 24,
    "papers": 34
   },
   {
    "name": "Qifeng Chen",
    "id": "2287417243",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Xiaolong Wang",
    "id": "2145748143",
    "h_index": 7,
    "papers": 8
   }
  ],
  "comment": "Project page: https://touchdexterity.github.io",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "sim2real",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2303.10880v4",
  "pdf_url": "https://arxiv.org/pdf/2303.10880v4",
  "html_url": "https://arxiv.org/html/2303.10880v4",
  "code_url": "https://touchdexterity.github.io",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.71
 },
 {
  "id": "2303.08774",
  "slug": "gpt-4-technical-report",
  "title": "GPT-4 Technical Report",
  "abstract": "We report the development of GPT-4, a large-scale, multimodal model which can accept image and text inputs and produce text outputs. While less capable than humans in many real-world scenarios, GPT-4 exhibits human-level performance on various professional and academic benchmarks, including passing a simulated bar exam with a score around the top 10% of test takers. GPT-4 is a Transformer-based model pre-trained to predict the next token in a document. The post-training alignment process results in improved performance on measures of factuality and adherence to desired behavior. A core component of this project was developing infrastructure and optimization methods that behave predictably across a wide range of scales. This allowed us to accurately predict some aspects of GPT-4's performance based on models trained with no more than 1/1,000th the compute of GPT-4.",
  "published": "2023-03-15",
  "updated": "2024-03-04",
  "year": "2023",
  "authors": [
   " OpenAI",
   "Josh Achiam",
   "Steven Adler",
   "Sandhini Agarwal",
   "Lama Ahmad",
   "Ilge Akkaya",
   "Florencia Leoni Aleman",
   "Diogo Almeida",
   "Janko Altenschmidt",
   "Sam Altman",
   "Shyamal Anadkat",
   "Red Avila",
   "Igor Babuschkin",
   "Suchir Balaji",
   "Valerie Balcom",
   "Paul Baltescu",
   "Haiming Bao",
   "Mohammad Bavarian",
   "Jeff Belgum",
   "Irwan Bello",
   "Jake Berdine",
   "Gabriel Bernadett-Shapiro",
   "Christopher Berner",
   "Lenny Bogdonoff",
   "Oleg Boiko",
   "Madelaine Boyd",
   "Anna-Luisa Brakman",
   "Greg Brockman",
   "Tim Brooks",
   "Miles Brundage",
   "Kevin Button",
   "Trevor Cai",
   "Rosie Campbell",
   "Andrew Cann",
   "Brittany Carey",
   "Chelsea Carlson",
   "Rory Carmichael",
   "Brooke Chan",
   "Che Chang",
   "Fotis Chantzis",
   "Derek Chen",
   "Sully Chen",
   "Ruby Chen",
   "Jason Chen",
   "Mark Chen",
   "Ben Chess",
   "Chester Cho",
   "Casey Chu",
   "Hyung Won Chung",
   "Dave Cummings",
   "Jeremiah Currier",
   "Yunxing Dai",
   "Cory Decareaux",
   "Thomas Degry",
   "Noah Deutsch",
   "Damien Deville",
   "Arka Dhar",
   "David Dohan",
   "Steve Dowling",
   "Sheila Dunning",
   "Adrien Ecoffet",
   "Atty Eleti",
   "Tyna Eloundou",
   "David Farhi",
   "Liam Fedus",
   "Niko Felix",
   "Sim\u00f3n Posada Fishman",
   "Juston Forte",
   "Isabella Fulford",
   "Leo Gao",
   "Elie Georges",
   "Christian Gibson",
   "Vik Goel",
   "Tarun Gogineni",
   "Gabriel Goh",
   "Rapha Gontijo-Lopes",
   "Jonathan Gordon",
   "Morgan Grafstein",
   "Scott Gray",
   "Ryan Greene",
   "Joshua Gross",
   "Shixiang Shane Gu",
   "Yufei Guo",
   "Chris Hallacy",
   "Jesse Han",
   "Jeff Harris",
   "Yuchen He",
   "Mike Heaton",
   "Johannes Heidecke",
   "Chris Hesse",
   "Alan Hickey",
   "Wade Hickey",
   "Peter Hoeschele",
   "Brandon Houghton",
   "Kenny Hsu",
   "Shengli Hu",
   "Xin Hu",
   "Joost Huizinga",
   "Shantanu Jain",
   "Shawn Jain",
   "Joanne Jang",
   "Angela Jiang",
   "Roger Jiang",
   "Haozhun Jin",
   "Denny Jin",
   "Shino Jomoto",
   "Billie Jonn",
   "Heewoo Jun",
   "Tomer Kaftan",
   "\u0141ukasz Kaiser",
   "Ali Kamali",
   "Ingmar Kanitscheider",
   "Nitish Shirish Keskar",
   "Tabarak Khan",
   "Logan Kilpatrick",
   "Jong Wook Kim",
   "Christina Kim",
   "Yongjik Kim",
   "Jan Hendrik Kirchner",
   "Jamie Kiros",
   "Matt Knight",
   "Daniel Kokotajlo",
   "\u0141ukasz Kondraciuk",
   "Andrew Kondrich",
   "Aris Konstantinidis",
   "Kyle Kosic",
   "Gretchen Krueger",
   "Vishal Kuo",
   "Michael Lampe",
   "Ikai Lan",
   "Teddy Lee",
   "Jan Leike",
   "Jade Leung",
   "Daniel Levy",
   "Chak Ming Li",
   "Rachel Lim",
   "Molly Lin",
   "Stephanie Lin",
   "Mateusz Litwin",
   "Theresa Lopez",
   "Ryan Lowe",
   "Patricia Lue",
   "Anna Makanju",
   "Kim Malfacini",
   "Sam Manning",
   "Todor Markov",
   "Yaniv Markovski",
   "Bianca Martin",
   "Katie Mayer",
   "Andrew Mayne",
   "Bob McGrew",
   "Scott Mayer McKinney",
   "Christine McLeavey",
   "Paul McMillan",
   "Jake McNeil",
   "David Medina",
   "Aalok Mehta",
   "Jacob Menick",
   "Luke Metz",
   "Andrey Mishchenko",
   "Pamela Mishkin",
   "Vinnie Monaco",
   "Evan Morikawa",
   "Daniel Mossing",
   "Tong Mu",
   "Mira Murati",
   "Oleg Murk",
   "David M\u00e9ly",
   "Ashvin Nair",
   "Reiichiro Nakano",
   "Rajeev Nayak",
   "Arvind Neelakantan",
   "Richard Ngo",
   "Hyeonwoo Noh",
   "Long Ouyang",
   "Cullen O'Keefe",
   "Jakub Pachocki",
   "Alex Paino",
   "Joe Palermo",
   "Ashley Pantuliano",
   "Giambattista Parascandolo",
   "Joel Parish",
   "Emy Parparita",
   "Alex Passos",
   "Mikhail Pavlov",
   "Andrew Peng",
   "Adam Perelman",
   "Filipe de Avila Belbute Peres",
   "Michael Petrov",
   "Henrique Ponde de Oliveira Pinto",
   " Michael",
   " Pokorny",
   "Michelle Pokrass",
   "Vitchyr H. Pong",
   "Tolly Powell",
   "Alethea Power",
   "Boris Power",
   "Elizabeth Proehl",
   "Raul Puri",
   "Alec Radford",
   "Jack Rae",
   "Aditya Ramesh",
   "Cameron Raymond",
   "Francis Real",
   "Kendra Rimbach",
   "Carl Ross",
   "Bob Rotsted",
   "Henri Roussez",
   "Nick Ryder",
   "Mario Saltarelli",
   "Ted Sanders",
   "Shibani Santurkar",
   "Girish Sastry",
   "Heather Schmidt",
   "David Schnurr",
   "John Schulman",
   "Daniel Selsam",
   "Kyla Sheppard",
   "Toki Sherbakov",
   "Jessica Shieh",
   "Sarah Shoker",
   "Pranav Shyam",
   "Szymon Sidor",
   "Eric Sigler",
   "Maddie Simens",
   "Jordan Sitkin",
   "Katarina Slama",
   "Ian Sohl",
   "Benjamin Sokolowsky",
   "Yang Song",
   "Natalie Staudacher",
   "Felipe Petroski Such",
   "Natalie Summers",
   "Ilya Sutskever",
   "Jie Tang",
   "Nikolas Tezak",
   "Madeleine B. Thompson",
   "Phil Tillet",
   "Amin Tootoonchian",
   "Elizabeth Tseng",
   "Preston Tuggle",
   "Nick Turley",
   "Jerry Tworek",
   "Juan Felipe Cer\u00f3n Uribe",
   "Andrea Vallone",
   "Arun Vijayvergiya",
   "Chelsea Voss",
   "Carroll Wainwright",
   "Justin Jay Wang",
   "Alvin Wang",
   "Ben Wang",
   "Jonathan Ward",
   "Jason Wei",
   "CJ Weinmann",
   "Akila Welihinda",
   "Peter Welinder",
   "Jiayi Weng",
   "Lilian Weng",
   "Matt Wiethoff",
   "Dave Willner",
   "Clemens Winter",
   "Samuel Wolrich",
   "Hannah Wong",
   "Lauren Workman",
   "Sherwin Wu",
   "Jeff Wu",
   "Michael Wu",
   "Kai Xiao",
   "Tao Xu",
   "Sarah Yoo",
   "Kevin Yu",
   "Qiming Yuan",
   "Wojciech Zaremba",
   "Rowan Zellers",
   "Chong Zhang",
   "Marvin Zhang",
   "Shengjia Zhao",
   "Tianhao Zheng",
   "Juntang Zhuang",
   "William Zhuk",
   "Barret Zoph"
  ],
  "author_count": 281,
  "categories": [
   "cs.CL",
   "cs.AI"
  ],
  "primary_category": "cs.CL",
  "venue": "",
  "venue_source": "",
  "citations": 26786,
  "influential_citations": 3248,
  "tldr": "GPT-4, a large-scale, multimodal model which can accept image and text inputs and produce text outputs, is developed, a Transformer-based model pre-trained to predict the next token in a document which exhibits human-level performance on various professional and academic benchmarks.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "OpenAI Josh Achiam",
    "id": "2275249853",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Steven Adler",
    "id": "2275250875",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "S. Agarwal",
    "id": "144517868",
    "h_index": 20,
    "papers": 72
   },
   {
    "name": "L. Ahmad",
    "id": "2274773568",
    "h_index": 14,
    "papers": 28
   },
   {
    "name": "Ilge Akkaya",
    "id": "2258629",
    "h_index": 13,
    "papers": 48
   },
   {
    "name": "Florencia Leoni Aleman",
    "id": "2275244794",
    "h_index": 3,
    "papers": 15
   },
   {
    "name": "D. Almeida",
    "id": "2275252021",
    "h_index": 5,
    "papers": 26
   },
   {
    "name": "Janko Altenschmidt",
    "id": "2275252424",
    "h_index": 4,
    "papers": 14
   },
   {
    "name": "S. Altman",
    "id": "2275245579",
    "h_index": 8,
    "papers": 31
   },
   {
    "name": "Shyamal Anadkat",
    "id": "2275246437",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Red Avila",
    "id": "2275139370",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Igor Babuschkin",
    "id": "2256699302",
    "h_index": 8,
    "papers": 13
   },
   {
    "name": "S. Balaji",
    "id": "2054519183",
    "h_index": 7,
    "papers": 20
   },
   {
    "name": "Valerie Balcom",
    "id": "2275251659",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Paul Baltescu",
    "id": "47626612",
    "h_index": 6,
    "papers": 14
   },
   {
    "name": "Haim-ing Bao",
    "id": "2275198557",
    "h_index": 5,
    "papers": 24
   },
   {
    "name": "Mo Bavarian",
    "id": "2275251620",
    "h_index": 12,
    "papers": 83
   },
   {
    "name": "Jeff Belgum",
    "id": "2275245092",
    "h_index": 6,
    "papers": 24
   },
   {
    "name": "Irwan Bello",
    "id": "4689792",
    "h_index": 12,
    "papers": 14
   },
   {
    "name": "Jake Berdine",
    "id": "2275245414",
    "h_index": 6,
    "papers": 25
   },
   {
    "name": "Gabriel Bernadett-Shapiro",
    "id": "2275245581",
    "h_index": 7,
    "papers": 27
   },
   {
    "name": "Christopher Berner",
    "id": "133740015",
    "h_index": 8,
    "papers": 53
   },
   {
    "name": "Lenny Bogdonoff",
    "id": "2275251674",
    "h_index": 6,
    "papers": 26
   },
   {
    "name": "O. Boiko",
    "id": "2275246071",
    "h_index": 7,
    "papers": 38
   },
   {
    "name": "Made-laine Boyd",
    "id": "2275248137",
    "h_index": 8,
    "papers": 39
   },
   {
    "name": "Anna-Luisa Brakman",
    "id": "2275245419",
    "h_index": 8,
    "papers": 29
   },
   {
    "name": "Greg Brockman",
    "id": "2065151121",
    "h_index": 11,
    "papers": 39
   },
   {
    "name": "Tim Brooks",
    "id": "2275219628",
    "h_index": 7,
    "papers": 26
   },
   {
    "name": "Miles Brundage",
    "id": "35167962",
    "h_index": 21,
    "papers": 31
   },
   {
    "name": "Kevin Button",
    "id": "2146257251",
    "h_index": 8,
    "papers": 34
   },
   {
    "name": "Trevor Cai",
    "id": "2275157286",
    "h_index": 7,
    "papers": 33
   },
   {
    "name": "Rosie Campbell",
    "id": "2274782053",
    "h_index": 8,
    "papers": 34
   },
   {
    "name": "Andrew Cann",
    "id": "2275245404",
    "h_index": 8,
    "papers": 40
   },
   {
    "name": "Brittany Carey",
    "id": "2275246368",
    "h_index": 7,
    "papers": 31
   },
   {
    "name": "Chelsea Carlson",
    "id": "2275120298",
    "h_index": 7,
    "papers": 31
   },
   {
    "name": "Rory Carmichael",
    "id": "144114446",
    "h_index": 11,
    "papers": 43
   },
   {
    "name": "Brooke Chan",
    "id": "1466431052",
    "h_index": 8,
    "papers": 39
   },
   {
    "name": "Che Chang",
    "id": "2275545855",
    "h_index": 8,
    "papers": 32
   },
   {
    "name": "Fotis Chantzis",
    "id": "2057091285",
    "h_index": 8,
    "papers": 31
   },
   {
    "name": "Derek Chen",
    "id": "2253841704",
    "h_index": 8,
    "papers": 36
   },
   {
    "name": "Sully Chen",
    "id": "2275188918",
    "h_index": 5,
    "papers": 32
   },
   {
    "name": "Ruby Chen",
    "id": "2275179180",
    "h_index": 8,
    "papers": 49
   },
   {
    "name": "Jason Chen",
    "id": "2275289833",
    "h_index": 7,
    "papers": 43
   },
   {
    "name": "Mark Chen",
    "id": "2108828435",
    "h_index": 16,
    "papers": 82
   },
   {
    "name": "Benjamin Chess",
    "id": "1490681878",
    "h_index": 8,
    "papers": 63
   },
   {
    "name": "Chester Cho",
    "id": "2275251158",
    "h_index": 7,
    "papers": 29
   },
   {
    "name": "Casey Chu",
    "id": "2276186593",
    "h_index": 4,
    "papers": 14
   },
   {
    "name": "Hyung Won Chung",
    "id": "2275839391",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Dave Cummings",
    "id": "2275231534",
    "h_index": 7,
    "papers": 28
   },
   {
    "name": "Jeremiah Currier",
    "id": "49645091",
    "h_index": 8,
    "papers": 29
   },
   {
    "name": "Yunxing Dai",
    "id": "2276187456",
    "h_index": 7,
    "papers": 30
   },
   {
    "name": "Cory Decareaux",
    "id": "2275251205",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Thomas Degry",
    "id": "2275244920",
    "h_index": 5,
    "papers": 17
   },
   {
    "name": "Noah Deutsch",
    "id": "2275247090",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Damien Deville",
    "id": "2275251200",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Arka Dhar",
    "id": "2275244298",
    "h_index": 6,
    "papers": 19
   },
   {
    "name": "David Dohan",
    "id": "35363891",
    "h_index": 24,
    "papers": 73
   },
   {
    "name": "Steve Dowling",
    "id": "2275252295",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Sheila Dunning",
    "id": "2275245491",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Adrien Ecoffet",
    "id": "66821245",
    "h_index": 10,
    "papers": 19
   },
   {
    "name": "Atty Eleti",
    "id": "2275245457",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Tyna Eloundou",
    "id": "2146257131",
    "h_index": 11,
    "papers": 25
   },
   {
    "name": "David Farhi",
    "id": "2065430571",
    "h_index": 11,
    "papers": 32
   },
   {
    "name": "L. Fedus",
    "id": "2096916416",
    "h_index": 8,
    "papers": 35
   },
   {
    "name": "N. Felix",
    "id": "2275249996",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "S. Fishman",
    "id": "2275245820",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Juston Forte",
    "id": "2275244914",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Is-abella Fulford",
    "id": "2275251173",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Leo Gao",
    "id": "2027599537",
    "h_index": 17,
    "papers": 24
   },
   {
    "name": "Elie Georges",
    "id": "2275200811",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "C. Gibson",
    "id": "2275254804",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Vik Goel",
    "id": "2275144649",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Tarun Gogineni",
    "id": "2325028819",
    "h_index": 8,
    "papers": 14
   },
   {
    "name": "Gabriel Goh",
    "id": "2261041177",
    "h_index": 6,
    "papers": 29
   },
   {
    "name": "Raphael Gontijo-Lopes",
    "id": "2158366935",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "J. Gordon",
    "id": "2265066144",
    "h_index": 4,
    "papers": 16
   },
   {
    "name": "Morgan Grafstein",
    "id": "2275250003",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Scott Gray",
    "id": "145565184",
    "h_index": 14,
    "papers": 57
   },
   {
    "name": "Ryan Greene",
    "id": "2275247307",
    "h_index": 3,
    "papers": 10
   },
   {
    "name": "Joshua Gross",
    "id": "2275137274",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "S. Gu",
    "id": "2253699903",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Yufei Guo",
    "id": "2276101257",
    "h_index": 6,
    "papers": 23
   },
   {
    "name": "Chris Hallacy",
    "id": "2004021329",
    "h_index": 9,
    "papers": 35
   },
   {
    "name": "Jesse Han",
    "id": "2275540338",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "J. Harris",
    "id": "2275295848",
    "h_index": 5,
    "papers": 19
   },
   {
    "name": "Yuchen He",
    "id": "2275226809",
    "h_index": 8,
    "papers": 31
   },
   {
    "name": "Mike Heaton",
    "id": "2275245527",
    "h_index": 3,
    "papers": 13
   },
   {
    "name": "Johannes Heidecke",
    "id": "2151087994",
    "h_index": 19,
    "papers": 42
   },
   {
    "name": "Chris Hesse",
    "id": "2242286342",
    "h_index": 8,
    "papers": 71
   },
   {
    "name": "Alan Hickey",
    "id": "2226452668",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "W. Hickey",
    "id": "2275246148",
    "h_index": 8,
    "papers": 39
   },
   {
    "name": "P. Hoeschele",
    "id": "2275245339",
    "h_index": 4,
    "papers": 26
   },
   {
    "name": "Brandon Houghton",
    "id": "103681415",
    "h_index": 9,
    "papers": 16
   },
   {
    "name": "Kenny Hsu",
    "id": "2275214107",
    "h_index": 6,
    "papers": 20
   },
   {
    "name": "Shengli Hu",
    "id": "2275210604",
    "h_index": 6,
    "papers": 22
   },
   {
    "name": "Xin Hu",
    "id": "2275777049",
    "h_index": 9,
    "papers": 49
   },
   {
    "name": "Joost Huizinga",
    "id": "39378983",
    "h_index": 18,
    "papers": 36
   },
   {
    "name": "Shantanu Jain",
    "id": "2276187117",
    "h_index": 6,
    "papers": 19
   },
   {
    "name": "Shawn Jain",
    "id": "2171110177",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Joanne Jang",
    "id": "2151094350",
    "h_index": 8,
    "papers": 22
   },
   {
    "name": "Angela Jiang",
    "id": "2253471334",
    "h_index": 8,
    "papers": 35
   },
   {
    "name": "R. Jiang",
    "id": "2275172062",
    "h_index": 8,
    "papers": 46
   },
   {
    "name": "Haozhun Jin",
    "id": "2275752035",
    "h_index": 5,
    "papers": 15
   },
   {
    "name": "Denny Jin",
    "id": "2275203081",
    "h_index": 8,
    "papers": 50
   },
   {
    "name": "Shino Jomoto",
    "id": "2275250083",
    "h_index": 4,
    "papers": 25
   },
   {
    "name": "Billie Jonn",
    "id": "2275247096",
    "h_index": 3,
    "papers": 13
   },
   {
    "name": "Heewoo Jun",
    "id": "35450887",
    "h_index": 15,
    "papers": 25
   },
   {
    "name": "Tomer Kaftan",
    "id": "2403754",
    "h_index": 10,
    "papers": 23
   },
   {
    "name": "Lukasz Kaiser",
    "id": "2275230678",
    "h_index": 8,
    "papers": 55
   },
   {
    "name": "Ali Kamali",
    "id": "2275169038",
    "h_index": 6,
    "papers": 35
   },
   {
    "name": "I. Kanitscheider",
    "id": "3151440",
    "h_index": 15,
    "papers": 45
   },
   {
    "name": "N. Keskar",
    "id": "2844898",
    "h_index": 28,
    "papers": 56
   },
   {
    "name": "Tabarak Khan",
    "id": "2152264064",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Logan Kilpatrick",
    "id": "2314112114",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Jong Wook Kim",
    "id": "2260346092",
    "h_index": 4,
    "papers": 18
   },
   {
    "name": "Christina Kim",
    "id": "2149054292",
    "h_index": 7,
    "papers": 20
   },
   {
    "name": "Yongjik Kim",
    "id": "2275296777",
    "h_index": 9,
    "papers": 42
   },
   {
    "name": "Hendrik Kirchner",
    "id": "2275112980",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "J. Kiros",
    "id": "51131802",
    "h_index": 16,
    "papers": 35
   },
   {
    "name": "Matthew Knight",
    "id": "2146257375",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Daniel Kokotajlo",
    "id": "1485556711",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Lukasz Kondraciuk",
    "id": "2275246094",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Andrew Kondrich",
    "id": "1666171360",
    "h_index": 5,
    "papers": 16
   },
   {
    "name": "Aris Konstantinidis",
    "id": "2317096965",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Kyle Kosic",
    "id": "2275245594",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Gretchen Krueger",
    "id": "2064404342",
    "h_index": 13,
    "papers": 70
   },
   {
    "name": "Vishal Kuo",
    "id": "2275229877",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Michael Lampe",
    "id": "2275247085",
    "h_index": 11,
    "papers": 25
   },
   {
    "name": "Ikai Lan",
    "id": "2275246287",
    "h_index": 5,
    "papers": 16
   },
   {
    "name": "Teddy Lee",
    "id": "2274915115",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Jan Leike",
    "id": "2990741",
    "h_index": 30,
    "papers": 76
   },
   {
    "name": "Jade Leung",
    "id": "52152632",
    "h_index": 8,
    "papers": 10
   },
   {
    "name": "Daniel Levy",
    "id": "2275256930",
    "h_index": 4,
    "papers": 12
   },
   {
    "name": "Chak Li",
    "id": "2275285124",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Rachel Lim",
    "id": "2275176375",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Molly Lin",
    "id": "2275759230",
    "h_index": 6,
    "papers": 18
   },
   {
    "name": "Stephanie Lin",
    "id": "2253840098",
    "h_index": 9,
    "papers": 52
   },
   {
    "name": "Ma-teusz Litwin",
    "id": "1380985420",
    "h_index": 8,
    "papers": 49
   },
   {
    "name": "Theresa Lopez",
    "id": "2275248327",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Ryan Lowe",
    "id": "2257272397",
    "h_index": 5,
    "papers": 15
   },
   {
    "name": "Patricia Lue",
    "id": "2275245628",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "A. Makanju",
    "id": "119341078",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Kim Malfacini",
    "id": "2275245649",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Sam Manning",
    "id": "46430291",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Todor Markov",
    "id": "14113256",
    "h_index": 7,
    "papers": 12
   },
   {
    "name": "Yaniv Markovski",
    "id": "2275245336",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Bianca Martin",
    "id": "2114362965",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Katie Mayer",
    "id": "2275231822",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Andrew Mayne",
    "id": "2275247045",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Bob McGrew",
    "id": "39593364",
    "h_index": 14,
    "papers": 30
   },
   {
    "name": "S. McKinney",
    "id": "2047820455",
    "h_index": 16,
    "papers": 21
   },
   {
    "name": "Christine McLeavey",
    "id": "3028785",
    "h_index": 7,
    "papers": 17
   },
   {
    "name": "Paul McMillan",
    "id": "2274772421",
    "h_index": 4,
    "papers": 16
   },
   {
    "name": "Jake McNeil",
    "id": "2275234856",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "David Medina",
    "id": "2275210659",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Aalok Mehta",
    "id": "2275132306",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Jacob Menick",
    "id": "10698483",
    "h_index": 9,
    "papers": 26
   },
   {
    "name": "Luke Metz",
    "id": "2275246330",
    "h_index": 6,
    "papers": 22
   },
   {
    "name": "Andrey Mishchenko",
    "id": "2275252694",
    "h_index": 5,
    "papers": 21
   },
   {
    "name": "Pamela Mishkin",
    "id": "2051714782",
    "h_index": 17,
    "papers": 66
   },
   {
    "name": "Vinnie Monaco",
    "id": "2275245453",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Evan Morikawa",
    "id": "1404556973",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Daniel P. Mossing",
    "id": "3407880",
    "h_index": 10,
    "papers": 18
   },
   {
    "name": "Tong Mu",
    "id": "2275154456",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "M. Murati",
    "id": "2117715631",
    "h_index": 6,
    "papers": 26
   },
   {
    "name": "O. Murk",
    "id": "147746767",
    "h_index": 6,
    "papers": 17
   },
   {
    "name": "David M'ely",
    "id": "2275246116",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Ashvin Nair",
    "id": "3422774",
    "h_index": 18,
    "papers": 22
   },
   {
    "name": "Reiichiro Nakano",
    "id": "7406311",
    "h_index": 10,
    "papers": 44
   },
   {
    "name": "Rajeev Nayak",
    "id": "2057426488",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Arvind Neelakantan",
    "id": "2072676",
    "h_index": 24,
    "papers": 75
   },
   {
    "name": "R. Ngo",
    "id": "2273886618",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Hyeonwoo Noh",
    "id": "2275115983",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "O. Long",
    "id": "2257278245",
    "h_index": 11,
    "papers": 61
   },
   {
    "name": "Cullen O'Keefe",
    "id": "1435765036",
    "h_index": 9,
    "papers": 14
   },
   {
    "name": "J. Pachocki",
    "id": "2713380",
    "h_index": 25,
    "papers": 54
   },
   {
    "name": "A. Paino",
    "id": "34800652",
    "h_index": 10,
    "papers": 27
   },
   {
    "name": "Joe Palermo",
    "id": "2275244652",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Ashley Pantuliano",
    "id": "2275246178",
    "h_index": 6,
    "papers": 16
   },
   {
    "name": "Giambattista Parascandolo",
    "id": "50213542",
    "h_index": 22,
    "papers": 37
   },
   {
    "name": "J. Parish",
    "id": "2275245818",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "Emy Parparita",
    "id": "2275245435",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Alexandre Passos",
    "id": "2274774915",
    "h_index": 6,
    "papers": 17
   },
   {
    "name": "Mikhail Pavlov",
    "id": "2068123790",
    "h_index": 10,
    "papers": 29
   },
   {
    "name": "Andrew Peng",
    "id": "2275125663",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Adam Perelman",
    "id": "2275245529",
    "h_index": 6,
    "papers": 26
   },
   {
    "name": "Filipe de Avila Belbute Peres",
    "id": "2275250075",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Michael Petrov",
    "id": "2136008481",
    "h_index": 7,
    "papers": 27
   },
   {
    "name": "Henrique Pond\u00e9 de Oliveira Pinto",
    "id": "1463773776",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Michael Pokorny",
    "id": "2275246346",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Michelle Pokrass",
    "id": "2275246814",
    "h_index": 7,
    "papers": 33
   },
   {
    "name": "Vitchyr H. Pong",
    "id": "144401061",
    "h_index": 15,
    "papers": 36
   },
   {
    "name": "Tolly Powell",
    "id": "2275150061",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Alethea Power",
    "id": "146162186",
    "h_index": 9,
    "papers": 39
   },
   {
    "name": "Boris Power",
    "id": "2151088845",
    "h_index": 5,
    "papers": 20
   },
   {
    "name": "Elizabeth Proehl",
    "id": "2275243930",
    "h_index": 9,
    "papers": 37
   },
   {
    "name": "Raul Puri",
    "id": "2285654208",
    "h_index": 6,
    "papers": 34
   },
   {
    "name": "Alec Radford",
    "id": "38909097",
    "h_index": 33,
    "papers": 153
   },
   {
    "name": "Jack W. Rae",
    "id": "2275178294",
    "h_index": 7,
    "papers": 26
   },
   {
    "name": "Aditya Ramesh",
    "id": "2261024614",
    "h_index": 6,
    "papers": 24
   },
   {
    "name": "Cameron Raymond",
    "id": "2275225165",
    "h_index": 6,
    "papers": 20
   },
   {
    "name": "F. Real",
    "id": "2275252438",
    "h_index": 4,
    "papers": 20
   },
   {
    "name": "Kendra Rimbach",
    "id": "2275252095",
    "h_index": 6,
    "papers": 22
   },
   {
    "name": "Carl Ross",
    "id": "2275207240",
    "h_index": 7,
    "papers": 30
   },
   {
    "name": "Bob Rotsted",
    "id": "11150265",
    "h_index": 8,
    "papers": 34
   },
   {
    "name": "Henri Roussez",
    "id": "2275250007",
    "h_index": 8,
    "papers": 32
   },
   {
    "name": "N. Ryder",
    "id": "2260406867",
    "h_index": 11,
    "papers": 111
   },
   {
    "name": "M. Saltarelli",
    "id": "47204843",
    "h_index": 23,
    "papers": 82
   },
   {
    "name": "Ted Sanders",
    "id": "2275246803",
    "h_index": 9,
    "papers": 40
   },
   {
    "name": "Shibani Santurkar",
    "id": "2852106",
    "h_index": 31,
    "papers": 77
   },
   {
    "name": "G. Sastry",
    "id": "144864359",
    "h_index": 17,
    "papers": 89
   },
   {
    "name": "Heather Schmidt",
    "id": "2275265666",
    "h_index": 9,
    "papers": 43
   },
   {
    "name": "David Schnurr",
    "id": "2252874293",
    "h_index": 10,
    "papers": 37
   },
   {
    "name": "John Schulman",
    "id": "47971768",
    "h_index": 45,
    "papers": 69
   },
   {
    "name": "Daniel Selsam",
    "id": "2196579",
    "h_index": 13,
    "papers": 51
   },
   {
    "name": "Kyla Sheppard",
    "id": "2275244711",
    "h_index": 8,
    "papers": 29
   },
   {
    "name": "T. Sherbakov",
    "id": "102475503",
    "h_index": 9,
    "papers": 37
   },
   {
    "name": "Jessica Shieh",
    "id": "2275246834",
    "h_index": 8,
    "papers": 40
   },
   {
    "name": "S. Shoker",
    "id": "118335789",
    "h_index": 9,
    "papers": 44
   },
   {
    "name": "Pranav Shyam",
    "id": "67311962",
    "h_index": 18,
    "papers": 93
   },
   {
    "name": "Szymon Sidor",
    "id": "2700360",
    "h_index": 14,
    "papers": 54
   },
   {
    "name": "Eric Sigler",
    "id": "2064673055",
    "h_index": 9,
    "papers": 77
   },
   {
    "name": "M. Simens",
    "id": "2151735251",
    "h_index": 9,
    "papers": 53
   },
   {
    "name": "Jordan Sitkin",
    "id": "2275252299",
    "h_index": 8,
    "papers": 34
   },
   {
    "name": "Katarina Slama",
    "id": "2117680841",
    "h_index": 11,
    "papers": 44
   },
   {
    "name": "Ian Sohl",
    "id": "103422608",
    "h_index": 8,
    "papers": 43
   },
   {
    "name": "Benjamin Sokolowsky",
    "id": "2901424",
    "h_index": 8,
    "papers": 42
   },
   {
    "name": "Yang Song",
    "id": "2307592658",
    "h_index": 7,
    "papers": 30
   },
   {
    "name": "N. Staudacher",
    "id": "2275245668",
    "h_index": 9,
    "papers": 40
   },
   {
    "name": "F. Such",
    "id": "9927844",
    "h_index": 17,
    "papers": 40
   },
   {
    "name": "Natalie Summers",
    "id": "2275252251",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "I. Sutskever",
    "id": "1701686",
    "h_index": 75,
    "papers": 166
   },
   {
    "name": "Jie Tang",
    "id": "2275750817",
    "h_index": 8,
    "papers": 46
   },
   {
    "name": "N. Tezak",
    "id": "145950540",
    "h_index": 18,
    "papers": 50
   },
   {
    "name": "Madeleine B Thompson",
    "id": "2151289331",
    "h_index": 4,
    "papers": 17
   },
   {
    "name": "P. Tillet",
    "id": "2275252092",
    "h_index": 8,
    "papers": 29
   },
   {
    "name": "Amin Tootoonchian",
    "id": "2267339677",
    "h_index": 12,
    "papers": 33
   },
   {
    "name": "Elizabeth Tseng",
    "id": "2275249879",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Preston Tuggle",
    "id": "2275249709",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Nick Turley",
    "id": "2275244171",
    "h_index": 5,
    "papers": 14
   },
   {
    "name": "Jerry Tworek",
    "id": "2065005836",
    "h_index": 13,
    "papers": 61
   },
   {
    "name": "Juan Felipe Cer\u00f3n Uribe",
    "id": "2275203310",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Andrea Vallone",
    "id": "2275244586",
    "h_index": 11,
    "papers": 26
   },
   {
    "name": "Arun Vijayvergiya",
    "id": "2275245661",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Chelsea Voss",
    "id": "153387869",
    "h_index": 13,
    "papers": 18
   },
   {
    "name": "Carroll L. Wainwright",
    "id": "2275245962",
    "h_index": 8,
    "papers": 25
   },
   {
    "name": "Justin Wang",
    "id": "2305029155",
    "h_index": 8,
    "papers": 17
   },
   {
    "name": "Alvin Wang",
    "id": "3209373",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Ben Wang",
    "id": "2275189326",
    "h_index": 5,
    "papers": 19
   },
   {
    "name": "Jonathan Ward",
    "id": "2170081200",
    "h_index": 3,
    "papers": 11
   },
   {
    "name": "Jason Wei",
    "id": "2253952872",
    "h_index": 10,
    "papers": 39
   },
   {
    "name": "CJ Weinmann",
    "id": "2275244218",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Akila Welihinda",
    "id": "2275245663",
    "h_index": 5,
    "papers": 27
   },
   {
    "name": "Peter Welinder",
    "id": "2930640",
    "h_index": 17,
    "papers": 37
   },
   {
    "name": "Jiayi Weng",
    "id": "2275139180",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Lilian Weng",
    "id": "2065741038",
    "h_index": 9,
    "papers": 13
   },
   {
    "name": "Matt Wiethoff",
    "id": "2275252154",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Dave Willner",
    "id": "2275249733",
    "h_index": 2,
    "papers": 7
   },
   {
    "name": "Clemens Winter",
    "id": "2059411355",
    "h_index": 12,
    "papers": 79
   },
   {
    "name": "Samuel Wolrich",
    "id": "2275244177",
    "h_index": 7,
    "papers": 23
   },
   {
    "name": "Hannah Wong",
    "id": "2275225207",
    "h_index": 8,
    "papers": 36
   },
   {
    "name": "Lauren Workman",
    "id": "2275245771",
    "h_index": 8,
    "papers": 35
   },
   {
    "name": "Sherwin Wu",
    "id": "2275299848",
    "h_index": 8,
    "papers": 26
   },
   {
    "name": "Jeff Wu",
    "id": "2274911253",
    "h_index": 8,
    "papers": 43
   },
   {
    "name": "Michael Wu",
    "id": "2307456650",
    "h_index": 8,
    "papers": 44
   },
   {
    "name": "Kai Xiao",
    "id": "2275190169",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Tao Xu",
    "id": "2275452480",
    "h_index": 8,
    "papers": 36
   },
   {
    "name": "Sarah Yoo",
    "id": "2275310096",
    "h_index": 7,
    "papers": 24
   },
   {
    "name": "Kevin Yu",
    "id": "2275593618",
    "h_index": 8,
    "papers": 35
   },
   {
    "name": "Qim-ing Yuan",
    "id": "2275194186",
    "h_index": 8,
    "papers": 37
   },
   {
    "name": "Wojciech Zaremba",
    "id": "2563432",
    "h_index": 31,
    "papers": 40
   },
   {
    "name": "Rowan Zellers",
    "id": "49629836",
    "h_index": 8,
    "papers": 32
   },
   {
    "name": "Chong Zhang",
    "id": "2262080679",
    "h_index": 8,
    "papers": 35
   },
   {
    "name": "Marvin Zhang",
    "id": "2275288889",
    "h_index": 7,
    "papers": 35
   },
   {
    "name": "Shengjia Zhao",
    "id": "2275545682",
    "h_index": 7,
    "papers": 24
   },
   {
    "name": "Tianhao Zheng",
    "id": "2275257857",
    "h_index": 3,
    "papers": 12
   },
   {
    "name": "Juntang Zhuang",
    "id": "2275201537",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "William Zhuk",
    "id": "2275245715",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Barret Zoph",
    "id": "2368067",
    "h_index": 44,
    "papers": 67
   }
  ],
  "comment": "100 pages; updated authors list; fixed author names and added citation",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2303.08774v6",
  "pdf_url": "https://arxiv.org/pdf/2303.08774v6",
  "html_url": "https://arxiv.org/html/2303.08774v6",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.0
 },
 {
  "id": "2303.07109",
  "slug": "transformer-based-world-models-are-happy-with-100k-interactions",
  "title": "Transformer-based World Models Are Happy With 100k Interactions",
  "abstract": "Deep neural networks have been successful in many reinforcement learning settings. However, compared to human learners they are overly data hungry. To build a sample-efficient world model, we apply a transformer to real-world episodes in an autoregressive manner: not only the compact latent states and the taken actions but also the experienced or predicted rewards are fed into the transformer, so that it can attend flexibly to all three modalities at different time steps. The transformer allows our world model to access previous states directly, instead of viewing them through a compressed recurrent state. By utilizing the Transformer-XL architecture, it is able to learn long-term dependencies while staying computationally efficient. Our transformer-based world model (TWM) generates meaningful, new experience, which is used to train a policy that outperforms previous model-free and model-based reinforcement learning algorithms on the Atari 100k benchmark.",
  "published": "2023-03-13",
  "updated": "2023-03-13",
  "year": "2023",
  "authors": [
   "Jan Robine",
   "Marc H\u00f6ftmann",
   "Tobias Uelwer",
   "Stefan Harmeling"
  ],
  "author_count": 4,
  "categories": [
   "cs.LG",
   "cs.AI",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 164,
  "influential_citations": 14,
  "tldr": "The transformer-based world model (TWM) generates meaningful, new experience, which is used to train a policy that outperforms previous model-free and model-based reinforcement learning algorithms on the Atari 100k benchmark.",
  "doi": "10.48550/arXiv.2303.07109",
  "oa_pdf": "http://arxiv.org/pdf/2303.07109",
  "s2_authors": [
   {
    "name": "Jan Robine",
    "id": "1994259894",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Marc Hoftmann",
    "id": "2211431015",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Tobias Uelwer",
    "id": "148250214",
    "h_index": 10,
    "papers": 21
   },
   {
    "name": "S. Harmeling",
    "id": "1734990",
    "h_index": 36,
    "papers": 97
   }
  ],
  "comment": "Published as a conference paper at ICLR 2023. Code is available at https://github.com/jrobine/twm",
  "topics": [
   "world-models",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2303.07109v1",
  "pdf_url": "https://arxiv.org/pdf/2303.07109v1",
  "html_url": "https://arxiv.org/html/2303.07109v1",
  "code_url": "https://github.com/jrobine/twm",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.72
 },
 {
  "id": "2303.05499",
  "slug": "grounding-dino-marrying-dino-with-grounded-pre-training-for-open-set-o",
  "title": "Grounding DINO: Marrying DINO with Grounded Pre-Training for Open-Set Object Detection",
  "abstract": "In this paper, we present an open-set object detector, called Grounding DINO, by marrying Transformer-based detector DINO with grounded pre-training, which can detect arbitrary objects with human inputs such as category names or referring expressions. The key solution of open-set object detection is introducing language to a closed-set detector for open-set concept generalization. To effectively fuse language and vision modalities, we conceptually divide a closed-set detector into three phases and propose a tight fusion solution, which includes a feature enhancer, a language-guided query selection, and a cross-modality decoder for cross-modality fusion. While previous works mainly evaluate open-set object detection on novel categories, we propose to also perform evaluations on referring expression comprehension for objects specified with attributes. Grounding DINO performs remarkably well on all three settings, including benchmarks on COCO, LVIS, ODinW, and RefCOCO/+/g. Grounding DINO achieves a $52.5$ AP on the COCO detection zero-shot transfer benchmark, i.e., without any training data from COCO. It sets a new record on the ODinW zero-shot benchmark with a mean $26.1$ AP. Code will be available at \\url{https://github.com/IDEA-Research/GroundingDINO}.",
  "published": "2023-03-09",
  "updated": "2024-07-19",
  "year": "2023",
  "authors": [
   "Shilong Liu",
   "Zhaoyang Zeng",
   "Tianhe Ren",
   "Feng Li",
   "Hao Zhang",
   "Jie Yang",
   "Qing Jiang",
   "Chunyuan Li",
   "Jianwei Yang",
   "Hang Su",
   "Jun Zhu",
   "Lei Zhang"
  ],
  "author_count": 12,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "ECCV",
  "venue_source": "semantic-scholar",
  "citations": 4800,
  "influential_citations": 659,
  "tldr": "An open-set object detector, called Grounding DINO, is presented by marrying Transformer-based detector DINO with grounded pre-training, which can detect arbitrary objects with human inputs such as category names or referring expressions, and performs remarkably well on all three settings.",
  "doi": "10.48550/arXiv.2303.05499",
  "oa_pdf": "http://arxiv.org/pdf/2303.05499",
  "s2_authors": [
   {
    "name": "Shilong Liu",
    "id": "8602739",
    "h_index": 20,
    "papers": 36
   },
   {
    "name": "Zhaoyang Zeng",
    "id": "2075413603",
    "h_index": 15,
    "papers": 28
   },
   {
    "name": "Tianhe Ren",
    "id": "2143150727",
    "h_index": 19,
    "papers": 34
   },
   {
    "name": "Feng Li",
    "id": "2152978390",
    "h_index": 26,
    "papers": 42
   },
   {
    "name": "Hao Zhang",
    "id": "2315254849",
    "h_index": 22,
    "papers": 36
   },
   {
    "name": "Jie Yang",
    "id": "2146105297",
    "h_index": 10,
    "papers": 22
   },
   {
    "name": "Chun-yue Li",
    "id": "2109738542",
    "h_index": 15,
    "papers": 18
   },
   {
    "name": "Jianwei Yang",
    "id": "120157163",
    "h_index": 40,
    "papers": 54
   },
   {
    "name": "Hang Su",
    "id": "2093561216",
    "h_index": 52,
    "papers": 136
   },
   {
    "name": "Jun-Juan Zhu",
    "id": "89006344",
    "h_index": 12,
    "papers": 19
   },
   {
    "name": "Lei Zhang",
    "id": "2152832911",
    "h_index": 15,
    "papers": 32
   }
  ],
  "comment": "Code will be available at https://github.com/IDEA-Research/GroundingDINO",
  "topics": [
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2303.05499v5",
  "pdf_url": "https://arxiv.org/pdf/2303.05499v5",
  "html_url": "https://arxiv.org/html/2303.05499v5",
  "code_url": "https://github.com/IDEA-Research/GroundingDINO",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.5
 },
 {
  "id": "2303.04137",
  "slug": "diffusion-policy-visuomotor-policy-learning-via-action-diffusion",
  "title": "Diffusion Policy: Visuomotor Policy Learning via Action Diffusion",
  "abstract": "This paper introduces Diffusion Policy, a new way of generating robot behavior by representing a robot's visuomotor policy as a conditional denoising diffusion process. We benchmark Diffusion Policy across 12 different tasks from 4 different robot manipulation benchmarks and find that it consistently outperforms existing state-of-the-art robot learning methods with an average improvement of 46.9%. Diffusion Policy learns the gradient of the action-distribution score function and iteratively optimizes with respect to this gradient field during inference via a series of stochastic Langevin dynamics steps. We find that the diffusion formulation yields powerful advantages when used for robot policies, including gracefully handling multimodal action distributions, being suitable for high-dimensional action spaces, and exhibiting impressive training stability. To fully unlock the potential of diffusion models for visuomotor policy learning on physical robots, this paper presents a set of key technical contributions including the incorporation of receding horizon control, visual conditioning, and the time-series diffusion transformer. We hope this work will help motivate a new generation of policy learning techniques that are able to leverage the powerful generative modeling capabilities of diffusion models. Code, data, and training details is publicly available diffusion-policy.cs.columbia.edu",
  "published": "2023-03-07",
  "updated": "2024-03-14",
  "year": "2023",
  "authors": [
   "Cheng Chi",
   "Zhenjia Xu",
   "Siyuan Feng",
   "Eric Cousineau",
   "Yilun Du",
   "Benjamin Burchfiel",
   "Russ Tedrake",
   "Shuran Song"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "RSS",
  "venue_source": "semantic-scholar",
  "citations": 3987,
  "influential_citations": 826,
  "tldr": "This paper introduces Diffusion Policy, a new way of generating robot behavior by representing a robot\u2019s visuomotor policy as a conditional denoising diffusion process that consistently outperforms existing state-of-the-art robot learning methods.",
  "doi": "10.1177/02783649241273668",
  "oa_pdf": "http://arxiv.org/pdf/2303.04137",
  "s2_authors": [
   {
    "name": "Cheng Chi",
    "id": "46859937",
    "h_index": 17,
    "papers": 83
   },
   {
    "name": "S. Feng",
    "id": "2480008",
    "h_index": 21,
    "papers": 27
   },
   {
    "name": "Yilun Du",
    "id": "15394275",
    "h_index": 48,
    "papers": 86
   },
   {
    "name": "Zhenjia Xu",
    "id": "74498275",
    "h_index": 15,
    "papers": 22
   },
   {
    "name": "Eric Cousineau",
    "id": "2090529",
    "h_index": 12,
    "papers": 16
   },
   {
    "name": "B. Burchfiel",
    "id": "2302757",
    "h_index": 18,
    "papers": 30
   },
   {
    "name": "Shuran Song",
    "id": "3340170",
    "h_index": 59,
    "papers": 90
   }
  ],
  "comment": "An extended journal version of the original RSS2023 paper",
  "topics": [
   "imitation-diffusion",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2303.04137v5",
  "pdf_url": "https://arxiv.org/pdf/2303.04137v5",
  "html_url": "https://arxiv.org/html/2303.04137v5",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "2303.03486",
  "slug": "sampling-based-exploration-for-reinforcement-learning-of-dexterous-man",
  "title": "Sampling-based Exploration for Reinforcement Learning of Dexterous Manipulation",
  "abstract": "In this paper, we present a novel method for achieving dexterous manipulation of complex objects, while simultaneously securing the object without the use of passive support surfaces. We posit that a key difficulty for training such policies in a Reinforcement Learning framework is the difficulty of exploring the problem state space, as the accessible regions of this space form a complex structure along manifolds of a high-dimensional space. To address this challenge, we use two versions of the non-holonomic Rapidly-Exploring Random Trees algorithm; one version is more general, but requires explicit use of the environment's transition function, while the second version uses manipulation-specific kinematic constraints to attain better sample efficiency. In both cases, we use states found via sampling-based exploration to generate reset distributions that enable training control policies under full dynamic constraints via model-free Reinforcement Learning. We show that these policies are effective at manipulation problems of higher difficulty than previously shown, and also transfer effectively to real robots. Videos of the real-hand demonstrations can be found on the project website: https://sbrl.cs.columbia.edu/",
  "published": "2023-03-06",
  "updated": "2023-05-23",
  "year": "2023",
  "authors": [
   "Gagan Khandate",
   "Siqi Shang",
   "Eric T. Chang",
   "Tristan Luca Saidi",
   "Yang Liu",
   "Seth Matthew Dennis",
   "Johnson Adams",
   "Matei Ciocarlie"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "RSS",
  "venue_source": "semantic-scholar",
  "citations": 46,
  "influential_citations": 1,
  "tldr": "This paper presents a novel method for achieving dexterous manipulation of complex objects, while simultaneously securing the object without the use of passive support surfaces using the non-holonomic Rapidly-Exploring Random Trees algorithm.",
  "doi": "10.48550/arXiv.2303.03486",
  "oa_pdf": "http://arxiv.org/pdf/2303.03486",
  "s2_authors": [
   {
    "name": "Gagan Khandate",
    "id": "1389555439",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Siqi Shang",
    "id": "2210857801",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Eric Chang",
    "id": "1708592",
    "h_index": 33,
    "papers": 157
   },
   {
    "name": "Tristan Luca Saidi",
    "id": "2323789069",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Johnson Adams",
    "id": "2210858319",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "M. Ciocarlie",
    "id": "3145951",
    "h_index": 34,
    "papers": 128
   }
  ],
  "comment": "10 pages, 7 figures, accepted at Robotics Science & Systems 2023",
  "topics": [
   "dexterous-manipulation",
   "rl-control",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2303.03486v3",
  "pdf_url": "https://arxiv.org/pdf/2303.03486v3",
  "html_url": "https://arxiv.org/html/2303.03486v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.17
 },
 {
  "id": "2303.03378",
  "slug": "palm-e-an-embodied-multimodal-language-model",
  "title": "PaLM-E: An Embodied Multimodal Language Model",
  "abstract": "Large language models excel at a wide range of complex tasks. However, enabling general inference in the real world, e.g., for robotics problems, raises the challenge of grounding. We propose embodied language models to directly incorporate real-world continuous sensor modalities into language models and thereby establish the link between words and percepts. Input to our embodied language model are multi-modal sentences that interleave visual, continuous state estimation, and textual input encodings. We train these encodings end-to-end, in conjunction with a pre-trained large language model, for multiple embodied tasks including sequential robotic manipulation planning, visual question answering, and captioning. Our evaluations show that PaLM-E, a single large embodied multimodal model, can address a variety of embodied reasoning tasks, from a variety of observation modalities, on multiple embodiments, and further, exhibits positive transfer: the model benefits from diverse joint training across internet-scale language, vision, and visual-language domains. Our largest model, PaLM-E-562B with 562B parameters, in addition to being trained on robotics tasks, is a visual-language generalist with state-of-the-art performance on OK-VQA, and retains generalist language capabilities with increasing scale.",
  "published": "2023-03-06",
  "updated": "2023-03-06",
  "year": "2023",
  "authors": [
   "Danny Driess",
   "Fei Xia",
   "Mehdi S. M. Sajjadi",
   "Corey Lynch",
   "Aakanksha Chowdhery",
   "Brian Ichter",
   "Ayzaan Wahid",
   "Jonathan Tompson",
   "Quan Vuong",
   "Tianhe Yu",
   "Wenlong Huang",
   "Yevgen Chebotar",
   "Pierre Sermanet",
   "Daniel Duckworth",
   "Sergey Levine",
   "Vincent Vanhoucke",
   "Karol Hausman",
   "Marc Toussaint",
   "Klaus Greff",
   "Andy Zeng",
   "Igor Mordatch",
   "Pete Florence"
  ],
  "author_count": 22,
  "categories": [
   "cs.LG",
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.LG",
  "venue": "ICML",
  "venue_source": "semantic-scholar",
  "citations": 3056,
  "influential_citations": 116,
  "tldr": "This work proposes embodied language models to directly incorporate real-world continuous sensor modalities into language models and thereby establish the link between words and percepts to enable general inference in the real world.",
  "doi": "10.48550/arXiv.2303.03378",
  "oa_pdf": "http://arxiv.org/pdf/2303.03378",
  "s2_authors": [
   {
    "name": "Danny Driess",
    "id": "2283848260",
    "h_index": 27,
    "papers": 35
   },
   {
    "name": "F. Xia",
    "id": "144956443",
    "h_index": 25,
    "papers": 31
   },
   {
    "name": "Mehdi S. M. Sajjadi",
    "id": "2283034",
    "h_index": 24,
    "papers": 45
   },
   {
    "name": "Corey Lynch",
    "id": "32245472",
    "h_index": 20,
    "papers": 27
   },
   {
    "name": "A. Chowdhery",
    "id": "2841893",
    "h_index": 30,
    "papers": 82
   },
   {
    "name": "Brian Ichter",
    "id": "2704814",
    "h_index": 37,
    "papers": 60
   },
   {
    "name": "Ayzaan Wahid",
    "id": "88728227",
    "h_index": 21,
    "papers": 27
   },
   {
    "name": "Jonathan Tompson",
    "id": "2704494",
    "h_index": 43,
    "papers": 71
   },
   {
    "name": "Q. Vuong",
    "id": "144579461",
    "h_index": 23,
    "papers": 40
   },
   {
    "name": "Tianhe Yu",
    "id": "10909315",
    "h_index": 31,
    "papers": 48
   },
   {
    "name": "Wenlong Huang",
    "id": "2158105356",
    "h_index": 10,
    "papers": 11
   },
   {
    "name": "Yevgen Chebotar",
    "id": "2527420",
    "h_index": 33,
    "papers": 57
   },
   {
    "name": "P. Sermanet",
    "id": "3142556",
    "h_index": 39,
    "papers": 77
   },
   {
    "name": "Daniel Duckworth",
    "id": "40620532",
    "h_index": 18,
    "papers": 37
   },
   {
    "name": "S. Levine",
    "id": "1736651",
    "h_index": 167,
    "papers": 551
   },
   {
    "name": "Vincent Vanhoucke",
    "id": "2657155",
    "h_index": 34,
    "papers": 60
   },
   {
    "name": "Karol Hausman",
    "id": "1944801",
    "h_index": 47,
    "papers": 122
   },
   {
    "name": "Marc Toussaint",
    "id": "144918851",
    "h_index": 50,
    "papers": 292
   },
   {
    "name": "Klaus Greff",
    "id": "3035541",
    "h_index": 28,
    "papers": 49
   },
   {
    "name": "Andy Zeng",
    "id": "38591293",
    "h_index": 34,
    "papers": 50
   },
   {
    "name": "Igor Mordatch",
    "id": "2080746",
    "h_index": 34,
    "papers": 49
   },
   {
    "name": "Peter R. Florence",
    "id": "47686265",
    "h_index": 30,
    "papers": 36
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2303.03378v1",
  "pdf_url": "https://arxiv.org/pdf/2303.03378v1",
  "html_url": "https://arxiv.org/html/2303.03378v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "2303.02646",
  "slug": "seq2seq-imitation-learning-for-tactile-feedback-based-manipulation",
  "title": "Seq2Seq Imitation Learning for Tactile Feedback-based Manipulation",
  "abstract": "Robot control for tactile feedback-based manipulation can be difficult due to the modeling of physical contacts, partial observability of the environment, and noise in perception and control. This work focuses on solving partial observability of contact-rich manipulation tasks as a Sequence-to-Sequence (Seq2Seq)} Imitation Learning (IL) problem. The proposed Seq2Seq model produces a robot-environment interaction sequence to estimate the partially observable environment state variables. Then, the observed interaction sequence is transformed to a control sequence for the task itself. The proposed Seq2Seq IL for tactile feedback-based manipulation is experimentally validated on a door-open task in a simulated environment and a snap-on insertion task with a real robot. The model is able to learn both tasks from only 50 expert demonstrations, while state-of-the-art reinforcement learning and imitation learning methods fail.",
  "published": "2023-03-05",
  "updated": "2023-03-05",
  "year": "2023",
  "authors": [
   "Wenyan Yang",
   "Alexandre Angleraud",
   "Roel S. Pieters",
   "Joni Pajarinen",
   "Joni-Kristian K\u00e4m\u00e4r\u00e4inen"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA",
  "venue_source": "semantic-scholar",
  "citations": 13,
  "influential_citations": 0,
  "tldr": "This work focuses on solving partial observability of contact-rich manipulation tasks as a Sequence-to-Sequence (Seq2Seq) Imitation Learning (IL) problem and is able to learn both tasks from only 50 expert demonstrations while state-of-the-art reinforcement learning and imitation learning methods fail.",
  "doi": "10.1109/ICRA48891.2023.10161145",
  "oa_pdf": "http://arxiv.org/pdf/2303.02646",
  "s2_authors": [
   {
    "name": "Wenyan Yang",
    "id": "1962348671",
    "h_index": 7,
    "papers": 24
   },
   {
    "name": "Alexandre Angleraud",
    "id": "51886590",
    "h_index": 8,
    "papers": 26
   },
   {
    "name": "R. Pieters",
    "id": "33000070",
    "h_index": 13,
    "papers": 64
   },
   {
    "name": "J. Pajarinen",
    "id": "34906504",
    "h_index": 21,
    "papers": 152
   },
   {
    "name": "Joni-Kristian K\u00e4m\u00e4r\u00e4inen",
    "id": "1381913868",
    "h_index": 17,
    "papers": 82
   }
  ],
  "comment": "",
  "topics": [
   "tactile",
   "imitation-diffusion",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2303.02646v1",
  "pdf_url": "https://arxiv.org/pdf/2303.02646v1",
  "html_url": "https://arxiv.org/html/2303.02646v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.65
 },
 {
  "id": "2303.01469",
  "slug": "consistency-models",
  "title": "Consistency Models",
  "abstract": "Diffusion models have significantly advanced the fields of image, audio, and video generation, but they depend on an iterative sampling process that causes slow generation. To overcome this limitation, we propose consistency models, a new family of models that generate high quality samples by directly mapping noise to data. They support fast one-step generation by design, while still allowing multistep sampling to trade compute for sample quality. They also support zero-shot data editing, such as image inpainting, colorization, and super-resolution, without requiring explicit training on these tasks. Consistency models can be trained either by distilling pre-trained diffusion models, or as standalone generative models altogether. Through extensive experiments, we demonstrate that they outperform existing distillation techniques for diffusion models in one- and few-step sampling, achieving the new state-of-the-art FID of 3.55 on CIFAR-10 and 6.20 on ImageNet 64x64 for one-step generation. When trained in isolation, consistency models become a new family of generative models that can outperform existing one-step, non-adversarial generative models on standard benchmarks such as CIFAR-10, ImageNet 64x64 and LSUN 256x256.",
  "published": "2023-03-02",
  "updated": "2023-05-31",
  "year": "2023",
  "authors": [
   "Yang Song",
   "Prafulla Dhariwal",
   "Mark Chen",
   "Ilya Sutskever"
  ],
  "author_count": 4,
  "categories": [
   "cs.LG",
   "cs.CV",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "ICML",
  "venue_source": "semantic-scholar",
  "citations": 2130,
  "influential_citations": 296,
  "tldr": "Consistency models are proposed, a new family of models that generate high quality samples by directly mapping noise to data that can outperform existing one-step, non-adversarial generative models on standard benchmarks such as CIFAR-10, ImageNet 64x64 and LSUN 256x256.",
  "doi": "10.1007/978-1-4842-1329-2_9",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yang Song",
    "id": "2157995251",
    "h_index": 8,
    "papers": 8
   },
   {
    "name": "Prafulla Dhariwal",
    "id": "6515819",
    "h_index": 20,
    "papers": 43
   },
   {
    "name": "Mark Chen",
    "id": "2108828435",
    "h_index": 16,
    "papers": 82
   },
   {
    "name": "I. Sutskever",
    "id": "1701686",
    "h_index": 75,
    "papers": 166
   }
  ],
  "comment": "ICML 2023",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2303.01469v2",
  "pdf_url": "https://arxiv.org/pdf/2303.01469v2",
  "html_url": "https://arxiv.org/html/2303.01469v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "2303.00905",
  "slug": "open-world-object-manipulation-using-pre-trained-vision-language-model",
  "title": "Open-World Object Manipulation using Pre-trained Vision-Language Models",
  "abstract": "For robots to follow instructions from people, they must be able to connect the rich semantic information in human vocabulary, e.g. \"can you get me the pink stuffed whale?\" to their sensory observations and actions. This brings up a notably difficult challenge for robots: while robot learning approaches allow robots to learn many different behaviors from first-hand experience, it is impractical for robots to have first-hand experiences that span all of this semantic information. We would like a robot's policy to be able to perceive and pick up the pink stuffed whale, even if it has never seen any data interacting with a stuffed whale before. Fortunately, static data on the internet has vast semantic information, and this information is captured in pre-trained vision-language models. In this paper, we study whether we can interface robot policies with these pre-trained models, with the aim of allowing robots to complete instructions involving object categories that the robot has never seen first-hand. We develop a simple approach, which we call Manipulation of Open-World Objects (MOO), which leverages a pre-trained vision-language model to extract object-identifying information from the language command and image, and conditions the robot policy on the current image, the instruction, and the extracted object information. In a variety of experiments on a real mobile manipulator, we find that MOO generalizes zero-shot to a wide range of novel object categories and environments. In addition, we show how MOO generalizes to other, non-language-based input modalities to specify the object of interest such as finger pointing, and how it can be further extended to enable open-world navigation and manipulation. The project's website and evaluation videos can be found at https://robot-moo.github.io/",
  "published": "2023-03-02",
  "updated": "2023-10-25",
  "year": "2023",
  "authors": [
   "Austin Stone",
   "Ted Xiao",
   "Yao Lu",
   "Keerthana Gopalakrishnan",
   "Kuang-Huei Lee",
   "Quan Vuong",
   "Paul Wohlhart",
   "Sean Kirmani",
   "Brianna Zitkovich",
   "Fei Xia",
   "Chelsea Finn",
   "Karol Hausman"
  ],
  "author_count": 12,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 242,
  "influential_citations": 5,
  "tldr": "Manipulation of Open-World Objects (MOO) is developed, which leverages a pre-trained vision-language model to extract object-identifying information from the language command and image, and conditions the robot policy on the current image, the instruction, and the extracted object information.",
  "doi": "10.48550/arXiv.2303.00905",
  "oa_pdf": "http://arxiv.org/pdf/2303.00905",
  "s2_authors": [
   {
    "name": "Austin Stone",
    "id": "2056868723",
    "h_index": 14,
    "papers": 19
   },
   {
    "name": "Ted Xiao",
    "id": "9961095",
    "h_index": 33,
    "papers": 45
   },
   {
    "name": "Yao Lu",
    "id": "2161346119",
    "h_index": 16,
    "papers": 22
   },
   {
    "name": "K. Gopalakrishnan",
    "id": "2161342233",
    "h_index": 17,
    "papers": 25
   },
   {
    "name": "Kuang-Huei Lee",
    "id": "2145145412",
    "h_index": 15,
    "papers": 19
   },
   {
    "name": "Q. Vuong",
    "id": "144579461",
    "h_index": 23,
    "papers": 40
   },
   {
    "name": "Paul Wohlhart",
    "id": "3202367",
    "h_index": 24,
    "papers": 47
   },
   {
    "name": "Brianna Zitkovich",
    "id": "2196524598",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "F. Xia",
    "id": "144956443",
    "h_index": 25,
    "papers": 31
   },
   {
    "name": "Chelsea Finn",
    "id": "46881670",
    "h_index": 94,
    "papers": 240
   },
   {
    "name": "Karol Hausman",
    "id": "1944801",
    "h_index": 47,
    "papers": 122
   }
  ],
  "comment": "Accepted at the 7th Conference on Robot Learning (CoRL 2023)",
  "topics": [
   "navigation",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2303.00905v2",
  "pdf_url": "https://arxiv.org/pdf/2303.00905v2",
  "html_url": "https://arxiv.org/html/2303.00905v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.89
 },
 {
  "id": "2303.00848",
  "slug": "understanding-diffusion-objectives-as-the-elbo-with-simple-data-augmen",
  "title": "Understanding Diffusion Objectives as the ELBO with Simple Data Augmentation",
  "abstract": "To achieve the highest perceptual quality, state-of-the-art diffusion models are optimized with objectives that typically look very different from the maximum likelihood and the Evidence Lower Bound (ELBO) objectives. In this work, we reveal that diffusion model objectives are actually closely related to the ELBO. Specifically, we show that all commonly used diffusion model objectives equate to a weighted integral of ELBOs over different noise levels, where the weighting depends on the specific objective used. Under the condition of monotonic weighting, the connection is even closer: the diffusion objective then equals the ELBO, combined with simple data augmentation, namely Gaussian noise perturbation. We show that this condition holds for a number of state-of-the-art diffusion models. In experiments, we explore new monotonic weightings and demonstrate their effectiveness, achieving state-of-the-art FID scores on the high-resolution ImageNet benchmark.",
  "published": "2023-03-01",
  "updated": "2023-09-25",
  "year": "2023",
  "authors": [
   "Diederik P. Kingma",
   "Ruiqi Gao"
  ],
  "author_count": 2,
  "categories": [
   "cs.LG",
   "cs.AI",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 340,
  "influential_citations": 29,
  "tldr": "This work reveals that diffusion model objectives are actually closely related to the ELBO, and shows that all commonly used diffusion model objective equate to a weighted integral of ELBOs over different noise levels, where the weighting depends on the specific objective used.",
  "doi": "10.52202/075280-2858",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Diederik P. Kingma",
    "id": "1726807",
    "h_index": 35,
    "papers": 46
   },
   {
    "name": "Ruiqi Gao",
    "id": "9659905",
    "h_index": 23,
    "papers": 50
   }
  ],
  "comment": "",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2303.00848v7",
  "pdf_url": "https://arxiv.org/pdf/2303.00848v7",
  "html_url": "https://arxiv.org/html/2303.00848v7",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.03
 },
 {
  "id": "2302.12766",
  "slug": "language-driven-representation-learning-for-robotics",
  "title": "Language-Driven Representation Learning for Robotics",
  "abstract": "Recent work in visual representation learning for robotics demonstrates the viability of learning from large video datasets of humans performing everyday tasks. Leveraging methods such as masked autoencoding and contrastive learning, these representations exhibit strong transfer to policy learning for visuomotor control. But, robot learning encompasses a diverse set of problems beyond control including grasp affordance prediction, language-conditioned imitation learning, and intent scoring for human-robot collaboration, amongst others. First, we demonstrate that existing representations yield inconsistent results across these tasks: masked autoencoding approaches pick up on low-level spatial features at the cost of high-level semantics, while contrastive learning approaches capture the opposite. We then introduce Voltron, a framework for language-driven representation learning from human videos and associated captions. Voltron trades off language-conditioned visual reconstruction to learn low-level visual patterns, and visually-grounded language generation to encode high-level semantics. We also construct a new evaluation suite spanning five distinct robot learning problems $\\unicode{x2013}$ a unified platform for holistically evaluating visual representations for robotics. Through comprehensive, controlled experiments across all five problems, we find that Voltron's language-driven representations outperform the prior state-of-the-art, especially on targeted problems requiring higher-level features.",
  "published": "2023-02-24",
  "updated": "2023-02-24",
  "year": "2023",
  "authors": [
   "Siddharth Karamcheti",
   "Suraj Nair",
   "Annie S. Chen",
   "Thomas Kollar",
   "Chelsea Finn",
   "Dorsa Sadigh",
   "Percy Liang"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CL",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "RSS",
  "venue_source": "semantic-scholar",
  "citations": 222,
  "influential_citations": 11,
  "tldr": "Voltron is introduced, a framework for language-driven representation learning from human videos and associated captions that outperform the prior state-of-the-art, especially on targeted problems requiring higher-level features.",
  "doi": "10.48550/arXiv.2302.12766",
  "oa_pdf": "http://arxiv.org/pdf/2302.12766",
  "s2_authors": [
   {
    "name": "Siddharth Karamcheti",
    "id": "10737060",
    "h_index": 21,
    "papers": 38
   },
   {
    "name": "Suraj Nair",
    "id": "4734949",
    "h_index": 20,
    "papers": 36
   },
   {
    "name": "Annie S. Chen",
    "id": "2111073657",
    "h_index": 12,
    "papers": 23
   },
   {
    "name": "T. Kollar",
    "id": "2836353",
    "h_index": 31,
    "papers": 67
   },
   {
    "name": "Chelsea Finn",
    "id": "46881670",
    "h_index": 94,
    "papers": 240
   },
   {
    "name": "Dorsa Sadigh",
    "id": "1779671",
    "h_index": 69,
    "papers": 226
   },
   {
    "name": "Percy Liang",
    "id": "145419642",
    "h_index": 104,
    "papers": 226
   }
  ],
  "comment": "30 Pages, 15 Figures",
  "topics": [
   "dexterous-manipulation",
   "egocentric-data",
   "imitation-diffusion",
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2302.12766v1",
  "pdf_url": "https://arxiv.org/pdf/2302.12766v1",
  "html_url": "https://arxiv.org/html/2302.12766v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.85
 },
 {
  "id": "2302.12422",
  "slug": "mimicplay-long-horizon-imitation-learning-by-watching-human-play",
  "title": "MimicPlay: Long-Horizon Imitation Learning by Watching Human Play",
  "abstract": "Imitation learning from human demonstrations is a promising paradigm for teaching robots manipulation skills in the real world. However, learning complex long-horizon tasks often requires an unattainable amount of demonstrations. To reduce the high data requirement, we resort to human play data - video sequences of people freely interacting with the environment using their hands. Even with different morphologies, we hypothesize that human play data contain rich and salient information about physical interactions that can readily facilitate robot policy learning. Motivated by this, we introduce a hierarchical learning framework named MimicPlay that learns latent plans from human play data to guide low-level visuomotor control trained on a small number of teleoperated demonstrations. With systematic evaluations of 14 long-horizon manipulation tasks in the real world, we show that MimicPlay outperforms state-of-the-art imitation learning methods in task success rate, generalization ability, and robustness to disturbances. Code and videos are available at https://mimic-play.github.io",
  "published": "2023-02-24",
  "updated": "2023-10-13",
  "year": "2023",
  "authors": [
   "Chen Wang",
   "Linxi Fan",
   "Jiankai Sun",
   "Ruohan Zhang",
   "Li Fei-Fei",
   "Danfei Xu",
   "Yuke Zhu",
   "Anima Anandkumar"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 355,
  "influential_citations": 19,
  "tldr": "A hierarchical learning framework named MimicPlay is introduced that learns latent plans from human play data to guide low-level visuomotor control trained on a small number of teleoperated demonstrations and outperforms state-of-the-art imitation learning methods in task success rate, generalization ability, and robustness to disturbances.",
  "doi": "10.48550/arXiv.2302.12422",
  "oa_pdf": "https://arxiv.org/pdf/2302.12422",
  "s2_authors": [
   {
    "name": "Chen Wang",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Linxi (Jim) Fan",
    "id": "3275727",
    "h_index": 18,
    "papers": 22
   },
   {
    "name": "Jiankai Sun",
    "id": "2282025",
    "h_index": 22,
    "papers": 61
   },
   {
    "name": "Ruohan Zhang",
    "id": "2657185",
    "h_index": 16,
    "papers": 46
   },
   {
    "name": "Li Fei-Fei",
    "id": "48004138",
    "h_index": 143,
    "papers": 606
   },
   {
    "name": "Danfei Xu",
    "id": "2068265",
    "h_index": 33,
    "papers": 45
   },
   {
    "name": "Yuke Zhu",
    "id": "2117748",
    "h_index": 57,
    "papers": 130
   },
   {
    "name": "Anima Anandkumar",
    "id": "47627049",
    "h_index": 41,
    "papers": 119
   }
  ],
  "comment": "7th Conference on Robot Learning (CoRL 2023 oral presentation)",
  "topics": [
   "egocentric-data",
   "imitation-diffusion",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2302.12422v2",
  "pdf_url": "https://arxiv.org/pdf/2302.12422v2",
  "html_url": "https://arxiv.org/html/2302.12422v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.05
 },
 {
  "id": "2302.12237",
  "slug": "learning-neural-volumetric-representations-of-dynamic-humans-in-minute",
  "title": "Learning Neural Volumetric Representations of Dynamic Humans in Minutes",
  "abstract": "This paper addresses the challenge of quickly reconstructing free-viewpoint videos of dynamic humans from sparse multi-view videos. Some recent works represent the dynamic human as a canonical neural radiance field (NeRF) and a motion field, which are learned from videos through differentiable rendering. But the per-scene optimization generally requires hours. Other generalizable NeRF models leverage learned prior from datasets and reduce the optimization time by only finetuning on new scenes at the cost of visual fidelity. In this paper, we propose a novel method for learning neural volumetric videos of dynamic humans from sparse view videos in minutes with competitive visual quality. Specifically, we define a novel part-based voxelized human representation to better distribute the representational power of the network to different human parts. Furthermore, we propose a novel 2D motion parameterization scheme to increase the convergence rate of deformation field learning. Experiments demonstrate that our model can be learned 100 times faster than prior per-scene optimization methods while being competitive in the rendering quality. Training our model on a $512 \\times 512$ video with 100 frames typically takes about 5 minutes on a single RTX 3090 GPU. The code will be released on our project page: https://zju3dv.github.io/instant_nvr",
  "published": "2023-02-23",
  "updated": "2023-02-24",
  "year": "2023",
  "authors": [
   "Chen Geng",
   "Sida Peng",
   "Zhen Xu",
   "Hujun Bao",
   "Xiaowei Zhou"
  ],
  "author_count": 5,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.GR"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 95,
  "influential_citations": 14,
  "tldr": "A novel method for learning neural volumetric representations of dynamic humans in minutes with competitive visual quality is proposed and a novel part-based voxelized human representation is defined to better distribute the representational power of the network to different human parts.",
  "doi": "10.1109/CVPR52729.2023.00846",
  "oa_pdf": "https://arxiv.org/pdf/2302.12237",
  "s2_authors": [
   {
    "name": "Chen Geng",
    "id": "2158857804",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Sida Peng",
    "id": "2072712025",
    "h_index": 41,
    "papers": 95
   },
   {
    "name": "Zhenqi Xu",
    "id": "3414892",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "H. Bao",
    "id": "1679542",
    "h_index": 71,
    "papers": 405
   },
   {
    "name": "Xiaowei Zhou",
    "id": "145453113",
    "h_index": 51,
    "papers": 95
   }
  ],
  "comment": "Project page: https://zju3dv.github.io/instant_nvr",
  "topics": [
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2302.12237v2",
  "pdf_url": "https://arxiv.org/pdf/2302.12237v2",
  "html_url": "https://arxiv.org/html/2302.12237v2",
  "code_url": "https://zju3dv.github.io/instant_nvr",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.48
 },
 {
  "id": "2302.06692",
  "slug": "guiding-pretraining-in-reinforcement-learning-with-large-language-mode",
  "title": "Guiding Pretraining in Reinforcement Learning with Large Language Models",
  "abstract": "Reinforcement learning algorithms typically struggle in the absence of a dense, well-shaped reward function. Intrinsically motivated exploration methods address this limitation by rewarding agents for visiting novel states or transitions, but these methods offer limited benefits in large environments where most discovered novelty is irrelevant for downstream tasks. We describe a method that uses background knowledge from text corpora to shape exploration. This method, called ELLM (Exploring with LLMs) rewards an agent for achieving goals suggested by a language model prompted with a description of the agent's current state. By leveraging large-scale language model pretraining, ELLM guides agents toward human-meaningful and plausibly useful behaviors without requiring a human in the loop. We evaluate ELLM in the Crafter game environment and the Housekeep robotic simulator, showing that ELLM-trained agents have better coverage of common-sense behaviors during pretraining and usually match or improve performance on a range of downstream tasks. Code available at https://github.com/yuqingd/ellm.",
  "published": "2023-02-13",
  "updated": "2023-09-15",
  "year": "2023",
  "authors": [
   "Yuqing Du",
   "Olivia Watkins",
   "Zihan Wang",
   "C\u00e9dric Colas",
   "Trevor Darrell",
   "Pieter Abbeel",
   "Abhishek Gupta",
   "Jacob Andreas"
  ],
  "author_count": 8,
  "categories": [
   "cs.LG",
   "cs.AI",
   "cs.CL"
  ],
  "primary_category": "cs.LG",
  "venue": "ICML",
  "venue_source": "semantic-scholar",
  "citations": 286,
  "influential_citations": 27,
  "tldr": "A method that uses background knowledge from text corpora to shape exploration and rewards an agent for achieving goals suggested by a language model prompted with a description of the agent's current state, called ELLM (Exploring with LLMs).",
  "doi": "10.48550/arXiv.2302.06692",
  "oa_pdf": "https://arxiv.org/pdf/2302.06692",
  "s2_authors": [
   {
    "name": "Yuqing Du",
    "id": "144894286",
    "h_index": 16,
    "papers": 27
   },
   {
    "name": "Olivia Watkins",
    "id": "145695607",
    "h_index": 9,
    "papers": 12
   },
   {
    "name": "Wang",
    "id": "2140051107",
    "h_index": 9,
    "papers": 19
   },
   {
    "name": "C\u00e9dric Colas",
    "id": "102281182",
    "h_index": 20,
    "papers": 46
   },
   {
    "name": "Trevor Darrell",
    "id": "1753210",
    "h_index": 158,
    "papers": 630
   },
   {
    "name": "P. Abbeel",
    "id": "1689992",
    "h_index": 165,
    "papers": 596
   },
   {
    "name": "Abhishek Gupta",
    "id": "144150274",
    "h_index": 33,
    "papers": 287
   },
   {
    "name": "Jacob Andreas",
    "id": "2112400",
    "h_index": 53,
    "papers": 93
   }
  ],
  "comment": "ICML 2023",
  "topics": [
   "sim2real",
   "rl-control",
   "navigation",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2302.06692v2",
  "pdf_url": "https://arxiv.org/pdf/2302.06692v2",
  "html_url": "https://arxiv.org/html/2302.06692v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.96
 },
 {
  "id": "2302.05442",
  "slug": "scaling-vision-transformers-to-22-billion-parameters",
  "title": "Scaling Vision Transformers to 22 Billion Parameters",
  "abstract": "The scaling of Transformers has driven breakthrough capabilities for language models. At present, the largest large language models (LLMs) contain upwards of 100B parameters. Vision Transformers (ViT) have introduced the same architecture to image and video modelling, but these have not yet been successfully scaled to nearly the same degree; the largest dense ViT contains 4B parameters (Chen et al., 2022). We present a recipe for highly efficient and stable training of a 22B-parameter ViT (ViT-22B) and perform a wide variety of experiments on the resulting model. When evaluated on downstream tasks (often with a lightweight linear model on frozen features), ViT-22B demonstrates increasing performance with scale. We further observe other interesting benefits of scale, including an improved tradeoff between fairness and performance, state-of-the-art alignment to human visual perception in terms of shape/texture bias, and improved robustness. ViT-22B demonstrates the potential for \"LLM-like\" scaling in vision, and provides key steps towards getting there.",
  "published": "2023-02-10",
  "updated": "2023-02-10",
  "year": "2023",
  "authors": [
   "Mostafa Dehghani",
   "Josip Djolonga",
   "Basil Mustafa",
   "Piotr Padlewski",
   "Jonathan Heek",
   "Justin Gilmer",
   "Andreas Steiner",
   "Mathilde Caron",
   "Robert Geirhos",
   "Ibrahim Alabdulmohsin",
   "Rodolphe Jenatton",
   "Lucas Beyer",
   "Michael Tschannen",
   "Anurag Arnab",
   "Xiao Wang",
   "Carlos Riquelme",
   "Matthias Minderer",
   "Joan Puigcerver",
   "Utku Evci",
   "Manoj Kumar",
   "Sjoerd van Steenkiste",
   "Gamaleldin F. Elsayed",
   "Aravindh Mahendran",
   "Fisher Yu",
   "Avital Oliver",
   "Fantine Huot",
   "Jasmijn Bastings",
   "Mark Patrick Collier",
   "Alexey Gritsenko",
   "Vighnesh Birodkar",
   "Cristina Vasconcelos",
   "Yi Tay",
   "Thomas Mensink",
   "Alexander Kolesnikov",
   "Filip Paveti\u0107",
   "Dustin Tran",
   "Thomas Kipf",
   "Mario Lu\u010di\u0107",
   "Xiaohua Zhai",
   "Daniel Keysers",
   "Jeremiah Harmsen",
   "Neil Houlsby"
  ],
  "author_count": 42,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.CV",
  "venue": "ICML",
  "venue_source": "semantic-scholar",
  "citations": 924,
  "influential_citations": 53,
  "tldr": "A recipe for highly efficient and stable training of a 22B-parameter ViT (ViT-22B) and a wide variety of experiments on the resulting model, which demonstrates the potential for \"LLM-like\"scaling in vision, and provides key steps towards getting there.",
  "doi": "10.48550/arXiv.2302.05442",
  "oa_pdf": "http://arxiv.org/pdf/2302.05442",
  "s2_authors": [
   {
    "name": "Mostafa Dehghani",
    "id": "3226635",
    "h_index": 42,
    "papers": 94
   },
   {
    "name": "J. Djolonga",
    "id": "2941141",
    "h_index": 24,
    "papers": 41
   },
   {
    "name": "Basil Mustafa",
    "id": "40608942",
    "h_index": 27,
    "papers": 40
   },
   {
    "name": "Piotr Padlewski",
    "id": "31148950",
    "h_index": 9,
    "papers": 11
   },
   {
    "name": "J. Heek",
    "id": "151488492",
    "h_index": 12,
    "papers": 19
   },
   {
    "name": "J. Gilmer",
    "id": "2058362",
    "h_index": 32,
    "papers": 58
   },
   {
    "name": "A. Steiner",
    "id": "2079614268",
    "h_index": 15,
    "papers": 23
   },
   {
    "name": "Mathilde Caron",
    "id": "2062862676",
    "h_index": 21,
    "papers": 29
   },
   {
    "name": "Robert Geirhos",
    "id": "1949747",
    "h_index": 20,
    "papers": 38
   },
   {
    "name": "Ibrahim M. Alabdulmohsin",
    "id": "2922782",
    "h_index": 22,
    "papers": 60
   },
   {
    "name": "Rodolphe Jenatton",
    "id": "2068720",
    "h_index": 34,
    "papers": 59
   },
   {
    "name": "Lucas Beyer",
    "id": "39611591",
    "h_index": 39,
    "papers": 59
   },
   {
    "name": "Michael Tschannen",
    "id": "143902495",
    "h_index": 37,
    "papers": 72
   },
   {
    "name": "Anurag Arnab",
    "id": "31638576",
    "h_index": 27,
    "papers": 45
   },
   {
    "name": "Xiao Wang",
    "id": "144129720",
    "h_index": 66,
    "papers": 713
   },
   {
    "name": "C. Riquelme",
    "id": "145814174",
    "h_index": 22,
    "papers": 33
   },
   {
    "name": "M. Minderer",
    "id": "46352821",
    "h_index": 22,
    "papers": 29
   },
   {
    "name": "J. Puigcerver",
    "id": "1794202",
    "h_index": 26,
    "papers": 46
   },
   {
    "name": "Utku Evci",
    "id": "3399348",
    "h_index": 19,
    "papers": 33
   },
   {
    "name": "Manoj Kumar",
    "id": "2157851754",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Sjoerd van Steenkiste",
    "id": "3440930",
    "h_index": 22,
    "papers": 40
   },
   {
    "name": "Gamaleldin F. Elsayed",
    "id": "7843061",
    "h_index": 19,
    "papers": 43
   },
   {
    "name": "Aravindh Mahendran",
    "id": "32694028",
    "h_index": 18,
    "papers": 34
   },
   {
    "name": "F. Yu",
    "id": "1807197",
    "h_index": 60,
    "papers": 105
   },
   {
    "name": "Avital Oliver",
    "id": "35679876",
    "h_index": 9,
    "papers": 14
   },
   {
    "name": "Fantine Huot",
    "id": "2174667321",
    "h_index": 11,
    "papers": 26
   },
   {
    "name": "Jasmijn Bastings",
    "id": "1994065972",
    "h_index": 14,
    "papers": 22
   },
   {
    "name": "Mark Collier",
    "id": "153247100",
    "h_index": 13,
    "papers": 30
   },
   {
    "name": "A. Gritsenko",
    "id": "2194424",
    "h_index": 23,
    "papers": 32
   },
   {
    "name": "Vighnesh Birodkar",
    "id": "3468723",
    "h_index": 11,
    "papers": 19
   },
   {
    "name": "C. Vasconcelos",
    "id": "2901520",
    "h_index": 15,
    "papers": 48
   },
   {
    "name": "Yi Tay",
    "id": "97947517",
    "h_index": 9,
    "papers": 14
   },
   {
    "name": "Thomas Mensink",
    "id": "1722052",
    "h_index": 31,
    "papers": 97
   },
   {
    "name": "Alexander Kolesnikov",
    "id": "144629422",
    "h_index": 30,
    "papers": 50
   },
   {
    "name": "Filip Paveti'c",
    "id": "2170163036",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Dustin Tran",
    "id": "47497262",
    "h_index": 38,
    "papers": 68
   },
   {
    "name": "Thomas Kipf",
    "id": "41016725",
    "h_index": 30,
    "papers": 56
   },
   {
    "name": "Mario Luvci'c",
    "id": "2170162986",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Xiaohua Zhai",
    "id": "2743563",
    "h_index": 40,
    "papers": 71
   },
   {
    "name": "Daniel Keysers",
    "id": "2064752644",
    "h_index": 8,
    "papers": 15
   },
   {
    "name": "Jeremiah Harmsen",
    "id": "2066076307",
    "h_index": 8,
    "papers": 8
   },
   {
    "name": "N. Houlsby",
    "id": "2815290",
    "h_index": 52,
    "papers": 102
   }
  ],
  "comment": "",
  "topics": [
   "video-generation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2302.05442v1",
  "pdf_url": "https://arxiv.org/pdf/2302.05442v1",
  "html_url": "https://arxiv.org/html/2302.05442v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.47
 },
 {
  "id": "2302.02662",
  "slug": "grounding-large-language-models-in-interactive-environments-with-onlin",
  "title": "Grounding Large Language Models in Interactive Environments with Online Reinforcement Learning",
  "abstract": "Recent works successfully leveraged Large Language Models' (LLM) abilities to capture abstract knowledge about world's physics to solve decision-making problems. Yet, the alignment between LLMs' knowledge and the environment can be wrong and limit functional competence due to lack of grounding. In this paper, we study an approach (named GLAM) to achieve this alignment through functional grounding: we consider an agent using an LLM as a policy that is progressively updated as the agent interacts with the environment, leveraging online Reinforcement Learning to improve its performance to solve goals. Using an interactive textual environment designed to study higher-level forms of functional grounding, and a set of spatial and navigation tasks, we study several scientific questions: 1) Can LLMs boost sample efficiency for online learning of various RL tasks? 2) How can it boost different forms of generalization? 3) What is the impact of online learning? We study these questions by functionally grounding several variants (size, architecture) of FLAN-T5.",
  "published": "2023-02-06",
  "updated": "2026-01-30",
  "year": "2023",
  "authors": [
   "Thomas Carta",
   "Cl\u00e9ment Romac",
   "Thomas Wolf",
   "Sylvain Lamprier",
   "Olivier Sigaud",
   "Pierre-Yves Oudeyer"
  ],
  "author_count": 6,
  "categories": [
   "cs.LG"
  ],
  "primary_category": "cs.LG",
  "venue": "ICML",
  "venue_source": "semantic-scholar",
  "citations": 285,
  "influential_citations": 30,
  "tldr": "This paper considers an agent using an LLM as a policy that is progressively updated as the agent interacts with the environment, leveraging online Reinforcement Learning to improve its performance to solve goals.",
  "doi": "10.48550/arXiv.2302.02662",
  "oa_pdf": "https://arxiv.org/pdf/2302.02662",
  "s2_authors": [
   {
    "name": "Thomas Carta",
    "id": "2003745905",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Cl\u00e9ment Romac",
    "id": "112906667",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Thomas Wolf",
    "id": "1407538116",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "S. Lamprier",
    "id": "1782552",
    "h_index": 24,
    "papers": 75
   },
   {
    "name": "Olivier Sigaud",
    "id": "97009622",
    "h_index": 17,
    "papers": 69
   },
   {
    "name": "P. Oudeyer",
    "id": "1720664",
    "h_index": 56,
    "papers": 370
   }
  ],
  "comment": "The associated code can be found at https://github.com/flowersteam/Grounding_LLMs_with_online_RL. This is an extended version of the paper published at ICML 2023: https://proceedings.mlr.press/v202/carta23a",
  "topics": [
   "rl-control",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2302.02662v5",
  "pdf_url": "https://arxiv.org/pdf/2302.02662v5",
  "html_url": "https://arxiv.org/html/2302.02662v5",
  "code_url": "https://github.com/flowersteam/Grounding_LLMs_with_online_RL.",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.96
 },
 {
  "id": "2302.02801",
  "slug": "lampp-language-models-as-probabilistic-priors-for-perception-and-actio",
  "title": "LaMPP: Language Models as Probabilistic Priors for Perception and Action",
  "abstract": "Language models trained on large text corpora encode rich distributional information about real-world environments and action sequences. This information plays a crucial role in current approaches to language processing tasks like question answering and instruction generation. We describe how to leverage language models for *non-linguistic* perception and control tasks. Our approach casts labeling and decision-making as inference in probabilistic graphical models in which language models parameterize prior distributions over labels, decisions and parameters, making it possible to integrate uncertain observations and incomplete background knowledge in a principled way. Applied to semantic segmentation, household navigation, and activity recognition tasks, this approach improves predictions on rare, out-of-distribution, and structurally novel inputs.",
  "published": "2023-02-03",
  "updated": "2023-02-03",
  "year": "2023",
  "authors": [
   "Belinda Z. Li",
   "William Chen",
   "Pratyusha Sharma",
   "Jacob Andreas"
  ],
  "author_count": 4,
  "categories": [
   "cs.LG",
   "cs.CL"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 17,
  "influential_citations": 0,
  "tldr": "This work describes how to leverage language models for *non-linguistic* perception and control tasks by casting labeling and decision-making as inference in probabilistic graphical models in which language models parameterize prior distributions over labels, decisions and parameters.",
  "doi": "10.48550/arXiv.2302.02801",
  "oa_pdf": "http://arxiv.org/pdf/2302.02801",
  "s2_authors": [
   {
    "name": "Belinda Z. Li",
    "id": "46708422",
    "h_index": 17,
    "papers": 29
   },
   {
    "name": "William Chen",
    "id": "2144302389",
    "h_index": 12,
    "papers": 16
   },
   {
    "name": "Pratyusha Sharma",
    "id": "50465425",
    "h_index": 16,
    "papers": 25
   },
   {
    "name": "Jacob Andreas",
    "id": "2112400",
    "h_index": 53,
    "papers": 93
   }
  ],
  "comment": "12 pages, 4 tables, 4 figures",
  "topics": [
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2302.02801v1",
  "pdf_url": "https://arxiv.org/pdf/2302.02801v1",
  "html_url": "https://arxiv.org/html/2302.02801v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.26
 },
 {
  "id": "2302.00763",
  "slug": "collaborating-with-language-models-for-embodied-reasoning",
  "title": "Collaborating with language models for embodied reasoning",
  "abstract": "Reasoning in a complex and ambiguous environment is a key goal for Reinforcement Learning (RL) agents. While some sophisticated RL agents can successfully solve difficult tasks, they require a large amount of training data and often struggle to generalize to new unseen environments and new tasks. On the other hand, Large Scale Language Models (LSLMs) have exhibited strong reasoning ability and the ability to to adapt to new tasks through in-context learning. However, LSLMs do not inherently have the ability to interrogate or intervene on the environment. In this work, we investigate how to combine these complementary abilities in a single system consisting of three parts: a Planner, an Actor, and a Reporter. The Planner is a pre-trained language model that can issue commands to a simple embodied agent (the Actor), while the Reporter communicates with the Planner to inform its next command. We present a set of tasks that require reasoning, test this system's ability to generalize zero-shot and investigate failure cases, and demonstrate how components of this system can be trained with reinforcement-learning to improve performance.",
  "published": "2023-02-01",
  "updated": "2023-02-01",
  "year": "2023",
  "authors": [
   "Ishita Dasgupta",
   "Christine Kaeser-Chen",
   "Kenneth Marino",
   "Arun Ahuja",
   "Sheila Babayan",
   "Felix Hill",
   "Rob Fergus"
  ],
  "author_count": 7,
  "categories": [
   "cs.LG",
   "cs.AI",
   "cs.CL"
  ],
  "primary_category": "cs.LG",
  "venue": "NeurIPS 2022",
  "venue_source": "arxiv-comment",
  "citations": 91,
  "influential_citations": 4,
  "tldr": "This work investigates how to combine complementary abilities in a single system consisting of a Planner, an Actor, and a Reporter, and presents a set of tasks that require reasoning, test this system's ability to generalize zero-shot and investigate failure cases, and demonstrates how components of this system can be trained with reinforcement-learning to improve performance.",
  "doi": "10.48550/arXiv.2302.00763",
  "oa_pdf": "http://arxiv.org/pdf/2302.00763",
  "s2_authors": [
   {
    "name": "Ishita Dasgupta",
    "id": "46745316",
    "h_index": 19,
    "papers": 52
   },
   {
    "name": "Christine Kaeser-Chen",
    "id": "1403585268",
    "h_index": 9,
    "papers": 16
   },
   {
    "name": "Kenneth Marino",
    "id": "35789996",
    "h_index": 10,
    "papers": 16
   },
   {
    "name": "Arun Ahuja",
    "id": "37968006",
    "h_index": 22,
    "papers": 49
   },
   {
    "name": "Sheila Babayan",
    "id": "2189420976",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Felix Hill",
    "id": "145783676",
    "h_index": 42,
    "papers": 77
   },
   {
    "name": "R. Fergus",
    "id": "2276554",
    "h_index": 78,
    "papers": 125
   }
  ],
  "comment": "Presented at NeurIPS 2022 Language and Reinforcement Learning Workshop (best paper) and NeurIPS 2022 Foundation Models for Decision Making Workshop. 4 pages main; 14 pages total (including references and appendix); 3 figures",
  "topics": [
   "rl-control",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2302.00763v1",
  "pdf_url": "https://arxiv.org/pdf/2302.00763v1",
  "html_url": "https://arxiv.org/html/2302.00763v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.46
 },
 {
  "id": "2302.00111",
  "slug": "learning-universal-policies-via-text-guided-video-generation",
  "title": "Learning Universal Policies via Text-Guided Video Generation",
  "abstract": "A goal of artificial intelligence is to construct an agent that can solve a wide variety of tasks. Recent progress in text-guided image synthesis has yielded models with an impressive ability to generate complex novel images, exhibiting combinatorial generalization across domains. Motivated by this success, we investigate whether such tools can be used to construct more general-purpose agents. Specifically, we cast the sequential decision making problem as a text-conditioned video generation problem, where, given a text-encoded specification of a desired goal, a planner synthesizes a set of future frames depicting its planned actions in the future, after which control actions are extracted from the generated video. By leveraging text as the underlying goal specification, we are able to naturally and combinatorially generalize to novel goals. The proposed policy-as-video formulation can further represent environments with different state and action spaces in a unified space of images, which, for example, enables learning and generalization across a variety of robot manipulation tasks. Finally, by leveraging pretrained language embeddings and widely available videos from the internet, the approach enables knowledge transfer through predicting highly realistic video plans for real robots.",
  "published": "2023-01-31",
  "updated": "2023-11-20",
  "year": "2023",
  "authors": [
   "Yilun Du",
   "Mengjiao Yang",
   "Bo Dai",
   "Hanjun Dai",
   "Ofir Nachum",
   "Joshua B. Tenenbaum",
   "Dale Schuurmans",
   "Pieter Abbeel"
  ],
  "author_count": 8,
  "categories": [
   "cs.AI"
  ],
  "primary_category": "cs.AI",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 644,
  "influential_citations": 47,
  "tldr": "This work casts the sequential decision making problem as a text-conditioned video generation problem, where a planner synthesizes a set of future frames depicting its planned actions in the future, after which control actions are extracted from the generated video.",
  "doi": "10.48550/arXiv.2302.00111",
  "oa_pdf": "http://arxiv.org/pdf/2302.00111",
  "s2_authors": [
   {
    "name": "Yilun Du",
    "id": "15394275",
    "h_index": 48,
    "papers": 86
   },
   {
    "name": "Mengjiao Yang",
    "id": "2111076891",
    "h_index": 13,
    "papers": 17
   },
   {
    "name": "Bo Dai",
    "id": "144445937",
    "h_index": 50,
    "papers": 152
   },
   {
    "name": "H. Dai",
    "id": "2791430",
    "h_index": 39,
    "papers": 87
   },
   {
    "name": "Ofir Nachum",
    "id": "7624658",
    "h_index": 47,
    "papers": 92
   },
   {
    "name": "J. Tenenbaum",
    "id": "1763295",
    "h_index": 140,
    "papers": 785
   },
   {
    "name": "Dale Schuurmans",
    "id": "1714772",
    "h_index": 65,
    "papers": 264
   },
   {
    "name": "P. Abbeel",
    "id": "1689992",
    "h_index": 165,
    "papers": 596
   }
  ],
  "comment": "NeurIPS 2023, Project Website: https://universal-policy.github.io/",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2302.00111v3",
  "pdf_url": "https://arxiv.org/pdf/2302.00111v3",
  "html_url": "https://arxiv.org/html/2302.00111v3",
  "code_url": "https://universal-policy.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.31
 },
 {
  "id": "2301.12597",
  "slug": "blip-2-bootstrapping-language-image-pre-training-with-frozen-image-enc",
  "title": "BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models",
  "abstract": "The cost of vision-and-language pre-training has become increasingly prohibitive due to end-to-end training of large-scale models. This paper proposes BLIP-2, a generic and efficient pre-training strategy that bootstraps vision-language pre-training from off-the-shelf frozen pre-trained image encoders and frozen large language models. BLIP-2 bridges the modality gap with a lightweight Querying Transformer, which is pre-trained in two stages. The first stage bootstraps vision-language representation learning from a frozen image encoder. The second stage bootstraps vision-to-language generative learning from a frozen language model. BLIP-2 achieves state-of-the-art performance on various vision-language tasks, despite having significantly fewer trainable parameters than existing methods. For example, our model outperforms Flamingo80B by 8.7% on zero-shot VQAv2 with 54x fewer trainable parameters. We also demonstrate the model's emerging capabilities of zero-shot image-to-text generation that can follow natural language instructions.",
  "published": "2023-01-30",
  "updated": "2023-06-15",
  "year": "2023",
  "authors": [
   "Junnan Li",
   "Dongxu Li",
   "Silvio Savarese",
   "Steven Hoi"
  ],
  "author_count": 4,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "ICML",
  "venue_source": "semantic-scholar",
  "citations": 9007,
  "influential_citations": 995,
  "tldr": "BLIP-2 achieves state-of-the-art performance on various vision-language tasks, despite having significantly fewer trainable parameters than existing methods, and is demonstrated's emerging capabilities of zero-shot image-to-text generation that can follow natural language instructions.",
  "doi": "10.48550/arXiv.2301.12597",
  "oa_pdf": "http://arxiv.org/pdf/2301.12597",
  "s2_authors": [
   {
    "name": "Junnan Li",
    "id": "49299019",
    "h_index": 18,
    "papers": 24
   },
   {
    "name": "Dongxu Li",
    "id": "2981509",
    "h_index": 21,
    "papers": 28
   },
   {
    "name": "S. Savarese",
    "id": "1702137",
    "h_index": 115,
    "papers": 346
   },
   {
    "name": "Steven C. H. Hoi",
    "id": "2184854289",
    "h_index": 13,
    "papers": 22
   }
  ],
  "comment": "",
  "topics": [
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2301.12597v3",
  "pdf_url": "https://arxiv.org/pdf/2301.12597v3",
  "html_url": "https://arxiv.org/html/2301.12597v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "2301.11093",
  "slug": "simple-diffusion-end-to-end-diffusion-for-high-resolution-images",
  "title": "Simple diffusion: End-to-end diffusion for high resolution images",
  "abstract": "Currently, applying diffusion models in pixel space of high resolution images is difficult. Instead, existing approaches focus on diffusion in lower dimensional spaces (latent diffusion), or have multiple super-resolution levels of generation referred to as cascades. The downside is that these approaches add additional complexity to the diffusion framework. This paper aims to improve denoising diffusion for high resolution images while keeping the model as simple as possible. The paper is centered around the research question: How can one train a standard denoising diffusion models on high resolution images, and still obtain performance comparable to these alternate approaches? The four main findings are: 1) the noise schedule should be adjusted for high resolution images, 2) It is sufficient to scale only a particular part of the architecture, 3) dropout should be added at specific locations in the architecture, and 4) downsampling is an effective strategy to avoid high resolution feature maps. Combining these simple yet effective techniques, we achieve state-of-the-art on image generation among diffusion models without sampling modifiers on ImageNet.",
  "published": "2023-01-26",
  "updated": "2023-12-12",
  "year": "2023",
  "authors": [
   "Emiel Hoogeboom",
   "Jonathan Heek",
   "Tim Salimans"
  ],
  "author_count": 3,
  "categories": [
   "cs.CV",
   "cs.LG",
   "stat.ML"
  ],
  "primary_category": "cs.CV",
  "venue": "ICML",
  "venue_source": "semantic-scholar",
  "citations": 465,
  "influential_citations": 38,
  "tldr": "The four main findings are: 1) the noise schedule should be adjusted for high resolution images, 2) It is sufficient to scale only a particular part of the architecture, 3) dropout should be added at specific locations in the Architecture, and 4) downsampling is an effective strategy to avoid high resolution feature maps.",
  "doi": "10.48550/arXiv.2301.11093",
  "oa_pdf": "http://arxiv.org/pdf/2301.11093",
  "s2_authors": [
   {
    "name": "E. Hoogeboom",
    "id": "65928943",
    "h_index": 19,
    "papers": 39
   },
   {
    "name": "J. Heek",
    "id": "151488492",
    "h_index": 12,
    "papers": 19
   },
   {
    "name": "Tim Salimans",
    "id": "2887364",
    "h_index": 36,
    "papers": 66
   }
  ],
  "comment": "",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2301.11093v2",
  "pdf_url": "https://arxiv.org/pdf/2301.11093v2",
  "html_url": "https://arxiv.org/html/2301.11093v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.17
 },
 {
  "id": "2301.07302",
  "slug": "pirlnav-pretraining-with-imitation-and-rl-finetuning-for-objectnav",
  "title": "PIRLNav: Pretraining with Imitation and RL Finetuning for ObjectNav",
  "abstract": "We study ObjectGoal Navigation -- where a virtual robot situated in a new environment is asked to navigate to an object. Prior work has shown that imitation learning (IL) using behavior cloning (BC) on a dataset of human demonstrations achieves promising results. However, this has limitations -- 1) BC policies generalize poorly to new states, since the training mimics actions not their consequences, and 2) collecting demonstrations is expensive. On the other hand, reinforcement learning (RL) is trivially scalable, but requires careful reward engineering to achieve desirable behavior. We present PIRLNav, a two-stage learning scheme for BC pretraining on human demonstrations followed by RL-finetuning. This leads to a policy that achieves a success rate of $65.0\\%$ on ObjectNav ($+5.0\\%$ absolute over previous state-of-the-art). Using this BC$\\rightarrow$RL training recipe, we present a rigorous empirical analysis of design choices. First, we investigate whether human demonstrations can be replaced with `free' (automatically generated) sources of demonstrations, e.g. shortest paths (SP) or task-agnostic frontier exploration (FE) trajectories. We find that BC$\\rightarrow$RL on human demonstrations outperforms BC$\\rightarrow$RL on SP and FE trajectories, even when controlled for same BC-pretraining success on train, and even on a subset of val episodes where BC-pretraining success favors the SP or FE policies. Next, we study how RL-finetuning performance scales with the size of the BC pretraining dataset. We find that as we increase the size of BC-pretraining dataset and get to high BC accuracies, improvements from RL-finetuning are smaller, and that $90\\%$ of the performance of our best BC$\\rightarrow$RL policy can be achieved with less than half the number of BC demonstrations. Finally, we analyze failure modes of our ObjectNav policies, and present guidelines for further improving them.",
  "published": "2023-01-18",
  "updated": "2023-03-26",
  "year": "2023",
  "authors": [
   "Ram Ramrakhya",
   "Dhruv Batra",
   "Erik Wijmans",
   "Abhishek Das"
  ],
  "author_count": 4,
  "categories": [
   "cs.LG",
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.LG",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 142,
  "influential_citations": 18,
  "tldr": "PIRLNav is presented, a two-stage learning scheme for BC pretraining on human demonstrations followed by RL-finetuning that outperforms BC\u2192RL on SP and FE trajectories, and investigates whether human demonstrations can be replaced with \u2018free\u2019 sources of demonstrations, e.g. shortest paths or task-agnostic frontier exploration trajectories.",
  "doi": "10.1109/CVPR52729.2023.01716",
  "oa_pdf": "https://arxiv.org/pdf/2301.07302",
  "s2_authors": [
   {
    "name": "Ram Ramrakhya",
    "id": "80155427",
    "h_index": 7,
    "papers": 8
   },
   {
    "name": "Dhruv Batra",
    "id": "1746610",
    "h_index": 87,
    "papers": 327
   },
   {
    "name": "Erik Wijmans",
    "id": "2065670046",
    "h_index": 15,
    "papers": 20
   },
   {
    "name": "Abhishek Das",
    "id": "1410472160",
    "h_index": 10,
    "papers": 43
   }
  ],
  "comment": "8 pages + supplement",
  "topics": [
   "egocentric-data",
   "imitation-diffusion",
   "rl-control",
   "navigation",
   "foundation-pretraining",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2301.07302v2",
  "pdf_url": "https://arxiv.org/pdf/2301.07302v2",
  "html_url": "https://arxiv.org/html/2301.07302v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.66
 },
 {
  "id": "2301.04195",
  "slug": "orbit-a-unified-simulation-framework-for-interactive-robot-learning-en",
  "title": "Orbit: A Unified Simulation Framework for Interactive Robot Learning Environments",
  "abstract": "We present Orbit, a unified and modular framework for robot learning powered by NVIDIA Isaac Sim. It offers a modular design to easily and efficiently create robotic environments with photo-realistic scenes and high-fidelity rigid and deformable body simulation. With Orbit, we provide a suite of benchmark tasks of varying difficulty -- from single-stage cabinet opening and cloth folding to multi-stage tasks such as room reorganization. To support working with diverse observations and action spaces, we include fixed-arm and mobile manipulators with different physically-based sensors and motion generators. Orbit allows training reinforcement learning policies and collecting large demonstration datasets from hand-crafted or expert solutions in a matter of minutes by leveraging GPU-based parallelization. In summary, we offer an open-sourced framework that readily comes with 16 robotic platforms, 4 sensor modalities, 10 motion generators, more than 20 benchmark tasks, and wrappers to 4 learning libraries. With this framework, we aim to support various research areas, including representation learning, reinforcement learning, imitation learning, and task and motion planning. We hope it helps establish interdisciplinary collaborations in these communities, and its modularity makes it easily extensible for more tasks and applications in the future.",
  "published": "2023-01-10",
  "updated": "2024-02-16",
  "year": "2023",
  "authors": [
   "Mayank Mittal",
   "Calvin Yu",
   "Qinxi Yu",
   "Jingzhou Liu",
   "Nikita Rudin",
   "David Hoeller",
   "Jia Lin Yuan",
   "Ritvik Singh",
   "Yunrong Guo",
   "Hammad Mazhar",
   "Ajay Mandlekar",
   "Buck Babich",
   "Gavriel State",
   "Marco Hutter",
   "Animesh Garg"
  ],
  "author_count": 15,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "RA-L",
  "venue_source": "semantic-scholar",
  "citations": 561,
  "influential_citations": 33,
  "tldr": "Orbit is presented, a unified and modular framework for robot learning powered by Nvidia Isaac Sim that readily comes with 16 robotic platforms, 4 sensor modalities, 10 motion generators, more than 20 benchmark tasks, and wrappers to 4 learning libraries to support various research areas.",
  "doi": "10.1109/LRA.2023.3270034",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Mayank Mittal",
    "id": "2061780867",
    "h_index": 13,
    "papers": 30
   },
   {
    "name": "C. Yu",
    "id": "40592238",
    "h_index": 21,
    "papers": 108
   },
   {
    "name": "Qinxi Yu",
    "id": "2199976211",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Jason Jingzhou Liu",
    "id": "2108367188",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "N. Rudin",
    "id": "2113243810",
    "h_index": 17,
    "papers": 17
   },
   {
    "name": "David Hoeller",
    "id": "71054073",
    "h_index": 18,
    "papers": 19
   },
   {
    "name": "Jianping Yuan",
    "id": "2118573822",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Pooria Poorsarvi Tehrani",
    "id": "2199940593",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Ritvik Singh",
    "id": "81452558",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Yunrong Guo",
    "id": "2029890590",
    "h_index": 13,
    "papers": 14
   },
   {
    "name": "H. Mazhar",
    "id": "3342503",
    "h_index": 18,
    "papers": 63
   },
   {
    "name": "A. Mandlekar",
    "id": "49686756",
    "h_index": 36,
    "papers": 67
   },
   {
    "name": "Buck Babich",
    "id": "1956025034",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Gavriel State",
    "id": "82261827",
    "h_index": 10,
    "papers": 14
   },
   {
    "name": "Marco Hutter",
    "id": "14349870",
    "h_index": 80,
    "papers": 279
   },
   {
    "name": "Animesh Garg",
    "id": "1873736",
    "h_index": 60,
    "papers": 163
   }
  ],
  "comment": "Project website: https://isaac-orbit.github.io/",
  "topics": [
   "imitation-diffusion",
   "rl-control"
  ],
  "orgs": [
   "NVIDIA"
  ],
  "abs_url": "https://arxiv.org/abs/2301.04195v2",
  "pdf_url": "https://arxiv.org/pdf/2301.04195v2",
  "html_url": "https://arxiv.org/html/2301.04195v2",
  "code_url": "https://isaac-orbit.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.75
 },
 {
  "id": "2301.04104",
  "slug": "mastering-diverse-domains-through-world-models",
  "title": "Mastering Diverse Domains through World Models",
  "abstract": "Developing a general algorithm that learns to solve tasks across a wide range of applications has been a fundamental challenge in artificial intelligence. Although current reinforcement learning algorithms can be readily applied to tasks similar to what they have been developed for, configuring them for new application domains requires significant human expertise and experimentation. We present DreamerV3, a general algorithm that outperforms specialized methods across over 150 diverse tasks, with a single configuration. Dreamer learns a model of the environment and improves its behavior by imagining future scenarios. Robustness techniques based on normalization, balancing, and transformations enable stable learning across domains. Applied out of the box, Dreamer is the first algorithm to collect diamonds in Minecraft from scratch without human data or curricula. This achievement has been posed as a significant challenge in artificial intelligence that requires exploring farsighted strategies from pixels and sparse rewards in an open world. Our work allows solving challenging control problems without extensive experimentation, making reinforcement learning broadly applicable.",
  "published": "2023-01-10",
  "updated": "2024-04-17",
  "year": "2023",
  "authors": [
   "Danijar Hafner",
   "Jurgis Pasukonis",
   "Jimmy Ba",
   "Timothy Lillicrap"
  ],
  "author_count": 4,
  "categories": [
   "cs.AI",
   "cs.LG",
   "stat.ML"
  ],
  "primary_category": "cs.AI",
  "venue": "",
  "venue_source": "",
  "citations": 1333,
  "influential_citations": 179,
  "tldr": "DreamerV3 is presented, a general algorithm that outperforms specialized methods across over 150 diverse tasks, with a single configuration, and is the first algorithm to collect diamonds in Minecraft from scratch without human data or curricula.",
  "doi": "10.48550/arXiv.2301.04104",
  "oa_pdf": "http://arxiv.org/pdf/2301.04104",
  "s2_authors": [
   {
    "name": "Danijar Hafner",
    "id": "35006479",
    "h_index": 25,
    "papers": 47
   },
   {
    "name": "J. Pa\u0161ukonis",
    "id": "31143488",
    "h_index": 8,
    "papers": 20
   },
   {
    "name": "Jimmy Ba",
    "id": "2503659",
    "h_index": 46,
    "papers": 84
   },
   {
    "name": "T. Lillicrap",
    "id": "2542999",
    "h_index": 69,
    "papers": 154
   }
  ],
  "comment": "Website: https://danijar.com/dreamerv3",
  "topics": [
   "world-models",
   "rl-control",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2301.04104v2",
  "pdf_url": "https://arxiv.org/pdf/2301.04104v2",
  "html_url": "https://arxiv.org/html/2301.04104v2",
  "code_url": null,
  "presented": [
   {
    "club": "saturday-robotics",
    "club_name": "Saturday Robotics",
    "session_number": 7,
    "session_title": "Robotics & World Models Reading Club 07: Learning to Dream: World Models, Imagination, Path to Foundation Models for Control \u2014 Los Altos",
    "date_text": "Saturday, May 9, 2026 | 2:00 PM \u2013 5:00 PM",
    "city": "",
    "url": "https://lu.ma/srhe0vuo",
    "listed_as": "Mastering Diverse Domains through World Models (2023)"
   }
  ],
  "club_note": "Unified training recipe across many domains with fixed hyperparameters",
  "featured": true,
  "signal": 6.0
 },
 {
  "id": "2301.00704",
  "slug": "muse-text-to-image-generation-via-masked-generative-transformers",
  "title": "Muse: Text-To-Image Generation via Masked Generative Transformers",
  "abstract": "We present Muse, a text-to-image Transformer model that achieves state-of-the-art image generation performance while being significantly more efficient than diffusion or autoregressive models. Muse is trained on a masked modeling task in discrete token space: given the text embedding extracted from a pre-trained large language model (LLM), Muse is trained to predict randomly masked image tokens. Compared to pixel-space diffusion models, such as Imagen and DALL-E 2, Muse is significantly more efficient due to the use of discrete tokens and requiring fewer sampling iterations; compared to autoregressive models, such as Parti, Muse is more efficient due to the use of parallel decoding. The use of a pre-trained LLM enables fine-grained language understanding, translating to high-fidelity image generation and the understanding of visual concepts such as objects, their spatial relationships, pose, cardinality etc. Our 900M parameter model achieves a new SOTA on CC3M, with an FID score of 6.06. The Muse 3B parameter model achieves an FID of 7.88 on zero-shot COCO evaluation, along with a CLIP score of 0.32. Muse also directly enables a number of image editing applications without the need to fine-tune or invert the model: inpainting, outpainting, and mask-free editing. More results are available at https://muse-model.github.io",
  "published": "2023-01-02",
  "updated": "2023-01-02",
  "year": "2023",
  "authors": [
   "Huiwen Chang",
   "Han Zhang",
   "Jarred Barber",
   "AJ Maschinot",
   "Jose Lezama",
   "Lu Jiang",
   "Ming-Hsuan Yang",
   "Kevin Murphy",
   "William T. Freeman",
   "Michael Rubinstein",
   "Yuanzhen Li",
   "Dilip Krishnan"
  ],
  "author_count": 12,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.CV",
  "venue": "ICML",
  "venue_source": "semantic-scholar",
  "citations": 792,
  "influential_citations": 52,
  "tldr": "",
  "doi": "10.48550/arXiv.2301.00704",
  "oa_pdf": "http://arxiv.org/pdf/2301.00704",
  "s2_authors": [
   {
    "name": "Huiwen Chang",
    "id": "2914394",
    "h_index": 28,
    "papers": 38
   },
   {
    "name": "Han Zhang",
    "id": "2146204239",
    "h_index": 8,
    "papers": 10
   },
   {
    "name": "Jarred Barber",
    "id": "152630175",
    "h_index": 9,
    "papers": 18
   },
   {
    "name": "AJ Maschinot",
    "id": "2199119286",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Jos\u00e9 Lezama",
    "id": "143923528",
    "h_index": 15,
    "papers": 33
   },
   {
    "name": "Lu Jiang",
    "id": "39978626",
    "h_index": 48,
    "papers": 78
   },
   {
    "name": "Ming Yang",
    "id": "152790163",
    "h_index": 26,
    "papers": 73
   },
   {
    "name": "K. Murphy",
    "id": "1702318",
    "h_index": 52,
    "papers": 68
   },
   {
    "name": "W. Freeman",
    "id": "1768236",
    "h_index": 119,
    "papers": 343
   },
   {
    "name": "Michael Rubinstein",
    "id": "144544291",
    "h_index": 36,
    "papers": 52
   },
   {
    "name": "Yuanzhen Li",
    "id": "2167749913",
    "h_index": 20,
    "papers": 29
   },
   {
    "name": "Dilip Krishnan",
    "id": "1707347",
    "h_index": 36,
    "papers": 101
   }
  ],
  "comment": "",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2301.00704v1",
  "pdf_url": "https://arxiv.org/pdf/2301.00704v1",
  "html_url": "https://arxiv.org/html/2301.00704v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.4
 },
 {
  "id": "2212.10846",
  "slug": "from-images-to-textual-prompts-zero-shot-vqa-with-frozen-large-languag",
  "title": "From Images to Textual Prompts: Zero-shot VQA with Frozen Large Language Models",
  "abstract": "Large language models (LLMs) have demonstrated excellent zero-shot generalization to new language tasks. However, effective utilization of LLMs for zero-shot visual question-answering (VQA) remains challenging, primarily due to the modality disconnection and task disconnection between LLM and VQA task. End-to-end training on vision and language data may bridge the disconnections, but is inflexible and computationally expensive. To address this issue, we propose \\emph{Img2Prompt}, a plug-and-play module that provides the prompts that can bridge the aforementioned modality and task disconnections, so that LLMs can perform zero-shot VQA tasks without end-to-end training. In order to provide such prompts, we further employ LLM-agnostic models to provide prompts that can describe image content and self-constructed question-answer pairs, which can effectively guide LLM to perform zero-shot VQA tasks. Img2Prompt offers the following benefits: 1) It can flexibly work with various LLMs to perform VQA. 2)~Without the needing of end-to-end training, it significantly reduces the cost of deploying LLM for zero-shot VQA tasks. 3) It achieves comparable or better performance than methods relying on end-to-end training. For example, we outperform Flamingo \\cite{Deepmind:Flamingo2022} by 5.6\\% on VQAv2. On the challenging A-OKVQA dataset, our method even outperforms few-shot methods by as much as 20\\%.",
  "published": "2022-12-21",
  "updated": "2023-05-08",
  "year": "2022",
  "authors": [
   "Jiaxian Guo",
   "Junnan Li",
   "Dongxu Li",
   "Anthony Meng Huat Tiong",
   "Boyang Li",
   "Dacheng Tao",
   "Steven C. H. Hoi"
  ],
  "author_count": 7,
  "categories": [
   "cs.CV",
   "cs.MM"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 195,
  "influential_citations": 16,
  "tldr": "Img2LLM is a plug-and-play module that provides LLM prompts to enable LLMs to perform zeroshot VQA tasks without end-to-end training and eliminates the need to specialize LLMs using end-to-end finetuning and serve highly specialized LLMs to end users, thereby reducing cost.",
  "doi": "10.1109/CVPR52729.2023.01046",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jiaxian Guo",
    "id": "15563286",
    "h_index": 10,
    "papers": 13
   },
   {
    "name": "Junnan Li",
    "id": "49299019",
    "h_index": 18,
    "papers": 24
   },
   {
    "name": "Dongxu Li",
    "id": "2981509",
    "h_index": 21,
    "papers": 28
   },
   {
    "name": "A. Tiong",
    "id": "73137089",
    "h_index": 12,
    "papers": 22
   },
   {
    "name": "Boyang Albert Li",
    "id": "1728712",
    "h_index": 27,
    "papers": 121
   },
   {
    "name": "Dacheng Tao",
    "id": "2140448089",
    "h_index": 30,
    "papers": 70
   },
   {
    "name": "Steven C. H. Hoi",
    "id": "2184854289",
    "h_index": 13,
    "papers": 22
   }
  ],
  "comment": "CVPR 2023 Camera Ready Version",
  "topics": [],
  "orgs": [
   "Google DeepMind"
  ],
  "abs_url": "https://arxiv.org/abs/2212.10846v3",
  "pdf_url": "https://arxiv.org/pdf/2212.10846v3",
  "html_url": "https://arxiv.org/html/2212.10846v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.29
 },
 {
  "id": "2212.09748",
  "slug": "scalable-diffusion-models-with-transformers",
  "title": "Scalable Diffusion Models with Transformers",
  "abstract": "We explore a new class of diffusion models based on the transformer architecture. We train latent diffusion models of images, replacing the commonly-used U-Net backbone with a transformer that operates on latent patches. We analyze the scalability of our Diffusion Transformers (DiTs) through the lens of forward pass complexity as measured by Gflops. We find that DiTs with higher Gflops -- through increased transformer depth/width or increased number of input tokens -- consistently have lower FID. In addition to possessing good scalability properties, our largest DiT-XL/2 models outperform all prior diffusion models on the class-conditional ImageNet 512x512 and 256x256 benchmarks, achieving a state-of-the-art FID of 2.27 on the latter.",
  "published": "2022-12-19",
  "updated": "2023-03-02",
  "year": "2022",
  "authors": [
   "William Peebles",
   "Saining Xie"
  ],
  "author_count": 2,
  "categories": [
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.CV",
  "venue": "ICCV",
  "venue_source": "semantic-scholar",
  "citations": 6982,
  "influential_citations": 821,
  "tldr": "A new class of diffusion models based on the transformer architecture is explored, replacing the commonly-used U-Net backbone with a transformer that operates on latent patches that outperform all prior diffusion models on the class-conditional ImageNet 512\u00d7512 and 256\u00d7256 benchmarks.",
  "doi": "10.1109/ICCV51070.2023.00387",
  "oa_pdf": "https://arxiv.org/pdf/2212.09748",
  "s2_authors": [
   {
    "name": "William S. Peebles",
    "id": "35235273",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Saining Xie",
    "id": "1817030",
    "h_index": 32,
    "papers": 46
   }
  ],
  "comment": "Code, project page and videos available at https://www.wpeebles.com/DiT",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2212.09748v2",
  "pdf_url": "https://arxiv.org/pdf/2212.09748v2",
  "html_url": "https://arxiv.org/html/2212.09748v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "2212.08333",
  "slug": "anygrasp-robust-and-efficient-grasp-perception-in-spatial-and-temporal",
  "title": "AnyGrasp: Robust and Efficient Grasp Perception in Spatial and Temporal Domains",
  "abstract": "As the basis for prehensile manipulation, it is vital to enable robots to grasp as robustly as humans. Our innate grasping system is prompt, accurate, flexible, and continuous across spatial and temporal domains. Few existing methods cover all these properties for robot grasping. In this paper, we propose AnyGrasp for grasp perception to enable robots these abilities using a parallel gripper. Specifically, we develop a dense supervision strategy with real perception and analytic labels in the spatial-temporal domain. Additional awareness of objects' center-of-mass is incorporated into the learning process to help improve grasping stability. Utilization of grasp correspondence across observations enables dynamic grasp tracking. Our model can efficiently generate accurate, 7-DoF, dense, and temporally-smooth grasp poses and works robustly against large depth-sensing noise. Using AnyGrasp, we achieve a 93.3% success rate when clearing bins with over 300 unseen objects, which is on par with human subjects under controlled conditions. Over 900 mean-picks-per-hour is reported on a single-arm system. For dynamic grasping, we demonstrate catching swimming robot fish in the water. Our project page is at https://graspnet.net/anygrasp.html",
  "published": "2022-12-16",
  "updated": "2023-06-06",
  "year": "2022",
  "authors": [
   "Hao-Shu Fang",
   "Chenxi Wang",
   "Hongjie Fang",
   "Minghao Gou",
   "Jirong Liu",
   "Hengxu Yan",
   "Wenhai Liu",
   "Yichen Xie",
   "Cewu Lu"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "T-RO",
  "venue_source": "semantic-scholar",
  "citations": 496,
  "influential_citations": 59,
  "tldr": "This article proposes AnyGrasp for grasp perception to enable robots to grasp as robustly as humans using a parallel gripper and develops a dense supervision strategy with real perception and analytic labels in the spatial\u2013temporal domain.",
  "doi": "10.1109/TRO.2023.3281153",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Haoshu Fang",
    "id": "122851212",
    "h_index": 34,
    "papers": 61
   },
   {
    "name": "Chenxi Wang",
    "id": "2109436791",
    "h_index": 11,
    "papers": 20
   },
   {
    "name": "Hongjie Fang",
    "id": "2152115958",
    "h_index": 16,
    "papers": 31
   },
   {
    "name": "Minghao Gou",
    "id": "66176207",
    "h_index": 11,
    "papers": 17
   },
   {
    "name": "Jirong Liu",
    "id": "2135259062",
    "h_index": 7,
    "papers": 12
   },
   {
    "name": "Hengxu Yan",
    "id": "2196614660",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Wenhai Liu",
    "id": "46641885",
    "h_index": 11,
    "papers": 23
   },
   {
    "name": "Yichen Xie",
    "id": "2149182167",
    "h_index": 10,
    "papers": 14
   },
   {
    "name": "Cewu Lu",
    "id": "2281998765",
    "h_index": 24,
    "papers": 45
   }
  ],
  "comment": "Paper accepted to T-RO. Project page is at https://graspnet.net/anygrasp.html",
  "topics": [
   "dexterous-manipulation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2212.08333v2",
  "pdf_url": "https://arxiv.org/pdf/2212.08333v2",
  "html_url": "https://arxiv.org/html/2212.08333v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.2
 },
 {
  "id": "2212.08073",
  "slug": "constitutional-ai-harmlessness-from-ai-feedback",
  "title": "Constitutional AI: Harmlessness from AI Feedback",
  "abstract": "As AI systems become more capable, we would like to enlist their help to supervise other AIs. We experiment with methods for training a harmless AI assistant through self-improvement, without any human labels identifying harmful outputs. The only human oversight is provided through a list of rules or principles, and so we refer to the method as 'Constitutional AI'. The process involves both a supervised learning and a reinforcement learning phase. In the supervised phase we sample from an initial model, then generate self-critiques and revisions, and then finetune the original model on revised responses. In the RL phase, we sample from the finetuned model, use a model to evaluate which of the two samples is better, and then train a preference model from this dataset of AI preferences. We then train with RL using the preference model as the reward signal, i.e. we use 'RL from AI Feedback' (RLAIF). As a result we are able to train a harmless but non-evasive AI assistant that engages with harmful queries by explaining its objections to them. Both the SL and RL methods can leverage chain-of-thought style reasoning to improve the human-judged performance and transparency of AI decision making. These methods make it possible to control AI behavior more precisely and with far fewer human labels.",
  "published": "2022-12-15",
  "updated": "2022-12-15",
  "year": "2022",
  "authors": [
   "Yuntao Bai",
   "Saurav Kadavath",
   "Sandipan Kundu",
   "Amanda Askell",
   "Jackson Kernion",
   "Andy Jones",
   "Anna Chen",
   "Anna Goldie",
   "Azalia Mirhoseini",
   "Cameron McKinnon",
   "Carol Chen",
   "Catherine Olsson",
   "Christopher Olah",
   "Danny Hernandez",
   "Dawn Drain",
   "Deep Ganguli",
   "Dustin Li",
   "Eli Tran-Johnson",
   "Ethan Perez",
   "Jamie Kerr",
   "Jared Mueller",
   "Jeffrey Ladish",
   "Joshua Landau",
   "Kamal Ndousse",
   "Kamile Lukosuite",
   "Liane Lovitt",
   "Michael Sellitto",
   "Nelson Elhage",
   "Nicholas Schiefer",
   "Noemi Mercado",
   "Nova DasSarma",
   "Robert Lasenby",
   "Robin Larson",
   "Sam Ringer",
   "Scott Johnston",
   "Shauna Kravec",
   "Sheer El Showk",
   "Stanislav Fort",
   "Tamera Lanham",
   "Timothy Telleen-Lawton",
   "Tom Conerly",
   "Tom Henighan",
   "Tristan Hume",
   "Samuel R. Bowman",
   "Zac Hatfield-Dodds",
   "Ben Mann",
   "Dario Amodei",
   "Nicholas Joseph",
   "Sam McCandlish",
   "Tom Brown",
   "Jared Kaplan"
  ],
  "author_count": 51,
  "categories": [
   "cs.CL",
   "cs.AI"
  ],
  "primary_category": "cs.CL",
  "venue": "",
  "venue_source": "",
  "citations": 3587,
  "influential_citations": 250,
  "tldr": "This work experiments with methods for training a harmless AI assistant through self-improvement, without any human labels identifying harmful outputs, and makes it possible to control AI behavior more precisely and with far fewer human labels.",
  "doi": "10.48550/arXiv.2212.08073",
  "oa_pdf": "http://arxiv.org/pdf/2212.08073",
  "s2_authors": [
   {
    "name": "Yuntao Bai",
    "id": "1486307451",
    "h_index": 22,
    "papers": 50
   },
   {
    "name": "Saurav Kadavath",
    "id": "148070327",
    "h_index": 16,
    "papers": 42
   },
   {
    "name": "Sandipan Kundu",
    "id": "2158813858",
    "h_index": 11,
    "papers": 44
   },
   {
    "name": "Amanda Askell",
    "id": "119609682",
    "h_index": 18,
    "papers": 29
   },
   {
    "name": "John Kernion",
    "id": "1583434563",
    "h_index": 17,
    "papers": 39
   },
   {
    "name": "Andy Jones",
    "id": "2149890773",
    "h_index": 13,
    "papers": 34
   },
   {
    "name": "A. Chen",
    "id": "2111074159",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Anna Goldie",
    "id": "46684455",
    "h_index": 20,
    "papers": 48
   },
   {
    "name": "Azalia Mirhoseini",
    "id": "1861312",
    "h_index": 36,
    "papers": 127
   },
   {
    "name": "C. McKinnon",
    "id": "2190108315",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "Carol Chen",
    "id": "2183729976",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Catherine Olsson",
    "id": "2061321863",
    "h_index": 20,
    "papers": 27
   },
   {
    "name": "Chris Olah",
    "id": "2287268442",
    "h_index": 16,
    "papers": 26
   },
   {
    "name": "Danny Hernandez",
    "id": "39182747",
    "h_index": 18,
    "papers": 33
   },
   {
    "name": "Dawn Drain",
    "id": "1943097969",
    "h_index": 19,
    "papers": 37
   },
   {
    "name": "Deep Ganguli",
    "id": "2081806483",
    "h_index": 24,
    "papers": 44
   },
   {
    "name": "Dustin Li",
    "id": "2108506462",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "Eli Tran-Johnson",
    "id": "2175781319",
    "h_index": 8,
    "papers": 14
   },
   {
    "name": "E. Perez",
    "id": "47635264",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Jamie Kerr",
    "id": "2067765208",
    "h_index": 7,
    "papers": 12
   },
   {
    "name": "J. Mueller",
    "id": "2190111475",
    "h_index": 7,
    "papers": 12
   },
   {
    "name": "Jeffrey Ladish",
    "id": "70988670",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "J. Landau",
    "id": "2068044809",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Kamal Ndousse",
    "id": "1978097132",
    "h_index": 17,
    "papers": 35
   },
   {
    "name": "Kamil\u0117 Luko\u0161i\u016bt\u0117",
    "id": "2105347564",
    "h_index": 8,
    "papers": 13
   },
   {
    "name": "Liane Lovitt",
    "id": "2154608229",
    "h_index": 14,
    "papers": 24
   },
   {
    "name": "M. Sellitto",
    "id": "2054578129",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "Nelson Elhage",
    "id": "2866708",
    "h_index": 16,
    "papers": 25
   },
   {
    "name": "Nicholas Schiefer",
    "id": "2833768",
    "h_index": 16,
    "papers": 28
   },
   {
    "name": "Noem'i Mercado",
    "id": "2190107517",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Nova Dassarma",
    "id": "2142833890",
    "h_index": 14,
    "papers": 27
   },
   {
    "name": "R. Lasenby",
    "id": "3112577",
    "h_index": 21,
    "papers": 40
   },
   {
    "name": "Robin Larson",
    "id": "48810415",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "Sam Ringer",
    "id": "1380664820",
    "h_index": 8,
    "papers": 14
   },
   {
    "name": "Scott Johnston",
    "id": "2154610174",
    "h_index": 12,
    "papers": 16
   },
   {
    "name": "S. Kravec",
    "id": "49604482",
    "h_index": 16,
    "papers": 28
   },
   {
    "name": "S. E. Showk",
    "id": "2154609053",
    "h_index": 6,
    "papers": 6
   },
   {
    "name": "Stanislav Fort",
    "id": "30176974",
    "h_index": 22,
    "papers": 37
   },
   {
    "name": "Tamera Lanham",
    "id": "46239941",
    "h_index": 9,
    "papers": 12
   },
   {
    "name": "Timothy Telleen-Lawton",
    "id": "1419532638",
    "h_index": 7,
    "papers": 14
   },
   {
    "name": "Tom Conerly",
    "id": "2154608209",
    "h_index": 8,
    "papers": 13
   },
   {
    "name": "T. Henighan",
    "id": "103143311",
    "h_index": 22,
    "papers": 86
   },
   {
    "name": "Tristan Hume",
    "id": "2162194147",
    "h_index": 13,
    "papers": 24
   },
   {
    "name": "Sam Bowman",
    "id": "1799822",
    "h_index": 24,
    "papers": 53
   },
   {
    "name": "Zac Hatfield-Dodds",
    "id": "1573482302",
    "h_index": 18,
    "papers": 36
   },
   {
    "name": "Benjamin Mann",
    "id": "2056658938",
    "h_index": 16,
    "papers": 29
   },
   {
    "name": "Dario Amodei",
    "id": "2698777",
    "h_index": 30,
    "papers": 61
   },
   {
    "name": "Nicholas Joseph",
    "id": "2117706920",
    "h_index": 18,
    "papers": 26
   },
   {
    "name": "Sam McCandlish",
    "id": "52238703",
    "h_index": 30,
    "papers": 45
   },
   {
    "name": "Tom B. Brown",
    "id": "31035595",
    "h_index": 25,
    "papers": 30
   },
   {
    "name": "Jared Kaplan",
    "id": "2053807409",
    "h_index": 33,
    "papers": 86
   }
  ],
  "comment": "",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2212.08073v1",
  "pdf_url": "https://arxiv.org/pdf/2212.08073v1",
  "html_url": "https://arxiv.org/html/2212.08073v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.0
 },
 {
  "id": "2212.08051",
  "slug": "objaverse-a-universe-of-annotated-3d-objects",
  "title": "Objaverse: A Universe of Annotated 3D Objects",
  "abstract": "Massive data corpora like WebText, Wikipedia, Conceptual Captions, WebImageText, and LAION have propelled recent dramatic progress in AI. Large neural models trained on such datasets produce impressive results and top many of today's benchmarks. A notable omission within this family of large-scale datasets is 3D data. Despite considerable interest and potential applications in 3D vision, datasets of high-fidelity 3D models continue to be mid-sized with limited diversity of object categories. Addressing this gap, we present Objaverse 1.0, a large dataset of objects with 800K+ (and growing) 3D models with descriptive captions, tags, and animations. Objaverse improves upon present day 3D repositories in terms of scale, number of categories, and in the visual diversity of instances within a category. We demonstrate the large potential of Objaverse via four diverse applications: training generative 3D models, improving tail category segmentation on the LVIS benchmark, training open-vocabulary object-navigation models for Embodied AI, and creating a new benchmark for robustness analysis of vision models. Objaverse can open new directions for research and enable new applications across the field of AI.",
  "published": "2022-12-15",
  "updated": "2022-12-15",
  "year": "2022",
  "authors": [
   "Matt Deitke",
   "Dustin Schwenk",
   "Jordi Salvador",
   "Luca Weihs",
   "Oscar Michel",
   "Eli VanderBilt",
   "Ludwig Schmidt",
   "Kiana Ehsani",
   "Aniruddha Kembhavi",
   "Ali Farhadi"
  ],
  "author_count": 10,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.GR",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 1805,
  "influential_citations": 293,
  "tldr": "The large potential of Objaverse is demonstrated via four diverse applications: training generative 3D models, improving tail category segmentation on the LVIS benchmark, training open-vocabulary object-navigation models for Embodied AI, and creating a new benchmark for robustness analysis of vision models.",
  "doi": "10.1109/CVPR52729.2023.01263",
  "oa_pdf": "https://arxiv.org/pdf/2212.08051",
  "s2_authors": [
   {
    "name": "Matt Deitke",
    "id": "1632916259",
    "h_index": 11,
    "papers": 14
   },
   {
    "name": "Dustin Schwenk",
    "id": "34846449",
    "h_index": 9,
    "papers": 14
   },
   {
    "name": "J. Salvador",
    "id": "145704057",
    "h_index": 15,
    "papers": 51
   },
   {
    "name": "Luca Weihs",
    "id": "20745881",
    "h_index": 31,
    "papers": 58
   },
   {
    "name": "Oscar Michel",
    "id": "2196005933",
    "h_index": 9,
    "papers": 13
   },
   {
    "name": "Eli VanderBilt",
    "id": "1632920625",
    "h_index": 13,
    "papers": 19
   },
   {
    "name": "Ludwig Schmidt",
    "id": "152772922",
    "h_index": 48,
    "papers": 84
   },
   {
    "name": "Kiana Ehsani",
    "id": "2883417",
    "h_index": 26,
    "papers": 42
   },
   {
    "name": "Aniruddha Kembhavi",
    "id": "2684226",
    "h_index": 49,
    "papers": 119
   },
   {
    "name": "Ali Farhadi",
    "id": "143787583",
    "h_index": 78,
    "papers": 208
   }
  ],
  "comment": "Website: objaverse.allenai.org",
  "topics": [
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2212.08051v1",
  "pdf_url": "https://arxiv.org/pdf/2212.08051v1",
  "html_url": "https://arxiv.org/html/2212.08051v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "2212.06817",
  "slug": "rt-1-robotics-transformer-for-real-world-control-at-scale",
  "title": "RT-1: Robotics Transformer for Real-World Control at Scale",
  "abstract": "By transferring knowledge from large, diverse, task-agnostic datasets, modern machine learning models can solve specific downstream tasks either zero-shot or with small task-specific datasets to a high level of performance. While this capability has been demonstrated in other fields such as computer vision, natural language processing or speech recognition, it remains to be shown in robotics, where the generalization capabilities of the models are particularly critical due to the difficulty of collecting real-world robotic data. We argue that one of the keys to the success of such general robotic models lies with open-ended task-agnostic training, combined with high-capacity architectures that can absorb all of the diverse, robotic data. In this paper, we present a model class, dubbed Robotics Transformer, that exhibits promising scalable model properties. We verify our conclusions in a study of different model classes and their ability to generalize as a function of the data size, model size, and data diversity based on a large-scale data collection on real robots performing real-world tasks. The project's website and videos can be found at robotics-transformer1.github.io",
  "published": "2022-12-13",
  "updated": "2023-08-11",
  "year": "2022",
  "authors": [
   "Anthony Brohan",
   "Noah Brown",
   "Justice Carbajal",
   "Yevgen Chebotar",
   "Joseph Dabis",
   "Chelsea Finn",
   "Keerthana Gopalakrishnan",
   "Karol Hausman",
   "Alex Herzog",
   "Jasmine Hsu",
   "Julian Ibarz",
   "Brian Ichter",
   "Alex Irpan",
   "Tomas Jackson",
   "Sally Jesmonth",
   "Nikhil J Joshi",
   "Ryan Julian",
   "Dmitry Kalashnikov",
   "Yuheng Kuang",
   "Isabel Leal",
   "Kuang-Huei Lee",
   "Sergey Levine",
   "Yao Lu",
   "Utsav Malla",
   "Deeksha Manjunath",
   "Igor Mordatch",
   "Ofir Nachum",
   "Carolina Parada",
   "Jodilyn Peralta",
   "Emily Perez",
   "Karl Pertsch",
   "Jornell Quiambao",
   "Kanishka Rao",
   "Michael Ryoo",
   "Grecia Salazar",
   "Pannag Sanketi",
   "Kevin Sayed",
   "Jaspiar Singh",
   "Sumedh Sontakke",
   "Austin Stone",
   "Clayton Tan",
   "Huong Tran",
   "Vincent Vanhoucke",
   "Steve Vega",
   "Quan Vuong",
   "Fei Xia",
   "Ted Xiao",
   "Peng Xu",
   "Sichun Xu",
   "Tianhe Yu",
   "Brianna Zitkovich"
  ],
  "author_count": 51,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CL",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "RSS",
  "venue_source": "semantic-scholar",
  "citations": 2724,
  "influential_citations": 204,
  "tldr": "This paper presents a model class, dubbed Robotics Transformer, that exhibits promising scalable model properties and verify the conclusions in a study of different model classes and their ability to generalize as a function of the data size, model size, and data diversity based on a large-scale data collection on real robots performing real-world tasks.",
  "doi": "10.48550/arXiv.2212.06817",
  "oa_pdf": "https://arxiv.org/pdf/2212.06817",
  "s2_authors": [
   {
    "name": "Anthony Brohan",
    "id": "118025075",
    "h_index": 12,
    "papers": 19
   },
   {
    "name": "Noah Brown",
    "id": "2161343011",
    "h_index": 10,
    "papers": 16
   },
   {
    "name": "Justice Carbajal",
    "id": "2196517336",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Yevgen Chebotar",
    "id": "2527420",
    "h_index": 33,
    "papers": 57
   },
   {
    "name": "Joseph Dabis",
    "id": "2196517328",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Chelsea Finn",
    "id": "46881670",
    "h_index": 94,
    "papers": 240
   },
   {
    "name": "K. Gopalakrishnan",
    "id": "2161342233",
    "h_index": 17,
    "papers": 25
   },
   {
    "name": "Karol Hausman",
    "id": "1944801",
    "h_index": 47,
    "papers": 122
   },
   {
    "name": "Alexander Herzog",
    "id": "1505793452",
    "h_index": 22,
    "papers": 35
   },
   {
    "name": "Jasmine Hsu",
    "id": "2726592",
    "h_index": 15,
    "papers": 20
   },
   {
    "name": "Julian Ibarz",
    "id": "46920727",
    "h_index": 22,
    "papers": 35
   },
   {
    "name": "Brian Ichter",
    "id": "2704814",
    "h_index": 37,
    "papers": 60
   },
   {
    "name": "A. Irpan",
    "id": "17818078",
    "h_index": 22,
    "papers": 32
   },
   {
    "name": "Tomas Jackson",
    "id": "2175779811",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Sally Jesmonth",
    "id": "2161341920",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Nikhil J. Joshi",
    "id": "2052368480",
    "h_index": 11,
    "papers": 14
   },
   {
    "name": "Ryan C. Julian",
    "id": "144885996",
    "h_index": 19,
    "papers": 34
   },
   {
    "name": "Dmitry Kalashnikov",
    "id": "48313860",
    "h_index": 21,
    "papers": 34
   },
   {
    "name": "Yuheng Kuang",
    "id": "2161342687",
    "h_index": 9,
    "papers": 14
   },
   {
    "name": "Isabel Leal",
    "id": "2057988112",
    "h_index": 12,
    "papers": 15
   },
   {
    "name": "Kuang-Huei Lee",
    "id": "2145145412",
    "h_index": 15,
    "papers": 19
   },
   {
    "name": "S. Levine",
    "id": "1736651",
    "h_index": 167,
    "papers": 551
   },
   {
    "name": "Yao Lu",
    "id": "2161346119",
    "h_index": 16,
    "papers": 22
   },
   {
    "name": "U. Malla",
    "id": "51225879",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "D. Manjunath",
    "id": "2064295888",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Igor Mordatch",
    "id": "2080746",
    "h_index": 34,
    "papers": 49
   },
   {
    "name": "Ofir Nachum",
    "id": "7624658",
    "h_index": 47,
    "papers": 92
   },
   {
    "name": "Carolina Parada",
    "id": "2057314286",
    "h_index": 23,
    "papers": 27
   },
   {
    "name": "Jodilyn Peralta",
    "id": "2195951533",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Emily Perez",
    "id": "2125121323",
    "h_index": 1,
    "papers": 3
   },
   {
    "name": "Karl Pertsch",
    "id": "31719101",
    "h_index": 32,
    "papers": 56
   },
   {
    "name": "Jornell Quiambao",
    "id": "2161342191",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Kanishka Rao",
    "id": "2251957",
    "h_index": 31,
    "papers": 42
   },
   {
    "name": "M. Ryoo",
    "id": "1766489",
    "h_index": 50,
    "papers": 164
   },
   {
    "name": "Grecia Salazar",
    "id": "2196524735",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Pannag R. Sanketi",
    "id": "2840758",
    "h_index": 22,
    "papers": 44
   },
   {
    "name": "Kevin Sayed",
    "id": "2196510642",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Jaspiar Singh",
    "id": "2196040785",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "S. Sontakke",
    "id": "34365421",
    "h_index": 10,
    "papers": 20
   },
   {
    "name": "Austin Stone",
    "id": "2056868723",
    "h_index": 14,
    "papers": 19
   },
   {
    "name": "Clayton Tan",
    "id": "2161386250",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Huong Tran",
    "id": "2195355151",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Vincent Vanhoucke",
    "id": "2657155",
    "h_index": 34,
    "papers": 60
   },
   {
    "name": "S. Vega",
    "id": "2195627101",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Q. Vuong",
    "id": "144579461",
    "h_index": 23,
    "papers": 40
   },
   {
    "name": "F. Xia",
    "id": "144956443",
    "h_index": 25,
    "papers": 31
   },
   {
    "name": "Ted Xiao",
    "id": "9961095",
    "h_index": 33,
    "papers": 45
   },
   {
    "name": "Peng Xu",
    "id": "2153917744",
    "h_index": 17,
    "papers": 22
   },
   {
    "name": "Sichun Xu",
    "id": "3068504",
    "h_index": 11,
    "papers": 16
   },
   {
    "name": "Tianhe Yu",
    "id": "10909315",
    "h_index": 31,
    "papers": 48
   },
   {
    "name": "Brianna Zitkovich",
    "id": "2196524598",
    "h_index": 4,
    "papers": 5
   }
  ],
  "comment": "See website at robotics-transformer1.github.io",
  "topics": [
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2212.06817v2",
  "pdf_url": "https://arxiv.org/pdf/2212.06817v2",
  "html_url": "https://arxiv.org/html/2212.06817v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "2212.05711",
  "slug": "cacti-a-framework-for-scalable-multi-task-multi-scene-visual-imitation",
  "title": "CACTI: A Framework for Scalable Multi-Task Multi-Scene Visual Imitation Learning",
  "abstract": "Large-scale training have propelled significant progress in various sub-fields of AI such as computer vision and natural language processing. However, building robot learning systems at a comparable scale remains challenging. To develop robots that can perform a wide range of skills and adapt to new scenarios, efficient methods for collecting vast and diverse amounts of data on physical robot systems are required, as well as the capability to train high-capacity policies using such datasets. In this work, we propose a framework for scaling robot learning, with specific focus on multi-task and multi-scene manipulation in kitchen environments, both in simulation and in the real world. Our proposed framework, CACTI, comprises four stages that separately handle: data collection, data augmentation, visual representation learning, and imitation policy training, to enable scalability in robot learning . We make use of state-of-the-art generative models as part of the data augmentation stage, and use pre-trained out-of-domain visual representations to improve training efficiency. Experimental results demonstrate the effectiveness of our approach. On a real robot setup, CACTI enables efficient training of a single policy that can perform 10 manipulation tasks involving kitchen objects, and is robust to varying layouts of distractors. In a simulated kitchen environment, CACTI trains a single policy to perform 18 semantic tasks across 100 layout variations for each individual task. We will release the simulation task benchmark and augmented datasets in both real and simulated environments to facilitate future research.",
  "published": "2022-12-12",
  "updated": "2023-02-16",
  "year": "2022",
  "authors": [
   "Zhao Mandi",
   "Homanga Bharadhwaj",
   "Vincent Moens",
   "Shuran Song",
   "Aravind Rajeswaran",
   "Vikash Kumar"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 105,
  "influential_citations": 3,
  "tldr": "This work proposes a framework for scaling robot learning, CACTI, which comprises four stages that separately handle: data collection, data augmentation, visual representation learning, and imitation policy training, to enable scalability in robot learning.",
  "doi": "10.48550/arXiv.2212.05711",
  "oa_pdf": "http://arxiv.org/pdf/2212.05711",
  "s2_authors": [
   {
    "name": "Zhao Mandi",
    "id": "2126966292",
    "h_index": 6,
    "papers": 6
   },
   {
    "name": "Homanga Bharadhwaj",
    "id": "51113848",
    "h_index": 23,
    "papers": 59
   },
   {
    "name": "V. Moens",
    "id": "12887111",
    "h_index": 7,
    "papers": 19
   },
   {
    "name": "Shuran Song",
    "id": "2110601402",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "A. Rajeswaran",
    "id": "19275599",
    "h_index": 34,
    "papers": 58
   },
   {
    "name": "Vikash Kumar",
    "id": "2109446216",
    "h_index": 38,
    "papers": 76
   }
  ],
  "comment": "",
  "topics": [
   "imitation-diffusion",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2212.05711v2",
  "pdf_url": "https://arxiv.org/pdf/2212.05711v2",
  "html_url": "https://arxiv.org/html/2212.05711v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.03
 },
 {
  "id": "2212.05199",
  "slug": "magvit-masked-generative-video-transformer",
  "title": "MAGVIT: Masked Generative Video Transformer",
  "abstract": "We introduce the MAsked Generative VIdeo Transformer, MAGVIT, to tackle various video synthesis tasks with a single model. We introduce a 3D tokenizer to quantize a video into spatial-temporal visual tokens and propose an embedding method for masked video token modeling to facilitate multi-task learning. We conduct extensive experiments to demonstrate the quality, efficiency, and flexibility of MAGVIT. Our experiments show that (i) MAGVIT performs favorably against state-of-the-art approaches and establishes the best-published FVD on three video generation benchmarks, including the challenging Kinetics-600. (ii) MAGVIT outperforms existing methods in inference time by two orders of magnitude against diffusion models and by 60x against autoregressive models. (iii) A single MAGVIT model supports ten diverse generation tasks and generalizes across videos from different visual domains. The source code and trained models will be released to the public at https://magvit.cs.cmu.edu.",
  "published": "2022-12-10",
  "updated": "2023-04-05",
  "year": "2022",
  "authors": [
   "Lijun Yu",
   "Yong Cheng",
   "Kihyuk Sohn",
   "Jos\u00e9 Lezama",
   "Han Zhang",
   "Huiwen Chang",
   "Alexander G. Hauptmann",
   "Ming-Hsuan Yang",
   "Yuan Hao",
   "Irfan Essa",
   "Lu Jiang"
  ],
  "author_count": 11,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 402,
  "influential_citations": 35,
  "tldr": "A 3D tokenizer to quantize a video into spatial-temporal visual tokens and propose an embedding method for masked video token modeling to facilitate multi-task learning are introduced.",
  "doi": "10.1109/CVPR52729.2023.01008",
  "oa_pdf": "https://arxiv.org/pdf/2212.05199",
  "s2_authors": [
   {
    "name": "Lijun Yu",
    "id": "8547960",
    "h_index": 14,
    "papers": 24
   },
   {
    "name": "Yong Cheng",
    "id": "2198464317",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Kihyuk Sohn",
    "id": "1729571",
    "h_index": 40,
    "papers": 85
   },
   {
    "name": "Jos\u00e9 Lezama",
    "id": "143923528",
    "h_index": 15,
    "papers": 33
   },
   {
    "name": "Han Zhang",
    "id": "2146204239",
    "h_index": 8,
    "papers": 10
   },
   {
    "name": "Huiwen Chang",
    "id": "2914394",
    "h_index": 28,
    "papers": 38
   },
   {
    "name": "A. Hauptmann",
    "id": "145788702",
    "h_index": 30,
    "papers": 58
   },
   {
    "name": "Ming-Hsuan Yang",
    "id": "37144787",
    "h_index": 72,
    "papers": 223
   },
   {
    "name": "Yuan Hao",
    "id": "2153968179",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Irfan Essa",
    "id": "145955800",
    "h_index": 22,
    "papers": 56
   },
   {
    "name": "Lu Jiang",
    "id": "39978626",
    "h_index": 48,
    "papers": 78
   }
  ],
  "comment": "CVPR 2023 highlight",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2212.05199v2",
  "pdf_url": "https://arxiv.org/pdf/2212.05199v2",
  "html_url": "https://arxiv.org/html/2212.05199v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.11
 },
 {
  "id": "2212.05108",
  "slug": "visuotactile-affordances-for-cloth-manipulation-with-local-control",
  "title": "Visuotactile Affordances for Cloth Manipulation with Local Control",
  "abstract": "Cloth in the real world is often crumpled, self-occluded, or folded in on itself such that key regions, such as corners, are not directly graspable, making manipulation difficult. We propose a system that leverages visual and tactile perception to unfold the cloth via grasping and sliding on edges. By doing so, the robot is able to grasp two adjacent corners, enabling subsequent manipulation tasks like folding or hanging. As components of this system, we develop tactile perception networks that classify whether an edge is grasped and estimate the pose of the edge. We use the edge classification network to supervise a visuotactile edge grasp affordance network that can grasp edges with a 90% success rate. Once an edge is grasped, we demonstrate that the robot can slide along the cloth to the adjacent corner using tactile pose estimation/control in real time. See http://nehasunil.com/visuotactile/visuotactile.html for videos.",
  "published": "2022-12-09",
  "updated": "2022-12-09",
  "year": "2022",
  "authors": [
   "Neha Sunil",
   "Shaoxiong Wang",
   "Yu She",
   "Edward Adelson",
   "Alberto Rodriguez"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.LG",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 55,
  "influential_citations": 2,
  "tldr": "This work proposes a system that leverages visual and tactile perception to unfold the cloth via grasping and sliding on edges, and demonstrates that the robot can slide along the cloth to the adjacent corner using tactile pose estimation/control in real time.",
  "doi": "10.48550/arXiv.2212.05108",
  "oa_pdf": "http://arxiv.org/pdf/2212.05108",
  "s2_authors": [
   {
    "name": "N. Sunil",
    "id": "1389062449",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Shaoxiong Wang",
    "id": "7488549",
    "h_index": 14,
    "papers": 16
   },
   {
    "name": "Y. She",
    "id": "2392034",
    "h_index": 17,
    "papers": 43
   },
   {
    "name": "E. Adelson",
    "id": "145358192",
    "h_index": 87,
    "papers": 272
   },
   {
    "name": "Alberto Rodriguez",
    "id": "152532021",
    "h_index": 49,
    "papers": 113
   }
  ],
  "comment": "Accepted at CoRL 2022. Project website: http://nehasunil.com/visuotactile/visuotactile.html",
  "topics": [
   "dexterous-manipulation",
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2212.05108v1",
  "pdf_url": "https://arxiv.org/pdf/2212.05108v1",
  "html_url": "https://arxiv.org/html/2212.05108v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.25
 },
 {
  "id": "2212.04498",
  "slug": "videodex-learning-dexterity-from-internet-videos",
  "title": "VideoDex: Learning Dexterity from Internet Videos",
  "abstract": "To build general robotic agents that can operate in many environments, it is often imperative for the robot to collect experience in the real world. However, this is often not feasible due to safety, time, and hardware restrictions. We thus propose leveraging the next best thing as real-world experience: internet videos of humans using their hands. Visual priors, such as visual features, are often learned from videos, but we believe that more information from videos can be utilized as a stronger prior. We build a learning algorithm, VideoDex, that leverages visual, action, and physical priors from human video datasets to guide robot behavior. These actions and physical priors in the neural network dictate the typical human behavior for a particular robot task. We test our approach on a robot arm and dexterous hand-based system and show strong results on various manipulation tasks, outperforming various state-of-the-art methods. Videos at https://video-dex.github.io",
  "published": "2022-12-08",
  "updated": "2022-12-08",
  "year": "2022",
  "authors": [
   "Kenneth Shaw",
   "Shikhar Bahl",
   "Deepak Pathak"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 167,
  "influential_citations": 3,
  "tldr": "A learning algorithm that leverages visual, action, and physical priors from human video datasets to guide robot behavior, and shows strong results on various manipulation tasks, outperforming various state-of-the-art methods.",
  "doi": "10.48550/arXiv.2212.04498",
  "oa_pdf": "https://arxiv.org/pdf/2212.04498",
  "s2_authors": [
   {
    "name": "Kenneth Shaw",
    "id": "2072761493",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Shikhar Bahl",
    "id": "8527563",
    "h_index": 20,
    "papers": 38
   },
   {
    "name": "Deepak Pathak",
    "id": "38236002",
    "h_index": 34,
    "papers": 78
   }
  ],
  "comment": "Accepted at CoRL 2022. Website at https://video-dex.github.io",
  "topics": [
   "dexterous-manipulation",
   "egocentric-data",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2212.04498v1",
  "pdf_url": "https://arxiv.org/pdf/2212.04498v1",
  "html_url": "https://arxiv.org/html/2212.04498v1",
  "code_url": "https://video-dex.github.io",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.73
 },
 {
  "id": "2212.04356",
  "slug": "robust-speech-recognition-via-large-scale-weak-supervision",
  "title": "Robust Speech Recognition via Large-Scale Weak Supervision",
  "abstract": "We study the capabilities of speech processing systems trained simply to predict large amounts of transcripts of audio on the internet. When scaled to 680,000 hours of multilingual and multitask supervision, the resulting models generalize well to standard benchmarks and are often competitive with prior fully supervised results but in a zero-shot transfer setting without the need for any fine-tuning. When compared to humans, the models approach their accuracy and robustness. We are releasing models and inference code to serve as a foundation for further work on robust speech processing.",
  "published": "2022-12-06",
  "updated": "2022-12-06",
  "year": "2022",
  "authors": [
   "Alec Radford",
   "Jong Wook Kim",
   "Tao Xu",
   "Greg Brockman",
   "Christine McLeavey",
   "Ilya Sutskever"
  ],
  "author_count": 6,
  "categories": [
   "eess.AS",
   "cs.CL",
   "cs.LG",
   "cs.SD"
  ],
  "primary_category": "eess.AS",
  "venue": "ICML",
  "venue_source": "semantic-scholar",
  "citations": 8223,
  "influential_citations": 1137,
  "tldr": "When scaled to 680,000 hours of multilingual and multitask supervision, the resulting models generalize well to standard benchmarks and are often competitive with prior fully supervised results but in a zero-shot transfer setting without the need for any fine-tuning.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Alec Radford",
    "id": "38909097",
    "h_index": 33,
    "papers": 153
   },
   {
    "name": "Jong Wook Kim",
    "id": "2110935237",
    "h_index": 10,
    "papers": 12
   },
   {
    "name": "Tao Xu",
    "id": "2118717067",
    "h_index": 7,
    "papers": 7
   },
   {
    "name": "Greg Brockman",
    "id": "2065151121",
    "h_index": 11,
    "papers": 39
   },
   {
    "name": "Christine McLeavey",
    "id": "3028785",
    "h_index": 7,
    "papers": 17
   },
   {
    "name": "I. Sutskever",
    "id": "1701686",
    "h_index": 75,
    "papers": 166
   }
  ],
  "comment": "",
  "topics": [
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2212.04356v1",
  "pdf_url": "https://arxiv.org/pdf/2212.04356v1",
  "html_url": "https://arxiv.org/html/2212.04356v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "2212.00794",
  "slug": "scaling-language-image-pre-training-via-masking",
  "title": "Scaling Language-Image Pre-training via Masking",
  "abstract": "We present Fast Language-Image Pre-training (FLIP), a simple and more efficient method for training CLIP. Our method randomly masks out and removes a large portion of image patches during training. Masking allows us to learn from more image-text pairs given the same wall-clock time and contrast more samples per iteration with similar memory footprint. It leads to a favorable trade-off between accuracy and training time. In our experiments on 400 million image-text pairs, FLIP improves both accuracy and speed over the no-masking baseline. On a large diversity of downstream tasks, FLIP dominantly outperforms the CLIP counterparts trained on the same data. Facilitated by the speedup, we explore the scaling behavior of increasing the model size, data size, or training length, and report encouraging results and comparisons. We hope that our work will foster future research on scaling vision-language learning.",
  "published": "2022-12-01",
  "updated": "2023-03-30",
  "year": "2022",
  "authors": [
   "Yanghao Li",
   "Haoqi Fan",
   "Ronghang Hu",
   "Christoph Feichtenhofer",
   "Kaiming He"
  ],
  "author_count": 5,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 436,
  "influential_citations": 46,
  "tldr": "",
  "doi": "10.1109/CVPR52729.2023.02240",
  "oa_pdf": "https://arxiv.org/pdf/2212.00794",
  "s2_authors": [
   {
    "name": "Yanghao Li",
    "id": "2359205979",
    "h_index": 12,
    "papers": 14
   },
   {
    "name": "Haoqi Fan",
    "id": "146884473",
    "h_index": 27,
    "papers": 36
   },
   {
    "name": "Ronghang Hu",
    "id": "2874347",
    "h_index": 23,
    "papers": 33
   },
   {
    "name": "Christoph Feichtenhofer",
    "id": "2322150",
    "h_index": 36,
    "papers": 57
   },
   {
    "name": "Kaiming He",
    "id": "2058350112",
    "h_index": 9,
    "papers": 11
   }
  ],
  "comment": "Tech report; arXiv v2: update scaling results and add code repo",
  "topics": [
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2212.00794v2",
  "pdf_url": "https://arxiv.org/pdf/2212.00794v2",
  "html_url": "https://arxiv.org/html/2212.00794v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.14
 },
 {
  "id": "2211.03726",
  "slug": "tap-vid-a-benchmark-for-tracking-any-point-in-a-video",
  "title": "TAP-Vid: A Benchmark for Tracking Any Point in a Video",
  "abstract": "Generic motion understanding from video involves not only tracking objects, but also perceiving how their surfaces deform and move. This information is useful to make inferences about 3D shape, physical properties and object interactions. While the problem of tracking arbitrary physical points on surfaces over longer video clips has received some attention, no dataset or benchmark for evaluation existed, until now. In this paper, we first formalize the problem, naming it tracking any point (TAP). We introduce a companion benchmark, TAP-Vid, which is composed of both real-world videos with accurate human annotations of point tracks, and synthetic videos with perfect ground-truth point tracks. Central to the construction of our benchmark is a novel semi-automatic crowdsourced pipeline which uses optical flow estimates to compensate for easier, short-term motion like camera shake, allowing annotators to focus on harder sections of video. We validate our pipeline on synthetic data and propose a simple end-to-end point tracking model TAP-Net, showing that it outperforms all prior methods on our benchmark when trained on synthetic data.",
  "published": "2022-11-07",
  "updated": "2023-03-31",
  "year": "2022",
  "authors": [
   "Carl Doersch",
   "Ankush Gupta",
   "Larisa Markeeva",
   "Adri\u00e0 Recasens",
   "Lucas Smaira",
   "Yusuf Aytar",
   "Jo\u00e3o Carreira",
   "Andrew Zisserman",
   "Yi Yang"
  ],
  "author_count": 9,
  "categories": [
   "cs.CV",
   "stat.ML"
  ],
  "primary_category": "cs.CV",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 319,
  "influential_citations": 77,
  "tldr": "A novel semi-automatic crowdsourced pipeline which uses optical flow estimates to compensate for easier, short-term motion like camera shake, allowing annotators to focus on harder sections of video, and proposes a simple end-to-end point tracking model TAP-Net, which outperforms all prior methods on the authors' benchmark when trained on synthetic data.",
  "doi": "10.48550/arXiv.2211.03726",
  "oa_pdf": "http://arxiv.org/pdf/2211.03726",
  "s2_authors": [
   {
    "name": "Carl Doersch",
    "id": "2786693",
    "h_index": 33,
    "papers": 57
   },
   {
    "name": "Ankush Gupta",
    "id": "2110759501",
    "h_index": 19,
    "papers": 37
   },
   {
    "name": "L. Markeeva",
    "id": "72361999",
    "h_index": 11,
    "papers": 26
   },
   {
    "name": "Adri\u00e0 Recasens",
    "id": "39257069",
    "h_index": 20,
    "papers": 33
   },
   {
    "name": "Lucas Smaira",
    "id": "1466466597",
    "h_index": 10,
    "papers": 11
   },
   {
    "name": "Y. Aytar",
    "id": "3152281",
    "h_index": 30,
    "papers": 57
   },
   {
    "name": "Jo\u00e3o Carreira",
    "id": "35681810",
    "h_index": 38,
    "papers": 67
   },
   {
    "name": "Andrew Zisserman",
    "id": "1688869",
    "h_index": 191,
    "papers": 839
   },
   {
    "name": "Yezhou Yang",
    "id": "7607499",
    "h_index": 71,
    "papers": 628
   }
  ],
  "comment": "Published in NeurIPS Datasets and Benchmarks track, 2022",
  "topics": [
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2211.03726v2",
  "pdf_url": "https://arxiv.org/pdf/2211.03726v2",
  "html_url": "https://arxiv.org/html/2211.03726v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.01
 },
 {
  "id": "2210.13702",
  "slug": "dextreme-transfer-of-agile-in-hand-manipulation-from-simulation-to-rea",
  "title": "DeXtreme: Transfer of Agile In-hand Manipulation from Simulation to Reality",
  "abstract": "Recent work has demonstrated the ability of deep reinforcement learning (RL) algorithms to learn complex robotic behaviours in simulation, including in the domain of multi-fingered manipulation. However, such models can be challenging to transfer to the real world due to the gap between simulation and reality. In this paper, we present our techniques to train a) a policy that can perform robust dexterous manipulation on an anthropomorphic robot hand and b) a robust pose estimator suitable for providing reliable real-time information on the state of the object being manipulated. Our policies are trained to adapt to a wide range of conditions in simulation. Consequently, our vision-based policies significantly outperform the best vision policies in the literature on the same reorientation task and are competitive with policies that are given privileged state information via motion capture systems. Our work reaffirms the possibilities of sim-to-real transfer for dexterous manipulation in diverse kinds of hardware and simulator setups, and in our case, with the Allegro Hand and Isaac Gym GPU-based simulation. Furthermore, it opens up possibilities for researchers to achieve such results with commonly-available, affordable robot hands and cameras. Videos of the resulting policy and supplementary information, including experiments and demos, can be found at https://dextreme.org/",
  "published": "2022-10-25",
  "updated": "2024-01-02",
  "year": "2022",
  "authors": [
   "Ankur Handa",
   "Arthur Allshire",
   "Viktor Makoviychuk",
   "Aleksei Petrenko",
   "Ritvik Singh",
   "Jingzhou Liu",
   "Denys Makoviichuk",
   "Karl Van Wyk",
   "Alexander Zhurkevich",
   "Balakumar Sundaralingam",
   "Yashraj Narang",
   "Jean-Francois Lafleche",
   "Dieter Fox",
   "Gavriel State"
  ],
  "author_count": 14,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA",
  "venue_source": "semantic-scholar",
  "citations": 245,
  "influential_citations": 14,
  "tldr": "The techniques to train a policy that can perform robust dexterous manipulation on an anthropomorphic robot hand and a robust pose estimator suitable for providing reliable real-time information on the state of the object being manipulated are presented.",
  "doi": "10.1109/ICRA48891.2023.10160216",
  "oa_pdf": "https://arxiv.org/pdf/2210.13702",
  "s2_authors": [
   {
    "name": "Ankur Handa",
    "id": "34653454",
    "h_index": 33,
    "papers": 55
   },
   {
    "name": "Arthur Allshire",
    "id": "2061149217",
    "h_index": 9,
    "papers": 12
   },
   {
    "name": "Viktor Makoviychuk",
    "id": "79875630",
    "h_index": 18,
    "papers": 21
   },
   {
    "name": "Aleksei Petrenko",
    "id": "90219258",
    "h_index": 11,
    "papers": 12
   },
   {
    "name": "Ritvik Singh",
    "id": "81452558",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Jason Jingzhou Liu",
    "id": "2108367188",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Denys Makoviichuk",
    "id": "2026863982",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Karl Van Wyk",
    "id": "2423933",
    "h_index": 20,
    "papers": 54
   },
   {
    "name": "Alexander Zhurkevich",
    "id": "2382152713",
    "h_index": 1,
    "papers": 4
   },
   {
    "name": "Balakumar Sundaralingam",
    "id": "32469503",
    "h_index": 26,
    "papers": 47
   },
   {
    "name": "Yashraj S. Narang",
    "id": "5046361",
    "h_index": 20,
    "papers": 36
   },
   {
    "name": "Jean-Francois Lafleche",
    "id": "52529783",
    "h_index": 8,
    "papers": 10
   },
   {
    "name": "D. Fox",
    "id": "145197953",
    "h_index": 133,
    "papers": 428
   },
   {
    "name": "Gavriel State",
    "id": "82261827",
    "h_index": 10,
    "papers": 14
   }
  ],
  "comment": "28 pages. A smaller version of this paper is accepted to ICRA 2023",
  "topics": [
   "dexterous-manipulation",
   "egocentric-data",
   "sim2real",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2210.13702v2",
  "pdf_url": "https://arxiv.org/pdf/2210.13702v2",
  "html_url": "https://arxiv.org/html/2210.13702v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.89
 },
 {
  "id": "2210.13382",
  "slug": "emergent-world-representations-exploring-a-sequence-model-trained-on-a",
  "title": "Emergent World Representations: Exploring a Sequence Model Trained on a Synthetic Task",
  "abstract": "Language models show a surprising range of capabilities, but the source of their apparent competence is unclear. Do these networks just memorize a collection of surface statistics, or do they rely on internal representations of the process that generates the sequences they see? We investigate this question by applying a variant of the GPT model to the task of predicting legal moves in a simple board game, Othello. Although the network has no a priori knowledge of the game or its rules, we uncover evidence of an emergent nonlinear internal representation of the board state. Interventional experiments indicate this representation can be used to control the output of the network and create \"latent saliency maps\" that can help explain predictions in human terms.",
  "published": "2022-10-24",
  "updated": "2024-06-26",
  "year": "2022",
  "authors": [
   "Kenneth Li",
   "Aspen K. Hopkins",
   "David Bau",
   "Fernanda Vi\u00e9gas",
   "Hanspeter Pfister",
   "Martin Wattenberg"
  ],
  "author_count": 6,
  "categories": [
   "cs.LG",
   "cs.AI",
   "cs.CL"
  ],
  "primary_category": "cs.LG",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 528,
  "influential_citations": 33,
  "tldr": "Although the network has no a priori knowledge of the game or its rules, it is uncovered evidence of an emergent nonlinear internal representation of the board state that can help explain predictions in human terms.",
  "doi": "10.48550/arXiv.2210.13382",
  "oa_pdf": "http://arxiv.org/pdf/2210.13382",
  "s2_authors": [
   {
    "name": "Kenneth Li",
    "id": "2149140708",
    "h_index": 7,
    "papers": 9
   },
   {
    "name": "Aspen K. Hopkins",
    "id": "1934459700",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "David Bau",
    "id": "144159726",
    "h_index": 37,
    "papers": 85
   },
   {
    "name": "Fernanda Vi'egas",
    "id": "2064951472",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "H. Pfister",
    "id": "143758236",
    "h_index": 81,
    "papers": 378
   },
   {
    "name": "M. Wattenberg",
    "id": "145233583",
    "h_index": 68,
    "papers": 189
   }
  ],
  "comment": "ICLR 2023 oral (notable-top-5%): https://openreview.net/forum?id=DeG07_TcZvT ; code: https://github.com/likenneth/othello_world",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2210.13382v5",
  "pdf_url": "https://arxiv.org/pdf/2210.13382v5",
  "html_url": "https://arxiv.org/html/2210.13382v5",
  "code_url": "https://github.com/likenneth/othello_world",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.22
 },
 {
  "id": "2210.12649",
  "slug": "anticipative-feature-fusion-transformer-for-multi-modal-action-anticip",
  "title": "Anticipative Feature Fusion Transformer for Multi-Modal Action Anticipation",
  "abstract": "Although human action anticipation is a task which is inherently multi-modal, state-of-the-art methods on well known action anticipation datasets leverage this data by applying ensemble methods and averaging scores of unimodal anticipation networks. In this work we introduce transformer based modality fusion techniques, which unify multi-modal data at an early stage. Our Anticipative Feature Fusion Transformer (AFFT) proves to be superior to popular score fusion approaches and presents state-of-the-art results outperforming previous methods on EpicKitchens-100 and EGTEA Gaze+. Our model is easily extensible and allows for adding new modalities without architectural changes. Consequently, we extracted audio features on EpicKitchens-100 which we add to the set of commonly used features in the community.",
  "published": "2022-10-23",
  "updated": "2022-10-23",
  "year": "2022",
  "authors": [
   "Zeyun Zhong",
   "David Schneider",
   "Michael Voit",
   "Rainer Stiefelhagen",
   "J\u00fcrgen Beyerer"
  ],
  "author_count": 5,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "WACV 2023",
  "venue_source": "arxiv-comment",
  "citations": 71,
  "influential_citations": 11,
  "tldr": "The Anticipative Feature Fusion Transformer (AFFT) proves to be superior to popular score fusion approaches and presents state-of-the-art results outperforming previous methods on EpicKitchens-100 and EGTEA Gaze+.",
  "doi": "10.1109/WACV56688.2023.00601",
  "oa_pdf": "https://arxiv.org/pdf/2210.12649",
  "s2_authors": [
   {
    "name": "Zeyun Zhong",
    "id": "2188735605",
    "h_index": 6,
    "papers": 25
   },
   {
    "name": "David Schneider",
    "id": "2118894575",
    "h_index": 7,
    "papers": 29
   },
   {
    "name": "M. Voit",
    "id": "1430776903",
    "h_index": 15,
    "papers": 71
   },
   {
    "name": "R. Stiefelhagen",
    "id": "1742325",
    "h_index": 72,
    "papers": 474
   },
   {
    "name": "J. Beyerer",
    "id": "2154675890",
    "h_index": 10,
    "papers": 41
   }
  ],
  "comment": "Accepted to WACV 2023",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2210.12649v1",
  "pdf_url": "https://arxiv.org/pdf/2210.12649v1",
  "html_url": "https://arxiv.org/html/2210.12649v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.36
 },
 {
  "id": "2210.11339",
  "slug": "viola-imitation-learning-for-vision-based-manipulation-with-object-pro",
  "title": "VIOLA: Imitation Learning for Vision-Based Manipulation with Object Proposal Priors",
  "abstract": "We introduce VIOLA, an object-centric imitation learning approach to learning closed-loop visuomotor policies for robot manipulation. Our approach constructs object-centric representations based on general object proposals from a pre-trained vision model. VIOLA uses a transformer-based policy to reason over these representations and attend to the task-relevant visual factors for action prediction. Such object-based structural priors improve deep imitation learning algorithm's robustness against object variations and environmental perturbations. We quantitatively evaluate VIOLA in simulation and on real robots. VIOLA outperforms the state-of-the-art imitation learning methods by $45.8\\%$ in success rate. It has also been deployed successfully on a physical robot to solve challenging long-horizon tasks, such as dining table arrangement and coffee making. More videos and model details can be found in supplementary material and the project website: https://ut-austin-rpl.github.io/VIOLA .",
  "published": "2022-10-20",
  "updated": "2023-03-08",
  "year": "2022",
  "authors": [
   "Yifeng Zhu",
   "Abhishek Joshi",
   "Peter Stone",
   "Yuke Zhu"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 233,
  "influential_citations": 18,
  "tldr": "VIOLA, an object-centric imitation learning approach to learning closed-loop visuomotor policies for robot manipulation, outperforms the state-of-the-art imitation learning methods by $45.8\\% in success rate and has been deployed successfully on a physical robot to solve challenging long-horizon tasks.",
  "doi": "10.48550/arXiv.2210.11339",
  "oa_pdf": "http://arxiv.org/pdf/2210.11339",
  "s2_authors": [
   {
    "name": "Yifeng Zhu",
    "id": "1557295600",
    "h_index": 14,
    "papers": 23
   },
   {
    "name": "Abhishek Joshi",
    "id": "2066732439",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "P. Stone",
    "id": "144848112",
    "h_index": 95,
    "papers": 704
   },
   {
    "name": "Yuke Zhu",
    "id": "2117748",
    "h_index": 57,
    "papers": 130
   }
  ],
  "comment": "Published at the 6th Conference on Robot Learning",
  "topics": [
   "imitation-diffusion",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2210.11339v2",
  "pdf_url": "https://arxiv.org/pdf/2210.11339v2",
  "html_url": "https://arxiv.org/html/2210.11339v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.37
 },
 {
  "id": "2210.10044",
  "slug": "deep-whole-body-control-learning-a-unified-policy-for-manipulation-and",
  "title": "Deep Whole-Body Control: Learning a Unified Policy for Manipulation and Locomotion",
  "abstract": "An attached arm can significantly increase the applicability of legged robots to several mobile manipulation tasks that are not possible for the wheeled or tracked counterparts. The standard hierarchical control pipeline for such legged manipulators is to decouple the controller into that of manipulation and locomotion. However, this is ineffective. It requires immense engineering to support coordination between the arm and legs, and error can propagate across modules causing non-smooth unnatural motions. It is also biological implausible given evidence for strong motor synergies across limbs. In this work, we propose to learn a unified policy for whole-body control of a legged manipulator using reinforcement learning. We propose Regularized Online Adaptation to bridge the Sim2Real gap for high-DoF control, and Advantage Mixing exploiting the causal dependency in the action space to overcome local minima during training the whole-body system. We also present a simple design for a low-cost legged manipulator, and find that our unified policy can demonstrate dynamic and agile behaviors across several task setups. Videos are at https://maniploco.github.io",
  "published": "2022-10-18",
  "updated": "2022-10-18",
  "year": "2022",
  "authors": [
   "Zipeng Fu",
   "Xuxin Cheng",
   "Deepak Pathak"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 282,
  "influential_citations": 22,
  "tldr": "This work proposes to learn a unified policy for whole- body control of a legged manipulator using reinforcement learning, and proposes Regularized Online Adaptation to bridge the Sim2Real gap for high-DoF control, and Advantage Mixing exploiting the causal dependency in the action space to overcome local minima during training the whole-body system.",
  "doi": "10.48550/arXiv.2210.10044",
  "oa_pdf": "http://arxiv.org/pdf/2210.10044",
  "s2_authors": [
   {
    "name": "Zipeng Fu",
    "id": "89704471",
    "h_index": 13,
    "papers": 18
   },
   {
    "name": "Xuxin Cheng",
    "id": "90080090",
    "h_index": 13,
    "papers": 13
   },
   {
    "name": "Deepak Pathak",
    "id": "2004879394",
    "h_index": 24,
    "papers": 32
   }
  ],
  "comment": "CoRL 2022 (Oral). Project website at https://maniploco.github.io",
  "topics": [
   "humanoids",
   "sim2real",
   "rl-control",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2210.10044v1",
  "pdf_url": "https://arxiv.org/pdf/2210.10044v1",
  "html_url": "https://arxiv.org/html/2210.10044v1",
  "code_url": "https://maniploco.github.io",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.95
 },
 {
  "id": "2210.08402",
  "slug": "laion-5b-an-open-large-scale-dataset-for-training-next-generation-imag",
  "title": "LAION-5B: An open large-scale dataset for training next generation image-text models",
  "abstract": "Groundbreaking language-vision architectures like CLIP and DALL-E proved the utility of training on large amounts of noisy image-text data, without relying on expensive accurate labels used in standard vision unimodal supervised learning. The resulting models showed capabilities of strong text-guided image generation and transfer to downstream tasks, while performing remarkably at zero-shot classification with noteworthy out-of-distribution robustness. Since then, large-scale language-vision models like ALIGN, BASIC, GLIDE, Flamingo and Imagen made further improvements. Studying the training and capabilities of such models requires datasets containing billions of image-text pairs. Until now, no datasets of this size have been made openly available for the broader research community. To address this problem and democratize research on large-scale multi-modal models, we present LAION-5B - a dataset consisting of 5.85 billion CLIP-filtered image-text pairs, of which 2.32B contain English language. We show successful replication and fine-tuning of foundational models like CLIP, GLIDE and Stable Diffusion using the dataset, and discuss further experiments enabled with an openly available dataset of this scale. Additionally we provide several nearest neighbor indices, an improved web-interface for dataset exploration and subset generation, and detection scores for watermark, NSFW, and toxic content detection. Announcement page https://laion.ai/laion-5b-a-new-era-of-open-large-scale-multi-modal-datasets/",
  "published": "2022-10-16",
  "updated": "2022-10-16",
  "year": "2022",
  "authors": [
   "Christoph Schuhmann",
   "Romain Beaumont",
   "Richard Vencu",
   "Cade Gordon",
   "Ross Wightman",
   "Mehdi Cherti",
   "Theo Coombes",
   "Aarush Katta",
   "Clayton Mullis",
   "Mitchell Wortsman",
   "Patrick Schramowski",
   "Srivatsa Kundurthy",
   "Katherine Crowson",
   "Ludwig Schmidt",
   "Robert Kaczmarczyk",
   "Jenia Jitsev"
  ],
  "author_count": 16,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.CV",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 5496,
  "influential_citations": 507,
  "tldr": "This work presents LAION-5B - a dataset consisting of 5.85 billion CLIP-filtered image-text pairs, of which 2.32B contain English language, and shows successful replication and fine-tuning of foundational models like CLIP, GLIDE and Stable Diffusion using the dataset, and discusses further experiments enabled with an openly available dataset of this scale.",
  "doi": "10.48550/arXiv.2210.08402",
  "oa_pdf": "http://arxiv.org/pdf/2210.08402",
  "s2_authors": [
   {
    "name": "Christoph Schuhmann",
    "id": "2137341362",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "R. Beaumont",
    "id": "2125377840",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "R. Vencu",
    "id": "2137306046",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Cade Gordon",
    "id": "2007745319",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Ross Wightman",
    "id": "2113839396",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Mehdi Cherti",
    "id": "40063601",
    "h_index": 9,
    "papers": 17
   },
   {
    "name": "Theo Coombes",
    "id": "2137419050",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Aarush Katta",
    "id": "2137380426",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Clayton Mullis",
    "id": "2137306237",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Mitchell Wortsman",
    "id": "52193502",
    "h_index": 25,
    "papers": 31
   },
   {
    "name": "P. Schramowski",
    "id": "40896023",
    "h_index": 27,
    "papers": 95
   },
   {
    "name": "Srivatsa Kundurthy",
    "id": "2165304339",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Katherine Crowson",
    "id": "104998244",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "Ludwig Schmidt",
    "id": "152772922",
    "h_index": 48,
    "papers": 84
   },
   {
    "name": "R. Kaczmarczyk",
    "id": "8095484",
    "h_index": 11,
    "papers": 57
   },
   {
    "name": "J. Jitsev",
    "id": "2191688",
    "h_index": 19,
    "papers": 75
   }
  ],
  "comment": "36th Conference on Neural Information Processing Systems (NeurIPS 2022), Track on Datasets and Benchmarks. OpenReview: https://openreview.net/forum?id=M3Y74vmsMcY",
  "topics": [
   "navigation",
   "video-generation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2210.08402v1",
  "pdf_url": "https://arxiv.org/pdf/2210.08402v1",
  "html_url": "https://arxiv.org/html/2210.08402v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "2210.06463",
  "slug": "holo-dex-teaching-dexterity-with-immersive-mixed-reality",
  "title": "Holo-Dex: Teaching Dexterity with Immersive Mixed Reality",
  "abstract": "A fundamental challenge in teaching robots is to provide an effective interface for human teachers to demonstrate useful skills to a robot. This challenge is exacerbated in dexterous manipulation, where teaching high-dimensional, contact-rich behaviors often require esoteric teleoperation tools. In this work, we present Holo-Dex, a framework for dexterous manipulation that places a teacher in an immersive mixed reality through commodity VR headsets. The high-fidelity hand pose estimator onboard the headset is used to teleoperate the robot and collect demonstrations for a variety of general-purpose dexterous tasks. Given these demonstrations, we use powerful feature learning combined with non-parametric imitation to train dexterous skills. Our experiments on six common dexterous tasks, including in-hand rotation, spinning, and bottle opening, indicate that Holo-Dex can both collect high-quality demonstration data and train skills in a matter of hours. Finally, we find that our trained skills can exhibit generalization on objects not seen in training. Videos of Holo-Dex are available at https://holo-dex.github.io.",
  "published": "2022-10-12",
  "updated": "2022-10-12",
  "year": "2022",
  "authors": [
   "Sridhar Pandian Arunachalam",
   "Irmak G\u00fczey",
   "Soumith Chintala",
   "Lerrel Pinto"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.HC",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA",
  "venue_source": "semantic-scholar",
  "citations": 106,
  "influential_citations": 1,
  "tldr": "Holo \u2212 Dex is presented, a framework for dexter-ous manipulation that places a teacher in an immersive mixed reality through commodity VR headsets and can both collect high-quality demonstration data and train skills in a matter of hours.",
  "doi": "10.1109/ICRA48891.2023.10160547",
  "oa_pdf": "https://arxiv.org/pdf/2210.06463",
  "s2_authors": [
   {
    "name": "Sridhar Pandian Arunachalam",
    "id": "2144825236",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Irmak Guzey",
    "id": "2143167646",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Soumith Chintala",
    "id": "2127604",
    "h_index": 32,
    "papers": 49
   },
   {
    "name": "Lerrel Pinto",
    "id": "34026610",
    "h_index": 41,
    "papers": 70
   }
  ],
  "comment": "Data, code and videos are available at https://holo-dex.github.io",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2210.06463v1",
  "pdf_url": "https://arxiv.org/pdf/2210.06463v1",
  "html_url": "https://arxiv.org/html/2210.06463v1",
  "code_url": "https://holo-dex.github.io",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.53
 },
 {
  "id": "2210.06407",
  "slug": "interactive-language-talking-to-robots-in-real-time",
  "title": "Interactive Language: Talking to Robots in Real Time",
  "abstract": "We present a framework for building interactive, real-time, natural language-instructable robots in the real world, and we open source related assets (dataset, environment, benchmark, and policies). Trained with behavioral cloning on a dataset of hundreds of thousands of language-annotated trajectories, a produced policy can proficiently execute an order of magnitude more commands than previous works: specifically we estimate a 93.5% success rate on a set of 87,000 unique natural language strings specifying raw end-to-end visuo-linguo-motor skills in the real world. We find that the same policy is capable of being guided by a human via real-time language to address a wide range of precise long-horizon rearrangement goals, e.g. \"make a smiley face out of blocks\". The dataset we release comprises nearly 600,000 language-labeled trajectories, an order of magnitude larger than prior available datasets. We hope the demonstrated results and associated assets enable further advancement of helpful, capable, natural-language-interactable robots. See videos at https://interactive-language.github.io.",
  "published": "2022-10-12",
  "updated": "2022-10-12",
  "year": "2022",
  "authors": [
   "Corey Lynch",
   "Ayzaan Wahid",
   "Jonathan Tompson",
   "Tianli Ding",
   "James Betker",
   "Robert Baruch",
   "Travis Armstrong",
   "Pete Florence"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "RA-L",
  "venue_source": "semantic-scholar",
  "citations": 348,
  "influential_citations": 23,
  "tldr": "It is found that the same policy trained with behavioral cloning is capable of being guided by a human via real-time language to address a wide range of precise long-horizon rearrangement goals, e.g.\"make a smiley face out of blocks\".",
  "doi": "10.48550/arXiv.2210.06407",
  "oa_pdf": "http://arxiv.org/pdf/2210.06407",
  "s2_authors": [
   {
    "name": "Corey Lynch",
    "id": "32245472",
    "h_index": 20,
    "papers": 27
   },
   {
    "name": "Ayzaan Wahid",
    "id": "88728227",
    "h_index": 21,
    "papers": 27
   },
   {
    "name": "Jonathan Tompson",
    "id": "2704494",
    "h_index": 43,
    "papers": 71
   },
   {
    "name": "Tianli Ding",
    "id": "95691186",
    "h_index": 15,
    "papers": 24
   },
   {
    "name": "James Betker",
    "id": "2187579768",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Robert Baruch",
    "id": "1659197538",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Travis Armstrong",
    "id": "2057103965",
    "h_index": 9,
    "papers": 11
   },
   {
    "name": "Peter R. Florence",
    "id": "47686265",
    "h_index": 30,
    "papers": 36
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2210.06407v1",
  "pdf_url": "https://arxiv.org/pdf/2210.06407v1",
  "html_url": "https://arxiv.org/html/2210.06407v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.04
 },
 {
  "id": "2210.04887",
  "slug": "in-hand-object-rotation-via-rapid-motor-adaptation",
  "title": "In-Hand Object Rotation via Rapid Motor Adaptation",
  "abstract": "Generalized in-hand manipulation has long been an unsolved challenge of robotics. As a small step towards this grand goal, we demonstrate how to design and learn a simple adaptive controller to achieve in-hand object rotation using only fingertips. The controller is trained entirely in simulation on only cylindrical objects, which then - without any fine-tuning - can be directly deployed to a real robot hand to rotate dozens of objects with diverse sizes, shapes, and weights over the z-axis. This is achieved via rapid online adaptation of the controller to the object properties using only proprioception history. Furthermore, natural and stable finger gaits automatically emerge from training the control policy via reinforcement learning. Code and more videos are available at https://haozhi.io/hora",
  "published": "2022-10-10",
  "updated": "2022-10-10",
  "year": "2022",
  "authors": [
   "Haozhi Qi",
   "Ashish Kumar",
   "Roberto Calandra",
   "Yi Ma",
   "Jitendra Malik"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 203,
  "influential_citations": 16,
  "tldr": "This paper demonstrates how to design and learn a simple adaptive controller to achieve in-hand object rotation using only fingertips via rapid online adaptation of the controller to the object properties using only proprioception history.",
  "doi": "10.48550/arXiv.2210.04887",
  "oa_pdf": "http://arxiv.org/pdf/2210.04887",
  "s2_authors": [
   {
    "name": "Haozhi Qi",
    "id": "7217794",
    "h_index": 13,
    "papers": 18
   },
   {
    "name": "Ashish Kumar",
    "id": "2118069675",
    "h_index": 15,
    "papers": 32
   },
   {
    "name": "R. Calandra",
    "id": "35159852",
    "h_index": 40,
    "papers": 75
   },
   {
    "name": "Yinsong Ma",
    "id": "1959069289",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "J. Malik",
    "id": "153652147",
    "h_index": 64,
    "papers": 99
   }
  ],
  "comment": "CoRL 2022. Code and Website: https://haozhi.io/hora",
  "topics": [
   "dexterous-manipulation",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2210.04887v1",
  "pdf_url": "https://arxiv.org/pdf/2210.04887v1",
  "html_url": "https://arxiv.org/html/2210.04887v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.81
 },
 {
  "id": "2210.04026",
  "slug": "enhancing-generalizable-6d-pose-tracking-of-an-in-hand-object-with-tac",
  "title": "Enhancing Generalizable 6D Pose Tracking of an In-Hand Object with Tactile Sensing",
  "abstract": "When manipulating an object to accomplish complex tasks, humans rely on both vision and touch to keep track of the object's 6D pose. However, most existing object pose tracking systems in robotics rely exclusively on visual signals, which hinder a robot's ability to manipulate objects effectively. To address this limitation, we introduce TEG-Track, a tactile-enhanced 6D pose tracking system that can track previously unseen objects held in hand. From consecutive tactile signals, TEG-Track optimizes object velocities from marker flows when slippage does not occur, or regresses velocities using a slippage estimation network when slippage is detected. The estimated object velocities are integrated into a geometric-kinematic optimization scheme to enhance existing visual pose trackers. To evaluate our method and to facilitate future research, we construct a real-world dataset for visual-tactile in-hand object pose tracking. Experimental results demonstrate that TEG-Track consistently enhances state-of-the-art generalizable 6D pose trackers in synthetic and real-world scenarios. Our code and dataset are available at https://github.com/leolyliu/TEG-Track.",
  "published": "2022-10-08",
  "updated": "2023-12-23",
  "year": "2022",
  "authors": [
   "Yun Liu",
   "Xiaomeng Xu",
   "Weihang Chen",
   "Haocheng Yuan",
   "He Wang",
   "Jing Xu",
   "Rui Chen",
   "Li Yi"
  ],
  "author_count": 8,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "RA-L",
  "venue_source": "semantic-scholar",
  "citations": 30,
  "influential_citations": 1,
  "tldr": "TEG-Track is introduced, a tactile-enhanced 6D pose tracking system that can track previously unseen objects held in hand that consistently enhances state-of-the-art generalizable 6D pose trackers in synthetic and real-world scenarios.",
  "doi": "10.1109/LRA.2023.3337690",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yun Liu",
    "id": "2279787165",
    "h_index": 10,
    "papers": 16
   },
   {
    "name": "Xiaomeng Xu",
    "id": "2158827128",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Weihang Chen",
    "id": "2162451452",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Hao Yuan",
    "id": "2166719596",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "He Wang",
    "id": "2149698886",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "Jing Xu",
    "id": "2155954613",
    "h_index": 14,
    "papers": 20
   },
   {
    "name": "Rui Chen",
    "id": "2118229411",
    "h_index": 13,
    "papers": 31
   },
   {
    "name": "Li Yi",
    "id": "2027660032",
    "h_index": 17,
    "papers": 24
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2210.04026v2",
  "pdf_url": "https://arxiv.org/pdf/2210.04026v2",
  "html_url": "https://arxiv.org/html/2210.04026v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.99
 },
 {
  "id": "2210.03629",
  "slug": "react-synergizing-reasoning-and-acting-in-language-models",
  "title": "ReAct: Synergizing Reasoning and Acting in Language Models",
  "abstract": "While large language models (LLMs) have demonstrated impressive capabilities across tasks in language understanding and interactive decision making, their abilities for reasoning (e.g. chain-of-thought prompting) and acting (e.g. action plan generation) have primarily been studied as separate topics. In this paper, we explore the use of LLMs to generate both reasoning traces and task-specific actions in an interleaved manner, allowing for greater synergy between the two: reasoning traces help the model induce, track, and update action plans as well as handle exceptions, while actions allow it to interface with external sources, such as knowledge bases or environments, to gather additional information. We apply our approach, named ReAct, to a diverse set of language and decision making tasks and demonstrate its effectiveness over state-of-the-art baselines, as well as improved human interpretability and trustworthiness over methods without reasoning or acting components. Concretely, on question answering (HotpotQA) and fact verification (Fever), ReAct overcomes issues of hallucination and error propagation prevalent in chain-of-thought reasoning by interacting with a simple Wikipedia API, and generates human-like task-solving trajectories that are more interpretable than baselines without reasoning traces. On two interactive decision making benchmarks (ALFWorld and WebShop), ReAct outperforms imitation and reinforcement learning methods by an absolute success rate of 34% and 10% respectively, while being prompted with only one or two in-context examples. Project site with code: https://react-lm.github.io",
  "published": "2022-10-06",
  "updated": "2023-03-10",
  "year": "2022",
  "authors": [
   "Shunyu Yao",
   "Jeffrey Zhao",
   "Dian Yu",
   "Nan Du",
   "Izhak Shafran",
   "Karthik Narasimhan",
   "Yuan Cao"
  ],
  "author_count": 7,
  "categories": [
   "cs.CL",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.CL",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 10552,
  "influential_citations": 1096,
  "tldr": "The use of LLMs are explored to generate both reasoning traces and task-specific actions in an interleaved manner, allowing for greater synergy between the two: reasoning traces help the model induce, track, and update action plans as well as handle exceptions, while actions allow it to interface with external sources to gather additional information.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Shunyu Yao",
    "id": "2093302161",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Jeffrey Zhao",
    "id": "2144551262",
    "h_index": 11,
    "papers": 20
   },
   {
    "name": "Dian Yu",
    "id": "150978762",
    "h_index": 16,
    "papers": 31
   },
   {
    "name": "Nan Du",
    "id": "2140321952",
    "h_index": 17,
    "papers": 22
   },
   {
    "name": "Izhak Shafran",
    "id": "1697494",
    "h_index": 34,
    "papers": 127
   },
   {
    "name": "Karthik Narasimhan",
    "id": "144958935",
    "h_index": 34,
    "papers": 70
   },
   {
    "name": "Yuan Cao",
    "id": "145144022",
    "h_index": 29,
    "papers": 61
   }
  ],
  "comment": "v3 is the ICLR camera ready version with some typos fixed. Project site with code: https://react-lm.github.io",
  "topics": [
   "rl-control",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2210.03629v3",
  "pdf_url": "https://arxiv.org/pdf/2210.03629v3",
  "html_url": "https://arxiv.org/html/2210.03629v3",
  "code_url": "https://react-lm.github.io",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.5
 },
 {
  "id": "2210.03109",
  "slug": "real-world-robot-learning-with-masked-visual-pre-training",
  "title": "Real-World Robot Learning with Masked Visual Pre-training",
  "abstract": "In this work, we explore self-supervised visual pre-training on images from diverse, in-the-wild videos for real-world robotic tasks. Like prior work, our visual representations are pre-trained via a masked autoencoder (MAE), frozen, and then passed into a learnable control module. Unlike prior work, we show that the pre-trained representations are effective across a range of real-world robotic tasks and embodiments. We find that our encoder consistently outperforms CLIP (up to 75%), supervised ImageNet pre-training (up to 81%), and training from scratch (up to 81%). Finally, we train a 307M parameter vision transformer on a massive collection of 4.5M images from the Internet and egocentric videos, and demonstrate clearly the benefits of scaling visual pre-training for robot learning.",
  "published": "2022-10-06",
  "updated": "2022-10-06",
  "year": "2022",
  "authors": [
   "Ilija Radosavovic",
   "Tete Xiao",
   "Stephen James",
   "Pieter Abbeel",
   "Jitendra Malik",
   "Trevor Darrell"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 361,
  "influential_citations": 28,
  "tldr": "This work explores self-supervised visual pre-training on images from diverse, in-the-wild videos for real-world robotic tasks via a masked autoencoder, frozen, and then passed into a learnable control module to show that the pre-trained representations are effective across a range of real- world robotic tasks and embodiments.",
  "doi": "10.48550/arXiv.2210.03109",
  "oa_pdf": "http://arxiv.org/pdf/2210.03109",
  "s2_authors": [
   {
    "name": "Ilija Radosavovic",
    "id": "30407997",
    "h_index": 20,
    "papers": 22
   },
   {
    "name": "Tete Xiao",
    "id": "15727192",
    "h_index": 17,
    "papers": 21
   },
   {
    "name": "Stephen James",
    "id": "2055291154",
    "h_index": 22,
    "papers": 37
   },
   {
    "name": "P. Abbeel",
    "id": "1689992",
    "h_index": 165,
    "papers": 596
   },
   {
    "name": "J. Malik",
    "id": "153652147",
    "h_index": 64,
    "papers": 99
   },
   {
    "name": "Trevor Darrell",
    "id": "1753210",
    "h_index": 158,
    "papers": 630
   }
  ],
  "comment": "CoRL 2022; Project page: https://tetexiao.com/projects/real-mvp",
  "topics": [
   "egocentric-data",
   "foundation-pretraining",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2210.03109v1",
  "pdf_url": "https://arxiv.org/pdf/2210.03109v1",
  "html_url": "https://arxiv.org/html/2210.03109v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.06
 },
 {
  "id": "2210.03094",
  "slug": "vima-general-robot-manipulation-with-multimodal-prompts",
  "title": "VIMA: General Robot Manipulation with Multimodal Prompts",
  "abstract": "Prompt-based learning has emerged as a successful paradigm in natural language processing, where a single general-purpose language model can be instructed to perform any task specified by input prompts. Yet task specification in robotics comes in various forms, such as imitating one-shot demonstrations, following language instructions, and reaching visual goals. They are often considered different tasks and tackled by specialized models. We show that a wide spectrum of robot manipulation tasks can be expressed with multimodal prompts, interleaving textual and visual tokens. Accordingly, we develop a new simulation benchmark that consists of thousands of procedurally-generated tabletop tasks with multimodal prompts, 600K+ expert trajectories for imitation learning, and a four-level evaluation protocol for systematic generalization. We design a transformer-based robot agent, VIMA, that processes these prompts and outputs motor actions autoregressively. VIMA features a recipe that achieves strong model scalability and data efficiency. It outperforms alternative designs in the hardest zero-shot generalization setting by up to $2.9\\times$ task success rate given the same training data. With $10\\times$ less training data, VIMA still performs $2.7\\times$ better than the best competing variant. Code and video demos are available at https://vimalabs.github.io/",
  "published": "2022-10-06",
  "updated": "2023-05-28",
  "year": "2022",
  "authors": [
   "Yunfan Jiang",
   "Agrim Gupta",
   "Zichen Zhang",
   "Guanzhi Wang",
   "Yongqiang Dou",
   "Yanjun Chen",
   "Li Fei-Fei",
   "Anima Anandkumar",
   "Yuke Zhu",
   "Linxi Fan"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "ICML 2023",
  "venue_source": "arxiv-comment",
  "citations": 597,
  "influential_citations": 47,
  "tldr": "It is shown that a wide spectrum of robot manipulation tasks can be expressed with multimodal prompts, interleaving textual and visual tokens, and designed a transformer-based robot agent, VIMA, that processes these prompts and outputs motor actions autoregressively.",
  "doi": "10.48550/arXiv.2210.03094",
  "oa_pdf": "http://arxiv.org/pdf/2210.03094",
  "s2_authors": [
   {
    "name": "Yunfan Jiang",
    "id": "2171112793",
    "h_index": 11,
    "papers": 16
   },
   {
    "name": "Agrim Gupta",
    "id": "2315452357",
    "h_index": 8,
    "papers": 10
   },
   {
    "name": "Zichen Zhang",
    "id": "5630943",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Guanzhi Wang",
    "id": "96374437",
    "h_index": 16,
    "papers": 22
   },
   {
    "name": "Yongqiang Dou",
    "id": "1768148923",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Yanjun Chen",
    "id": "2187067176",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Li Fei-Fei",
    "id": "48004138",
    "h_index": 143,
    "papers": 606
   },
   {
    "name": "Anima Anandkumar",
    "id": "47627049",
    "h_index": 41,
    "papers": 119
   },
   {
    "name": "Yuke Zhu",
    "id": "2117748",
    "h_index": 57,
    "papers": 130
   },
   {
    "name": "Linxi (Jim) Fan",
    "id": "3275727",
    "h_index": 18,
    "papers": 22
   }
  ],
  "comment": "ICML 2023 Camera-ready version. Project website: https://vimalabs.github.io/",
  "topics": [
   "sim2real",
   "imitation-diffusion"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2210.03094v2",
  "pdf_url": "https://arxiv.org/pdf/2210.03094v2",
  "html_url": "https://arxiv.org/html/2210.03094v2",
  "code_url": "https://vimalabs.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.28
 },
 {
  "id": "2210.02747",
  "slug": "flow-matching-for-generative-modeling",
  "title": "Flow Matching for Generative Modeling",
  "abstract": "We introduce a new paradigm for generative modeling built on Continuous Normalizing Flows (CNFs), allowing us to train CNFs at unprecedented scale. Specifically, we present the notion of Flow Matching (FM), a simulation-free approach for training CNFs based on regressing vector fields of fixed conditional probability paths. Flow Matching is compatible with a general family of Gaussian probability paths for transforming between noise and data samples -- which subsumes existing diffusion paths as specific instances. Interestingly, we find that employing FM with diffusion paths results in a more robust and stable alternative for training diffusion models. Furthermore, Flow Matching opens the door to training CNFs with other, non-diffusion probability paths. An instance of particular interest is using Optimal Transport (OT) displacement interpolation to define the conditional probability paths. These paths are more efficient than diffusion paths, provide faster training and sampling, and result in better generalization. Training CNFs using Flow Matching on ImageNet leads to consistently better performance than alternative diffusion-based methods in terms of both likelihood and sample quality, and allows fast and reliable sample generation using off-the-shelf numerical ODE solvers.",
  "published": "2022-10-06",
  "updated": "2023-02-08",
  "year": "2022",
  "authors": [
   "Yaron Lipman",
   "Ricky T. Q. Chen",
   "Heli Ben-Hamu",
   "Maximilian Nickel",
   "Matt Le"
  ],
  "author_count": 5,
  "categories": [
   "cs.LG",
   "cs.AI",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 5895,
  "influential_citations": 829,
  "tldr": "This work presents the notion of Flow Matching (FM), a simulation-free approach for training CNFs based on regressing vector fields of fixed conditional probability paths, which is compatible with a general family of Gaussian probability paths for transforming between noise and data samples.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Y. Lipman",
    "id": "3232072",
    "h_index": 62,
    "papers": 143
   },
   {
    "name": "Ricky T. Q. Chen",
    "id": "51466615",
    "h_index": 22,
    "papers": 31
   },
   {
    "name": "Heli Ben-Hamu",
    "id": "1405681873",
    "h_index": 14,
    "papers": 17
   },
   {
    "name": "Maximilian Nickel",
    "id": "1729762",
    "h_index": 32,
    "papers": 70
   },
   {
    "name": "Matt Le",
    "id": "123542315",
    "h_index": 7,
    "papers": 8
   }
  ],
  "comment": "",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2210.02747v2",
  "pdf_url": "https://arxiv.org/pdf/2210.02747v2",
  "html_url": "https://arxiv.org/html/2210.02747v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "2210.02399",
  "slug": "phenaki-variable-length-video-generation-from-open-domain-textual-desc",
  "title": "Phenaki: Variable Length Video Generation From Open Domain Textual Description",
  "abstract": "We present Phenaki, a model capable of realistic video synthesis, given a sequence of textual prompts. Generating videos from text is particularly challenging due to the computational cost, limited quantities of high quality text-video data and variable length of videos. To address these issues, we introduce a new model for learning video representation which compresses the video to a small representation of discrete tokens. This tokenizer uses causal attention in time, which allows it to work with variable-length videos. To generate video tokens from text we are using a bidirectional masked transformer conditioned on pre-computed text tokens. The generated video tokens are subsequently de-tokenized to create the actual video. To address data issues, we demonstrate how joint training on a large corpus of image-text pairs as well as a smaller number of video-text examples can result in generalization beyond what is available in the video datasets. Compared to the previous video generation methods, Phenaki can generate arbitrary long videos conditioned on a sequence of prompts (i.e. time variable text or a story) in open domain. To the best of our knowledge, this is the first time a paper studies generating videos from time variable prompts. In addition, compared to the per-frame baselines, the proposed video encoder-decoder computes fewer tokens per video but results in better spatio-temporal consistency.",
  "published": "2022-10-05",
  "updated": "2022-10-05",
  "year": "2022",
  "authors": [
   "Ruben Villegas",
   "Mohammad Babaeizadeh",
   "Pieter-Jan Kindermans",
   "Hernan Moraldo",
   "Han Zhang",
   "Mohammad Taghi Saffar",
   "Santiago Castro",
   "Julius Kunze",
   "Dumitru Erhan"
  ],
  "author_count": 9,
  "categories": [
   "cs.CV",
   "cs.AI"
  ],
  "primary_category": "cs.CV",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 586,
  "influential_citations": 35,
  "tldr": "This paper presents Phenaki, a model capable of realistic video synthesis, given a sequence of textual prompts, and demonstrates how joint training on a large corpus of image-text pairs as well as a smaller number of video-text examples can result in generalization beyond what is available in the video datasets.",
  "doi": "10.48550/arXiv.2210.02399",
  "oa_pdf": "http://arxiv.org/pdf/2210.02399",
  "s2_authors": [
   {
    "name": "Ruben Villegas",
    "id": "144543406",
    "h_index": 18,
    "papers": 25
   },
   {
    "name": "M. Babaeizadeh",
    "id": "3365707",
    "h_index": 15,
    "papers": 24
   },
   {
    "name": "Pieter-Jan Kindermans",
    "id": "2113697",
    "h_index": 26,
    "papers": 60
   },
   {
    "name": "Hernan Moraldo",
    "id": "2186980336",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Han Zhang",
    "id": "2146205308",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "M. Saffar",
    "id": "2814161",
    "h_index": 9,
    "papers": 21
   },
   {
    "name": "Santiago Castro",
    "id": "145691225",
    "h_index": 11,
    "papers": 21
   },
   {
    "name": "Julius Kunze",
    "id": "32768402",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "D. Erhan",
    "id": "1761978",
    "h_index": 37,
    "papers": 60
   }
  ],
  "comment": "",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2210.02399v1",
  "pdf_url": "https://arxiv.org/pdf/2210.02399v1",
  "html_url": "https://arxiv.org/html/2210.02399v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.27
 },
 {
  "id": "2210.00066",
  "slug": "improving-policy-learning-via-language-dynamics-distillation",
  "title": "Improving Policy Learning via Language Dynamics Distillation",
  "abstract": "Recent work has shown that augmenting environments with language descriptions improves policy learning. However, for environments with complex language abstractions, learning how to ground language to observations is difficult due to sparse, delayed rewards. We propose Language Dynamics Distillation (LDD), which pretrains a model to predict environment dynamics given demonstrations with language descriptions, and then fine-tunes these language-aware pretrained representations via reinforcement learning (RL). In this way, the model is trained to both maximize expected reward and retain knowledge about how language relates to environment dynamics. On SILG, a benchmark of five tasks with language descriptions that evaluate distinct generalization challenges on unseen environments (NetHack, ALFWorld, RTFM, Messenger, and Touchdown), LDD outperforms tabula-rasa RL, VAE pretraining, and methods that learn from unlabeled demonstrations in inverse RL and reward shaping with pretrained experts. In our analyses, we show that language descriptions in demonstrations improve sample-efficiency and generalization across environments, and that dynamics modelling with expert demonstrations is more effective than with non-experts.",
  "published": "2022-09-30",
  "updated": "2022-09-30",
  "year": "2022",
  "authors": [
   "Victor Zhong",
   "Jesse Mu",
   "Luke Zettlemoyer",
   "Edward Grefenstette",
   "Tim Rockt\u00e4schel"
  ],
  "author_count": 5,
  "categories": [
   "cs.LG",
   "cs.AI",
   "cs.CL"
  ],
  "primary_category": "cs.LG",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 16,
  "influential_citations": 0,
  "tldr": "This work proposes Language Dynamics Distillation (LDD), which pretrains a model to predict environment dynamics given demonstrations with language descriptions, and then fine-tunes these language-aware pretrained representations via reinforcement learning (RL).",
  "doi": "10.48550/arXiv.2210.00066",
  "oa_pdf": "http://arxiv.org/pdf/2210.00066",
  "s2_authors": [
   {
    "name": "Victor Zhong",
    "id": "3428769",
    "h_index": 25,
    "papers": 41
   },
   {
    "name": "Jesse Mu",
    "id": "24835910",
    "h_index": 12,
    "papers": 27
   },
   {
    "name": "Luke Zettlemoyer",
    "id": "1982950",
    "h_index": 118,
    "papers": 278
   },
   {
    "name": "Edward Grefenstette",
    "id": "1864353",
    "h_index": 47,
    "papers": 98
   },
   {
    "name": "Tim Rocktaschel",
    "id": "1389854357",
    "h_index": 22,
    "papers": 42
   }
  ],
  "comment": "Accepted to NeurIPS 2022. 16 pages, 12 figures",
  "topics": [
   "rl-control",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2210.00066v1",
  "pdf_url": "https://arxiv.org/pdf/2210.00066v1",
  "html_url": "https://arxiv.org/html/2210.00066v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.73
 },
 {
  "id": "2210.00030",
  "slug": "vip-towards-universal-visual-reward-and-representation-via-value-impli",
  "title": "VIP: Towards Universal Visual Reward and Representation via Value-Implicit Pre-Training",
  "abstract": "Reward and representation learning are two long-standing challenges for learning an expanding set of robot manipulation skills from sensory observations. Given the inherent cost and scarcity of in-domain, task-specific robot data, learning from large, diverse, offline human videos has emerged as a promising path towards acquiring a generally useful visual representation for control; however, how these human videos can be used for general-purpose reward learning remains an open question. We introduce $\\textbf{V}$alue-$\\textbf{I}$mplicit $\\textbf{P}$re-training (VIP), a self-supervised pre-trained visual representation capable of generating dense and smooth reward functions for unseen robotic tasks. VIP casts representation learning from human videos as an offline goal-conditioned reinforcement learning problem and derives a self-supervised dual goal-conditioned value-function objective that does not depend on actions, enabling pre-training on unlabeled human videos. Theoretically, VIP can be understood as a novel implicit time contrastive objective that generates a temporally smooth embedding, enabling the value function to be implicitly defined via the embedding distance, which can then be used to construct the reward for any goal-image specified downstream task. Trained on large-scale Ego4D human videos and without any fine-tuning on in-domain, task-specific data, VIP's frozen representation can provide dense visual reward for an extensive set of simulated and $\\textbf{real-robot}$ tasks, enabling diverse reward-based visual control methods and significantly outperforming all prior pre-trained representations. Notably, VIP can enable simple, $\\textbf{few-shot}$ offline RL on a suite of real-world robot tasks with as few as 20 trajectories.",
  "published": "2022-09-30",
  "updated": "2023-03-07",
  "year": "2022",
  "authors": [
   "Yecheng Jason Ma",
   "Shagun Sodhani",
   "Dinesh Jayaraman",
   "Osbert Bastani",
   "Vikash Kumar",
   "Amy Zhang"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 516,
  "influential_citations": 37,
  "tldr": "Trained on large-scale Ego4D human videos and without any fine-tuning on in-domain, task-specific data, VIP's frozen representation can provide dense visual reward for an extensive set of simulated and real-robot tasks, enabling diverse reward-based visual control methods and significantly outperforming all prior pre-trained representations.",
  "doi": "10.48550/arXiv.2210.00030",
  "oa_pdf": "http://arxiv.org/pdf/2210.00030",
  "s2_authors": [
   {
    "name": "Y. Ma",
    "id": "2130215451",
    "h_index": 16,
    "papers": 23
   },
   {
    "name": "Shagun Sodhani",
    "id": "2462516",
    "h_index": 19,
    "papers": 49
   },
   {
    "name": "Dinesh Jayaraman",
    "id": "144348441",
    "h_index": 33,
    "papers": 72
   },
   {
    "name": "O. Bastani",
    "id": "1697444",
    "h_index": 40,
    "papers": 186
   },
   {
    "name": "Vikash Kumar",
    "id": "2109446216",
    "h_index": 38,
    "papers": 76
   },
   {
    "name": "Amy Zhang",
    "id": "2111672235",
    "h_index": 28,
    "papers": 54
   }
  ],
  "comment": "ICLR 2023, Notable-Top-25% (Spotlight). Project website: https://sites.google.com/view/vip-rl",
  "topics": [
   "egocentric-data",
   "rl-control",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2210.00030v2",
  "pdf_url": "https://arxiv.org/pdf/2210.00030v2",
  "html_url": "https://arxiv.org/html/2210.00030v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.21
 },
 {
  "id": "2209.12152",
  "slug": "all-are-worth-words-a-vit-backbone-for-diffusion-models",
  "title": "All are Worth Words: A ViT Backbone for Diffusion Models",
  "abstract": "Vision transformers (ViT) have shown promise in various vision tasks while the U-Net based on a convolutional neural network (CNN) remains dominant in diffusion models. We design a simple and general ViT-based architecture (named U-ViT) for image generation with diffusion models. U-ViT is characterized by treating all inputs including the time, condition and noisy image patches as tokens and employing long skip connections between shallow and deep layers. We evaluate U-ViT in unconditional and class-conditional image generation, as well as text-to-image generation tasks, where U-ViT is comparable if not superior to a CNN-based U-Net of a similar size. In particular, latent diffusion models with U-ViT achieve record-breaking FID scores of 2.29 in class-conditional image generation on ImageNet 256x256, and 5.48 in text-to-image generation on MS-COCO, among methods without accessing large external datasets during the training of generative models. Our results suggest that, for diffusion-based image modeling, the long skip connection is crucial while the down-sampling and up-sampling operators in CNN-based U-Net are not always necessary. We believe that U-ViT can provide insights for future research on backbones in diffusion models and benefit generative modeling on large scale cross-modality datasets.",
  "published": "2022-09-25",
  "updated": "2023-03-25",
  "year": "2022",
  "authors": [
   "Fan Bao",
   "Shen Nie",
   "Kaiwen Xue",
   "Yue Cao",
   "Chongxuan Li",
   "Hang Su",
   "Jun Zhu"
  ],
  "author_count": 7,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 648,
  "influential_citations": 71,
  "tldr": "The results suggest that, for diffusion-based image modeling, the long skip connection is crucial while the down-sampling and upsampling operators in CNN-based U-Net are not always necessary.",
  "doi": "10.1109/CVPR52729.2023.02171",
  "oa_pdf": "http://arxiv.org/pdf/2209.12152",
  "s2_authors": [
   {
    "name": "Fan Bao",
    "id": "2071898125",
    "h_index": 22,
    "papers": 36
   },
   {
    "name": "Shen Nie",
    "id": "2191077545",
    "h_index": 10,
    "papers": 16
   },
   {
    "name": "Kaiwen Xue",
    "id": "2054220757",
    "h_index": 7,
    "papers": 12
   },
   {
    "name": "Yue Cao",
    "id": "1491647685",
    "h_index": 13,
    "papers": 18
   },
   {
    "name": "Chongxuan Li",
    "id": "2399563",
    "h_index": 34,
    "papers": 74
   },
   {
    "name": "Hang Su",
    "id": "2093561216",
    "h_index": 52,
    "papers": 136
   },
   {
    "name": "Jun Zhu",
    "id": "2155220672",
    "h_index": 22,
    "papers": 29
   }
  ],
  "comment": "Accepted to CVPR 2023",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2209.12152v4",
  "pdf_url": "https://arxiv.org/pdf/2209.12152v4",
  "html_url": "https://arxiv.org/html/2209.12152v4",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.31
 },
 {
  "id": "2209.11302",
  "slug": "progprompt-generating-situated-robot-task-plans-using-large-language-m",
  "title": "ProgPrompt: Generating Situated Robot Task Plans using Large Language Models",
  "abstract": "Task planning can require defining myriad domain knowledge about the world in which a robot needs to act. To ameliorate that effort, large language models (LLMs) can be used to score potential next actions during task planning, and even generate action sequences directly, given an instruction in natural language with no additional domain information. However, such methods either require enumerating all possible next steps for scoring, or generate free-form text that may contain actions not possible on a given robot in its current context. We present a programmatic LLM prompt structure that enables plan generation functional across situated environments, robot capabilities, and tasks. Our key insight is to prompt the LLM with program-like specifications of the available actions and objects in an environment, as well as with example programs that can be executed. We make concrete recommendations about prompt structure and generation constraints through ablation experiments, demonstrate state of the art success rates in VirtualHome household tasks, and deploy our method on a physical robot arm for tabletop tasks. Website at progprompt.github.io",
  "published": "2022-09-22",
  "updated": "2022-09-22",
  "year": "2022",
  "authors": [
   "Ishika Singh",
   "Valts Blukis",
   "Arsalan Mousavian",
   "Ankit Goyal",
   "Danfei Xu",
   "Jonathan Tremblay",
   "Dieter Fox",
   "Jesse Thomason",
   "Animesh Garg"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CL",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA",
  "venue_source": "semantic-scholar",
  "citations": 1067,
  "influential_citations": 55,
  "tldr": "This work presents a programmatic LLM prompt structure that enables plan generation functional across situated environments, robot capabilities, and tasks, and makes concrete recommendations about prompt structure and generation constraints through ablation experiments.",
  "doi": "10.1109/ICRA48891.2023.10161317",
  "oa_pdf": "https://arxiv.org/pdf/2209.11302",
  "s2_authors": [
   {
    "name": "Ishika Singh",
    "id": "144399662",
    "h_index": 9,
    "papers": 21
   },
   {
    "name": "Valts Blukis",
    "id": "32481910",
    "h_index": 13,
    "papers": 20
   },
   {
    "name": "A. Mousavian",
    "id": "3040583",
    "h_index": 36,
    "papers": 57
   },
   {
    "name": "Ankit Goyal",
    "id": "47989608",
    "h_index": 18,
    "papers": 30
   },
   {
    "name": "Danfei Xu",
    "id": "2068265",
    "h_index": 33,
    "papers": 45
   },
   {
    "name": "Jonathan Tremblay",
    "id": "31943350",
    "h_index": 27,
    "papers": 56
   },
   {
    "name": "D. Fox",
    "id": "145197953",
    "h_index": 133,
    "papers": 428
   },
   {
    "name": "Jesse Thomason",
    "id": "2665873",
    "h_index": 27,
    "papers": 63
   },
   {
    "name": "Animesh Garg",
    "id": "1873736",
    "h_index": 60,
    "papers": 163
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2209.11302v1",
  "pdf_url": "https://arxiv.org/pdf/2209.11302v1",
  "html_url": "https://arxiv.org/html/2209.11302v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "2209.09002",
  "slug": "movq-modulating-quantized-vectors-for-high-fidelity-image-generation",
  "title": "MoVQ: Modulating Quantized Vectors for High-Fidelity Image Generation",
  "abstract": "Although two-stage Vector Quantized (VQ) generative models allow for synthesizing high-fidelity and high-resolution images, their quantization operator encodes similar patches within an image into the same index, resulting in a repeated artifact for similar adjacent regions using existing decoder architectures. To address this issue, we propose to incorporate the spatially conditional normalization to modulate the quantized vectors so as to insert spatially variant information to the embedded index maps, encouraging the decoder to generate more photorealistic images. Moreover, we use multichannel quantization to increase the recombination capability of the discrete codes without increasing the cost of model and codebook. Additionally, to generate discrete tokens at the second stage, we adopt a Masked Generative Image Transformer (MaskGIT) to learn an underlying prior distribution in the compressed latent space, which is much faster than the conventional autoregressive model. Experiments on two benchmark datasets demonstrate that our proposed modulated VQGAN is able to greatly improve the reconstructed image quality as well as provide high-fidelity image generation.",
  "published": "2022-09-19",
  "updated": "2022-09-19",
  "year": "2022",
  "authors": [
   "Chuanxia Zheng",
   "Long Tung Vuong",
   "Jianfei Cai",
   "Dinh Phung"
  ],
  "author_count": 4,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 156,
  "influential_citations": 7,
  "tldr": "This work proposes to incorporate the spatially conditional normalization to modulate the quantized vectors so as to insert spatially variant information to the embedded index maps, encouraging the decoder to generate more photorealistic images.",
  "doi": "10.48550/arXiv.2209.09002",
  "oa_pdf": "http://arxiv.org/pdf/2209.09002",
  "s2_authors": [
   {
    "name": "Chuanxia Zheng",
    "id": "4978976",
    "h_index": 19,
    "papers": 40
   },
   {
    "name": "L. Vuong",
    "id": "2142454775",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Jianfei Cai",
    "id": "2152629962",
    "h_index": 19,
    "papers": 34
   },
   {
    "name": "Dinh Q. Phung",
    "id": "1400659302",
    "h_index": 26,
    "papers": 147
   }
  ],
  "comment": "",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2209.09002v1",
  "pdf_url": "https://arxiv.org/pdf/2209.09002v1",
  "html_url": "https://arxiv.org/html/2209.09002v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.7
 },
 {
  "id": "2209.07753",
  "slug": "code-as-policies-language-model-programs-for-embodied-control",
  "title": "Code as Policies: Language Model Programs for Embodied Control",
  "abstract": "Large language models (LLMs) trained on code completion have been shown to be capable of synthesizing simple Python programs from docstrings [1]. We find that these code-writing LLMs can be re-purposed to write robot policy code, given natural language commands. Specifically, policy code can express functions or feedback loops that process perception outputs (e.g.,from object detectors [2], [3]) and parameterize control primitive APIs. When provided as input several example language commands (formatted as comments) followed by corresponding policy code (via few-shot prompting), LLMs can take in new commands and autonomously re-compose API calls to generate new policy code respectively. By chaining classic logic structures and referencing third-party libraries (e.g., NumPy, Shapely) to perform arithmetic, LLMs used in this way can write robot policies that (i) exhibit spatial-geometric reasoning, (ii) generalize to new instructions, and (iii) prescribe precise values (e.g., velocities) to ambiguous descriptions (\"faster\") depending on context (i.e., behavioral commonsense). This paper presents code as policies: a robot-centric formulation of language model generated programs (LMPs) that can represent reactive policies (e.g., impedance controllers), as well as waypoint-based policies (vision-based pick and place, trajectory-based control), demonstrated across multiple real robot platforms. Central to our approach is prompting hierarchical code-gen (recursively defining undefined functions), which can write more complex code and also improves state-of-the-art to solve 39.8% of problems on the HumanEval [1] benchmark. Code and videos are available at https://code-as-policies.github.io",
  "published": "2022-09-16",
  "updated": "2023-05-25",
  "year": "2022",
  "authors": [
   "Jacky Liang",
   "Wenlong Huang",
   "Fei Xia",
   "Peng Xu",
   "Karol Hausman",
   "Brian Ichter",
   "Pete Florence",
   "Andy Zeng"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA",
  "venue_source": "semantic-scholar",
  "citations": 1786,
  "influential_citations": 111,
  "tldr": "Code as Policies is presented, a robot-centric formulation of language model generated programs (LMPs) that can represent reactive policies (e.g., impedance controllers), as well as waypoint-based policies (vision-based pick and place, trajectory-based control), demonstrated across multiple real robot platforms.",
  "doi": "10.1109/ICRA48891.2023.10160591",
  "oa_pdf": "https://arxiv.org/pdf/2209.07753",
  "s2_authors": [
   {
    "name": "Jacky Liang",
    "id": "6454541",
    "h_index": 23,
    "papers": 40
   },
   {
    "name": "Wenlong Huang",
    "id": "2158105356",
    "h_index": 10,
    "papers": 11
   },
   {
    "name": "F. Xia",
    "id": "144956443",
    "h_index": 25,
    "papers": 31
   },
   {
    "name": "Peng Xu",
    "id": "2153917744",
    "h_index": 17,
    "papers": 22
   },
   {
    "name": "Karol Hausman",
    "id": "1944801",
    "h_index": 47,
    "papers": 122
   },
   {
    "name": "Brian Ichter",
    "id": "2704814",
    "h_index": 37,
    "papers": 60
   },
   {
    "name": "Peter R. Florence",
    "id": "47686265",
    "h_index": 30,
    "papers": 36
   },
   {
    "name": "Andy Zeng",
    "id": "38591293",
    "h_index": 34,
    "papers": 50
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2209.07753v4",
  "pdf_url": "https://arxiv.org/pdf/2209.07753v4",
  "html_url": "https://arxiv.org/html/2209.07753v4",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "2209.05451",
  "slug": "perceiver-actor-a-multi-task-transformer-for-robotic-manipulation",
  "title": "Perceiver-Actor: A Multi-Task Transformer for Robotic Manipulation",
  "abstract": "Transformers have revolutionized vision and natural language processing with their ability to scale with large datasets. But in robotic manipulation, data is both limited and expensive. Can manipulation still benefit from Transformers with the right problem formulation? We investigate this question with PerAct, a language-conditioned behavior-cloning agent for multi-task 6-DoF manipulation. PerAct encodes language goals and RGB-D voxel observations with a Perceiver Transformer, and outputs discretized actions by ``detecting the next best voxel action''. Unlike frameworks that operate on 2D images, the voxelized 3D observation and action space provides a strong structural prior for efficiently learning 6-DoF actions. With this formulation, we train a single multi-task Transformer for 18 RLBench tasks (with 249 variations) and 7 real-world tasks (with 18 variations) from just a few demonstrations per task. Our results show that PerAct significantly outperforms unstructured image-to-action agents and 3D ConvNet baselines for a wide range of tabletop tasks.",
  "published": "2022-09-12",
  "updated": "2022-11-11",
  "year": "2022",
  "authors": [
   "Mohit Shridhar",
   "Lucas Manuelli",
   "Dieter Fox"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CL",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 871,
  "influential_citations": 99,
  "tldr": "PerAct is investigated, a language-conditioned behavior-cloning agent for multi-task 6-DoF manipulation that significantly outperforms unstructured image-to-action agents and 3D ConvNet baselines for a wide range of tabletop tasks.",
  "doi": "10.48550/arXiv.2209.05451",
  "oa_pdf": "https://arxiv.org/pdf/2209.05451",
  "s2_authors": [
   {
    "name": "Mohit Shridhar",
    "id": "33516562",
    "h_index": 15,
    "papers": 23
   },
   {
    "name": "Lucas Manuelli",
    "id": "2033958",
    "h_index": 11,
    "papers": 21
   },
   {
    "name": "D. Fox",
    "id": "145197953",
    "h_index": 133,
    "papers": 428
   }
  ],
  "comment": "CoRL 2022. Project Website: https://peract.github.io/",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2209.05451v2",
  "pdf_url": "https://arxiv.org/pdf/2209.05451v2",
  "html_url": "https://arxiv.org/html/2209.05451v2",
  "code_url": "https://peract.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.44
 },
 {
  "id": "2209.05022",
  "slug": "poseit-a-visual-tactile-dataset-of-holding-poses-for-grasp-stability-a",
  "title": "PoseIt: A Visual-Tactile Dataset of Holding Poses for Grasp Stability Analysis",
  "abstract": "When humans grasp objects in the real world, we often move our arms to hold the object in a different pose where we can use it. In contrast, typical lab settings only study the stability of the grasp immediately after lifting, without any subsequent re-positioning of the arm. However, the grasp stability could vary widely based on the object's holding pose, as the gravitational torque and gripper contact forces could change completely. To facilitate the study of how holding poses affect grasp stability, we present PoseIt, a novel multi-modal dataset that contains visual and tactile data collected from a full cycle of grasping an object, re-positioning the arm to one of the sampled poses, and shaking the object. Using data from PoseIt, we can formulate and tackle the task of predicting whether a grasped object is stable in a particular held pose. We train an LSTM classifier that achieves 85% accuracy on the proposed task. Our experimental results show that multi-modal models trained on PoseIt achieve higher accuracy than using solely vision or tactile data and that our classifiers can also generalize to unseen objects and poses.",
  "published": "2022-09-12",
  "updated": "2022-09-12",
  "year": "2022",
  "authors": [
   "Shubham Kanitkar",
   "Helen Jiang",
   "Wenzhen Yuan"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS",
  "venue_source": "semantic-scholar",
  "citations": 24,
  "influential_citations": 0,
  "tldr": "PoseIt is presented, a novel multi-modal dataset that contains visual and tactile data collected from a full cycle of grasping an object, re-positioning the arm to one of the sampled poses, and shaking the object, and an LSTM classifier is trained that achieves 85% accuracy on the proposed task.",
  "doi": "10.1109/IROS47612.2022.9981562",
  "oa_pdf": "https://arxiv.org/pdf/2209.05022",
  "s2_authors": [
   {
    "name": "Shubham Kanitkar",
    "id": "2284807151",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Helen Jiang",
    "id": "4910251",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Wenzhen Yuan",
    "id": "3333169",
    "h_index": 27,
    "papers": 41
   }
  ],
  "comment": "8 pages, 7 figures, IEEE/RSJ International Conference on Intelligent Robots and Systems 2022",
  "topics": [
   "dexterous-manipulation",
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2209.05022v1",
  "pdf_url": "https://arxiv.org/pdf/2209.05022v1",
  "html_url": "https://arxiv.org/html/2209.05022v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.9
 },
 {
  "id": "2209.04439",
  "slug": "improved-masked-image-generation-with-token-critic",
  "title": "Improved Masked Image Generation with Token-Critic",
  "abstract": "Non-autoregressive generative transformers recently demonstrated impressive image generation performance, and orders of magnitude faster sampling than their autoregressive counterparts. However, optimal parallel sampling from the true joint distribution of visual tokens remains an open challenge. In this paper we introduce Token-Critic, an auxiliary model to guide the sampling of a non-autoregressive generative transformer. Given a masked-and-reconstructed real image, the Token-Critic model is trained to distinguish which visual tokens belong to the original image and which were sampled by the generative transformer. During non-autoregressive iterative sampling, Token-Critic is used to select which tokens to accept and which to reject and resample. Coupled with Token-Critic, a state-of-the-art generative transformer significantly improves its performance, and outperforms recent diffusion models and GANs in terms of the trade-off between generated image quality and diversity, in the challenging class-conditional ImageNet generation.",
  "published": "2022-09-09",
  "updated": "2022-09-09",
  "year": "2022",
  "authors": [
   "Jos\u00e9 Lezama",
   "Huiwen Chang",
   "Lu Jiang",
   "Irfan Essa"
  ],
  "author_count": 4,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "ECCV",
  "venue_source": "semantic-scholar",
  "citations": 75,
  "influential_citations": 9,
  "tldr": "This paper introduces Token-Critic, an auxiliary model to guide the sampling of a non-autoregressive generative transformer that significantly improves its performance, and outperforms recent diffusion models and GANs in terms of the trade-off between generated image quality and diversity, in the challenging class-conditional ImageNet generation.",
  "doi": "10.48550/arXiv.2209.04439",
  "oa_pdf": "http://arxiv.org/pdf/2209.04439",
  "s2_authors": [
   {
    "name": "Jos\u00e9 Lezama",
    "id": "143923528",
    "h_index": 15,
    "papers": 33
   },
   {
    "name": "Huiwen Chang",
    "id": "2914394",
    "h_index": 28,
    "papers": 38
   },
   {
    "name": "Lu Jiang",
    "id": "39978626",
    "h_index": 48,
    "papers": 78
   },
   {
    "name": "Irfan Essa",
    "id": "145955800",
    "h_index": 22,
    "papers": 56
   }
  ],
  "comment": "Accepted to ECCV 2022",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2209.04439v1",
  "pdf_url": "https://arxiv.org/pdf/2209.04439v1",
  "html_url": "https://arxiv.org/html/2209.04439v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.38
 },
 {
  "id": "2209.03320",
  "slug": "what-does-a-platypus-look-like-generating-customized-prompts-for-zero",
  "title": "What does a platypus look like? Generating customized prompts for zero-shot image classification",
  "abstract": "Open-vocabulary models are a promising new paradigm for image classification. Unlike traditional classification models, open-vocabulary models classify among any arbitrary set of categories specified with natural language during inference. This natural language, called \"prompts\", typically consists of a set of hand-written templates (e.g., \"a photo of a {}\") which are completed with each of the category names. This work introduces a simple method to generate higher accuracy prompts, without relying on any explicit knowledge of the task domain and with far fewer hand-constructed sentences. To achieve this, we combine open-vocabulary models with large language models (LLMs) to create Customized Prompts via Language models (CuPL, pronounced \"couple\"). In particular, we leverage the knowledge contained in LLMs in order to generate many descriptive sentences that contain important discriminating characteristics of the image categories. This allows the model to place a greater importance on these regions in the image when making predictions. We find that this straightforward and general approach improves accuracy on a range of zero-shot image classification benchmarks, including over one percentage point gain on ImageNet. Finally, this simple baseline requires no additional training and remains completely zero-shot. Code available at https://github.com/sarahpratt/CuPL.",
  "published": "2022-09-07",
  "updated": "2023-12-03",
  "year": "2022",
  "authors": [
   "Sarah Pratt",
   "Ian Covert",
   "Rosanne Liu",
   "Ali Farhadi"
  ],
  "author_count": 4,
  "categories": [
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.CV",
  "venue": "ICCV",
  "venue_source": "semantic-scholar",
  "citations": 374,
  "influential_citations": 75,
  "tldr": "This work introduces a simple method to generate higher accuracy prompts, without relying on any explicit knowledge of the task domain and with far fewer hand-constructed sentences, and improves accuracy on a range of zero-shot image classification benchmarks, including over one percentage point gain on ImageNet.",
  "doi": "10.1109/ICCV51070.2023.01438",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Sarah Pratt",
    "id": "4055152",
    "h_index": 8,
    "papers": 17
   },
   {
    "name": "Rosanne Liu",
    "id": "48757909",
    "h_index": 14,
    "papers": 23
   },
   {
    "name": "Ali Farhadi",
    "id": "143787583",
    "h_index": 78,
    "papers": 208
   }
  ],
  "comment": "ICCV 2023",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2209.03320v3",
  "pdf_url": "https://arxiv.org/pdf/2209.03320v3",
  "html_url": "https://arxiv.org/html/2209.03320v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.07
 },
 {
  "id": "2209.03003",
  "slug": "flow-straight-and-fast-learning-to-generate-and-transfer-data-with-rec",
  "title": "Flow Straight and Fast: Learning to Generate and Transfer Data with Rectified Flow",
  "abstract": "We present rectified flow, a surprisingly simple approach to learning (neural) ordinary differential equation (ODE) models to transport between two empirically observed distributions \u03c0_0 and \u03c0_1, hence providing a unified solution to generative modeling and domain transfer, among various other tasks involving distribution transport. The idea of rectified flow is to learn the ODE to follow the straight paths connecting the points drawn from \u03c0_0 and \u03c0_1 as much as possible. This is achieved by solving a straightforward nonlinear least squares optimization problem, which can be easily scaled to large models without introducing extra parameters beyond standard supervised learning. The straight paths are special and preferred because they are the shortest paths between two points, and can be simulated exactly without time discretization and hence yield computationally efficient models. We show that the procedure of learning a rectified flow from data, called rectification, turns an arbitrary coupling of \u03c0_0 and \u03c0_1 to a new deterministic coupling with provably non-increasing convex transport costs. In addition, recursively applying rectification allows us to obtain a sequence of flows with increasingly straight paths, which can be simulated accurately with coarse time discretization in the inference phase. In empirical studies, we show that rectified flow performs superbly on image generation, image-to-image translation, and domain adaptation. In particular, on image generation and translation, our method yields nearly straight flows that give high quality results even with a single Euler discretization step.",
  "published": "2022-09-07",
  "updated": "2022-09-07",
  "year": "2022",
  "authors": [
   "Xingchao Liu",
   "Chengyue Gong",
   "Qiang Liu"
  ],
  "author_count": 3,
  "categories": [
   "cs.LG"
  ],
  "primary_category": "cs.LG",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 3768,
  "influential_citations": 397,
  "tldr": "",
  "doi": "10.48550/arXiv.2209.03003",
  "oa_pdf": "http://arxiv.org/pdf/2209.03003",
  "s2_authors": [
   {
    "name": "Xingchao Liu",
    "id": "46521757",
    "h_index": 23,
    "papers": 37
   },
   {
    "name": "Chengyue Gong",
    "id": "29777869",
    "h_index": 27,
    "papers": 48
   },
   {
    "name": "Qiang Liu",
    "id": "2165566517",
    "h_index": 8,
    "papers": 12
   }
  ],
  "comment": "",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2209.03003v1",
  "pdf_url": "https://arxiv.org/pdf/2209.03003v1",
  "html_url": "https://arxiv.org/html/2209.03003v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "2209.02778",
  "slug": "multi-skill-mobile-manipulation-for-object-rearrangement",
  "title": "Multi-skill Mobile Manipulation for Object Rearrangement",
  "abstract": "We study a modular approach to tackle long-horizon mobile manipulation tasks for object rearrangement, which decomposes a full task into a sequence of subtasks. To tackle the entire task, prior work chains multiple stationary manipulation skills with a point-goal navigation skill, which are learned individually on subtasks. Although more effective than monolithic end-to-end RL policies, this framework suffers from compounding errors in skill chaining, e.g., navigating to a bad location where a stationary manipulation skill can not reach its target to manipulate. To this end, we propose that the manipulation skills should include mobility to have flexibility in interacting with the target object from multiple locations and at the same time the navigation skill could have multiple end points which lead to successful manipulation. We operationalize these ideas by implementing mobile manipulation skills rather than stationary ones and training a navigation skill trained with region goal instead of point goal. We evaluate our multi-skill mobile manipulation method M3 on 3 challenging long-horizon mobile manipulation tasks in the Home Assistant Benchmark (HAB), and show superior performance as compared to the baselines.",
  "published": "2022-09-06",
  "updated": "2022-09-06",
  "year": "2022",
  "authors": [
   "Jiayuan Gu",
   "Devendra Singh Chaplot",
   "Hao Su",
   "Jitendra Malik"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 85,
  "influential_citations": 9,
  "tldr": "This work proposes that the manipulation skills should include mobility to have flexibility in interacting with the target object from multiple locations and at the same time the navigation skill could have multiple end points which lead to successful manipulation.",
  "doi": "10.48550/arXiv.2209.02778",
  "oa_pdf": "http://arxiv.org/pdf/2209.02778",
  "s2_authors": [
   {
    "name": "Jiayuan Gu",
    "id": "30107062",
    "h_index": 12,
    "papers": 18
   },
   {
    "name": "D. Chaplot",
    "id": "2142753065",
    "h_index": 13,
    "papers": 18
   },
   {
    "name": "Hao Su",
    "id": "2087042750",
    "h_index": 21,
    "papers": 26
   },
   {
    "name": "J. Malik",
    "id": "153652147",
    "h_index": 64,
    "papers": 99
   }
  ],
  "comment": "Project website: https://sites.google.com/view/hab-m3",
  "topics": [
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2209.02778v1",
  "pdf_url": "https://arxiv.org/pdf/2209.02778v1",
  "html_url": "https://arxiv.org/html/2209.02778v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.43
 },
 {
  "id": "2209.00588",
  "slug": "transformers-are-sample-efficient-world-models",
  "title": "Transformers are Sample-Efficient World Models",
  "abstract": "Deep reinforcement learning agents are notoriously sample inefficient, which considerably limits their application to real-world problems. Recently, many model-based methods have been designed to address this issue, with learning in the imagination of a world model being one of the most prominent approaches. However, while virtually unlimited interaction with a simulated environment sounds appealing, the world model has to be accurate over extended periods of time. Motivated by the success of Transformers in sequence modeling tasks, we introduce IRIS, a data-efficient agent that learns in a world model composed of a discrete autoencoder and an autoregressive Transformer. With the equivalent of only two hours of gameplay in the Atari 100k benchmark, IRIS achieves a mean human normalized score of 1.046, and outperforms humans on 10 out of 26 games, setting a new state of the art for methods without lookahead search. To foster future research on Transformers and world models for sample-efficient reinforcement learning, we release our code and models at https://github.com/eloialonso/iris.",
  "published": "2022-09-01",
  "updated": "2023-03-01",
  "year": "2022",
  "authors": [
   "Vincent Micheli",
   "Eloi Alonso",
   "Fran\u00e7ois Fleuret"
  ],
  "author_count": 3,
  "categories": [
   "cs.LG",
   "cs.AI",
   "cs.CV"
  ],
  "primary_category": "cs.LG",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 357,
  "influential_citations": 37,
  "tldr": "IRIS is introduced, a data-efficient agent that learns in a world model composed of a discrete autoencoder and an autoregressive Transformer that outperforms humans on 10 out of 26 games, setting a new state of the art for methods without lookahead search.",
  "doi": "10.48550/arXiv.2209.00588",
  "oa_pdf": "http://arxiv.org/pdf/2209.00588",
  "s2_authors": [
   {
    "name": "Vincent Micheli",
    "id": "1491750155",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Eloi Alonso",
    "id": "144370326",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "Franccois Fleuret",
    "id": "116272138",
    "h_index": 19,
    "papers": 45
   }
  ],
  "comment": "ICLR 2023 (notable top 5%)",
  "topics": [
   "world-models",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2209.00588v2",
  "pdf_url": "https://arxiv.org/pdf/2209.00588v2",
  "html_url": "https://arxiv.org/html/2209.00588v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.05
 },
 {
  "id": "2208.12250",
  "slug": "grasp-d-differentiable-contact-rich-grasp-synthesis-for-multi-fingered",
  "title": "Grasp'D: Differentiable Contact-rich Grasp Synthesis for Multi-fingered Hands",
  "abstract": "The study of hand-object interaction requires generating viable grasp poses for high-dimensional multi-finger models, often relying on analytic grasp synthesis which tends to produce brittle and unnatural results. This paper presents Grasp'D, an approach for grasp synthesis with a differentiable contact simulation from both known models as well as visual inputs. We use gradient-based methods as an alternative to sampling-based grasp synthesis, which fails without simplifying assumptions, such as pre-specified contact locations and eigengrasps. Such assumptions limit grasp discovery and, in particular, exclude high-contact power grasps. In contrast, our simulation-based approach allows for stable, efficient, physically realistic, high-contact grasp synthesis, even for gripper morphologies with high-degrees of freedom. We identify and address challenges in making grasp simulation amenable to gradient-based optimization, such as non-smooth object surface geometry, contact sparsity, and a rugged optimization landscape. Grasp'D compares favorably to analytic grasp synthesis on human and robotic hand models, and resultant grasps achieve over 4x denser contact, leading to significantly higher grasp stability. Video and code available at https://graspd-eccv22.github.io/.",
  "published": "2022-08-25",
  "updated": "2022-08-26",
  "year": "2022",
  "authors": [
   "Dylan Turpin",
   "Liquan Wang",
   "Eric Heiden",
   "Yun-Chun Chen",
   "Miles Macklin",
   "Stavros Tsogkas",
   "Sven Dickinson",
   "Animesh Garg"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "ECCV",
  "venue_source": "semantic-scholar",
  "citations": 110,
  "influential_citations": 8,
  "tldr": "This paper presents Grasp'D, an approach for grasp synthesis with a differentiable contact simulation from both known models as well as visual inputs, and uses gradient-based methods as an alternative to sampling-based grasp synthesis, which fails without simplifying assumptions.",
  "doi": "10.48550/arXiv.2208.12250",
  "oa_pdf": "http://arxiv.org/pdf/2208.12250",
  "s2_authors": [
   {
    "name": "Dylan Turpin",
    "id": "1782348078",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Liquang Wang",
    "id": "2108908463",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Eric Heiden",
    "id": "6014852",
    "h_index": 16,
    "papers": 37
   },
   {
    "name": "Yun-Chun Chen",
    "id": "1965873910",
    "h_index": 13,
    "papers": 17
   },
   {
    "name": "M. Macklin",
    "id": "46637939",
    "h_index": 30,
    "papers": 53
   },
   {
    "name": "Stavros Tsogkas",
    "id": "2381485",
    "h_index": 17,
    "papers": 32
   },
   {
    "name": "S. Dickinson",
    "id": "1779136",
    "h_index": 44,
    "papers": 168
   },
   {
    "name": "Animesh Garg",
    "id": "1873736",
    "h_index": 60,
    "papers": 163
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2208.12250v2",
  "pdf_url": "https://arxiv.org/pdf/2208.12250v2",
  "html_url": "https://arxiv.org/html/2208.12250v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.55
 },
 {
  "id": "2208.12242",
  "slug": "dreambooth-fine-tuning-text-to-image-diffusion-models-for-subject-driv",
  "title": "DreamBooth: Fine Tuning Text-to-Image Diffusion Models for Subject-Driven Generation",
  "abstract": "Large text-to-image models achieved a remarkable leap in the evolution of AI, enabling high-quality and diverse synthesis of images from a given text prompt. However, these models lack the ability to mimic the appearance of subjects in a given reference set and synthesize novel renditions of them in different contexts. In this work, we present a new approach for \"personalization\" of text-to-image diffusion models. Given as input just a few images of a subject, we fine-tune a pretrained text-to-image model such that it learns to bind a unique identifier with that specific subject. Once the subject is embedded in the output domain of the model, the unique identifier can be used to synthesize novel photorealistic images of the subject contextualized in different scenes. By leveraging the semantic prior embedded in the model with a new autogenous class-specific prior preservation loss, our technique enables synthesizing the subject in diverse scenes, poses, views and lighting conditions that do not appear in the reference images. We apply our technique to several previously-unassailable tasks, including subject recontextualization, text-guided view synthesis, and artistic rendering, all while preserving the subject's key features. We also provide a new dataset and evaluation protocol for this new task of subject-driven generation. Project page: https://dreambooth.github.io/",
  "published": "2022-08-25",
  "updated": "2023-03-15",
  "year": "2022",
  "authors": [
   "Nataniel Ruiz",
   "Yuanzhen Li",
   "Varun Jampani",
   "Yael Pritch",
   "Michael Rubinstein",
   "Kfir Aberman"
  ],
  "author_count": 6,
  "categories": [
   "cs.CV",
   "cs.GR",
   "cs.LG"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 4484,
  "influential_citations": 746,
  "tldr": "This work presents a new approach for \u201cpersonalization\u201d of text-to-image diffusion models, and applies it to several previously-unassailable tasks, including subject recontextualization, text-guided view synthesis, and artistic rendering, all while preserving the subject's key features.",
  "doi": "10.1109/CVPR52729.2023.02155",
  "oa_pdf": "https://arxiv.org/pdf/2208.12242",
  "s2_authors": [
   {
    "name": "Nataniel Ruiz",
    "id": "31601235",
    "h_index": 14,
    "papers": 24
   },
   {
    "name": "Yuanzhen Li",
    "id": "2167749913",
    "h_index": 20,
    "papers": 29
   },
   {
    "name": "Varun Jampani",
    "id": "2131639924",
    "h_index": 34,
    "papers": 88
   },
   {
    "name": "Y. Pritch",
    "id": "1782328",
    "h_index": 24,
    "papers": 50
   },
   {
    "name": "Michael Rubinstein",
    "id": "144544291",
    "h_index": 36,
    "papers": 52
   },
   {
    "name": "Kfir Aberman",
    "id": "3451442",
    "h_index": 31,
    "papers": 60
   }
  ],
  "comment": "Published at CVPR 2023. Project page: https://dreambooth.github.io/",
  "topics": [
   "sim2real",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2208.12242v2",
  "pdf_url": "https://arxiv.org/pdf/2208.12242v2",
  "html_url": "https://arxiv.org/html/2208.12242v2",
  "code_url": "https://dreambooth.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.5
 },
 {
  "id": "2208.09743",
  "slug": "where-shall-i-touch-vision-guided-tactile-poking-for-transparent-objec",
  "title": "Where Shall I Touch? Vision-Guided Tactile Poking for Transparent Object Grasping",
  "abstract": "Picking up transparent objects is still a challenging task for robots. The visual properties of transparent objects such as reflection and refraction make the current grasping methods that rely on camera sensing fail to detect and localise them. However, humans can handle the transparent object well by first observing its coarse profile and then poking an area of interest to get a fine profile for grasping. Inspired by this, we propose a novel framework of vision-guided tactile poking for transparent objects grasping. In the proposed framework, a segmentation network is first used to predict the horizontal upper regions named as poking regions, where the robot can poke the object to obtain a good tactile reading while leading to minimal disturbance to the object's state. A poke is then performed with a high-resolution GelSight tactile sensor. Given the local profiles improved with the tactile reading, a heuristic grasp is planned for grasping the transparent object. To mitigate the limitations of real-world data collection and labelling for transparent objects, a large-scale realistic synthetic dataset was constructed. Extensive experiments demonstrate that our proposed segmentation network can predict the potential poking region with a high mean Average Precision (mAP) of 0.360, and the vision-guided tactile poking can enhance the grasping success rate significantly from 38.9% to 85.2%. Thanks to its simplicity, our proposed approach could also be adopted by other force or tactile sensors and could be used for grasping of other challenging objects. All the materials used in this paper are available at https://sites.google.com/view/tactilepoking.",
  "published": "2022-08-20",
  "updated": "2022-08-20",
  "year": "2022",
  "authors": [
   "Jiaqi Jiang",
   "Guanqun Cao",
   "Aaron Butterworth",
   "Thanh-Toan Do",
   "Shan Luo"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 44,
  "influential_citations": 1,
  "tldr": "A segmentation network is first used to predict the horizontal upper regions named as poking regions, where the robot can poke the object to obtain a good tactile reading, while leading to minimal disturbance to the object's state.",
  "doi": "10.1109/TMECH.2022.3201057",
  "oa_pdf": "https://arxiv.org/pdf/2208.09743",
  "s2_authors": [
   {
    "name": "Jiaqi Jiang",
    "id": "2149531489",
    "h_index": 11,
    "papers": 18
   },
   {
    "name": "Guanqun Cao",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Aaron Butterworth",
    "id": "2058156702",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Thanh-Toan Do",
    "id": "3354627",
    "h_index": 33,
    "papers": 94
   },
   {
    "name": "Shan Luo",
    "id": "145524951",
    "h_index": 26,
    "papers": 65
   }
  ],
  "comment": "11 pages, 11 figures, accepted by T-Mech",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2208.09743v1",
  "pdf_url": "https://arxiv.org/pdf/2208.09743v1",
  "html_url": "https://arxiv.org/html/2208.09743v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.65
 },
 {
  "id": "2208.07860",
  "slug": "a-walk-in-the-park-learning-to-walk-in-20-minutes-with-model-free-rein",
  "title": "A Walk in the Park: Learning to Walk in 20 Minutes With Model-Free Reinforcement Learning",
  "abstract": "Deep reinforcement learning is a promising approach to learning policies in uncontrolled environments that do not require domain knowledge. Unfortunately, due to sample inefficiency, deep RL applications have primarily focused on simulated environments. In this work, we demonstrate that the recent advancements in machine learning algorithms and libraries combined with a carefully tuned robot controller lead to learning quadruped locomotion in only 20 minutes in the real world. We evaluate our approach on several indoor and outdoor terrains which are known to be challenging for classical model-based controllers. We observe the robot to be able to learn walking gait consistently on all of these terrains. Finally, we evaluate our design decisions in a simulated environment.",
  "published": "2022-08-16",
  "updated": "2022-08-16",
  "year": "2022",
  "authors": [
   "Laura Smith",
   "Ilya Kostrikov",
   "Sergey Levine"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 144,
  "influential_citations": 2,
  "tldr": "This work demonstrates that the recent advancements in machine learning algorithms and libraries combined with a carefully tuned robot controller lead to learning quadruped locomotion in only 20 minutes in the real world.",
  "doi": "10.48550/arXiv.2208.07860",
  "oa_pdf": "http://arxiv.org/pdf/2208.07860",
  "s2_authors": [
   {
    "name": "Laura M. Smith",
    "id": "152447364",
    "h_index": 15,
    "papers": 16
   },
   {
    "name": "Ilya Kostrikov",
    "id": "2000906",
    "h_index": 29,
    "papers": 43
   },
   {
    "name": "Sergey Levine",
    "id": "2256675884",
    "h_index": 7,
    "papers": 7
   }
  ],
  "comment": "First two authors contributed equally. Project website: https://sites.google.com/berkeley.edu/walk-in-the-park",
  "topics": [
   "humanoids",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2208.07860v1",
  "pdf_url": "https://arxiv.org/pdf/2208.07860v1",
  "html_url": "https://arxiv.org/html/2208.07860v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.16
 },
 {
  "id": "2208.03374",
  "slug": "learning-to-generalize-with-object-centric-agents-in-the-open-world-su",
  "title": "Learning to Generalize with Object-centric Agents in the Open World Survival Game Crafter",
  "abstract": "Reinforcement learning agents must generalize beyond their training experience. Prior work has focused mostly on identical training and evaluation environments. Starting from the recently introduced Crafter benchmark, a 2D open world survival game, we introduce a new set of environments suitable for evaluating some agent's ability to generalize on previously unseen (numbers of) objects and to adapt quickly (meta-learning). In Crafter, the agents are evaluated by the number of unlocked achievements (such as collecting resources) when trained for 1M steps. We show that current agents struggle to generalize, and introduce novel object-centric agents that improve over strong baselines. We also provide critical insights of general interest for future work on Crafter through several experiments. We show that careful hyper-parameter tuning improves the PPO baseline agent by a large margin and that even feedforward agents can unlock almost all achievements by relying on the inventory display. We achieve new state-of-the-art performance on the original Crafter environment. Additionally, when trained beyond 1M steps, our tuned agents can unlock almost all achievements. We show that the recurrent PPO agents improve over feedforward ones, even with the inventory information removed. We introduce CrafterOOD, a set of 15 new environments that evaluate OOD generalization. On CrafterOOD, we show that the current agents fail to generalize, whereas our novel object-centric agents achieve state-of-the-art OOD generalization while also being interpretable. Our code is public.",
  "published": "2022-08-05",
  "updated": "2022-08-05",
  "year": "2022",
  "authors": [
   "Aleksandar Stani\u0107",
   "Yujin Tang",
   "David Ha",
   "J\u00fcrgen Schmidhuber"
  ],
  "author_count": 4,
  "categories": [
   "cs.LG",
   "cs.AI"
  ],
  "primary_category": "cs.LG",
  "venue": "IEEE Transactions on Games",
  "venue_source": "semantic-scholar",
  "citations": 16,
  "influential_citations": 2,
  "tldr": "It is shown that the current agents fail to generalize, whereas the novel object-centric agents achieve state-of-the-art OOD generalization while also being interpretable on CrafterOOD, a set of 15 new environments that evaluate OOD generalization.",
  "doi": "10.1109/TG.2023.3276849",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Aleksandar Stani'c",
    "id": "2064448632",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Yujin Tang",
    "id": "48066360",
    "h_index": 10,
    "papers": 15
   },
   {
    "name": "David Ha",
    "id": "1389041357",
    "h_index": 11,
    "papers": 15
   },
   {
    "name": "J. Schmidhuber",
    "id": "145341374",
    "h_index": 101,
    "papers": 467
   }
  ],
  "comment": "",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2208.03374v1",
  "pdf_url": "https://arxiv.org/pdf/2208.03374v1",
  "html_url": "https://arxiv.org/html/2208.03374v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.73
 },
 {
  "id": "2208.02957",
  "slug": "meaning-without-reference-in-large-language-models",
  "title": "Meaning without reference in large language models",
  "abstract": "The widespread success of large language models (LLMs) has been met with skepticism that they possess anything like human concepts or meanings. Contrary to claims that LLMs possess no meaning whatsoever, we argue that they likely capture important aspects of meaning, and moreover work in a way that approximates a compelling account of human cognition in which meaning arises from conceptual role. Because conceptual role is defined by the relationships between internal representational states, meaning cannot be determined from a model's architecture, training data, or objective function, but only by examination of how its internal states relate to each other. This approach may clarify why and how LLMs are so successful and suggest how they can be made more human-like.",
  "published": "2022-08-05",
  "updated": "2022-08-12",
  "year": "2022",
  "authors": [
   "Steven T. Piantadosi",
   "Felix Hill"
  ],
  "author_count": 2,
  "categories": [
   "cs.CL",
   "cs.AI"
  ],
  "primary_category": "cs.CL",
  "venue": "",
  "venue_source": "",
  "citations": 134,
  "influential_citations": 0,
  "tldr": "It is argued that large language models likely capture important aspects of meaning, and moreover work in a way that approximates a compelling account of human cognition in which meaning arises from conceptual role.",
  "doi": "10.48550/arXiv.2208.02957",
  "oa_pdf": "http://arxiv.org/pdf/2208.02957",
  "s2_authors": [
   {
    "name": "S. Piantadosi",
    "id": "6766605",
    "h_index": 35,
    "papers": 131
   },
   {
    "name": "Felix Hill",
    "id": "145783676",
    "h_index": 42,
    "papers": 77
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2208.02957v2",
  "pdf_url": "https://arxiv.org/pdf/2208.02957v2",
  "html_url": "https://arxiv.org/html/2208.02957v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.13
 },
 {
  "id": "2208.02885",
  "slug": "grasp-stability-prediction-with-sim-to-real-transfer-from-tactile-sens",
  "title": "Grasp Stability Prediction with Sim-to-Real Transfer from Tactile Sensing",
  "abstract": "Robot simulation has been an essential tool for data-driven manipulation tasks. However, most existing simulation frameworks lack either efficient and accurate models of physical interactions with tactile sensors or realistic tactile simulation. This makes the sim-to-real transfer for tactile-based manipulation tasks still challenging. In this work, we integrate simulation of robot dynamics and vision-based tactile sensors by modeling the physics of contact. This contact model uses simulated contact forces at the robot's end-effector to inform the generation of realistic tactile outputs. To eliminate the sim-to-real transfer gap, we calibrate our physics simulator of robot dynamics, contact model, and tactile optical simulator with real-world data, and then we demonstrate the effectiveness of our system on a zero-shot sim-to-real grasp stability prediction task where we achieve an average accuracy of 90.7% on various objects. Experiments reveal the potential of applying our simulation framework to more complicated manipulation tasks. We open-source our simulation framework at https://github.com/CMURoboTouch/Taxim/tree/taxim-robot.",
  "published": "2022-08-04",
  "updated": "2022-08-04",
  "year": "2022",
  "authors": [
   "Zilin Si",
   "Zirui Zhu",
   "Arpit Agarwal",
   "Stuart Anderson",
   "Wenzhen Yuan"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS",
  "venue_source": "semantic-scholar",
  "citations": 47,
  "influential_citations": 1,
  "tldr": "This work integrates simulation of robot dynamics and vision-based tactile sensors by modeling the physics of contact, and uses simulated contact forces at the robot's end-effector to inform the generation of realistic tactile outputs.",
  "doi": "10.1109/IROS47612.2022.9981863",
  "oa_pdf": "https://arxiv.org/pdf/2208.02885",
  "s2_authors": [
   {
    "name": "Zilin Si",
    "id": "2122346383",
    "h_index": 9,
    "papers": 12
   },
   {
    "name": "Zirui Zhu",
    "id": "2035699351",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Arpit Agarwal",
    "id": "2149663936",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Stuart Anderson",
    "id": "46635103",
    "h_index": 8,
    "papers": 10
   },
   {
    "name": "Wenzhen Yuan",
    "id": "3333169",
    "h_index": 27,
    "papers": 41
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "sim2real"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2208.02885v1",
  "pdf_url": "https://arxiv.org/pdf/2208.02885v1",
  "html_url": "https://arxiv.org/html/2208.02885v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.18
 },
 {
  "id": "2207.12598",
  "slug": "classifier-free-diffusion-guidance",
  "title": "Classifier-Free Diffusion Guidance",
  "abstract": "Classifier guidance is a recently introduced method to trade off mode coverage and sample fidelity in conditional diffusion models post training, in the same spirit as low temperature sampling or truncation in other types of generative models. Classifier guidance combines the score estimate of a diffusion model with the gradient of an image classifier and thereby requires training an image classifier separate from the diffusion model. It also raises the question of whether guidance can be performed without a classifier. We show that guidance can be indeed performed by a pure generative model without such a classifier: in what we call classifier-free guidance, we jointly train a conditional and an unconditional diffusion model, and we combine the resulting conditional and unconditional score estimates to attain a trade-off between sample quality and diversity similar to that obtained using classifier guidance.",
  "published": "2022-07-26",
  "updated": "2022-07-26",
  "year": "2022",
  "authors": [
   "Jonathan Ho",
   "Tim Salimans"
  ],
  "author_count": 2,
  "categories": [
   "cs.LG",
   "cs.AI"
  ],
  "primary_category": "cs.LG",
  "venue": "NeurIPS 2021",
  "venue_source": "arxiv-comment",
  "citations": 6968,
  "influential_citations": 1011,
  "tldr": "This work jointly train a conditional and an unconditional diffusion model, and combines the resulting conditional and unconditional score estimates to attain a trade-off between sample quality and diversity similar to that obtained using classifier guidance.",
  "doi": "10.48550/arXiv.2207.12598",
  "oa_pdf": "http://arxiv.org/pdf/2207.12598",
  "s2_authors": [
   {
    "name": "Jonathan Ho",
    "id": "2126278",
    "h_index": 25,
    "papers": 32
   }
  ],
  "comment": "A short version of this paper appeared in the NeurIPS 2021 Workshop on Deep Generative Models and Downstream Applications: https://openreview.net/pdf?id=qw8AKxfYbI",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2207.12598v1",
  "pdf_url": "https://arxiv.org/pdf/2207.12598v1",
  "html_url": "https://arxiv.org/html/2207.12598v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "2207.09450",
  "slug": "human-to-robot-imitation-in-the-wild",
  "title": "Human-to-Robot Imitation in the Wild",
  "abstract": "We approach the problem of learning by watching humans in the wild. While traditional approaches in Imitation and Reinforcement Learning are promising for learning in the real world, they are either sample inefficient or are constrained to lab settings. Meanwhile, there has been a lot of success in processing passive, unstructured human data. We propose tackling this problem via an efficient one-shot robot learning algorithm, centered around learning from a third-person perspective. We call our method WHIRL: In-the-Wild Human Imitating Robot Learning. WHIRL extracts a prior over the intent of the human demonstrator, using it to initialize our agent's policy. We introduce an efficient real-world policy learning scheme that improves using interactions. Our key contributions are a simple sampling-based policy optimization approach, a novel objective function for aligning human and robot videos as well as an exploration method to boost sample efficiency. We show one-shot generalization and success in real-world settings, including 20 different manipulation tasks in the wild. Videos and talk at https://human2robot.github.io",
  "published": "2022-07-19",
  "updated": "2022-07-19",
  "year": "2022",
  "authors": [
   "Shikhar Bahl",
   "Abhinav Gupta",
   "Deepak Pathak"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "RSS",
  "venue_source": "semantic-scholar",
  "citations": 254,
  "influential_citations": 7,
  "tldr": "A simple sampling-based policy optimization approach, a novel objective function for aligning human and robot videos as well as an exploration method to boost sample efficiency, and an efficient real-world policy learning scheme that improves using interactions are introduced.",
  "doi": "10.15607/rss.2022.xviii.026",
  "oa_pdf": "https://doi.org/10.15607/rss.2022.xviii.026",
  "s2_authors": [
   {
    "name": "Shikhar Bahl",
    "id": "8527563",
    "h_index": 20,
    "papers": 38
   },
   {
    "name": "Abhi Gupta",
    "id": "2117767136",
    "h_index": 16,
    "papers": 24
   },
   {
    "name": "Deepak Pathak",
    "id": "2004879394",
    "h_index": 24,
    "papers": 32
   }
  ],
  "comment": "Published at RSS 2022. Demos at https://human2robot.github.io",
  "topics": [
   "rl-control",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2207.09450v1",
  "pdf_url": "https://arxiv.org/pdf/2207.09450v1",
  "html_url": "https://arxiv.org/html/2207.09450v1",
  "code_url": "https://human2robot.github.io",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.91
 },
 {
  "id": "2207.05608",
  "slug": "inner-monologue-embodied-reasoning-through-planning-with-language-mode",
  "title": "Inner Monologue: Embodied Reasoning through Planning with Language Models",
  "abstract": "Recent works have shown how the reasoning capabilities of Large Language Models (LLMs) can be applied to domains beyond natural language processing, such as planning and interaction for robots. These embodied problems require an agent to understand many semantic aspects of the world: the repertoire of skills available, how these skills influence the world, and how changes to the world map back to the language. LLMs planning in embodied environments need to consider not just what skills to do, but also how and when to do them - answers that change over time in response to the agent's own choices. In this work, we investigate to what extent LLMs used in such embodied contexts can reason over sources of feedback provided through natural language, without any additional training. We propose that by leveraging environment feedback, LLMs are able to form an inner monologue that allows them to more richly process and plan in robotic control scenarios. We investigate a variety of sources of feedback, such as success detection, scene description, and human interaction. We find that closed-loop language feedback significantly improves high-level instruction completion on three domains, including simulated and real table top rearrangement tasks and long-horizon mobile manipulation tasks in a kitchen environment in the real world.",
  "published": "2022-07-12",
  "updated": "2022-07-12",
  "year": "2022",
  "authors": [
   "Wenlong Huang",
   "Fei Xia",
   "Ted Xiao",
   "Harris Chan",
   "Jacky Liang",
   "Pete Florence",
   "Andy Zeng",
   "Jonathan Tompson",
   "Igor Mordatch",
   "Yevgen Chebotar",
   "Pierre Sermanet",
   "Noah Brown",
   "Tomas Jackson",
   "Linda Luu",
   "Sergey Levine",
   "Karol Hausman",
   "Brian Ichter"
  ],
  "author_count": 17,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CL",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 1519,
  "influential_citations": 71,
  "tldr": "This work proposes that by leveraging environment feedback, LLMs are able to form an inner monologue that allows them to more richly process and plan in robotic control scenarios, and finds that closed-loop language feedback significantly improves high-level instruction completion on three domains.",
  "doi": "10.48550/arXiv.2207.05608",
  "oa_pdf": "http://arxiv.org/pdf/2207.05608",
  "s2_authors": [
   {
    "name": "Wenlong Huang",
    "id": "2158105356",
    "h_index": 10,
    "papers": 11
   },
   {
    "name": "F. Xia",
    "id": "144956443",
    "h_index": 25,
    "papers": 31
   },
   {
    "name": "Ted Xiao",
    "id": "9961095",
    "h_index": 33,
    "papers": 45
   },
   {
    "name": "Harris Chan",
    "id": "41228532",
    "h_index": 15,
    "papers": 39
   },
   {
    "name": "Jacky Liang",
    "id": "6454541",
    "h_index": 23,
    "papers": 40
   },
   {
    "name": "Peter R. Florence",
    "id": "47686265",
    "h_index": 30,
    "papers": 36
   },
   {
    "name": "Andy Zeng",
    "id": "38591293",
    "h_index": 34,
    "papers": 50
   },
   {
    "name": "Jonathan Tompson",
    "id": "2704494",
    "h_index": 43,
    "papers": 71
   },
   {
    "name": "Igor Mordatch",
    "id": "2080746",
    "h_index": 34,
    "papers": 49
   },
   {
    "name": "Yevgen Chebotar",
    "id": "2527420",
    "h_index": 33,
    "papers": 57
   },
   {
    "name": "P. Sermanet",
    "id": "3142556",
    "h_index": 39,
    "papers": 77
   },
   {
    "name": "Noah Brown",
    "id": "2161343011",
    "h_index": 10,
    "papers": 16
   },
   {
    "name": "Tomas Jackson",
    "id": "2175779811",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Linda Luu",
    "id": "13219952",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "S. Levine",
    "id": "1736651",
    "h_index": 167,
    "papers": 551
   },
   {
    "name": "Karol Hausman",
    "id": "1944801",
    "h_index": 47,
    "papers": 122
   },
   {
    "name": "Brian Ichter",
    "id": "2704814",
    "h_index": 37,
    "papers": 60
   }
  ],
  "comment": "Project website: https://innermonologue.github.io",
  "topics": [
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2207.05608v1",
  "pdf_url": "https://arxiv.org/pdf/2207.05608v1",
  "html_url": "https://arxiv.org/html/2207.05608v1",
  "code_url": "https://innermonologue.github.io",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.5
 },
 {
  "id": "2207.05053",
  "slug": "learning-continuous-grasping-function-with-a-dexterous-hand-from-human",
  "title": "Learning Continuous Grasping Function with a Dexterous Hand from Human Demonstrations",
  "abstract": "We propose to learn to generate grasping motion for manipulation with a dexterous hand using implicit functions. With continuous time inputs, the model can generate a continuous and smooth grasping plan. We name the proposed model Continuous Grasping Function (CGF). CGF is learned via generative modeling with a Conditional Variational Autoencoder using 3D human demonstrations. We will first convert the large-scale human-object interaction trajectories to robot demonstrations via motion retargeting, and then use these demonstrations to train CGF. During inference, we perform sampling with CGF to generate different grasping plans in the simulator and select the successful ones to transfer to the real robot. By training on diverse human data, our CGF allows generalization to manipulate multiple objects. Compared to previous planning algorithms, CGF is more efficient and achieves significant improvement on success rate when transferred to grasping with the real Allegro Hand. Our project page is available at https://jianglongye.com/cgf .",
  "published": "2022-07-11",
  "updated": "2023-03-19",
  "year": "2022",
  "authors": [
   "Jianglong Ye",
   "Jiashun Wang",
   "Binghao Huang",
   "Yuzhe Qin",
   "Xiaolong Wang"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "RA-L",
  "venue_source": "semantic-scholar",
  "citations": 80,
  "influential_citations": 1,
  "tldr": "The proposed model Continuous Grasping Function (CGF) is more efficient and achieves significant improvement on success rate when transferred to grasping with the real Allegro Hand.",
  "doi": "10.1109/LRA.2023.3261745",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jianglong Ye",
    "id": "2153258399",
    "h_index": 9,
    "papers": 17
   },
   {
    "name": "Jiashun Wang",
    "id": "2110144663",
    "h_index": 10,
    "papers": 13
   },
   {
    "name": "Binghao Huang",
    "id": "2175672218",
    "h_index": 8,
    "papers": 8
   },
   {
    "name": "Yuzhe Qin",
    "id": "12701031",
    "h_index": 24,
    "papers": 34
   },
   {
    "name": "Xiaolong Wang",
    "id": "2145748143",
    "h_index": 7,
    "papers": 8
   }
  ],
  "comment": "Accepted to RA-L 2023 & IROS 2023. Project page: https://jianglongye.com/cgf",
  "topics": [
   "dexterous-manipulation",
   "egocentric-data",
   "sim2real"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2207.05053v3",
  "pdf_url": "https://arxiv.org/pdf/2207.05053v3",
  "html_url": "https://arxiv.org/html/2207.05053v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.41
 },
 {
  "id": "2207.01780",
  "slug": "coderl-mastering-code-generation-through-pretrained-models-and-deep-re",
  "title": "CodeRL: Mastering Code Generation through Pretrained Models and Deep Reinforcement Learning",
  "abstract": "Program synthesis or code generation aims to generate a program that satisfies a problem specification. Recent approaches using large-scale pretrained language models (LMs) have shown promising results, yet they have some critical limitations. In particular, they often follow a standard supervised fine-tuning procedure to train a code generation model only from the pairs of natural-language problem descriptions and ground-truth programs. Such paradigm largely ignores some important but potentially useful signals in the problem specification such as unit tests, which thus often results in poor performance when solving complex unseen coding tasks. To address the limitations, we propose \"CodeRL\", a new framework for program synthesis tasks through pretrained LMs and deep reinforcement learning (RL). Specifically, during training, we treat the code-generating LM as an actor network, and introduce a critic network that is trained to predict the functional correctness of generated programs and provide dense feedback signals to the actor. During inference, we introduce a new generation procedure with a critical sampling strategy that allows a model to automatically regenerate programs based on feedback from example unit tests and critic scores. For the model backbones, we extended the encoder-decoder architecture of CodeT5 with enhanced learning objectives, larger model sizes, and better pretraining data. Our method not only achieves new SOTA results on the challenging APPS benchmark, but also shows strong zero-shot transfer capability with new SOTA results on the simpler MBPP benchmark.",
  "published": "2022-07-05",
  "updated": "2022-11-03",
  "year": "2022",
  "authors": [
   "Hung Le",
   "Yue Wang",
   "Akhilesh Deepak Gotmare",
   "Silvio Savarese",
   "Steven C. H. Hoi"
  ],
  "author_count": 5,
  "categories": [
   "cs.LG",
   "cs.CL",
   "cs.PL"
  ],
  "primary_category": "cs.LG",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 523,
  "influential_citations": 30,
  "tldr": "This work proposes \"CodeRL\", a new framework for program synthesis tasks through pretrained LMs and deep reinforcement learning (RL), which treats the code-generating LM as an actor network, and introduces a critic network that is trained to predict the functional correctness of generated programs and provide dense feedback signals to the actor.",
  "doi": "10.48550/arXiv.2207.01780",
  "oa_pdf": "https://arxiv.org/pdf/2207.01780",
  "s2_authors": [
   {
    "name": "Hung Le",
    "id": "2064728738",
    "h_index": 18,
    "papers": 32
   },
   {
    "name": "Yue Wang",
    "id": "49416727",
    "h_index": 16,
    "papers": 23
   },
   {
    "name": "Akhilesh Deepak Gotmare",
    "id": "144049726",
    "h_index": 12,
    "papers": 15
   },
   {
    "name": "S. Savarese",
    "id": "1702137",
    "h_index": 115,
    "papers": 346
   },
   {
    "name": "S. Hoi",
    "id": "1741126",
    "h_index": 87,
    "papers": 385
   }
  ],
  "comment": "An earlier version of the work was accepted to NeurIPS 2022",
  "topics": [
   "rl-control",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2207.01780v3",
  "pdf_url": "https://arxiv.org/pdf/2207.01780v3",
  "html_url": "https://arxiv.org/html/2207.01780v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.22
 },
 {
  "id": "2207.00195",
  "slug": "learning-diverse-and-physically-feasible-dexterous-grasps-with-generat",
  "title": "Learning Diverse and Physically Feasible Dexterous Grasps with Generative Model and Bilevel Optimization",
  "abstract": "To fully utilize the versatility of a multi-fingered dexterous robotic hand for executing diverse object grasps, one must consider the rich physical constraints introduced by hand-object interaction and object geometry. We propose an integrative approach of combining a generative model and a bilevel optimization (BO) to plan diverse grasp configurations on novel objects. First, a conditional variational autoencoder trained on merely six YCB objects predicts the finger placement directly from the object point cloud. The prediction is then used to seed a nonconvex BO that solves for a grasp configuration under collision, reachability, wrench closure, and friction constraints. Our method achieved an 86.7% success over 120 real world grasping trials on 20 household objects, including unseen and challenging geometries. Through quantitative empirical evaluations, we confirm that grasp configurations produced by our pipeline are indeed guaranteed to satisfy kinematic and dynamic constraints. A video summary of our results is available at youtu.be/9DTrImbN99I.",
  "published": "2022-07-01",
  "updated": "2022-12-24",
  "year": "2022",
  "authors": [
   "Albert Wu",
   "Michelle Guo",
   "C. Karen Liu"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 49,
  "influential_citations": 4,
  "tldr": "An integrative approach of combining a generative model and a bilevel optimization (BO) to plan diverse grasp configurations on novel objects to satisfy kinematic and dynamic constraints is proposed.",
  "doi": "10.48550/arXiv.2207.00195",
  "oa_pdf": "http://arxiv.org/pdf/2207.00195",
  "s2_authors": [
   {
    "name": "Albert Wu",
    "id": "46840821",
    "h_index": 13,
    "papers": 29
   },
   {
    "name": "Michelle Guo",
    "id": "2112543232",
    "h_index": 8,
    "papers": 14
   },
   {
    "name": "C. K. Liu",
    "id": "2278583770",
    "h_index": 5,
    "papers": 7
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2207.00195v2",
  "pdf_url": "https://arxiv.org/pdf/2207.00195v2",
  "html_url": "https://arxiv.org/html/2207.00195v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.2
 },
 {
  "id": "2206.14244",
  "slug": "masked-world-models-for-visual-control",
  "title": "Masked World Models for Visual Control",
  "abstract": "Visual model-based reinforcement learning (RL) has the potential to enable sample-efficient robot learning from visual observations. Yet the current approaches typically train a single model end-to-end for learning both visual representations and dynamics, making it difficult to accurately model the interaction between robots and small objects. In this work, we introduce a visual model-based RL framework that decouples visual representation learning and dynamics learning. Specifically, we train an autoencoder with convolutional layers and vision transformers (ViT) to reconstruct pixels given masked convolutional features, and learn a latent dynamics model that operates on the representations from the autoencoder. Moreover, to encode task-relevant information, we introduce an auxiliary reward prediction objective for the autoencoder. We continually update both autoencoder and dynamics model using online samples collected from environment interaction. We demonstrate that our decoupling approach achieves state-of-the-art performance on a variety of visual robotic tasks from Meta-world and RLBench, e.g., we achieve 81.7% success rate on 50 visual robotic manipulation tasks from Meta-world, while the baseline achieves 67.9%. Code is available on the project website: https://sites.google.com/view/mwm-rl.",
  "published": "2022-06-28",
  "updated": "2023-05-27",
  "year": "2022",
  "authors": [
   "Younggyo Seo",
   "Danijar Hafner",
   "Hao Liu",
   "Fangchen Liu",
   "Stephen James",
   "Kimin Lee",
   "Pieter Abbeel"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 222,
  "influential_citations": 14,
  "tldr": "A visual model-based RL framework that decouples visual representation learning and dynamics learning is introduced that achieves state-of-the-art performance on a variety of visual robotic tasks from Meta-world and RLBench.",
  "doi": "10.48550/arXiv.2206.14244",
  "oa_pdf": "http://arxiv.org/pdf/2206.14244",
  "s2_authors": [
   {
    "name": "Younggyo Seo",
    "id": "2067714176",
    "h_index": 20,
    "papers": 36
   },
   {
    "name": "Danijar Hafner",
    "id": "35006479",
    "h_index": 25,
    "papers": 47
   },
   {
    "name": "Hao Liu",
    "id": "2143855835",
    "h_index": 21,
    "papers": 27
   },
   {
    "name": "Fangchen Liu",
    "id": "32324034",
    "h_index": 14,
    "papers": 17
   },
   {
    "name": "Stephen James",
    "id": "2055291154",
    "h_index": 22,
    "papers": 37
   },
   {
    "name": "Kimin Lee",
    "id": "3436470",
    "h_index": 37,
    "papers": 69
   },
   {
    "name": "P. Abbeel",
    "id": "1689992",
    "h_index": 165,
    "papers": 596
   }
  ],
  "comment": "Project website: https://sites.google.com/view/mwm-rl. Accepted to CoRL 2022",
  "topics": [
   "world-models",
   "rl-control",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2206.14244v3",
  "pdf_url": "https://arxiv.org/pdf/2206.14244v3",
  "html_url": "https://arxiv.org/html/2206.14244v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.85
 },
 {
  "id": "2206.14176",
  "slug": "daydreamer-world-models-for-physical-robot-learning",
  "title": "DayDreamer: World Models for Physical Robot Learning",
  "abstract": "To solve tasks in complex environments, robots need to learn from experience. Deep reinforcement learning is a common approach to robot learning but requires a large amount of trial and error to learn, limiting its deployment in the physical world. As a consequence, many advances in robot learning rely on simulators. On the other hand, learning inside of simulators fails to capture the complexity of the real world, is prone to simulator inaccuracies, and the resulting behaviors do not adapt to changes in the world. The Dreamer algorithm has recently shown great promise for learning from small amounts of interaction by planning within a learned world model, outperforming pure reinforcement learning in video games. Learning a world model to predict the outcomes of potential actions enables planning in imagination, reducing the amount of trial and error needed in the real environment. However, it is unknown whether Dreamer can facilitate faster learning on physical robots. In this paper, we apply Dreamer to 4 robots to learn online and directly in the real world, without simulators. Dreamer trains a quadruped robot to roll off its back, stand up, and walk from scratch and without resets in only 1 hour. We then push the robot and find that Dreamer adapts within 10 minutes to withstand perturbations or quickly roll over and stand back up. On two different robotic arms, Dreamer learns to pick and place multiple objects directly from camera images and sparse rewards, approaching human performance. On a wheeled robot, Dreamer learns to navigate to a goal position purely from camera images, automatically resolving ambiguity about the robot orientation. Using the same hyperparameters across all experiments, we find that Dreamer is capable of online learning in the real world, establishing a strong baseline. We release our infrastructure for future applications of world models to robot learning.",
  "published": "2022-06-28",
  "updated": "2022-06-28",
  "year": "2022",
  "authors": [
   "Philipp Wu",
   "Alejandro Escontrela",
   "Danijar Hafner",
   "Ken Goldberg",
   "Pieter Abbeel"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 553,
  "influential_citations": 25,
  "tldr": "This paper applies Dreamer to 4 robots to learn online and directly in the real world, without simulators, and finds that Dreamer is capable of online learning in thereal world, establishing a strong baseline.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Philipp Wu",
    "id": "2108864104",
    "h_index": 12,
    "papers": 18
   },
   {
    "name": "Alejandro Escontrela",
    "id": "2008560299",
    "h_index": 11,
    "papers": 17
   },
   {
    "name": "Danijar Hafner",
    "id": "35006479",
    "h_index": 25,
    "papers": 47
   },
   {
    "name": "Ken Goldberg",
    "id": "144344283",
    "h_index": 90,
    "papers": 647
   },
   {
    "name": "P. Abbeel",
    "id": "1689992",
    "h_index": 165,
    "papers": 596
   }
  ],
  "comment": "Website: https://danijar.com/daydreamer",
  "topics": [
   "world-models",
   "humanoids",
   "sim2real",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2206.14176v1",
  "pdf_url": "https://arxiv.org/pdf/2206.14176v1",
  "html_url": "https://arxiv.org/html/2206.14176v1",
  "code_url": null,
  "presented": [
   {
    "club": "saturday-robotics",
    "club_name": "Saturday Robotics",
    "session_number": 7,
    "session_title": "Robotics & World Models Reading Club 07: Learning to Dream: World Models, Imagination, Path to Foundation Models for Control \u2014 Los Altos",
    "date_text": "Saturday, May 9, 2026 | 2:00 PM \u2013 5:00 PM",
    "city": "",
    "url": "https://lu.ma/srhe0vuo",
    "listed_as": "DayDreamer (2022)"
   }
  ],
  "club_note": "Real-world robot learning using Dreamer-style latent imagination",
  "featured": true,
  "signal": 7.24
 },
 {
  "id": "2206.11795",
  "slug": "video-pretraining-vpt-learning-to-act-by-watching-unlabeled-online-vid",
  "title": "Video PreTraining (VPT): Learning to Act by Watching Unlabeled Online Videos",
  "abstract": "Pretraining on noisy, internet-scale datasets has been heavily studied as a technique for training models with broad, general capabilities for text, images, and other modalities. However, for many sequential decision domains such as robotics, video games, and computer use, publicly available data does not contain the labels required to train behavioral priors in the same way. We extend the internet-scale pretraining paradigm to sequential decision domains through semi-supervised imitation learning wherein agents learn to act by watching online unlabeled videos. Specifically, we show that with a small amount of labeled data we can train an inverse dynamics model accurate enough to label a huge unlabeled source of online data -- here, online videos of people playing Minecraft -- from which we can then train a general behavioral prior. Despite using the native human interface (mouse and keyboard at 20Hz), we show that this behavioral prior has nontrivial zero-shot capabilities and that it can be fine-tuned, with both imitation learning and reinforcement learning, to hard-exploration tasks that are impossible to learn from scratch via reinforcement learning. For many tasks our models exhibit human-level performance, and we are the first to report computer agents that can craft diamond tools, which can take proficient humans upwards of 20 minutes (24,000 environment actions) of gameplay to accomplish.",
  "published": "2022-06-23",
  "updated": "2022-06-23",
  "year": "2022",
  "authors": [
   "Bowen Baker",
   "Ilge Akkaya",
   "Peter Zhokhov",
   "Joost Huizinga",
   "Jie Tang",
   "Adrien Ecoffet",
   "Brandon Houghton",
   "Raul Sampedro",
   "Jeff Clune"
  ],
  "author_count": 9,
  "categories": [
   "cs.LG",
   "cs.AI"
  ],
  "primary_category": "cs.LG",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 446,
  "influential_citations": 72,
  "tldr": "This work extends the internet-scale pretraining paradigm to sequential decision domains through semi-supervised imitation learning wherein agents learn to act by watching online unlabeled videos, and is the first to report computer agents that can craft diamond tools.",
  "doi": "10.48550/arXiv.2206.11795",
  "oa_pdf": "https://arxiv.org/pdf/2206.11795",
  "s2_authors": [
   {
    "name": "Bowen Baker",
    "id": "40566201",
    "h_index": 12,
    "papers": 18
   },
   {
    "name": "Ilge Akkaya",
    "id": "2258629",
    "h_index": 13,
    "papers": 48
   },
   {
    "name": "P. Zhokhov",
    "id": "6985635",
    "h_index": 8,
    "papers": 19
   },
   {
    "name": "Joost Huizinga",
    "id": "39378983",
    "h_index": 18,
    "papers": 36
   },
   {
    "name": "Jie Tang",
    "id": "2148911990",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Adrien Ecoffet",
    "id": "66821245",
    "h_index": 10,
    "papers": 19
   },
   {
    "name": "Brandon Houghton",
    "id": "103681415",
    "h_index": 9,
    "papers": 16
   },
   {
    "name": "Raul Sampedro",
    "id": "2076161792",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "J. Clune",
    "id": "2552141",
    "h_index": 53,
    "papers": 119
   }
  ],
  "comment": "",
  "topics": [
   "imitation-diffusion",
   "rl-control",
   "navigation",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2206.11795v1",
  "pdf_url": "https://arxiv.org/pdf/2206.11795v1",
  "html_url": "https://arxiv.org/html/2206.11795v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.15
 },
 {
  "id": "2206.10789",
  "slug": "scaling-autoregressive-models-for-content-rich-text-to-image-generatio",
  "title": "Scaling Autoregressive Models for Content-Rich Text-to-Image Generation",
  "abstract": "We present the Pathways Autoregressive Text-to-Image (Parti) model, which generates high-fidelity photorealistic images and supports content-rich synthesis involving complex compositions and world knowledge. Parti treats text-to-image generation as a sequence-to-sequence modeling problem, akin to machine translation, with sequences of image tokens as the target outputs rather than text tokens in another language. This strategy can naturally tap into the rich body of prior work on large language models, which have seen continued advances in capabilities and performance through scaling data and model sizes. Our approach is simple: First, Parti uses a Transformer-based image tokenizer, ViT-VQGAN, to encode images as sequences of discrete tokens. Second, we achieve consistent quality improvements by scaling the encoder-decoder Transformer model up to 20B parameters, with a new state-of-the-art zero-shot FID score of 7.23 and finetuned FID score of 3.22 on MS-COCO. Our detailed analysis on Localized Narratives as well as PartiPrompts (P2), a new holistic benchmark of over 1600 English prompts, demonstrate the effectiveness of Parti across a wide variety of categories and difficulty aspects. We also explore and highlight limitations of our models in order to define and exemplify key areas of focus for further improvements. See https://parti.research.google/ for high-resolution images.",
  "published": "2022-06-22",
  "updated": "2022-06-22",
  "year": "2022",
  "authors": [
   "Jiahui Yu",
   "Yuanzhong Xu",
   "Jing Yu Koh",
   "Thang Luong",
   "Gunjan Baid",
   "Zirui Wang",
   "Vijay Vasudevan",
   "Alexander Ku",
   "Yinfei Yang",
   "Burcu Karagol Ayan",
   "Ben Hutchinson",
   "Wei Han",
   "Zarana Parekh",
   "Xin Li",
   "Han Zhang",
   "Jason Baldridge",
   "Yonghui Wu"
  ],
  "author_count": 17,
  "categories": [
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.CV",
  "venue": "Trans. Mach. Learn. Res.",
  "venue_source": "semantic-scholar",
  "citations": 1557,
  "influential_citations": 117,
  "tldr": "The Pathways Autoregressive Text-to-Image (Parti) model is presented, which generates high-fidelity photorealistic images and supports content-rich synthesis involving complex compositions and world knowledge and explores and highlights limitations of the models.",
  "doi": "10.48550/arXiv.2206.10789",
  "oa_pdf": "https://arxiv.org/pdf/2206.10789",
  "s2_authors": [
   {
    "name": "Jiahui Yu",
    "id": "2338016295",
    "h_index": 7,
    "papers": 9
   },
   {
    "name": "Yuanzhong Xu",
    "id": "2145139570",
    "h_index": 18,
    "papers": 29
   },
   {
    "name": "Jing Yu Koh",
    "id": "23978705",
    "h_index": 17,
    "papers": 24
   },
   {
    "name": "Thang Luong",
    "id": "1821711",
    "h_index": 19,
    "papers": 37
   },
   {
    "name": "Gunjan Baid",
    "id": "1396954703",
    "h_index": 12,
    "papers": 16
   },
   {
    "name": "Zirui Wang",
    "id": "2331539",
    "h_index": 11,
    "papers": 23
   },
   {
    "name": "Vijay Vasudevan",
    "id": "2053781980",
    "h_index": 27,
    "papers": 34
   },
   {
    "name": "Alexander Ku",
    "id": "31702389",
    "h_index": 13,
    "papers": 18
   },
   {
    "name": "Yinfei Yang",
    "id": "2118771180",
    "h_index": 15,
    "papers": 20
   },
   {
    "name": "Burcu Karagol Ayan",
    "id": "143990191",
    "h_index": 9,
    "papers": 11
   },
   {
    "name": "Ben Hutchinson",
    "id": "2044655623",
    "h_index": 19,
    "papers": 22
   },
   {
    "name": "Wei Han",
    "id": "143911112",
    "h_index": 33,
    "papers": 39
   },
   {
    "name": "Zarana Parekh",
    "id": "27456119",
    "h_index": 9,
    "papers": 15
   },
   {
    "name": "Xin Li",
    "id": "2158973314",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Han Zhang",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Jason Baldridge",
    "id": "1387994164",
    "h_index": 49,
    "papers": 128
   },
   {
    "name": "Yonghui Wu",
    "id": "48607963",
    "h_index": 65,
    "papers": 89
   }
  ],
  "comment": "Preprint",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2206.10789v1",
  "pdf_url": "https://arxiv.org/pdf/2206.10789v1",
  "html_url": "https://arxiv.org/html/2206.10789v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "2206.08948",
  "slug": "cmt-deeplab-clustering-mask-transformers-for-panoptic-segmentation",
  "title": "CMT-DeepLab: Clustering Mask Transformers for Panoptic Segmentation",
  "abstract": "We propose Clustering Mask Transformer (CMT-DeepLab), a transformer-based framework for panoptic segmentation designed around clustering. It rethinks the existing transformer architectures used in segmentation and detection; CMT-DeepLab considers the object queries as cluster centers, which fill the role of grouping the pixels when applied to segmentation. The clustering is computed with an alternating procedure, by first assigning pixels to the clusters by their feature affinity, and then updating the cluster centers and pixel features. Together, these operations comprise the Clustering Mask Transformer (CMT) layer, which produces cross-attention that is denser and more consistent with the final segmentation task. CMT-DeepLab improves the performance over prior art significantly by 4.4% PQ, achieving a new state-of-the-art of 55.7% PQ on the COCO test-dev set.",
  "published": "2022-06-17",
  "updated": "2022-06-17",
  "year": "2022",
  "authors": [
   "Qihang Yu",
   "Huiyu Wang",
   "Dahun Kim",
   "Siyuan Qiao",
   "Maxwell Collins",
   "Yukun Zhu",
   "Hartwig Adam",
   "Alan Yuille",
   "Liang-Chieh Chen"
  ],
  "author_count": 9,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 114,
  "influential_citations": 4,
  "tldr": "Clustering Mask Transformer (CMT-DeepLab) is proposed, a transformer-based framework for panoptic segmentation designed around clustering that improves the performance over prior art significantly and achieves a new state-of-the-art of 55.7% PQ on the COCO test-dev set.",
  "doi": "10.1109/CVPR52688.2022.00259",
  "oa_pdf": "http://arxiv.org/pdf/2206.08948",
  "s2_authors": [
   {
    "name": "Qihang Yu",
    "id": "2156559",
    "h_index": 23,
    "papers": 41
   },
   {
    "name": "Huiyu Wang",
    "id": "46506170",
    "h_index": 20,
    "papers": 80
   },
   {
    "name": "Dahun Kim",
    "id": "24028009",
    "h_index": 17,
    "papers": 23
   },
   {
    "name": "Siyuan Qiao",
    "id": "2383133",
    "h_index": 24,
    "papers": 50
   },
   {
    "name": "Maxwell D. Collins",
    "id": "31604982",
    "h_index": 21,
    "papers": 33
   },
   {
    "name": "Yukun Zhu",
    "id": "1844940337",
    "h_index": 18,
    "papers": 22
   },
   {
    "name": "Hartwig Adam",
    "id": "2595180",
    "h_index": 44,
    "papers": 70
   },
   {
    "name": "A. Yuille",
    "id": "145081362",
    "h_index": 137,
    "papers": 757
   },
   {
    "name": "Liang-Chieh Chen",
    "id": "34192119",
    "h_index": 41,
    "papers": 57
   }
  ],
  "comment": "CVPR 2022 Oral",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2206.08948v1",
  "pdf_url": "https://arxiv.org/pdf/2206.08948v1",
  "html_url": "https://arxiv.org/html/2206.08948v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.56
 },
 {
  "id": "2206.08916",
  "slug": "unified-io-a-unified-model-for-vision-language-and-multi-modal-tasks",
  "title": "Unified-IO: A Unified Model for Vision, Language, and Multi-Modal Tasks",
  "abstract": "We propose Unified-IO, a model that performs a large variety of AI tasks spanning classical computer vision tasks, including pose estimation, object detection, depth estimation and image generation, vision-and-language tasks such as region captioning and referring expression, to natural language processing tasks such as question answering and paraphrasing. Developing a single unified model for such a large variety of tasks poses unique challenges due to the heterogeneous inputs and outputs pertaining to each task, including RGB images, per-pixel maps, binary masks, bounding boxes, and language. We achieve this unification by homogenizing every supported input and output into a sequence of discrete vocabulary tokens. This common representation across all tasks allows us to train a single transformer-based architecture, jointly on over 90 diverse datasets in the vision and language fields. Unified-IO is the first model capable of performing all 7 tasks on the GRIT benchmark and produces strong results across 16 diverse benchmarks like NYUv2-Depth, ImageNet, VQA2.0, OK-VQA, Swig, VizWizGround, BoolQ, and SciTail, with no task-specific fine-tuning. Code and demos for Unified-IO are available at: https://unified-io.allenai.org.",
  "published": "2022-06-17",
  "updated": "2022-10-04",
  "year": "2022",
  "authors": [
   "Jiasen Lu",
   "Christopher Clark",
   "Rowan Zellers",
   "Roozbeh Mottaghi",
   "Aniruddha Kembhavi"
  ],
  "author_count": 5,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 539,
  "influential_citations": 42,
  "tldr": "Unified-IO is the first model capable of performing all 7 tasks on the GRIT benchmark and produces strong results across 16 diverse benchmarks like NYUv2-Depth, ImageNet, VQA2.0, OK-VQA, Swig, VizWizGround, BoolQ, and SciTail, with no task-specific fine-tuning.",
  "doi": "10.48550/arXiv.2206.08916",
  "oa_pdf": "http://arxiv.org/pdf/2206.08916",
  "s2_authors": [
   {
    "name": "Jiasen Lu",
    "id": "2286022498",
    "h_index": 17,
    "papers": 38
   },
   {
    "name": "Christopher Clark",
    "id": "143997772",
    "h_index": 17,
    "papers": 23
   },
   {
    "name": "Rowan Zellers",
    "id": "2545335",
    "h_index": 25,
    "papers": 36
   },
   {
    "name": "Roozbeh Mottaghi",
    "id": "3012475",
    "h_index": 46,
    "papers": 102
   },
   {
    "name": "Aniruddha Kembhavi",
    "id": "2684226",
    "h_index": 49,
    "papers": 119
   }
  ],
  "comment": "",
  "topics": [
   "spatial-3d",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2206.08916v2",
  "pdf_url": "https://arxiv.org/pdf/2206.08916v2",
  "html_url": "https://arxiv.org/html/2206.08916v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.23
 },
 {
  "id": "2206.08853",
  "slug": "minedojo-building-open-ended-embodied-agents-with-internet-scale-knowl",
  "title": "MineDojo: Building Open-Ended Embodied Agents with Internet-Scale Knowledge",
  "abstract": "Autonomous agents have made great strides in specialist domains like Atari games and Go. However, they typically learn tabula rasa in isolated environments with limited and manually conceived objectives, thus failing to generalize across a wide spectrum of tasks and capabilities. Inspired by how humans continually learn and adapt in the open world, we advocate a trinity of ingredients for building generalist agents: 1) an environment that supports a multitude of tasks and goals, 2) a large-scale database of multimodal knowledge, and 3) a flexible and scalable agent architecture. We introduce MineDojo, a new framework built on the popular Minecraft game that features a simulation suite with thousands of diverse open-ended tasks and an internet-scale knowledge base with Minecraft videos, tutorials, wiki pages, and forum discussions. Using MineDojo's data, we propose a novel agent learning algorithm that leverages large pre-trained video-language models as a learned reward function. Our agent is able to solve a variety of open-ended tasks specified in free-form language without any manually designed dense shaping reward. We open-source the simulation suite, knowledge bases, algorithm implementation, and pretrained models (https://minedojo.org) to promote research towards the goal of generally capable embodied agents.",
  "published": "2022-06-17",
  "updated": "2022-11-22",
  "year": "2022",
  "authors": [
   "Linxi Fan",
   "Guanzhi Wang",
   "Yunfan Jiang",
   "Ajay Mandlekar",
   "Yuncong Yang",
   "Haoyi Zhu",
   "Andrew Tang",
   "De-An Huang",
   "Yuke Zhu",
   "Anima Anandkumar"
  ],
  "author_count": 10,
  "categories": [
   "cs.LG",
   "cs.AI",
   "cs.CL",
   "cs.CV"
  ],
  "primary_category": "cs.LG",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 608,
  "influential_citations": 84,
  "tldr": "This work introduces MineDojo, a new framework built on the popular Minecraft game that features a simulation suite with thousands of diverse open-ended tasks and an internet-scale knowledge base with Minecraft videos, tutorials, wiki pages, and forum discussions, and proposes a novel agent learning algorithm that leverages large pre-trained video-language models as a learned reward function.",
  "doi": "10.48550/arXiv.2206.08853",
  "oa_pdf": "https://arxiv.org/pdf/2206.08853",
  "s2_authors": [
   {
    "name": "Linxi (Jim) Fan",
    "id": "3275727",
    "h_index": 18,
    "papers": 22
   },
   {
    "name": "Guanzhi Wang",
    "id": "96374437",
    "h_index": 16,
    "papers": 22
   },
   {
    "name": "Yunfan Jiang",
    "id": "2171112793",
    "h_index": 11,
    "papers": 16
   },
   {
    "name": "A. Mandlekar",
    "id": "49686756",
    "h_index": 36,
    "papers": 67
   },
   {
    "name": "Yuncong Yang",
    "id": "2178644022",
    "h_index": 6,
    "papers": 12
   },
   {
    "name": "Haoyi Zhu",
    "id": "2171155650",
    "h_index": 9,
    "papers": 11
   },
   {
    "name": "Andy Tang",
    "id": "2309480619",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "De-An Huang",
    "id": "38485317",
    "h_index": 32,
    "papers": 52
   },
   {
    "name": "Yuke Zhu",
    "id": "2117748",
    "h_index": 57,
    "papers": 130
   },
   {
    "name": "Anima Anandkumar",
    "id": "47627049",
    "h_index": 41,
    "papers": 119
   }
  ],
  "comment": "Outstanding Paper Award at NeurIPS 2022. Project website: https://minedojo.org",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2206.08853v2",
  "pdf_url": "https://arxiv.org/pdf/2206.08853v2",
  "html_url": "https://arxiv.org/html/2206.08853v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.28
 },
 {
  "id": "2206.04452",
  "slug": "draft-and-revise-effective-image-generation-with-contextual-rq-transfo",
  "title": "Draft-and-Revise: Effective Image Generation with Contextual RQ-Transformer",
  "abstract": "Although autoregressive models have achieved promising results on image generation, their unidirectional generation process prevents the resultant images from fully reflecting global contexts. To address the issue, we propose an effective image generation framework of Draft-and-Revise with Contextual RQ-transformer to consider global contexts during the generation process. As a generalized VQ-VAE, RQ-VAE first represents a high-resolution image as a sequence of discrete code stacks. After code stacks in the sequence are randomly masked, Contextual RQ-Transformer is trained to infill the masked code stacks based on the unmasked contexts of the image. Then, Contextual RQ-Transformer uses our two-phase decoding, Draft-and-Revise, and generates an image, while exploiting the global contexts of the image during the generation process. Specifically. in the draft phase, our model first focuses on generating diverse images despite rather low quality. Then, in the revise phase, the model iteratively improves the quality of images, while preserving the global contexts of generated images. In experiments, our method achieves state-of-the-art results on conditional image generation. We also validate that the Draft-and-Revise decoding can achieve high performance by effectively controlling the quality-diversity trade-off in image generation.",
  "published": "2022-06-09",
  "updated": "2022-06-09",
  "year": "2022",
  "authors": [
   "Doyup Lee",
   "Chiheon Kim",
   "Saehoon Kim",
   "Minsu Cho",
   "Wook-Shin Han"
  ],
  "author_count": 5,
  "categories": [
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.CV",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 35,
  "influential_citations": 5,
  "tldr": "An effective image generation framework of Draft-and-Revise with Contextual RQ-transformer to consider global contexts during the generation process to achieve state-of-the-art results on conditional image generation.",
  "doi": "10.48550/arXiv.2206.04452",
  "oa_pdf": "https://arxiv.org/pdf/2206.04452",
  "s2_authors": [
   {
    "name": "Doyup Lee",
    "id": "2154633624",
    "h_index": 10,
    "papers": 17
   },
   {
    "name": "Chiheon Kim",
    "id": "25004333",
    "h_index": 12,
    "papers": 20
   },
   {
    "name": "Saehoon Kim",
    "id": "1898376",
    "h_index": 19,
    "papers": 43
   },
   {
    "name": "Minsu Cho",
    "id": "72643925",
    "h_index": 48,
    "papers": 145
   },
   {
    "name": "Wook-Shin Han",
    "id": "144422954",
    "h_index": 32,
    "papers": 128
   }
  ],
  "comment": "20 pages, 11 figures",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2206.04452v1",
  "pdf_url": "https://arxiv.org/pdf/2206.04452v1",
  "html_url": "https://arxiv.org/html/2206.04452v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.06
 },
 {
  "id": "2206.04114",
  "slug": "deep-hierarchical-planning-from-pixels",
  "title": "Deep Hierarchical Planning from Pixels",
  "abstract": "Intelligent agents need to select long sequences of actions to solve complex tasks. While humans easily break down tasks into subgoals and reach them through millions of muscle commands, current artificial intelligence is limited to tasks with horizons of a few hundred decisions, despite large compute budgets. Research on hierarchical reinforcement learning aims to overcome this limitation but has proven to be challenging, current methods rely on manually specified goal spaces or subtasks, and no general solution exists. We introduce Director, a practical method for learning hierarchical behaviors directly from pixels by planning inside the latent space of a learned world model. The high-level policy maximizes task and exploration rewards by selecting latent goals and the low-level policy learns to achieve the goals. Despite operating in latent space, the decisions are interpretable because the world model can decode goals into images for visualization. Director outperforms exploration methods on tasks with sparse rewards, including 3D maze traversal with a quadruped robot from an egocentric camera and proprioception, without access to the global position or top-down view that was used by prior work. Director also learns successful behaviors across a wide range of environments, including visual control, Atari games, and DMLab levels.",
  "published": "2022-06-08",
  "updated": "2022-06-08",
  "year": "2022",
  "authors": [
   "Danijar Hafner",
   "Kuang-Huei Lee",
   "Ian Fischer",
   "Pieter Abbeel"
  ],
  "author_count": 4,
  "categories": [
   "cs.AI",
   "cs.LG",
   "cs.RO",
   "stat.ML"
  ],
  "primary_category": "cs.AI",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 145,
  "influential_citations": 13,
  "tldr": "Director is introduced, a practical method for learning hierarchical behaviors directly from pixels by planning inside the latent space of a learned world model, and the decisions are interpretable because the world model can decode goals into images for visualization.",
  "doi": "10.48550/arXiv.2206.04114",
  "oa_pdf": "https://arxiv.org/pdf/2206.04114",
  "s2_authors": [
   {
    "name": "Danijar Hafner",
    "id": "35006479",
    "h_index": 25,
    "papers": 47
   },
   {
    "name": "Kuang-Huei Lee",
    "id": "2145145412",
    "h_index": 15,
    "papers": 19
   },
   {
    "name": "Ian S. Fischer",
    "id": "33091759",
    "h_index": 18,
    "papers": 25
   },
   {
    "name": "P. Abbeel",
    "id": "1689992",
    "h_index": 165,
    "papers": 596
   }
  ],
  "comment": "Website: https://danijar.com/director",
  "topics": [
   "world-models",
   "humanoids",
   "egocentric-data",
   "rl-control",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2206.04114v1",
  "pdf_url": "https://arxiv.org/pdf/2206.04114v1",
  "html_url": "https://arxiv.org/html/2206.04114v1",
  "code_url": null,
  "presented": [
   {
    "club": "saturday-robotics",
    "club_name": "Saturday Robotics",
    "session_number": 7,
    "session_title": "Robotics & World Models Reading Club 07: Learning to Dream: World Models, Imagination, Path to Foundation Models for Control \u2014 Los Altos",
    "date_text": "Saturday, May 9, 2026 | 2:00 PM \u2013 5:00 PM",
    "city": "",
    "url": "https://lu.ma/srhe0vuo",
    "listed_as": "Director (2022)"
   }
  ],
  "club_note": "Hierarchical latent planning for long-horizon decision making",
  "featured": true,
  "signal": 6.66
 },
 {
  "id": "2206.01634",
  "slug": "reinforcement-learning-with-neural-radiance-fields",
  "title": "Reinforcement Learning with Neural Radiance Fields",
  "abstract": "It is a long-standing problem to find effective representations for training reinforcement learning (RL) agents. This paper demonstrates that learning state representations with supervision from Neural Radiance Fields (NeRFs) can improve the performance of RL compared to other learned representations or even low-dimensional, hand-engineered state information. Specifically, we propose to train an encoder that maps multiple image observations to a latent space describing the objects in the scene. The decoder built from a latent-conditioned NeRF serves as the supervision signal to learn the latent space. An RL algorithm then operates on the learned latent space as its state representation. We call this NeRF-RL. Our experiments indicate that NeRF as supervision leads to a latent space better suited for the downstream RL tasks involving robotic object manipulations like hanging mugs on hooks, pushing objects, or opening doors. Video: https://dannydriess.github.io/nerf-rl",
  "published": "2022-06-03",
  "updated": "2022-06-03",
  "year": "2022",
  "authors": [
   "Danny Driess",
   "Ingmar Schubert",
   "Pete Florence",
   "Yunzhu Li",
   "Marc Toussaint"
  ],
  "author_count": 5,
  "categories": [
   "cs.LG",
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.LG",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 71,
  "influential_citations": 3,
  "tldr": "This paper demonstrates that learning state representations with supervision from Neural Radiance Fields (NeRFs) can improve the performance of RL compared to other learned representations or even low-dimensional, hand-engineered state information.",
  "doi": "10.48550/arXiv.2206.01634",
  "oa_pdf": "http://arxiv.org/pdf/2206.01634",
  "s2_authors": [
   {
    "name": "Danny Driess",
    "id": "30837327",
    "h_index": 19,
    "papers": 31
   },
   {
    "name": "I. Schubert",
    "id": "2091687904",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Peter R. Florence",
    "id": "47686265",
    "h_index": 30,
    "papers": 36
   },
   {
    "name": "Yunzhu Li",
    "id": "3422021",
    "h_index": 25,
    "papers": 63
   },
   {
    "name": "Marc Toussaint",
    "id": "144918851",
    "h_index": 50,
    "papers": 292
   }
  ],
  "comment": "",
  "topics": [
   "rl-control",
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2206.01634v1",
  "pdf_url": "https://arxiv.org/pdf/2206.01634v1",
  "html_url": "https://arxiv.org/html/2206.01634v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.36
 },
 {
  "id": "2205.15361",
  "slug": "tubeformer-deeplab-video-mask-transformer",
  "title": "TubeFormer-DeepLab: Video Mask Transformer",
  "abstract": "We present TubeFormer-DeepLab, the first attempt to tackle multiple core video segmentation tasks in a unified manner. Different video segmentation tasks (e.g., video semantic/instance/panoptic segmentation) are usually considered as distinct problems. State-of-the-art models adopted in the separate communities have diverged, and radically different approaches dominate in each task. By contrast, we make a crucial observation that video segmentation tasks could be generally formulated as the problem of assigning different predicted labels to video tubes (where a tube is obtained by linking segmentation masks along the time axis) and the labels may encode different values depending on the target task. The observation motivates us to develop TubeFormer-DeepLab, a simple and effective video mask transformer model that is widely applicable to multiple video segmentation tasks. TubeFormer-DeepLab directly predicts video tubes with task-specific labels (either pure semantic categories, or both semantic categories and instance identities), which not only significantly simplifies video segmentation models, but also advances state-of-the-art results on multiple video segmentation benchmarks",
  "published": "2022-05-30",
  "updated": "2023-03-05",
  "year": "2022",
  "authors": [
   "Dahun Kim",
   "Jun Xie",
   "Huiyu Wang",
   "Siyuan Qiao",
   "Qihang Yu",
   "Hong-Seok Kim",
   "Hartwig Adam",
   "In So Kweon",
   "Liang-Chieh Chen"
  ],
  "author_count": 9,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 53,
  "influential_citations": 6,
  "tldr": "TubeFormer-DeepLab is presented, the first attempt to tackle multiple core video segmentation tasks in a unified manner and directly predicts video tubes with task-specific labels, which not only significantly simplifiesVideo segmentation models, but also advances state-of-the-art results on multiple video segmentations benchmarks.",
  "doi": "10.1109/CVPR52688.2022.01354",
  "oa_pdf": "https://arxiv.org/pdf/2205.15361",
  "s2_authors": [
   {
    "name": "Dahun Kim",
    "id": "24028009",
    "h_index": 17,
    "papers": 23
   },
   {
    "name": "Jun Xie",
    "id": "2109934951",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Huiyu Wang",
    "id": "1587922010",
    "h_index": 10,
    "papers": 13
   },
   {
    "name": "Siyuan Qiao",
    "id": "2383133",
    "h_index": 24,
    "papers": 50
   },
   {
    "name": "Qihang Yu",
    "id": "2156559",
    "h_index": 23,
    "papers": 41
   },
   {
    "name": "Hong-Seok Kim",
    "id": "2110138093",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Hartwig Adam",
    "id": "2595180",
    "h_index": 44,
    "papers": 70
   },
   {
    "name": "In-So Kweon",
    "id": "145017151",
    "h_index": 28,
    "papers": 64
   },
   {
    "name": "Liang-Chieh Chen",
    "id": "34192119",
    "h_index": 41,
    "papers": 57
   }
  ],
  "comment": "CVPR 2022; arXiv v2: add results on VIPSeg val/test sets and VSPW new test set",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2205.15361v2",
  "pdf_url": "https://arxiv.org/pdf/2205.15361v2",
  "html_url": "https://arxiv.org/html/2205.15361v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.23
 },
 {
  "id": "2205.06175",
  "slug": "a-generalist-agent",
  "title": "A Generalist Agent",
  "abstract": "Inspired by progress in large-scale language modeling, we apply a similar approach towards building a single generalist agent beyond the realm of text outputs. The agent, which we refer to as Gato, works as a multi-modal, multi-task, multi-embodiment generalist policy. The same network with the same weights can play Atari, caption images, chat, stack blocks with a real robot arm and much more, deciding based on its context whether to output text, joint torques, button presses, or other tokens. In this report we describe the model and the data, and document the current capabilities of Gato.",
  "published": "2022-05-12",
  "updated": "2022-11-11",
  "year": "2022",
  "authors": [
   "Scott Reed",
   "Konrad Zolna",
   "Emilio Parisotto",
   "Sergio Gomez Colmenarejo",
   "Alexander Novikov",
   "Gabriel Barth-Maron",
   "Mai Gimenez",
   "Yury Sulsky",
   "Jackie Kay",
   "Jost Tobias Springenberg",
   "Tom Eccles",
   "Jake Bruce",
   "Ali Razavi",
   "Ashley Edwards",
   "Nicolas Heess",
   "Yutian Chen",
   "Raia Hadsell",
   "Oriol Vinyals",
   "Mahyar Bordbar",
   "Nando de Freitas"
  ],
  "author_count": 20,
  "categories": [
   "cs.AI",
   "cs.CL",
   "cs.LG",
   "cs.RO"
  ],
  "primary_category": "cs.AI",
  "venue": "Trans. Mach. Learn. Res.",
  "venue_source": "semantic-scholar",
  "citations": 1140,
  "influential_citations": 78,
  "tldr": "This report describes the model and the data, and document the current capabilities of Gato, a single generalist agent that works as a multi-modal, multi-task,Multi-embodiment generalist policy beyond the realm of text outputs.",
  "doi": "10.48550/arXiv.2205.06175",
  "oa_pdf": "https://arxiv.org/pdf/2205.06175",
  "s2_authors": [
   {
    "name": "S. Reed",
    "id": "145577281",
    "h_index": 2,
    "papers": 15
   },
   {
    "name": "Konrad Zolna",
    "id": "7912420",
    "h_index": 21,
    "papers": 33
   },
   {
    "name": "Emilio Parisotto",
    "id": "3166516",
    "h_index": 26,
    "papers": 44
   },
   {
    "name": "Sergio Gomez Colmenarejo",
    "id": "2016840",
    "h_index": 21,
    "papers": 24
   },
   {
    "name": "Alexander Novikov",
    "id": "2050212830",
    "h_index": 21,
    "papers": 25
   },
   {
    "name": "Gabriel Barth-Maron",
    "id": "1403998955",
    "h_index": 15,
    "papers": 29
   },
   {
    "name": "Mai Gim\u00e9nez",
    "id": "2047713067",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Yury Sulsky",
    "id": "1390139201",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Jackie Kay",
    "id": "2059147422",
    "h_index": 9,
    "papers": 11
   },
   {
    "name": "Jost Tobias Springenberg",
    "id": "2060551",
    "h_index": 44,
    "papers": 93
   },
   {
    "name": "Tom Eccles",
    "id": "3241600",
    "h_index": 13,
    "papers": 30
   },
   {
    "name": "Jake Bruce",
    "id": "12139064",
    "h_index": 12,
    "papers": 22
   },
   {
    "name": "Ali Razavi",
    "id": "143653164",
    "h_index": 14,
    "papers": 17
   },
   {
    "name": "Ashley D. Edwards",
    "id": "48779623",
    "h_index": 7,
    "papers": 16
   },
   {
    "name": "N. Heess",
    "id": "2801204",
    "h_index": 73,
    "papers": 192
   },
   {
    "name": "Yutian Chen",
    "id": "2275897",
    "h_index": 25,
    "papers": 59
   },
   {
    "name": "R. Hadsell",
    "id": "2315504",
    "h_index": 51,
    "papers": 111
   },
   {
    "name": "O. Vinyals",
    "id": "1689108",
    "h_index": 103,
    "papers": 204
   },
   {
    "name": "Mahyar Bordbar",
    "id": "46232775",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Nando de Freitas",
    "id": "1737568",
    "h_index": 79,
    "papers": 193
   }
  ],
  "comment": "Published at TMLR, 42 pages",
  "topics": [
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2205.06175v3",
  "pdf_url": "https://arxiv.org/pdf/2205.06175v3",
  "html_url": "https://arxiv.org/html/2205.06175v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "2204.14198",
  "slug": "flamingo-a-visual-language-model-for-few-shot-learning",
  "title": "Flamingo: a Visual Language Model for Few-Shot Learning",
  "abstract": "Building models that can be rapidly adapted to novel tasks using only a handful of annotated examples is an open challenge for multimodal machine learning research. We introduce Flamingo, a family of Visual Language Models (VLM) with this ability. We propose key architectural innovations to: (i) bridge powerful pretrained vision-only and language-only models, (ii) handle sequences of arbitrarily interleaved visual and textual data, and (iii) seamlessly ingest images or videos as inputs. Thanks to their flexibility, Flamingo models can be trained on large-scale multimodal web corpora containing arbitrarily interleaved text and images, which is key to endow them with in-context few-shot learning capabilities. We perform a thorough evaluation of our models, exploring and measuring their ability to rapidly adapt to a variety of image and video tasks. These include open-ended tasks such as visual question-answering, where the model is prompted with a question which it has to answer; captioning tasks, which evaluate the ability to describe a scene or an event; and close-ended tasks such as multiple-choice visual question-answering. For tasks lying anywhere on this spectrum, a single Flamingo model can achieve a new state of the art with few-shot learning, simply by prompting the model with task-specific examples. On numerous benchmarks, Flamingo outperforms models fine-tuned on thousands of times more task-specific data.",
  "published": "2022-04-29",
  "updated": "2022-11-15",
  "year": "2022",
  "authors": [
   "Jean-Baptiste Alayrac",
   "Jeff Donahue",
   "Pauline Luc",
   "Antoine Miech",
   "Iain Barr",
   "Yana Hasson",
   "Karel Lenc",
   "Arthur Mensch",
   "Katie Millican",
   "Malcolm Reynolds",
   "Roman Ring",
   "Eliza Rutherford",
   "Serkan Cabi",
   "Tengda Han",
   "Zhitao Gong",
   "Sina Samangooei",
   "Marianne Monteiro",
   "Jacob Menick",
   "Sebastian Borgeaud",
   "Andrew Brock",
   "Aida Nematzadeh",
   "Sahand Sharifzadeh",
   "Mikolaj Binkowski",
   "Ricardo Barreira",
   "Oriol Vinyals",
   "Andrew Zisserman",
   "Karen Simonyan"
  ],
  "author_count": 27,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.CV",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 6468,
  "influential_citations": 421,
  "tldr": "This work introduces Flamingo, a family of Visual Language Models (VLM) with this ability to bridge powerful pretrained vision-only and language-only models, handle sequences of arbitrarily interleaved visual and textual data, and seamlessly ingest images or videos as inputs.",
  "doi": "10.52202/068431-1723",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jean-Baptiste Alayrac",
    "id": "2285263",
    "h_index": 33,
    "papers": 67
   },
   {
    "name": "Jeff Donahue",
    "id": "7408951",
    "h_index": 32,
    "papers": 50
   },
   {
    "name": "Pauline Luc",
    "id": "152831141",
    "h_index": 11,
    "papers": 17
   },
   {
    "name": "Antoine Miech",
    "id": "19200186",
    "h_index": 21,
    "papers": 42
   },
   {
    "name": "Iain Barr",
    "id": "2159207795",
    "h_index": 8,
    "papers": 21
   },
   {
    "name": "Yana Hasson",
    "id": "66535271",
    "h_index": 5,
    "papers": 15
   },
   {
    "name": "Karel Lenc",
    "id": "3257286",
    "h_index": 14,
    "papers": 26
   },
   {
    "name": "Arthur Mensch",
    "id": "1697879",
    "h_index": 14,
    "papers": 32
   },
   {
    "name": "Katie Millican",
    "id": "2143434227",
    "h_index": 11,
    "papers": 29
   },
   {
    "name": "Malcolm Reynolds",
    "id": "47447264",
    "h_index": 11,
    "papers": 12
   },
   {
    "name": "Roman Ring",
    "id": "81387328",
    "h_index": 9,
    "papers": 20
   },
   {
    "name": "Eliza Rutherford",
    "id": "2143538252",
    "h_index": 9,
    "papers": 27
   },
   {
    "name": "Serkan Cabi",
    "id": "12159303",
    "h_index": 19,
    "papers": 29
   },
   {
    "name": "Tengda Han",
    "id": "22237490",
    "h_index": 16,
    "papers": 33
   },
   {
    "name": "Zhitao Gong",
    "id": "48398849",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Sina Samangooei",
    "id": "2412073",
    "h_index": 16,
    "papers": 52
   },
   {
    "name": "Marianne Monteiro",
    "id": "49601928",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Jacob Menick",
    "id": "10698483",
    "h_index": 9,
    "papers": 26
   },
   {
    "name": "Sebastian Borgeaud",
    "id": "148016269",
    "h_index": 20,
    "papers": 52
   },
   {
    "name": "Andy Brock",
    "id": "2065040422",
    "h_index": 9,
    "papers": 17
   },
   {
    "name": "Aida Nematzadeh",
    "id": "3208081",
    "h_index": 14,
    "papers": 50
   },
   {
    "name": "Sahand Sharifzadeh",
    "id": "7782886",
    "h_index": 14,
    "papers": 28
   },
   {
    "name": "Mikolaj Binkowski",
    "id": "9961753",
    "h_index": 8,
    "papers": 11
   },
   {
    "name": "Ricardo Barreira",
    "id": "2026369796",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "O. Vinyals",
    "id": "1689108",
    "h_index": 103,
    "papers": 204
   },
   {
    "name": "Andrew Zisserman",
    "id": "1688869",
    "h_index": 191,
    "papers": 839
   },
   {
    "name": "K. Simonyan",
    "id": "34838386",
    "h_index": 65,
    "papers": 108
   }
  ],
  "comment": "54 pages. In Proceedings of Neural Information Processing Systems (NeurIPS) 2022",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2204.14198v2",
  "pdf_url": "https://arxiv.org/pdf/2204.14198v2",
  "html_url": "https://arxiv.org/html/2204.14198v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "2204.13662",
  "slug": "arctic-a-dataset-for-dexterous-bimanual-hand-object-manipulation",
  "title": "ARCTIC: A Dataset for Dexterous Bimanual Hand-Object Manipulation",
  "abstract": "Humans intuitively understand that inanimate objects do not move by themselves, but that state changes are typically caused by human manipulation (e.g., the opening of a book). This is not yet the case for machines. In part this is because there exist no datasets with ground-truth 3D annotations for the study of physically consistent and synchronised motion of hands and articulated objects. To this end, we introduce ARCTIC -- a dataset of two hands that dexterously manipulate objects, containing 2.1M video frames paired with accurate 3D hand and object meshes and detailed, dynamic contact information. It contains bi-manual articulation of objects such as scissors or laptops, where hand poses and object states evolve jointly in time. We propose two novel articulated hand-object interaction tasks: (1) Consistent motion reconstruction: Given a monocular video, the goal is to reconstruct two hands and articulated objects in 3D, so that their motions are spatio-temporally consistent. (2) Interaction field estimation: Dense relative hand-object distances must be estimated from images. We introduce two baselines ArcticNet and InterField, respectively and evaluate them qualitatively and quantitatively on ARCTIC. Our code and data are available at https://arctic.is.tue.mpg.de.",
  "published": "2022-04-28",
  "updated": "2023-04-23",
  "year": "2022",
  "authors": [
   "Zicong Fan",
   "Omid Taheri",
   "Dimitrios Tzionas",
   "Muhammed Kocabas",
   "Manuel Kaufmann",
   "Michael J. Black",
   "Otmar Hilliges"
  ],
  "author_count": 7,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 384,
  "influential_citations": 51,
  "tldr": "ARCTIC is introduced - a dataset of two hands that dexterously manipulate objects, containing 2 .1M video frames paired with accurate 3D hand and object meshes and detailed, dynamic contact information and two novel articulated hand-object interaction tasks.",
  "doi": "10.1109/CVPR52729.2023.01244",
  "oa_pdf": "https://www.research-collection.ethz.ch/bitstream/20.500.11850/642263/4/Fan_ARCTIC_A_Dataset_for_Dexterous_Bimanual_Hand-Object_Manipulation_CVPR_2023_paper.pdf",
  "s2_authors": [
   {
    "name": "Zicong Fan",
    "id": "27678755",
    "h_index": 10,
    "papers": 18
   },
   {
    "name": "Omid Taheri",
    "id": "48693082",
    "h_index": 11,
    "papers": 20
   },
   {
    "name": "Dimitrios Tzionas",
    "id": "1940674",
    "h_index": 26,
    "papers": 34
   },
   {
    "name": "Muhammed Kocabas",
    "id": "51131930",
    "h_index": 18,
    "papers": 23
   },
   {
    "name": "Manuel Kaufmann",
    "id": "35090707",
    "h_index": 13,
    "papers": 26
   },
   {
    "name": "Michael J. Black",
    "id": "2105795",
    "h_index": 138,
    "papers": 440
   },
   {
    "name": "Otmar Hilliges",
    "id": "1466533438",
    "h_index": 46,
    "papers": 122
   }
  ],
  "comment": "Project page: https://arctic.is.tue.mpg.de",
  "topics": [
   "dexterous-manipulation",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2204.13662v3",
  "pdf_url": "https://arxiv.org/pdf/2204.13662v3",
  "html_url": "https://arxiv.org/html/2204.13662v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.09
 },
 {
  "id": "2204.12490",
  "slug": "from-one-hand-to-multiple-hands-imitation-learning-for-dexterous-manip",
  "title": "From One Hand to Multiple Hands: Imitation Learning for Dexterous Manipulation from Single-Camera Teleoperation",
  "abstract": "We propose to perform imitation learning for dexterous manipulation with multi-finger robot hand from human demonstrations, and transfer the policy to the real robot hand. We introduce a novel single-camera teleoperation system to collect the 3D demonstrations efficiently with only an iPad and a computer. One key contribution of our system is that we construct a customized robot hand for each user in the physical simulator, which is a manipulator resembling the same kinematics structure and shape of the operator's hand. This provides an intuitive interface and avoid unstable human-robot hand retargeting for data collection, leading to large-scale and high quality data. Once the data is collected, the customized robot hand trajectories can be converted to different specified robot hands (models that are manufactured) to generate training demonstrations. With imitation learning using our data, we show large improvement over baselines with multiple complex manipulation tasks. Importantly, we show our learned policy is significantly more robust when transferring to the real robot. More videos can be found in the https://yzqin.github.io/dex-teleop-imitation .",
  "published": "2022-04-26",
  "updated": "2023-01-18",
  "year": "2022",
  "authors": [
   "Yuzhe Qin",
   "Hao Su",
   "Xiaolong Wang"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "RA-L",
  "venue_source": "semantic-scholar",
  "citations": 160,
  "influential_citations": 4,
  "tldr": "A novel single-camera teleoperation system to collect the 3D demonstrations efficiently with only an iPad and a computer and shows large improvement over baselines with multiple complex manipulation tasks.",
  "doi": "10.1109/LRA.2022.3196104",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yuzhe Qin",
    "id": "12701031",
    "h_index": 24,
    "papers": 34
   },
   {
    "name": "Hao Su",
    "id": "2087042750",
    "h_index": 21,
    "papers": 26
   },
   {
    "name": "Xiaolong Wang",
    "id": "122024152",
    "h_index": 42,
    "papers": 63
   }
  ],
  "comment": "https://yzqin.github.io/dex-teleop-imitation/",
  "topics": [
   "dexterous-manipulation",
   "egocentric-data",
   "sim2real",
   "imitation-diffusion",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2204.12490v2",
  "pdf_url": "https://arxiv.org/pdf/2204.12490v2",
  "html_url": "https://arxiv.org/html/2204.12490v2",
  "code_url": "https://yzqin.github.io/dex-teleop-imitation/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.71
 },
 {
  "id": "2204.08585",
  "slug": "information-prioritization-through-empowerment-in-visual-model-based-r",
  "title": "INFOrmation Prioritization through EmPOWERment in Visual Model-Based RL",
  "abstract": "Model-based reinforcement learning (RL) algorithms designed for handling complex visual observations typically learn some sort of latent state representation, either explicitly or implicitly. Standard methods of this sort do not distinguish between functionally relevant aspects of the state and irrelevant distractors, instead aiming to represent all available information equally. We propose a modified objective for model-based RL that, in combination with mutual information maximization, allows us to learn representations and dynamics for visual model-based RL without reconstruction in a way that explicitly prioritizes functionally relevant factors. The key principle behind our design is to integrate a term inspired by variational empowerment into a state-space model based on mutual information. This term prioritizes information that is correlated with action, thus ensuring that functionally relevant factors are captured first. Furthermore, the same empowerment term also promotes faster exploration during the RL process, especially for sparse-reward tasks where the reward signal is insufficient to drive exploration in the early stages of learning. We evaluate the approach on a suite of vision-based robot control tasks with natural video backgrounds, and show that the proposed prioritized information objective outperforms state-of-the-art model based RL approaches with higher sample efficiency and episodic returns. https://sites.google.com/view/information-empowerment",
  "published": "2022-04-18",
  "updated": "2022-04-18",
  "year": "2022",
  "authors": [
   "Homanga Bharadhwaj",
   "Mohammad Babaeizadeh",
   "Dumitru Erhan",
   "Sergey Levine"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 37,
  "influential_citations": 1,
  "tldr": "The key principle behind the design is to integrate a term inspired by variational empowerment into a state-space model based on mutual information that prioritizes information that is correlated with action, thus ensuring that functionally relevant factors are captured first.",
  "doi": "10.48550/arXiv.2204.08585",
  "oa_pdf": "http://arxiv.org/pdf/2204.08585",
  "s2_authors": [
   {
    "name": "Homanga Bharadhwaj",
    "id": "51113848",
    "h_index": 23,
    "papers": 59
   },
   {
    "name": "M. Babaeizadeh",
    "id": "3365707",
    "h_index": 15,
    "papers": 24
   },
   {
    "name": "D. Erhan",
    "id": "1761978",
    "h_index": 37,
    "papers": 60
   },
   {
    "name": "S. Levine",
    "id": "1736651",
    "h_index": 167,
    "papers": 551
   }
  ],
  "comment": "Published in International Conference on Learning Representations (ICLR 2022)",
  "topics": [
   "rl-control",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2204.08585v1",
  "pdf_url": "https://arxiv.org/pdf/2204.08585v1",
  "html_url": "https://arxiv.org/html/2204.08585v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.08
 },
 {
  "id": "2204.07153",
  "slug": "what-s-in-your-hands-3d-reconstruction-of-generic-objects-in-hands",
  "title": "What's in your hands? 3D Reconstruction of Generic Objects in Hands",
  "abstract": "Our work aims to reconstruct hand-held objects given a single RGB image. In contrast to prior works that typically assume known 3D templates and reduce the problem to 3D pose estimation, our work reconstructs generic hand-held object without knowing their 3D templates. Our key insight is that hand articulation is highly predictive of the object shape, and we propose an approach that conditionally reconstructs the object based on the articulation and the visual input. Given an image depicting a hand-held object, we first use off-the-shelf systems to estimate the underlying hand pose and then infer the object shape in a normalized hand-centric coordinate frame. We parameterized the object by signed distance which are inferred by an implicit network which leverages the information from both visual feature and articulation-aware coordinates to process a query point. We perform experiments across three datasets and show that our method consistently outperforms baselines and is able to reconstruct a diverse set of objects. We analyze the benefits and robustness of explicit articulation conditioning and also show that this allows the hand pose estimation to further improve in test-time optimization.",
  "published": "2022-04-14",
  "updated": "2022-04-14",
  "year": "2022",
  "authors": [
   "Yufei Ye",
   "Abhinav Gupta",
   "Shubham Tulsiani"
  ],
  "author_count": 3,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 123,
  "influential_citations": 23,
  "tldr": "The key insight is that hand articulation is highly predictive of the object shape, and this work proposes an approach that conditionally reconstructs the object based on the articulation and the visual input and allows the hand pose estimation to further improve in test-time optimization.",
  "doi": "10.1109/CVPR52688.2022.00387",
  "oa_pdf": "https://arxiv.org/pdf/2204.07153",
  "s2_authors": [
   {
    "name": "Yufei Ye",
    "id": "9653518",
    "h_index": 10,
    "papers": 18
   },
   {
    "name": "A. Gupta",
    "id": "1726095131",
    "h_index": 96,
    "papers": 211
   },
   {
    "name": "Shubham Tulsiani",
    "id": "2757335",
    "h_index": 45,
    "papers": 98
   }
  ],
  "comment": "accepted to CVPR 22",
  "topics": [
   "spatial-3d",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2204.07153v1",
  "pdf_url": "https://arxiv.org/pdf/2204.07153v1",
  "html_url": "https://arxiv.org/html/2204.07153v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.59
 },
 {
  "id": "2204.05080",
  "slug": "semantic-exploration-from-language-abstractions-and-pretrained-represe",
  "title": "Semantic Exploration from Language Abstractions and Pretrained Representations",
  "abstract": "Effective exploration is a challenge in reinforcement learning (RL). Novelty-based exploration methods can suffer in high-dimensional state spaces, such as continuous partially-observable 3D environments. We address this challenge by defining novelty using semantically meaningful state abstractions, which can be found in learned representations shaped by natural language. In particular, we evaluate vision-language representations, pretrained on natural image captioning datasets. We show that these pretrained representations drive meaningful, task-relevant exploration and improve performance on 3D simulated environments. We also characterize why and how language provides useful abstractions for exploration by considering the impacts of using representations from a pretrained model, a language oracle, and several ablations. We demonstrate the benefits of our approach in two very different task domains -- one that stresses the identification and manipulation of everyday objects, and one that requires navigational exploration in an expansive world. Our results suggest that using language-shaped representations could improve exploration for various algorithms and agents in challenging environments.",
  "published": "2022-04-08",
  "updated": "2023-04-26",
  "year": "2022",
  "authors": [
   "Allison C. Tam",
   "Neil C. Rabinowitz",
   "Andrew K. Lampinen",
   "Nicholas A. Roy",
   "Stephanie C. Y. Chan",
   "DJ Strouse",
   "Jane X. Wang",
   "Andrea Banino",
   "Felix Hill"
  ],
  "author_count": 9,
  "categories": [
   "cs.LG",
   "cs.AI"
  ],
  "primary_category": "cs.LG",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 85,
  "influential_citations": 4,
  "tldr": "This work evaluates vision-language representations, pretrained on natural image captioning datasets, and shows that these pretrained representations drive meaningful, task-relevant exploration and improve performance on 3D simulated environments.",
  "doi": "10.48550/arXiv.2204.05080",
  "oa_pdf": "http://arxiv.org/pdf/2204.05080",
  "s2_authors": [
   {
    "name": "Allison C. Tam",
    "id": "2143234885",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Neil C. Rabinowitz",
    "id": "3422052",
    "h_index": 23,
    "papers": 46
   },
   {
    "name": "Andrew Kyle Lampinen",
    "id": "32322945",
    "h_index": 29,
    "papers": 70
   },
   {
    "name": "Nicholas A. Roy",
    "id": "153676637",
    "h_index": 11,
    "papers": 19
   },
   {
    "name": "Stephanie C. Y. Chan",
    "id": "50328436",
    "h_index": 16,
    "papers": 20
   },
   {
    "name": "D. Strouse",
    "id": "69925460",
    "h_index": 14,
    "papers": 21
   },
   {
    "name": "Jane X. Wang",
    "id": "2116439278",
    "h_index": 22,
    "papers": 28
   },
   {
    "name": "Andrea Banino",
    "id": "4194027",
    "h_index": 17,
    "papers": 26
   },
   {
    "name": "Felix Hill",
    "id": "145783676",
    "h_index": 42,
    "papers": 77
   }
  ],
  "comment": "NeurIPS 2022",
  "topics": [
   "rl-control",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2204.05080v3",
  "pdf_url": "https://arxiv.org/pdf/2204.05080v3",
  "html_url": "https://arxiv.org/html/2204.05080v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.43
 },
 {
  "id": "2204.03698",
  "slug": "learning-purely-tactile-in-hand-manipulation-with-a-torque-controlled",
  "title": "Learning Purely Tactile In-Hand Manipulation with a Torque-Controlled Hand",
  "abstract": "We show that a purely tactile dextrous in-hand manipulation task with continuous regrasping, requiring permanent force closure, can be learned from scratch and executed robustly on a torque-controlled humanoid robotic hand. The task is rotating a cube without dropping it, but in contrast to OpenAI's seminal cube manipulation task, the palm faces downwards and no cameras but only the hand's position and torque sensing are used. Although the task seems simple, it combines for the first time all the challenges in execution as well as learning that are important for using in-hand manipulation in real-world applications. We efficiently train in a precisely modeled and identified rigid body simulation with off-policy deep reinforcement learning, significantly sped up by a domain adapted curriculum, leading to a moderate 600 CPU hours of training time. The resulting policy is robustly transferred to the real humanoid DLR Hand-II, e.g., reaching more than 46 full 2$\u03c0$ rotations of the cube in a single run and allowing for disturbances like different cube sizes, hand orientation, or pulling a finger.",
  "published": "2022-04-07",
  "updated": "2022-04-12",
  "year": "2022",
  "authors": [
   "Leon Sievers",
   "Johannes Pitz",
   "Berthold B\u00e4uml"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA",
  "venue_source": "semantic-scholar",
  "citations": 56,
  "influential_citations": 2,
  "tldr": "It is shown that a purely tactile dextrous in-hand manipulation task with continuous regrasping, requiring permanent force closure, can be learned from scratch and executed robustly on a torque-controlled humanoid robotic hand.",
  "doi": "10.1109/ICRA46639.2022.9812093",
  "oa_pdf": "https://arxiv.org/pdf/2204.03698",
  "s2_authors": [
   {
    "name": "Leon Sievers",
    "id": "1742398486",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Johannes Pitz",
    "id": "15944810",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "B. B\u00e4uml",
    "id": "2846100",
    "h_index": 19,
    "papers": 42
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "humanoids",
   "tactile",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2204.03698v2",
  "pdf_url": "https://arxiv.org/pdf/2204.03698v2",
  "html_url": "https://arxiv.org/html/2204.03698v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.26
 },
 {
  "id": "2204.02320",
  "slug": "learning-generalizable-dexterous-manipulation-from-human-grasp-afforda",
  "title": "Learning Generalizable Dexterous Manipulation from Human Grasp Affordance",
  "abstract": "Dexterous manipulation with a multi-finger hand is one of the most challenging problems in robotics. While recent progress in imitation learning has largely improved the sample efficiency compared to Reinforcement Learning, the learned policy can hardly generalize to manipulate novel objects, given limited expert demonstrations. In this paper, we propose to learn dexterous manipulation using large-scale demonstrations with diverse 3D objects in a category, which are generated from a human grasp affordance model. This generalizes the policy to novel object instances within the same category. To train the policy, we propose a novel imitation learning objective jointly with a geometric representation learning objective using our demonstrations. By experimenting with relocating diverse objects in simulation, we show that our approach outperforms baselines with a large margin when manipulating novel objects. We also ablate the importance on 3D object representation learning for manipulation. We include videos, code, and additional information on the project website - https://kristery.github.io/ILAD/ .",
  "published": "2022-04-05",
  "updated": "2022-06-29",
  "year": "2022",
  "authors": [
   "Yueh-Hua Wu",
   "Jiashun Wang",
   "Xiaolong Wang"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 96,
  "influential_citations": 6,
  "tldr": "This paper proposes to learn dexterous manipulation using large-scale demonstrations with diverse 3D objects in a category, which are generated from a human grasp affordance model, and ablate the importance on 3D object representation learning for manipulation.",
  "doi": "10.48550/arXiv.2204.02320",
  "oa_pdf": "http://arxiv.org/pdf/2204.02320",
  "s2_authors": [
   {
    "name": "Yueh-Hua Wu",
    "id": "31609618",
    "h_index": 10,
    "papers": 13
   },
   {
    "name": "Jiashun Wang",
    "id": "2110144663",
    "h_index": 10,
    "papers": 13
   },
   {
    "name": "Xiaolong Wang",
    "id": "122024152",
    "h_index": 42,
    "papers": 63
   }
  ],
  "comment": "project page: https://kristery.github.io/ILAD/",
  "topics": [
   "dexterous-manipulation",
   "imitation-diffusion",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2204.02320v4",
  "pdf_url": "https://arxiv.org/pdf/2204.02320v4",
  "html_url": "https://arxiv.org/html/2204.02320v4",
  "code_url": "https://kristery.github.io/ILAD/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.49
 },
 {
  "id": "2204.01696",
  "slug": "joint-hand-motion-and-interaction-hotspots-prediction-from-egocentric",
  "title": "Joint Hand Motion and Interaction Hotspots Prediction from Egocentric Videos",
  "abstract": "We propose to forecast future hand-object interactions given an egocentric video. Instead of predicting action labels or pixels, we directly predict the hand motion trajectory and the future contact points on the next active object (i.e., interaction hotspots). This relatively low-dimensional representation provides a concrete description of future interactions. To tackle this task, we first provide an automatic way to collect trajectory and hotspots labels on large-scale data. We then use this data to train an Object-Centric Transformer (OCT) model for prediction. Our model performs hand and object interaction reasoning via the self-attention mechanism in Transformers. OCT also provides a probabilistic framework to sample the future trajectory and hotspots to handle uncertainty in prediction. We perform experiments on the Epic-Kitchens-55, Epic-Kitchens-100, and EGTEA Gaze+ datasets, and show that OCT significantly outperforms state-of-the-art approaches by a large margin. Project page is available at https://stevenlsw.github.io/hoi-forecast .",
  "published": "2022-04-04",
  "updated": "2022-04-04",
  "year": "2022",
  "authors": [
   "Shaowei Liu",
   "Subarna Tripathi",
   "Somdeb Majumdar",
   "Xiaolong Wang"
  ],
  "author_count": 4,
  "categories": [
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 146,
  "influential_citations": 24,
  "tldr": "This work directly predicts the hand motion trajectory and the future contact points on the next active object (i.e., interaction hotspots) through the self-attention mechanism in Transformers, and provides a probabilistic framework to sample the future trajectory and hotspots to handle uncertainty in prediction.",
  "doi": "10.1109/CVPR52688.2022.00328",
  "oa_pdf": "https://arxiv.org/pdf/2204.01696",
  "s2_authors": [
   {
    "name": "Shao-Wei Liu",
    "id": "48641958",
    "h_index": 13,
    "papers": 52
   },
   {
    "name": "Subarna Tripathi",
    "id": "2906509",
    "h_index": 25,
    "papers": 97
   },
   {
    "name": "Somdeb Majumdar",
    "id": "2413238",
    "h_index": 9,
    "papers": 28
   },
   {
    "name": "Xiaolong Wang",
    "id": "1709719",
    "h_index": 79,
    "papers": 805
   }
  ],
  "comment": "CVPR 2022, Project page: https://stevenlsw.github.io/hoi-forecast",
  "topics": [
   "egocentric-data"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2204.01696v1",
  "pdf_url": "https://arxiv.org/pdf/2204.01696v1",
  "html_url": "https://arxiv.org/html/2204.01696v1",
  "code_url": "https://stevenlsw.github.io/hoi-forecast",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.67
 },
 {
  "id": "2204.01691",
  "slug": "do-as-i-can-not-as-i-say-grounding-language-in-robotic-affordances",
  "title": "Do As I Can, Not As I Say: Grounding Language in Robotic Affordances",
  "abstract": "Large language models can encode a wealth of semantic knowledge about the world. Such knowledge could be extremely useful to robots aiming to act upon high-level, temporally extended instructions expressed in natural language. However, a significant weakness of language models is that they lack real-world experience, which makes it difficult to leverage them for decision making within a given embodiment. For example, asking a language model to describe how to clean a spill might result in a reasonable narrative, but it may not be applicable to a particular agent, such as a robot, that needs to perform this task in a particular environment. We propose to provide real-world grounding by means of pretrained skills, which are used to constrain the model to propose natural language actions that are both feasible and contextually appropriate. The robot can act as the language model's \"hands and eyes,\" while the language model supplies high-level semantic knowledge about the task. We show how low-level skills can be combined with large language models so that the language model provides high-level knowledge about the procedures for performing complex and temporally-extended instructions, while value functions associated with these skills provide the grounding necessary to connect this knowledge to a particular physical environment. We evaluate our method on a number of real-world robotic tasks, where we show the need for real-world grounding and that this approach is capable of completing long-horizon, abstract, natural language instructions on a mobile manipulator. The project's website and the video can be found at https://say-can.github.io/.",
  "published": "2022-04-04",
  "updated": "2022-08-16",
  "year": "2022",
  "authors": [
   "Michael Ahn",
   "Anthony Brohan",
   "Noah Brown",
   "Yevgen Chebotar",
   "Omar Cortes",
   "Byron David",
   "Chelsea Finn",
   "Chuyuan Fu",
   "Keerthana Gopalakrishnan",
   "Karol Hausman",
   "Alex Herzog",
   "Daniel Ho",
   "Jasmine Hsu",
   "Julian Ibarz",
   "Brian Ichter",
   "Alex Irpan",
   "Eric Jang",
   "Rosario Jauregui Ruano",
   "Kyle Jeffrey",
   "Sally Jesmonth",
   "Nikhil J Joshi",
   "Ryan Julian",
   "Dmitry Kalashnikov",
   "Yuheng Kuang",
   "Kuang-Huei Lee",
   "Sergey Levine",
   "Yao Lu",
   "Linda Luu",
   "Carolina Parada",
   "Peter Pastor",
   "Jornell Quiambao",
   "Kanishka Rao",
   "Jarek Rettinghouse",
   "Diego Reyes",
   "Pierre Sermanet",
   "Nicolas Sievers",
   "Clayton Tan",
   "Alexander Toshev",
   "Vincent Vanhoucke",
   "Fei Xia",
   "Ted Xiao",
   "Peng Xu",
   "Sichun Xu",
   "Mengyuan Yan",
   "Andy Zeng"
  ],
  "author_count": 45,
  "categories": [
   "cs.RO",
   "cs.CL",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 3571,
  "influential_citations": 213,
  "tldr": "This work proposes to provide real-world grounding by means of pretrained skills, which are used to constrain the model to propose natural language actions that are both feasible and contextually appropriate, and shows how low-level skills can be combined with large language models so that the language model provides high-level knowledge about the procedures for performing complex and temporally extended instructions.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Michael Ahn",
    "id": "2106194123",
    "h_index": 10,
    "papers": 15
   },
   {
    "name": "Anthony Brohan",
    "id": "118025075",
    "h_index": 12,
    "papers": 19
   },
   {
    "name": "Noah Brown",
    "id": "2161343011",
    "h_index": 10,
    "papers": 16
   },
   {
    "name": "Yevgen Chebotar",
    "id": "2527420",
    "h_index": 33,
    "papers": 57
   },
   {
    "name": "Omar Cortes",
    "id": "2161341260",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Byron David",
    "id": "2131683144",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Chelsea Finn",
    "id": "46881670",
    "h_index": 94,
    "papers": 240
   },
   {
    "name": "K. Gopalakrishnan",
    "id": "2161342233",
    "h_index": 17,
    "papers": 25
   },
   {
    "name": "Karol Hausman",
    "id": "1944801",
    "h_index": 47,
    "papers": 122
   },
   {
    "name": "Alexander Herzog",
    "id": "1505793452",
    "h_index": 22,
    "papers": 35
   },
   {
    "name": "Daniel Ho",
    "id": "2056459339",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "Jasmine Hsu",
    "id": "2726592",
    "h_index": 15,
    "papers": 20
   },
   {
    "name": "Julian Ibarz",
    "id": "46920727",
    "h_index": 22,
    "papers": 35
   },
   {
    "name": "Brian Ichter",
    "id": "2704814",
    "h_index": 37,
    "papers": 60
   },
   {
    "name": "A. Irpan",
    "id": "17818078",
    "h_index": 22,
    "papers": 32
   },
   {
    "name": "Eric Jang",
    "id": "145116380",
    "h_index": 20,
    "papers": 30
   },
   {
    "name": "Rosario M Jauregui Ruano",
    "id": "2161341254",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Kyle Jeffrey",
    "id": "2294363149",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Sally Jesmonth",
    "id": "2161341920",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "N. Joshi",
    "id": "2641664",
    "h_index": 20,
    "papers": 85
   },
   {
    "name": "Ryan C. Julian",
    "id": "144885996",
    "h_index": 19,
    "papers": 34
   },
   {
    "name": "Dmitry Kalashnikov",
    "id": "48313860",
    "h_index": 21,
    "papers": 34
   },
   {
    "name": "Yuheng Kuang",
    "id": "2161342687",
    "h_index": 9,
    "papers": 14
   },
   {
    "name": "Kuang-Huei Lee",
    "id": "2145145412",
    "h_index": 15,
    "papers": 19
   },
   {
    "name": "S. Levine",
    "id": "1736651",
    "h_index": 167,
    "papers": 551
   },
   {
    "name": "Yao Lu",
    "id": "2161346119",
    "h_index": 16,
    "papers": 22
   },
   {
    "name": "Linda Luu",
    "id": "13219952",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Carolina Parada",
    "id": "2057314286",
    "h_index": 23,
    "papers": 27
   },
   {
    "name": "P. Pastor",
    "id": "143970835",
    "h_index": 32,
    "papers": 47
   },
   {
    "name": "Jornell Quiambao",
    "id": "2161342191",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Kanishka Rao",
    "id": "2251957",
    "h_index": 31,
    "papers": 42
   },
   {
    "name": "Jarek Rettinghouse",
    "id": "2161341616",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "D. Reyes",
    "id": "48674590",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "P. Sermanet",
    "id": "3142556",
    "h_index": 39,
    "papers": 77
   },
   {
    "name": "Nicolas Sievers",
    "id": "2153300433",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Clayton Tan",
    "id": "2161386250",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Alexander Toshev",
    "id": "1726415",
    "h_index": 42,
    "papers": 82
   },
   {
    "name": "Vincent Vanhoucke",
    "id": "2657155",
    "h_index": 34,
    "papers": 60
   },
   {
    "name": "F. Xia",
    "id": "144956443",
    "h_index": 25,
    "papers": 31
   },
   {
    "name": "Ted Xiao",
    "id": "9961095",
    "h_index": 33,
    "papers": 45
   },
   {
    "name": "Peng Xu",
    "id": "2153917744",
    "h_index": 17,
    "papers": 22
   },
   {
    "name": "Sichun Xu",
    "id": "3068504",
    "h_index": 11,
    "papers": 16
   },
   {
    "name": "Mengyuan Yan",
    "id": "3235234",
    "h_index": 13,
    "papers": 20
   }
  ],
  "comment": "See website at https://say-can.github.io/ V1. Initial Upload. V2. Added PaLM results. Added study about new capabilities (drawer manipulation, chain of thought prompting, multilingual instructions). Added an ablation study of language model size. Added an open-source version of \\algname on a simulated tabletop environment. Improved readability",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2204.01691v2",
  "pdf_url": "https://arxiv.org/pdf/2204.01691v2",
  "html_url": "https://arxiv.org/html/2204.01691v2",
  "code_url": "https://say-can.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.5
 },
 {
  "id": "2203.17138",
  "slug": "imitate-and-repurpose-learning-reusable-robot-movement-skills-from-hum",
  "title": "Imitate and Repurpose: Learning Reusable Robot Movement Skills From Human and Animal Behaviors",
  "abstract": "We investigate the use of prior knowledge of human and animal movement to learn reusable locomotion skills for real legged robots. Our approach builds upon previous work on imitating human or dog Motion Capture (MoCap) data to learn a movement skill module. Once learned, this skill module can be reused for complex downstream tasks. Importantly, due to the prior imposed by the MoCap data, our approach does not require extensive reward engineering to produce sensible and natural looking behavior at the time of reuse. This makes it easy to create well-regularized, task-oriented controllers that are suitable for deployment on real robots. We demonstrate how our skill module can be used for imitation, and train controllable walking and ball dribbling policies for both the ANYmal quadruped and OP3 humanoid. These policies are then deployed on hardware via zero-shot simulation-to-reality transfer. Accompanying videos are available at https://bit.ly/robot-npmp.",
  "published": "2022-03-31",
  "updated": "2022-03-31",
  "year": "2022",
  "authors": [
   "Steven Bohez",
   "Saran Tunyasuvunakool",
   "Philemon Brakel",
   "Fereshteh Sadeghi",
   "Leonard Hasenclever",
   "Yuval Tassa",
   "Emilio Parisotto",
   "Jan Humplik",
   "Tuomas Haarnoja",
   "Roland Hafner",
   "Markus Wulfmeier",
   "Michael Neunert",
   "Ben Moran",
   "Noah Siegel",
   "Andrea Huber",
   "Francesco Romano",
   "Nathan Batchelor",
   "Federico Casarini",
   "Josh Merel",
   "Raia Hadsell",
   "Nicolas Heess"
  ],
  "author_count": 21,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 65,
  "influential_citations": 4,
  "tldr": "This work investigates the use of prior knowledge of human and animal movement to learn reusable locomotion skills for real legged robots, and demonstrates how this skill module can be used for imitation, and train controllable walking and ball dribbling policies for both the ANYmal quadruped and OP3 humanoid.",
  "doi": "10.48550/arXiv.2203.17138",
  "oa_pdf": "http://arxiv.org/pdf/2203.17138",
  "s2_authors": [
   {
    "name": "Steven Bohez",
    "id": "1832575",
    "h_index": 20,
    "papers": 45
   },
   {
    "name": "S. Tunyasuvunakool",
    "id": "47985172",
    "h_index": 18,
    "papers": 26
   },
   {
    "name": "Philemon Brakel",
    "id": "2616163",
    "h_index": 22,
    "papers": 32
   },
   {
    "name": "Fereshteh Sadeghi",
    "id": "3253737",
    "h_index": 15,
    "papers": 31
   },
   {
    "name": "Leonard Hasenclever",
    "id": "40401956",
    "h_index": 28,
    "papers": 50
   },
   {
    "name": "Yuval Tassa",
    "id": "2109481",
    "h_index": 39,
    "papers": 67
   },
   {
    "name": "Emilio Parisotto",
    "id": "3166516",
    "h_index": 26,
    "papers": 44
   },
   {
    "name": "Jan Humplik",
    "id": "2066450521",
    "h_index": 11,
    "papers": 22
   },
   {
    "name": "Tuomas Haarnoja",
    "id": "2587648",
    "h_index": 15,
    "papers": 31
   },
   {
    "name": "Roland Hafner",
    "id": "49512734",
    "h_index": 21,
    "papers": 36
   },
   {
    "name": "Markus Wulfmeier",
    "id": "3331786",
    "h_index": 25,
    "papers": 64
   },
   {
    "name": "M. Neunert",
    "id": "2366050",
    "h_index": 29,
    "papers": 50
   },
   {
    "name": "Ben Moran",
    "id": "2160887670",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Noah Siegel",
    "id": "1500370330",
    "h_index": 15,
    "papers": 19
   },
   {
    "name": "Andrea Huber",
    "id": "2054151655",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Francesco Romano",
    "id": "145549279",
    "h_index": 17,
    "papers": 41
   },
   {
    "name": "Nathan Batchelor",
    "id": "2150504360",
    "h_index": 8,
    "papers": 10
   },
   {
    "name": "F. Casarini",
    "id": "2132053697",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "J. Merel",
    "id": "1879232",
    "h_index": 28,
    "papers": 51
   },
   {
    "name": "R. Hadsell",
    "id": "2315504",
    "h_index": 51,
    "papers": 111
   },
   {
    "name": "N. Heess",
    "id": "2801204",
    "h_index": 73,
    "papers": 192
   }
  ],
  "comment": "30 pages, 9 figures, 8 tables, 14 videos at https://bit.ly/robot-npmp , submitted to Science Robotics",
  "topics": [
   "humanoids",
   "egocentric-data"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2203.17138v1",
  "pdf_url": "https://arxiv.org/pdf/2203.17138v1",
  "html_url": "https://arxiv.org/html/2203.17138v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.82
 },
 {
  "id": "2203.15103",
  "slug": "adversarial-motion-priors-make-good-substitutes-for-complex-reward-fun",
  "title": "Adversarial Motion Priors Make Good Substitutes for Complex Reward Functions",
  "abstract": "Training a high-dimensional simulated agent with an under-specified reward function often leads the agent to learn physically infeasible strategies that are ineffective when deployed in the real world. To mitigate these unnatural behaviors, reinforcement learning practitioners often utilize complex reward functions that encourage physically plausible behaviors. However, a tedious labor-intensive tuning process is often required to create hand-designed rewards which might not easily generalize across platforms and tasks. We propose substituting complex reward functions with \"style rewards\" learned from a dataset of motion capture demonstrations. A learned style reward can be combined with an arbitrary task reward to train policies that perform tasks using naturalistic strategies. These natural strategies can also facilitate transfer to the real world. We build upon Adversarial Motion Priors -- an approach from the computer graphics domain that encodes a style reward from a dataset of reference motions -- to demonstrate that an adversarial approach to training policies can produce behaviors that transfer to a real quadrupedal robot without requiring complex reward functions. We also demonstrate that an effective style reward can be learned from a few seconds of motion capture data gathered from a German Shepherd and leads to energy-efficient locomotion strategies with natural gait transitions.",
  "published": "2022-03-28",
  "updated": "2022-03-28",
  "year": "2022",
  "authors": [
   "Alejandro Escontrela",
   "Xue Bin Peng",
   "Wenhao Yu",
   "Tingnan Zhang",
   "Atil Iscen",
   "Ken Goldberg",
   "Pieter Abbeel"
  ],
  "author_count": 7,
  "categories": [
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.AI",
  "venue": "IROS",
  "venue_source": "semantic-scholar",
  "citations": 191,
  "influential_citations": 17,
  "tldr": "It is demonstrated that an adversarial approach to training policies can produce behaviors that transfer to a real quadrupedal robot without requiring complex reward functions, and an effective style reward can be learned from a few seconds of motion capture data gathered from a German Shepherd.",
  "doi": "10.1109/IROS47612.2022.9981973",
  "oa_pdf": "https://arxiv.org/pdf/2203.15103",
  "s2_authors": [
   {
    "name": "Alejandro Escontrela",
    "id": "2008560299",
    "h_index": 11,
    "papers": 17
   },
   {
    "name": "Xue Bin Peng",
    "id": "2375236722",
    "h_index": 34,
    "papers": 41
   },
   {
    "name": "Wenhao Yu",
    "id": "70461341",
    "h_index": 35,
    "papers": 93
   },
   {
    "name": "Tingnan Zhang",
    "id": "28292148",
    "h_index": 26,
    "papers": 53
   },
   {
    "name": "Atil Iscen",
    "id": "2106754",
    "h_index": 21,
    "papers": 41
   },
   {
    "name": "Ken Goldberg",
    "id": "144344283",
    "h_index": 90,
    "papers": 647
   },
   {
    "name": "P. Abbeel",
    "id": "1689992",
    "h_index": 165,
    "papers": 596
   }
  ],
  "comment": "8 pages, 6 figures, 3 tables",
  "topics": [
   "humanoids",
   "egocentric-data",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2203.15103v1",
  "pdf_url": "https://arxiv.org/pdf/2203.15103v1",
  "html_url": "https://arxiv.org/html/2203.15103v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.78
 },
 {
  "id": "2203.13251",
  "slug": "dexterous-imitation-made-easy-a-learning-based-framework-for-efficient",
  "title": "Dexterous Imitation Made Easy: A Learning-Based Framework for Efficient Dexterous Manipulation",
  "abstract": "Optimizing behaviors for dexterous manipulation has been a longstanding challenge in robotics, with a variety of methods from model-based control to model-free reinforcement learning having been previously explored in literature. Perhaps one of the most powerful techniques to learn complex manipulation strategies is imitation learning. However, collecting and learning from demonstrations in dexterous manipulation is quite challenging. The complex, high-dimensional action-space involved with multi-finger control often leads to poor sample efficiency of learning-based methods. In this work, we propose 'Dexterous Imitation Made Easy' (DIME) a new imitation learning framework for dexterous manipulation. DIME only requires a single RGB camera to observe a human operator and teleoperate our robotic hand. Once demonstrations are collected, DIME employs standard imitation learning methods to train dexterous manipulation policies. On both simulation and real robot benchmarks we demonstrate that DIME can be used to solve complex, in-hand manipulation tasks such as 'flipping', 'spinning', and 'rotating' objects with the Allegro hand. Our framework along with pre-collected demonstrations is publicly available at https://nyu-robot-learning.github.io/dime.",
  "published": "2022-03-24",
  "updated": "2022-03-24",
  "year": "2022",
  "authors": [
   "Sridhar Pandian Arunachalam",
   "Sneha Silwal",
   "Ben Evans",
   "Lerrel Pinto"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA",
  "venue_source": "semantic-scholar",
  "citations": 151,
  "influential_citations": 2,
  "tldr": "\u2018Dexterous Imitation Made Easy\u2019 (DIME) is proposed, a new imitation learning framework for dexterous manipulation that only requires a single RGB camera that observes a human operator to teleoperate a robotic hand and employs state-of-the-art imitation learning methods to train dexterous manipulate policies.",
  "doi": "10.1109/ICRA48891.2023.10160275",
  "oa_pdf": "https://arxiv.org/pdf/2203.13251",
  "s2_authors": [
   {
    "name": "Sridhar Pandian Arunachalam",
    "id": "2144825236",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "S. Silwal",
    "id": "2159712500",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Ben Evans",
    "id": "2153473632",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Lerrel Pinto",
    "id": "34026610",
    "h_index": 41,
    "papers": 70
   }
  ],
  "comment": "The first two authors contributed equally",
  "topics": [
   "dexterous-manipulation",
   "imitation-diffusion",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2203.13251v1",
  "pdf_url": "https://arxiv.org/pdf/2203.13251v1",
  "html_url": "https://arxiv.org/html/2203.13251v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.68
 },
 {
  "id": "2203.13131",
  "slug": "make-a-scene-scene-based-text-to-image-generation-with-human-priors",
  "title": "Make-A-Scene: Scene-Based Text-to-Image Generation with Human Priors",
  "abstract": "Recent text-to-image generation methods provide a simple yet exciting conversion capability between text and image domains. While these methods have incrementally improved the generated image fidelity and text relevancy, several pivotal gaps remain unanswered, limiting applicability and quality. We propose a novel text-to-image method that addresses these gaps by (i) enabling a simple control mechanism complementary to text in the form of a scene, (ii) introducing elements that substantially improve the tokenization process by employing domain-specific knowledge over key image regions (faces and salient objects), and (iii) adapting classifier-free guidance for the transformer use case. Our model achieves state-of-the-art FID and human evaluation results, unlocking the ability to generate high fidelity images in a resolution of 512x512 pixels, significantly improving visual quality. Through scene controllability, we introduce several new capabilities: (i) Scene editing, (ii) text editing with anchor scenes, (iii) overcoming out-of-distribution text prompts, and (iv) story illustration generation, as demonstrated in the story we wrote.",
  "published": "2022-03-24",
  "updated": "2022-03-24",
  "year": "2022",
  "authors": [
   "Oran Gafni",
   "Adam Polyak",
   "Oron Ashual",
   "Shelly Sheynin",
   "Devi Parikh",
   "Yaniv Taigman"
  ],
  "author_count": 6,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.CL",
   "cs.GR",
   "cs.LG"
  ],
  "primary_category": "cs.CV",
  "venue": "ECCV",
  "venue_source": "semantic-scholar",
  "citations": 643,
  "influential_citations": 38,
  "tldr": "This work proposes a novel text-to-image method that addresses gaps in applicability and quality by enabling a simple control mechanism complementary to text in the form of a scene, and introducing elements that substantially improve the tokenization process by employing domain-specific knowledge over key image regions.",
  "doi": "10.48550/arXiv.2203.13131",
  "oa_pdf": "http://arxiv.org/pdf/2203.13131",
  "s2_authors": [
   {
    "name": "Oran Gafni",
    "id": "90840812",
    "h_index": 14,
    "papers": 16
   },
   {
    "name": "Adam Polyak",
    "id": "33964593",
    "h_index": 28,
    "papers": 38
   },
   {
    "name": "Oron Ashual",
    "id": "1388005058",
    "h_index": 8,
    "papers": 10
   },
   {
    "name": "Shelly Sheynin",
    "id": "2086827528",
    "h_index": 12,
    "papers": 14
   },
   {
    "name": "Devi Parikh",
    "id": "153432684",
    "h_index": 82,
    "papers": 238
   },
   {
    "name": "Yaniv Taigman",
    "id": "2188620",
    "h_index": 30,
    "papers": 41
   }
  ],
  "comment": "",
  "topics": [
   "video-generation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2203.13131v1",
  "pdf_url": "https://arxiv.org/pdf/2203.13131v1",
  "html_url": "https://arxiv.org/html/2203.13131v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.31
 },
 {
  "id": "2203.12601",
  "slug": "r3m-a-universal-visual-representation-for-robot-manipulation",
  "title": "R3M: A Universal Visual Representation for Robot Manipulation",
  "abstract": "We study how visual representations pre-trained on diverse human video data can enable data-efficient learning of downstream robotic manipulation tasks. Concretely, we pre-train a visual representation using the Ego4D human video dataset using a combination of time-contrastive learning, video-language alignment, and an L1 penalty to encourage sparse and compact representations. The resulting representation, R3M, can be used as a frozen perception module for downstream policy learning. Across a suite of 12 simulated robot manipulation tasks, we find that R3M improves task success by over 20% compared to training from scratch and by over 10% compared to state-of-the-art visual representations like CLIP and MoCo. Furthermore, R3M enables a Franka Emika Panda arm to learn a range of manipulation tasks in a real, cluttered apartment given just 20 demonstrations. Code and pre-trained models are available at https://tinyurl.com/robotr3m.",
  "published": "2022-03-23",
  "updated": "2022-11-18",
  "year": "2022",
  "authors": [
   "Suraj Nair",
   "Aravind Rajeswaran",
   "Vikash Kumar",
   "Chelsea Finn",
   "Abhinav Gupta"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 970,
  "influential_citations": 103,
  "tldr": "R3M enables a Franka Emika Panda arm to learn a range of manipulation tasks in a real, cluttered apartment given just 20 demonstrations and can be used as a frozen perception module for downstream policy learning.",
  "doi": "10.48550/arXiv.2203.12601",
  "oa_pdf": "https://arxiv.org/pdf/2203.12601",
  "s2_authors": [
   {
    "name": "Suraj Nair",
    "id": "4734949",
    "h_index": 20,
    "papers": 36
   },
   {
    "name": "A. Rajeswaran",
    "id": "19275599",
    "h_index": 34,
    "papers": 58
   },
   {
    "name": "Vikash Kumar",
    "id": "2109446216",
    "h_index": 38,
    "papers": 76
   },
   {
    "name": "Chelsea Finn",
    "id": "46881670",
    "h_index": 94,
    "papers": 240
   },
   {
    "name": "Abhi Gupta",
    "id": "2117767136",
    "h_index": 16,
    "papers": 24
   }
  ],
  "comment": "Conference on Robot Learning (CoRL) 2022",
  "topics": [
   "egocentric-data"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2203.12601v3",
  "pdf_url": "https://arxiv.org/pdf/2203.12601v3",
  "html_url": "https://arxiv.org/html/2203.12601v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.49
 },
 {
  "id": "2203.10638",
  "slug": "v2x-vit-vehicle-to-everything-cooperative-perception-with-vision-trans",
  "title": "V2X-ViT: Vehicle-to-Everything Cooperative Perception with Vision Transformer",
  "abstract": "In this paper, we investigate the application of Vehicle-to-Everything (V2X) communication to improve the perception performance of autonomous vehicles. We present a robust cooperative perception framework with V2X communication using a novel vision Transformer. Specifically, we build a holistic attention model, namely V2X-ViT, to effectively fuse information across on-road agents (i.e., vehicles and infrastructure). V2X-ViT consists of alternating layers of heterogeneous multi-agent self-attention and multi-scale window self-attention, which captures inter-agent interaction and per-agent spatial relationships. These key modules are designed in a unified Transformer architecture to handle common V2X challenges, including asynchronous information sharing, pose errors, and heterogeneity of V2X components. To validate our approach, we create a large-scale V2X perception dataset using CARLA and OpenCDA. Extensive experimental results demonstrate that V2X-ViT sets new state-of-the-art performance for 3D object detection and achieves robust performance even under harsh, noisy environments. The code is available at https://github.com/DerrickXuNu/v2x-vit.",
  "published": "2022-03-20",
  "updated": "2022-08-08",
  "year": "2022",
  "authors": [
   "Runsheng Xu",
   "Hao Xiang",
   "Zhengzhong Tu",
   "Xin Xia",
   "Ming-Hsuan Yang",
   "Jiaqi Ma"
  ],
  "author_count": 6,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "ECCV",
  "venue_source": "semantic-scholar",
  "citations": 717,
  "influential_citations": 166,
  "tldr": "A robust cooperative perception framework with V2X communication using a novel vision Transformer that sets new state-of-the-art performance for 3D object detection and achieves robust performance even under harsh, noisy environments.",
  "doi": "10.48550/arXiv.2203.10638",
  "oa_pdf": "http://arxiv.org/pdf/2203.10638",
  "s2_authors": [
   {
    "name": "Runsheng Xu",
    "id": "47462785",
    "h_index": 26,
    "papers": 41
   },
   {
    "name": "Hao Xiang",
    "id": "2119233630",
    "h_index": 15,
    "papers": 21
   },
   {
    "name": "Zhengzhong Tu",
    "id": "40992714",
    "h_index": 19,
    "papers": 40
   },
   {
    "name": "Xin Xia",
    "id": "2150058228",
    "h_index": 21,
    "papers": 33
   },
   {
    "name": "Ming-Hsuan Yang",
    "id": "37144787",
    "h_index": 72,
    "papers": 223
   },
   {
    "name": "Jiaqi Ma",
    "id": "2146393082",
    "h_index": 26,
    "papers": 52
   }
  ],
  "comment": "ECCV 2022. Code: https://github.com/DerrickXuNu/v2x-vit",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2203.10638v3",
  "pdf_url": "https://arxiv.org/pdf/2203.10638v3",
  "html_url": "https://arxiv.org/html/2203.10638v3",
  "code_url": "https://github.com/DerrickXuNu/v2x-vit",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.36
 },
 {
  "id": "2203.12677",
  "slug": "vision-based-manipulators-need-to-also-see-from-their-hands",
  "title": "Vision-Based Manipulators Need to Also See from Their Hands",
  "abstract": "We study how the choice of visual perspective affects learning and generalization in the context of physical manipulation from raw sensor observations. Compared with the more commonly used global third-person perspective, a hand-centric (eye-in-hand) perspective affords reduced observability, but we find that it consistently improves training efficiency and out-of-distribution generalization. These benefits hold across a variety of learning algorithms, experimental settings, and distribution shifts, and for both simulated and real robot apparatuses. However, this is only the case when hand-centric observability is sufficient; otherwise, including a third-person perspective is necessary for learning, but also harms out-of-distribution generalization. To mitigate this, we propose to regularize the third-person information stream via a variational information bottleneck. On six representative manipulation tasks with varying hand-centric observability adapted from the Meta-World benchmark, this results in a state-of-the-art reinforcement learning agent operating from both perspectives improving its out-of-distribution generalization on every task. While some practitioners have long put cameras in the hands of robots, our work systematically analyzes the benefits of doing so and provides simple and broadly applicable insights for improving end-to-end learned vision-based robotic manipulation.",
  "published": "2022-03-15",
  "updated": "2022-03-15",
  "year": "2022",
  "authors": [
   "Kyle Hsu",
   "Moo Jin Kim",
   "Rafael Rafailov",
   "Jiajun Wu",
   "Chelsea Finn"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 63,
  "influential_citations": 5,
  "tldr": "This work systematically analyzes the benefits of putting cameras in the hands of robots and provides simple and broadly applicable insights for improving end-to-end learned vision-based robotic manipulation.",
  "doi": "10.48550/arXiv.2203.12677",
  "oa_pdf": "http://arxiv.org/pdf/2203.12677",
  "s2_authors": [
   {
    "name": "Kyle Hsu",
    "id": "32028215",
    "h_index": 9,
    "papers": 11
   },
   {
    "name": "Moo Jin Kim",
    "id": "2159987907",
    "h_index": 10,
    "papers": 10
   },
   {
    "name": "Rafael Rafailov",
    "id": "102801230",
    "h_index": 25,
    "papers": 44
   },
   {
    "name": "Jiajun Wu",
    "id": "3045089",
    "h_index": 80,
    "papers": 228
   },
   {
    "name": "Chelsea Finn",
    "id": "46881670",
    "h_index": 94,
    "papers": 240
   }
  ],
  "comment": "First two authors contributed equally. ICLR 2022 (oral) camera-ready. 30 pages, 20 figures. Project website: https://sites.google.com/view/seeing-from-hands",
  "topics": [
   "dexterous-manipulation",
   "rl-control",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2203.12677v1",
  "pdf_url": "https://arxiv.org/pdf/2203.12677v1",
  "html_url": "https://arxiv.org/html/2203.12677v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.31
 },
 {
  "id": "2203.06972",
  "slug": "icub3-avatar-system-enabling-remote-fully-immersive-embodiment-of-huma",
  "title": "iCub3 Avatar System: Enabling Remote Fully-Immersive Embodiment of Humanoid Robots",
  "abstract": "We present an avatar system designed to facilitate the embodiment of humanoid robots by human operators, validated through iCub3, a humanoid developed at the Istituto Italiano di Tecnologia (IIT). More precisely, the contribution of the paper is twofold: first, we present the humanoid iCub3 as a robotic avatar which integrates the latest significant improvements after about fifteen years of development of the iCub series; second, we present a versatile avatar system enabling humans to embody humanoid robots encompassing aspects such as locomotion, manipulation, voice, and face expressions with comprehensive sensory feedback including visual, auditory, haptic, weight, and touch modalities. We validate the system by implementing several avatar architecture instances, each tailored to specific requirements. First, we evaluated the optimized architecture for verbal, non-verbal, and physical interactions with a remote recipient. This testing involved the operator in Genoa and the avatar in the Biennale di Venezia, Venice - about 290 Km away - thus allowing the operator to visit remotely the Italian art exhibition. Second, we evaluated the optimised architecture for recipient physical collaboration and public engagement on-stage, live, at the We Make Future show, a prominent world digital innovation festival. In this instance, the operator was situated in Genoa while the avatar operates in Rimini - about 300 Km away - interacting with a recipient who entrusted the avatar a payload to carry on stage before an audience of approximately 2000 spectators. Third, we present the architecture implemented by the iCub Team for the ANA Avatar XPrize competition.",
  "published": "2022-03-14",
  "updated": "2024-01-25",
  "year": "2022",
  "authors": [
   "Stefano Dafarra",
   "Ugo Pattacini",
   "Giulio Romualdi",
   "Lorenzo Rapetti",
   "Riccardo Grieco",
   "Kourosh Darvish",
   "Gianluca Milani",
   "Enrico Valli",
   "Ines Sorrentino",
   "Paolo Maria Viceconte",
   "Alessandro Scalzo",
   "Silvio Traversaro",
   "Carlotta Sartore",
   "Mohamed Elobaid",
   "Nuno Guedelha",
   "Connor Herron",
   "Alexander Leonessa",
   "Francesco Draicchio",
   "Giorgio Metta",
   "Marco Maggiali",
   "Daniele Pucci"
  ],
  "author_count": 21,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "Science Robotics",
  "venue_source": "semantic-scholar",
  "citations": 81,
  "influential_citations": 2,
  "tldr": "The humanoid iCub3 is presented as a robotic avatar that integrates the latest significant improvements after about 15 years of development of the iCub series and is validated through iCub3, a humanoid developed at the Istituto Italiano di Tecnologia.",
  "doi": "10.1126/scirobotics.adh3834",
  "oa_pdf": "https://hdl.handle.net/11573/1717558",
  "s2_authors": [
   {
    "name": "Stefano Dafarra",
    "id": "8458022",
    "h_index": 15,
    "papers": 43
   },
   {
    "name": "K. Darvish",
    "id": "34308317",
    "h_index": 16,
    "papers": 47
   },
   {
    "name": "Riccardo Grieco",
    "id": "2158817466",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Gianluca Milani",
    "id": "2158817562",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "U. Pattacini",
    "id": "2212120",
    "h_index": 24,
    "papers": 66
   },
   {
    "name": "Lorenzo Rapetti",
    "id": "27558790",
    "h_index": 13,
    "papers": 32
   },
   {
    "name": "Giulio Romualdi",
    "id": "51451149",
    "h_index": 12,
    "papers": 32
   },
   {
    "name": "Mattia Salvi",
    "id": "2143658899",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "A. Scalzo",
    "id": "34851539",
    "h_index": 9,
    "papers": 15
   },
   {
    "name": "Ines Sorrentino",
    "id": "1380588792",
    "h_index": 7,
    "papers": 15
   },
   {
    "name": "David Tom\u00e9",
    "id": "8039501",
    "h_index": 10,
    "papers": 37
   },
   {
    "name": "Silvio Traversaro",
    "id": "2255650",
    "h_index": 22,
    "papers": 85
   },
   {
    "name": "Enrico Valli",
    "id": "37137034",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Paolo Maria Viceconte",
    "id": "2086827197",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "G. Metta",
    "id": "1698471",
    "h_index": 64,
    "papers": 415
   },
   {
    "name": "M. Maggiali",
    "id": "1934462",
    "h_index": 20,
    "papers": 61
   },
   {
    "name": "D. Pucci",
    "id": "2202742",
    "h_index": 26,
    "papers": 139
   }
  ],
  "comment": "This is the author's version of the work. It is posted here by permission of the AAAS for personal use, not for redistribution. The definitive version was published in https://www.science.org/doi/10.1126/scirobotics.adh3834 on January 24th 2024, DOI: 10.1126/scirobotics.adh3834",
  "topics": [
   "humanoids"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2203.06972v2",
  "pdf_url": "https://arxiv.org/pdf/2203.06972v2",
  "html_url": "https://arxiv.org/html/2203.06972v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.41
 },
 {
  "id": "2203.06173",
  "slug": "masked-visual-pre-training-for-motor-control",
  "title": "Masked Visual Pre-training for Motor Control",
  "abstract": "This paper shows that self-supervised visual pre-training from real-world images is effective for learning motor control tasks from pixels. We first train the visual representations by masked modeling of natural images. We then freeze the visual encoder and train neural network controllers on top with reinforcement learning. We do not perform any task-specific fine-tuning of the encoder; the same visual representations are used for all motor control tasks. To the best of our knowledge, this is the first self-supervised model to exploit real-world images at scale for motor control. To accelerate progress in learning from pixels, we contribute a benchmark suite of hand-designed tasks varying in movements, scenes, and robots. Without relying on labels, state-estimation, or expert demonstrations, we consistently outperform supervised encoders by up to 80% absolute success rate, sometimes even matching the oracle state performance. We also find that in-the-wild images, e.g., from YouTube or Egocentric videos, lead to better visual representations for various manipulation tasks than ImageNet images.",
  "published": "2022-03-11",
  "updated": "2022-03-11",
  "year": "2022",
  "authors": [
   "Tete Xiao",
   "Ilija Radosavovic",
   "Trevor Darrell",
   "Jitendra Malik"
  ],
  "author_count": 4,
  "categories": [
   "cs.CV",
   "cs.LG",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 329,
  "influential_citations": 33,
  "tldr": "This paper shows that self-supervised visual pre-training from real-world images is effective for learning motor control tasks from pixels, and is the first self- supervised model to exploit real- world images at scale for motor control.",
  "doi": "10.48550/arXiv.2203.06173",
  "oa_pdf": "http://arxiv.org/pdf/2203.06173",
  "s2_authors": [
   {
    "name": "Tete Xiao",
    "id": "15727192",
    "h_index": 17,
    "papers": 21
   },
   {
    "name": "Ilija Radosavovic",
    "id": "30407997",
    "h_index": 20,
    "papers": 22
   },
   {
    "name": "Trevor Darrell",
    "id": "1753210",
    "h_index": 158,
    "papers": 630
   },
   {
    "name": "J. Malik",
    "id": "153652147",
    "h_index": 64,
    "papers": 99
   }
  ],
  "comment": "Code and videos at: https://tetexiao.com/projects/mvp",
  "topics": [
   "egocentric-data",
   "rl-control",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2203.06173v1",
  "pdf_url": "https://arxiv.org/pdf/2203.06173v1",
  "html_url": "https://arxiv.org/html/2203.06173v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.52
 },
 {
  "id": "2203.03605",
  "slug": "dino-detr-with-improved-denoising-anchor-boxes-for-end-to-end-object-d",
  "title": "DINO: DETR with Improved DeNoising Anchor Boxes for End-to-End Object Detection",
  "abstract": "We present DINO (\\textbf{D}ETR with \\textbf{I}mproved de\\textbf{N}oising anch\\textbf{O}r boxes), a state-of-the-art end-to-end object detector. % in this paper. DINO improves over previous DETR-like models in performance and efficiency by using a contrastive way for denoising training, a mixed query selection method for anchor initialization, and a look forward twice scheme for box prediction. DINO achieves $49.4$AP in $12$ epochs and $51.3$AP in $24$ epochs on COCO with a ResNet-50 backbone and multi-scale features, yielding a significant improvement of $\\textbf{+6.0}$\\textbf{AP} and $\\textbf{+2.7}$\\textbf{AP}, respectively, compared to DN-DETR, the previous best DETR-like model. DINO scales well in both model size and data size. Without bells and whistles, after pre-training on the Objects365 dataset with a SwinL backbone, DINO obtains the best results on both COCO \\texttt{val2017} ($\\textbf{63.2}$\\textbf{AP}) and \\texttt{test-dev} (\\textbf{$\\textbf{63.3}$AP}). Compared to other models on the leaderboard, DINO significantly reduces its model size and pre-training data size while achieving better results. Our code will be available at \\url{https://github.com/IDEACVR/DINO}.",
  "published": "2022-03-07",
  "updated": "2022-07-11",
  "year": "2022",
  "authors": [
   "Hao Zhang",
   "Feng Li",
   "Shilong Liu",
   "Lei Zhang",
   "Hang Su",
   "Jun Zhu",
   "Lionel M. Ni",
   "Heung-Yeung Shum"
  ],
  "author_count": 8,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 3062,
  "influential_citations": 350,
  "tldr": "DINO improves over previous DETR-like models in performance and efficiency by using a contrastive way for denoising training, a mixed query selection method for anchor initialization, and a look forward twice scheme for box prediction.",
  "doi": "10.48550/arXiv.2203.03605",
  "oa_pdf": "http://arxiv.org/pdf/2203.03605",
  "s2_authors": [
   {
    "name": "Hao Zhang",
    "id": "2315254849",
    "h_index": 22,
    "papers": 36
   },
   {
    "name": "Feng Li",
    "id": "2146312758",
    "h_index": 4,
    "papers": 9
   },
   {
    "name": "Shilong Liu",
    "id": "8602739",
    "h_index": 20,
    "papers": 36
   },
   {
    "name": "Lei Zhang",
    "id": "2152834943",
    "h_index": 11,
    "papers": 18
   },
   {
    "name": "Hang Su",
    "id": "2093561216",
    "h_index": 52,
    "papers": 136
   },
   {
    "name": "Jun-Juan Zhu",
    "id": "89006344",
    "h_index": 12,
    "papers": 19
   },
   {
    "name": "L. Ni",
    "id": "1726587",
    "h_index": 80,
    "papers": 511
   },
   {
    "name": "H. Shum",
    "id": "93596028",
    "h_index": 34,
    "papers": 149
   }
  ],
  "comment": "",
  "topics": [
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2203.03605v4",
  "pdf_url": "https://arxiv.org/pdf/2203.03605v4",
  "html_url": "https://arxiv.org/html/2203.03605v4",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "2203.02638",
  "slug": "safe-reinforcement-learning-for-legged-locomotion",
  "title": "Safe Reinforcement Learning for Legged Locomotion",
  "abstract": "Designing control policies for legged locomotion is complex due to the under-actuated and non-continuous robot dynamics. Model-free reinforcement learning provides promising tools to tackle this challenge. However, a major bottleneck of applying model-free reinforcement learning in real world is safety. In this paper, we propose a safe reinforcement learning framework that switches between a safe recovery policy that prevents the robot from entering unsafe states, and a learner policy that is optimized to complete the task. The safe recovery policy takes over the control when the learner policy violates safety constraints, and hands over the control back when there are no future safety violations. We design the safe recovery policy so that it ensures safety of legged locomotion while minimally intervening in the learning process. Furthermore, we theoretically analyze the proposed framework and provide an upper bound on the task performance. We verify the proposed framework in four locomotion tasks on a simulated and real quadrupedal robot: efficient gait, catwalk, two-leg balance, and pacing. On average, our method achieves 48.6% fewer falls and comparable or better rewards than the baseline methods in simulation. When deployed it on real-world quadruped robot, our training pipeline enables 34% improvement in energy efficiency for the efficient gait, 40.9% narrower of the feet placement in the catwalk, and two times more jumping duration in the two-leg balance. Our method achieves less than five falls over the duration of 115 minutes of hardware time.",
  "published": "2022-03-05",
  "updated": "2022-03-05",
  "year": "2022",
  "authors": [
   "Tsung-Yen Yang",
   "Tingnan Zhang",
   "Linda Luu",
   "Sehoon Ha",
   "Jie Tan",
   "Wenhao Yu"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS",
  "venue_source": "semantic-scholar",
  "citations": 61,
  "influential_citations": 3,
  "tldr": "This paper designs the safe recovery policy so that it ensures safety of quadruped locomotion while minimally intervening in the learning process, and theoretically analyze the proposed framework and provide an upper bound on the task performance.",
  "doi": "10.1109/IROS47612.2022.9982038",
  "oa_pdf": "http://arxiv.org/pdf/2203.02638",
  "s2_authors": [
   {
    "name": "Tsung-Yen Yang",
    "id": "11844404",
    "h_index": 12,
    "papers": 17
   },
   {
    "name": "Tingnan Zhang",
    "id": "28292148",
    "h_index": 26,
    "papers": 53
   },
   {
    "name": "Linda Luu",
    "id": "13219952",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Sehoon Ha",
    "id": "2248552",
    "h_index": 26,
    "papers": 69
   },
   {
    "name": "Jie Tan",
    "id": "1739176520",
    "h_index": 37,
    "papers": 68
   },
   {
    "name": "Wenhao Yu",
    "id": "70461341",
    "h_index": 35,
    "papers": 93
   }
  ],
  "comment": "Video is included in the submission and the project website: https://sites.google.com/view/saferlleggedlocomotion/",
  "topics": [
   "humanoids",
   "rl-control",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2203.02638v1",
  "pdf_url": "https://arxiv.org/pdf/2203.02638v1",
  "html_url": "https://arxiv.org/html/2203.02638v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.29
 },
 {
  "id": "2203.02155",
  "slug": "training-language-models-to-follow-instructions-with-human-feedback",
  "title": "Training language models to follow instructions with human feedback",
  "abstract": "Making language models bigger does not inherently make them better at following a user's intent. For example, large language models can generate outputs that are untruthful, toxic, or simply not helpful to the user. In other words, these models are not aligned with their users. In this paper, we show an avenue for aligning language models with user intent on a wide range of tasks by fine-tuning with human feedback. Starting with a set of labeler-written prompts and prompts submitted through the OpenAI API, we collect a dataset of labeler demonstrations of the desired model behavior, which we use to fine-tune GPT-3 using supervised learning. We then collect a dataset of rankings of model outputs, which we use to further fine-tune this supervised model using reinforcement learning from human feedback. We call the resulting models InstructGPT. In human evaluations on our prompt distribution, outputs from the 1.3B parameter InstructGPT model are preferred to outputs from the 175B GPT-3, despite having 100x fewer parameters. Moreover, InstructGPT models show improvements in truthfulness and reductions in toxic output generation while having minimal performance regressions on public NLP datasets. Even though InstructGPT still makes simple mistakes, our results show that fine-tuning with human feedback is a promising direction for aligning language models with human intent.",
  "published": "2022-03-04",
  "updated": "2022-03-04",
  "year": "2022",
  "authors": [
   "Long Ouyang",
   "Jeff Wu",
   "Xu Jiang",
   "Diogo Almeida",
   "Carroll L. Wainwright",
   "Pamela Mishkin",
   "Chong Zhang",
   "Sandhini Agarwal",
   "Katarina Slama",
   "Alex Ray",
   "John Schulman",
   "Jacob Hilton",
   "Fraser Kelton",
   "Luke Miller",
   "Maddie Simens",
   "Amanda Askell",
   "Peter Welinder",
   "Paul Christiano",
   "Jan Leike",
   "Ryan Lowe"
  ],
  "author_count": 20,
  "categories": [
   "cs.CL",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.CL",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 23603,
  "influential_citations": 2404,
  "tldr": "The results show that fine-tuning with human feedback is a promising direction for aligning language models with human intent and showing improvements in truthfulness and reductions in toxic output generation while having minimal performance regressions on public NLP datasets.",
  "doi": "10.52202/068431-2011",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Long Ouyang",
    "id": "31793034",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Jeff Wu",
    "id": "49387725",
    "h_index": 11,
    "papers": 12
   },
   {
    "name": "Xu Jiang",
    "id": "2115903168",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Diogo Almeida",
    "id": "2061137049",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Carroll L. Wainwright",
    "id": "2064084601",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Pamela Mishkin",
    "id": "2051714782",
    "h_index": 17,
    "papers": 66
   },
   {
    "name": "Chong Zhang",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "S. Agarwal",
    "id": "144517868",
    "h_index": 20,
    "papers": 72
   },
   {
    "name": "Katarina Slama",
    "id": "2117680841",
    "h_index": 11,
    "papers": 44
   },
   {
    "name": "Alex Ray",
    "id": "2064770039",
    "h_index": 7,
    "papers": 8
   },
   {
    "name": "John Schulman",
    "id": "47971768",
    "h_index": 45,
    "papers": 69
   },
   {
    "name": "Jacob Hilton",
    "id": "2052366271",
    "h_index": 9,
    "papers": 14
   },
   {
    "name": "Fraser Kelton",
    "id": "2151735262",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Luke E. Miller",
    "id": "2142365973",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "M. Simens",
    "id": "2151735251",
    "h_index": 9,
    "papers": 53
   },
   {
    "name": "Amanda Askell",
    "id": "119609682",
    "h_index": 18,
    "papers": 29
   },
   {
    "name": "Peter Welinder",
    "id": "2930640",
    "h_index": 17,
    "papers": 37
   },
   {
    "name": "P. Christiano",
    "id": "145791315",
    "h_index": 16,
    "papers": 22
   },
   {
    "name": "Jan Leike",
    "id": "2990741",
    "h_index": 30,
    "papers": 76
   },
   {
    "name": "Ryan J. Lowe",
    "id": "49407415",
    "h_index": 50,
    "papers": 192
   }
  ],
  "comment": "",
  "topics": [
   "rl-control",
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2203.02155v1",
  "pdf_url": "https://arxiv.org/pdf/2203.02155v1",
  "html_url": "https://arxiv.org/html/2203.02155v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "2203.01941",
  "slug": "autoregressive-image-generation-using-residual-quantization",
  "title": "Autoregressive Image Generation using Residual Quantization",
  "abstract": "For autoregressive (AR) modeling of high-resolution images, vector quantization (VQ) represents an image as a sequence of discrete codes. A short sequence length is important for an AR model to reduce its computational costs to consider long-range interactions of codes. However, we postulate that previous VQ cannot shorten the code sequence and generate high-fidelity images together in terms of the rate-distortion trade-off. In this study, we propose the two-stage framework, which consists of Residual-Quantized VAE (RQ-VAE) and RQ-Transformer, to effectively generate high-resolution images. Given a fixed codebook size, RQ-VAE can precisely approximate a feature map of an image and represent the image as a stacked map of discrete codes. Then, RQ-Transformer learns to predict the quantized feature vector at the next position by predicting the next stack of codes. Thanks to the precise approximation of RQ-VAE, we can represent a 256$\\times$256 image as 8$\\times$8 resolution of the feature map, and RQ-Transformer can efficiently reduce the computational costs. Consequently, our framework outperforms the existing AR models on various benchmarks of unconditional and conditional image generation. Our approach also has a significantly faster sampling speed than previous AR models to generate high-quality images.",
  "published": "2022-03-03",
  "updated": "2022-03-09",
  "year": "2022",
  "authors": [
   "Doyup Lee",
   "Chiheon Kim",
   "Saehoon Kim",
   "Minsu Cho",
   "Wook-Shin Han"
  ],
  "author_count": 5,
  "categories": [
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 911,
  "influential_citations": 110,
  "tldr": "This study proposes the two-stage framework, which consists of Residual-Quantized VAE (RQ-VAE) and RQ-Transformer, to effectively generate high-resolution images and out-performs the existing AR models on various benchmarks of unconditional and conditional image generation.",
  "doi": "10.1109/CVPR52688.2022.01123",
  "oa_pdf": "https://arxiv.org/pdf/2203.01941",
  "s2_authors": [
   {
    "name": "Doyup Lee",
    "id": "2154633624",
    "h_index": 10,
    "papers": 17
   },
   {
    "name": "Chiheon Kim",
    "id": "25004333",
    "h_index": 12,
    "papers": 20
   },
   {
    "name": "Saehoon Kim",
    "id": "1898376",
    "h_index": 19,
    "papers": 43
   },
   {
    "name": "Minsu Cho",
    "id": "2098812123",
    "h_index": 11,
    "papers": 20
   },
   {
    "name": "Wook-Shin Han",
    "id": "144422954",
    "h_index": 32,
    "papers": 128
   }
  ],
  "comment": "30 pages, 24 figures, accepted by CVPR 2022, the code is available at https://github.com/kakaobrain/rq-vae-transformer",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2203.01941v2",
  "pdf_url": "https://arxiv.org/pdf/2203.01941v2",
  "html_url": "https://arxiv.org/html/2203.01941v2",
  "code_url": "https://github.com/kakaobrain/rq-vae-transformer",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.46
 },
 {
  "id": "2203.01577",
  "slug": "hoi4d-a-4d-egocentric-dataset-for-category-level-human-object-interact",
  "title": "HOI4D: A 4D Egocentric Dataset for Category-Level Human-Object Interaction",
  "abstract": "We present HOI4D, a large-scale 4D egocentric dataset with rich annotations, to catalyze the research of category-level human-object interaction. HOI4D consists of 2.4M RGB-D egocentric video frames over 4000 sequences collected by 4 participants interacting with 800 different object instances from 16 categories over 610 different indoor rooms. Frame-wise annotations for panoptic segmentation, motion segmentation, 3D hand pose, category-level object pose and hand action have also been provided, together with reconstructed object meshes and scene point clouds. With HOI4D, we establish three benchmarking tasks to promote category-level HOI from 4D visual signals including semantic segmentation of 4D dynamic point cloud sequences, category-level object pose tracking, and egocentric action segmentation with diverse interaction targets. In-depth analysis shows HOI4D poses great challenges to existing methods and produces great research opportunities.",
  "published": "2022-03-03",
  "updated": "2024-01-03",
  "year": "2022",
  "authors": [
   "Yunze Liu",
   "Yun Liu",
   "Che Jiang",
   "Kangbo Lyu",
   "Weikang Wan",
   "Hao Shen",
   "Boqiang Liang",
   "Zhoujie Fu",
   "He Wang",
   "Li Yi"
  ],
  "author_count": 10,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 399,
  "influential_citations": 48,
  "tldr": "Three benchmarking tasks to pro-mote category-level HOI from 4D visual signals are established including semantic segmentation of 4D dynamic point cloud se-quences, category- level object pose tracking, and egocen-tric action segmentation with diverse interaction targets.",
  "doi": "10.1109/CVPR52688.2022.02034",
  "oa_pdf": "http://arxiv.org/pdf/2203.01577",
  "s2_authors": [
   {
    "name": "Yunze Liu",
    "id": "2117416146",
    "h_index": 11,
    "papers": 18
   },
   {
    "name": "Yun Liu",
    "id": "2279787165",
    "h_index": 10,
    "papers": 16
   },
   {
    "name": "Che Jiang",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Zhoujie Fu",
    "id": "2157046216",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Kangbo Lyu",
    "id": "2157042895",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Weikang Wan",
    "id": "51451566",
    "h_index": 7,
    "papers": 18
   },
   {
    "name": "Hao Shen",
    "id": "2110771204",
    "h_index": 5,
    "papers": 12
   },
   {
    "name": "Bo-Hua Liang",
    "id": "152724385",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "He Wang",
    "id": "2125571",
    "h_index": 23,
    "papers": 33
   },
   {
    "name": "Li Yi",
    "id": "2027660032",
    "h_index": 17,
    "papers": 24
   }
  ],
  "comment": "",
  "topics": [
   "egocentric-data",
   "spatial-3d",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2203.01577v4",
  "pdf_url": "https://arxiv.org/pdf/2203.01577v4",
  "html_url": "https://arxiv.org/html/2203.01577v4",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.1
 },
 {
  "id": "2202.12385",
  "slug": "a-collision-free-mpc-for-whole-body-dynamic-locomotion-and-manipulatio",
  "title": "A Collision-Free MPC for Whole-Body Dynamic Locomotion and Manipulation",
  "abstract": "In this paper, we present a real-time whole-body planner for collision-free legged mobile manipulation. We enforce both self-collision and environment-collision avoidance as soft constraints within a Model Predictive Control (MPC) scheme that solves a multi-contact optimal control problem. By penalizing the signed distances among a set of representative primitive collision bodies, the robot is able to safely execute a variety of dynamic maneuvers while preventing any self-collisions. Moreover, collision-free navigation and manipulation in both static and dynamic environments are made viable through efficient queries of distances and their gradients via a euclidean signed distance field. We demonstrate through a comparative study that our approach only slightly increases the computational complexity of the MPC planning. Finally, we validate the effectiveness of our framework through a set of hardware experiments involving dynamic mobile manipulation tasks with potential collisions, such as locomotion balancing with the swinging arm, weight throwing, and autonomous door opening.",
  "published": "2022-02-24",
  "updated": "2022-02-24",
  "year": "2022",
  "authors": [
   "Jia-Ruei Chiu",
   "Jean-Pierre Sleiman",
   "Mayank Mittal",
   "Farbod Farshidian",
   "Marco Hutter"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA",
  "venue_source": "semantic-scholar",
  "citations": 103,
  "influential_citations": 2,
  "tldr": "A real-time whole-body planner for collision-free legged mobile manipulation that enforce both self-collision and environment-collison avoidance as soft constraints within a Model Predictive Control (MPC) scheme that solves a multi-contact optimal control problem.",
  "doi": "10.1109/icra46639.2022.9812280",
  "oa_pdf": "https://arxiv.org/pdf/2202.12385",
  "s2_authors": [
   {
    "name": "Jiawei Chiu",
    "id": "1921358",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Jean-Pierre Sleiman",
    "id": "66242644",
    "h_index": 10,
    "papers": 16
   },
   {
    "name": "Mayank Mittal",
    "id": "2061780867",
    "h_index": 13,
    "papers": 30
   },
   {
    "name": "Farbod Farshidian",
    "id": "2583867",
    "h_index": 35,
    "papers": 67
   },
   {
    "name": "Marco Hutter",
    "id": "14349870",
    "h_index": 80,
    "papers": 279
   }
  ],
  "comment": "Accepted in IEEE International Conference on Robotics and Automation (ICRA) 2022 in Philadelphia (PA), USA",
  "topics": [
   "humanoids",
   "rl-control",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2202.12385v1",
  "pdf_url": "https://arxiv.org/pdf/2202.12385v1",
  "html_url": "https://arxiv.org/html/2202.12385v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.52
 },
 {
  "id": "2202.11271",
  "slug": "viking-vision-based-kilometer-scale-navigation-with-geographic-hints",
  "title": "ViKiNG: Vision-Based Kilometer-Scale Navigation with Geographic Hints",
  "abstract": "Robotic navigation has been approached as a problem of 3D reconstruction and planning, as well as an end-to-end learning problem. However, long-range navigation requires both planning and reasoning about local traversability, as well as being able to utilize general knowledge about global geography, in the form of a roadmap, GPS, or other side information providing important cues. In this work, we propose an approach that integrates learning and planning, and can utilize side information such as schematic roadmaps, satellite maps and GPS coordinates as a planning heuristic, without relying on them being accurate. Our method, ViKiNG, incorporates a local traversability model, which looks at the robot's current camera observation and a potential subgoal to infer how easily that subgoal can be reached, as well as a heuristic model, which looks at overhead maps for hints and attempts to evaluate the appropriateness of these subgoals in order to reach the goal. These models are used by a heuristic planner to identify the best waypoint in order to reach the final destination. Our method performs no explicit geometric reconstruction, utilizing only a topological representation of the environment. Despite having never seen trajectories longer than 80 meters in its training dataset, ViKiNG can leverage its image-based learned controller and goal-directed heuristic to navigate to goals up to 3 kilometers away in previously unseen environments, and exhibit complex behaviors such as probing potential paths and backtracking when they are found to be non-viable. ViKiNG is also robust to unreliable maps and GPS, since the low-level controller ultimately makes decisions based on egocentric image observations, using maps only as planning heuristics. For videos of our experiments, please check out our project page https://sites.google.com/view/viking-release.",
  "published": "2022-02-23",
  "updated": "2023-01-10",
  "year": "2022",
  "authors": [
   "Dhruv Shah",
   "Sergey Levine"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "RSS",
  "venue_source": "semantic-scholar",
  "citations": 109,
  "influential_citations": 3,
  "tldr": "The method, ViKiNG, incorporates a local traversability model, which looks at the robot's current camera observation and a potential subgoal to infer how easily that subgoal can be reached, as well as a heuristic model that looks at overhead maps for hints and attempts to evaluate the appropriateness of these subgoals in order to reach the goal.",
  "doi": "10.15607/RSS.2022.XVIII.019",
  "oa_pdf": "https://doi.org/10.15607/rss.2022.xviii.019",
  "s2_authors": [
   {
    "name": "Dhruv Shah",
    "id": "2322628540",
    "h_index": 29,
    "papers": 63
   },
   {
    "name": "S. Levine",
    "id": "1736651",
    "h_index": 167,
    "papers": 551
   }
  ],
  "comment": "Best Systems Paper Finalist at XVII Robotics: Science and Systems (RSS 2022), New York City, USA. Project page https://sites.google.com/view/viking-release",
  "topics": [
   "egocentric-data",
   "spatial-3d",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2202.11271v3",
  "pdf_url": "https://arxiv.org/pdf/2202.11271v3",
  "html_url": "https://arxiv.org/html/2202.11271v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.54
 },
 {
  "id": "2202.10448",
  "slug": "robotic-telekinesis-learning-a-robotic-hand-imitator-by-watching-human",
  "title": "Robotic Telekinesis: Learning a Robotic Hand Imitator by Watching Humans on Youtube",
  "abstract": "We build a system that enables any human to control a robot hand and arm, simply by demonstrating motions with their own hand. The robot observes the human operator via a single RGB camera and imitates their actions in real-time. Human hands and robot hands differ in shape, size, and joint structure, and performing this translation from a single uncalibrated camera is a highly underconstrained problem. Moreover, the retargeted trajectories must effectively execute tasks on a physical robot, which requires them to be temporally smooth and free of self-collisions. Our key insight is that while paired human-robot correspondence data is expensive to collect, the internet contains a massive corpus of rich and diverse human hand videos. We leverage this data to train a system that understands human hands and retargets a human video stream into a robot hand-arm trajectory that is smooth, swift, safe, and semantically similar to the guiding demonstration. We demonstrate that it enables previously untrained people to teleoperate a robot on various dexterous manipulation tasks. Our low-cost, glove-free, marker-free remote teleoperation system makes robot teaching more accessible and we hope that it can aid robots in learning to act autonomously in the real world. Videos at https://robotic-telekinesis.github.io/",
  "published": "2022-02-21",
  "updated": "2022-07-24",
  "year": "2022",
  "authors": [
   "Aravind Sivakumar",
   "Kenneth Shaw",
   "Deepak Pathak"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "RSS",
  "venue_source": "semantic-scholar",
  "citations": 162,
  "influential_citations": 11,
  "tldr": "This work builds a system that enables any human to control a robot hand and arm, simply by demonstrating motions with their own hand, and demonstrates that it enables previously untrained people to teleoperate a robot on various dexterous manipulation tasks.",
  "doi": "10.15607/rss.2022.xviii.023",
  "oa_pdf": "https://doi.org/10.15607/rss.2022.xviii.023",
  "s2_authors": [
   {
    "name": "Aravind Sivakumar",
    "id": "32088177",
    "h_index": 10,
    "papers": 19
   },
   {
    "name": "Kenneth Shaw",
    "id": "2072761493",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Deepak Pathak",
    "id": "2004879394",
    "h_index": 24,
    "papers": 32
   }
  ],
  "comment": "RSS 2022 final version. Website and demos at https://robotic-telekinesis.github.io/",
  "topics": [
   "dexterous-manipulation",
   "egocentric-data",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2202.10448v2",
  "pdf_url": "https://arxiv.org/pdf/2202.10448v2",
  "html_url": "https://arxiv.org/html/2202.10448v2",
  "code_url": "https://robotic-telekinesis.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.71
 },
 {
  "id": "2202.09481",
  "slug": "transdreamer-reinforcement-learning-with-transformer-world-models",
  "title": "TransDreamer: Reinforcement Learning with Transformer World Models",
  "abstract": "The Dreamer agent provides various benefits of Model-Based Reinforcement Learning (MBRL) such as sample efficiency, reusable knowledge, and safe planning. However, its world model and policy networks inherit the limitations of recurrent neural networks and thus an important question is how an MBRL framework can benefit from the recent advances of transformers and what the challenges are in doing so. In this paper, we propose a transformer-based MBRL agent, called TransDreamer. We first introduce the Transformer State-Space Model, a world model that leverages a transformer for dynamics predictions. We then share this world model with a transformer-based policy network and obtain stability in training a transformer-based RL agent. In experiments, we apply the proposed model to 2D visual RL and 3D first-person visual RL tasks both requiring long-range memory access for memory-based reasoning. We show that the proposed model outperforms Dreamer in these complex tasks.",
  "published": "2022-02-19",
  "updated": "2024-11-19",
  "year": "2022",
  "authors": [
   "Chang Chen",
   "Yi-Fu Wu",
   "Jaesik Yoon",
   "Sungjin Ahn"
  ],
  "author_count": 4,
  "categories": [
   "cs.LG"
  ],
  "primary_category": "cs.LG",
  "venue": "NeurIPS 2021",
  "venue_source": "arxiv-comment",
  "citations": 164,
  "influential_citations": 11,
  "tldr": "A transformer-based MBRL agent, called TransDreamer, is proposed that outperforms Dreamer in these complex tasks and is applied to 2D visual RL and 3D first-person visual RL tasks both requiring long-range memory access for memory-based reasoning.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Changgu Chen",
    "id": "1642920232",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Yi-Fu Wu",
    "id": "4197575",
    "h_index": 10,
    "papers": 17
   },
   {
    "name": "Jaesik Yoon",
    "id": "2115996713",
    "h_index": 8,
    "papers": 22
   },
   {
    "name": "Sungjin Ahn",
    "id": "1882851",
    "h_index": 17,
    "papers": 28
   }
  ],
  "comment": "Deep RL Workshop NeurIPS 2021",
  "topics": [
   "world-models",
   "egocentric-data",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2202.09481v2",
  "pdf_url": "https://arxiv.org/pdf/2202.09481v2",
  "html_url": "https://arxiv.org/html/2202.09481v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.72
 },
 {
  "id": "2202.10583",
  "slug": "minerl-diamond-2021-competition-overview-results-and-lessons-learned",
  "title": "MineRL Diamond 2021 Competition: Overview, Results, and Lessons Learned",
  "abstract": "Reinforcement learning competitions advance the field by providing appropriate scope and support to develop solutions toward a specific problem. To promote the development of more broadly applicable methods, organizers need to enforce the use of general techniques, the use of sample-efficient methods, and the reproducibility of the results. While beneficial for the research community, these restrictions come at a cost -- increased difficulty. If the barrier for entry is too high, many potential participants are demoralized. With this in mind, we hosted the third edition of the MineRL ObtainDiamond competition, MineRL Diamond 2021, with a separate track in which we permitted any solution to promote the participation of newcomers. With this track and more extensive tutorials and support, we saw an increased number of submissions. The participants of this easier track were able to obtain a diamond, and the participants of the harder track progressed the generalizable solutions in the same task.",
  "published": "2022-02-17",
  "updated": "2022-02-17",
  "year": "2022",
  "authors": [
   "Anssi Kanervisto",
   "Stephanie Milani",
   "Karolis Ramanauskas",
   "Nicholay Topin",
   "Zichuan Lin",
   "Junyou Li",
   "Jianing Shi",
   "Deheng Ye",
   "Qiang Fu",
   "Wei Yang",
   "Weijun Hong",
   "Zhongyue Huang",
   "Haicheng Chen",
   "Guangjun Zeng",
   "Yue Lin",
   "Vincent Micheli",
   "Eloi Alonso",
   "Fran\u00e7ois Fleuret",
   "Alexander Nikulin",
   "Yury Belousov",
   "Oleg Svidchenko",
   "Aleksei Shpilman"
  ],
  "author_count": 22,
  "categories": [
   "cs.LG",
   "cs.AI"
  ],
  "primary_category": "cs.LG",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 35,
  "influential_citations": 0,
  "tldr": "",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "A. Kanervisto",
    "id": "3469155",
    "h_index": 18,
    "papers": 43
   },
   {
    "name": "Stephanie Milani",
    "id": "144177520",
    "h_index": 14,
    "papers": 29
   },
   {
    "name": "Karolis Ramanauskas",
    "id": "25072106",
    "h_index": 9,
    "papers": 25
   },
   {
    "name": "Nicholay Topin",
    "id": "34887814",
    "h_index": 18,
    "papers": 28
   },
   {
    "name": "Zichuan Lin",
    "id": "41123614",
    "h_index": 12,
    "papers": 36
   },
   {
    "name": "Junyou Li",
    "id": "2282561331",
    "h_index": 11,
    "papers": 33
   },
   {
    "name": "Jianing Shi",
    "id": "2117868795",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Deheng Ye",
    "id": "2055648566",
    "h_index": 21,
    "papers": 44
   },
   {
    "name": "Qiang Fu",
    "id": "2091914469",
    "h_index": 19,
    "papers": 38
   },
   {
    "name": "Wei Yang",
    "id": "2005150594",
    "h_index": 18,
    "papers": 30
   },
   {
    "name": "Weijun Hong",
    "id": "2114947718",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Zhong-Hao Huang",
    "id": "102992587",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Haicheng Chen",
    "id": "2042696382",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "Guangjun Zeng",
    "id": "2022732",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Yue Lin",
    "id": "8708059",
    "h_index": 22,
    "papers": 115
   },
   {
    "name": "Vincent Micheli",
    "id": "1491750155",
    "h_index": 6,
    "papers": 11
   },
   {
    "name": "Eloi Alonso",
    "id": "144370326",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "Franccois Fleuret",
    "id": "116272138",
    "h_index": 19,
    "papers": 45
   },
   {
    "name": "Alexander Nikulin",
    "id": "2155646949",
    "h_index": 13,
    "papers": 24
   },
   {
    "name": "Yury Belousov",
    "id": "2106191576",
    "h_index": 7,
    "papers": 19
   },
   {
    "name": "Oleg Svidchenko",
    "id": "1434552251",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "A. Shpilman",
    "id": "35431760",
    "h_index": 10,
    "papers": 30
   }
  ],
  "comment": "Under review for PMLR volume on NeurIPS 2021 competitions",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2202.10583v1",
  "pdf_url": "https://arxiv.org/pdf/2202.10583v1",
  "html_url": "https://arxiv.org/html/2202.10583v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.06
 },
 {
  "id": "2202.08938",
  "slug": "improving-intrinsic-exploration-with-language-abstractions",
  "title": "Improving Intrinsic Exploration with Language Abstractions",
  "abstract": "Reinforcement learning (RL) agents are particularly hard to train when rewards are sparse. One common solution is to use intrinsic rewards to encourage agents to explore their environment. However, recent intrinsic exploration methods often use state-based novelty measures which reward low-level exploration and may not scale to domains requiring more abstract skills. Instead, we explore natural language as a general medium for highlighting relevant abstractions in an environment. Unlike previous work, we evaluate whether language can improve over existing exploration methods by directly extending (and comparing to) competitive intrinsic exploration baselines: AMIGo (Campero et al., 2021) and NovelD (Zhang et al., 2021). These language-based variants outperform their non-linguistic forms by 47-85% across 13 challenging tasks from the MiniGrid and MiniHack environment suites.",
  "published": "2022-02-17",
  "updated": "2022-11-21",
  "year": "2022",
  "authors": [
   "Jesse Mu",
   "Victor Zhong",
   "Roberta Raileanu",
   "Minqi Jiang",
   "Noah Goodman",
   "Tim Rockt\u00e4schel",
   "Edward Grefenstette"
  ],
  "author_count": 7,
  "categories": [
   "cs.LG",
   "cs.AI",
   "cs.CL"
  ],
  "primary_category": "cs.LG",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 80,
  "influential_citations": 2,
  "tldr": "This work evaluates whether language can improve over existing exploration methods by directly extending (and comparing to) competitive intrinsic exploration baselines: AMIGo (Campero et al, 2021) and NovelD (Zhang et al., 2021).",
  "doi": "10.52202/068431-2460",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jesse Mu",
    "id": "24835910",
    "h_index": 12,
    "papers": 27
   },
   {
    "name": "Victor Zhong",
    "id": "3428769",
    "h_index": 25,
    "papers": 41
   },
   {
    "name": "R. Raileanu",
    "id": "48647153",
    "h_index": 30,
    "papers": 99
   },
   {
    "name": "Minqi Jiang",
    "id": "2152154941",
    "h_index": 19,
    "papers": 30
   },
   {
    "name": "Noah D. Goodman",
    "id": "144002017",
    "h_index": 76,
    "papers": 301
   },
   {
    "name": "Tim Rocktaschel",
    "id": "1389854357",
    "h_index": 22,
    "papers": 42
   },
   {
    "name": "Edward Grefenstette",
    "id": "1864353",
    "h_index": 47,
    "papers": 98
   }
  ],
  "comment": "NeurIPS 2022",
  "topics": [
   "rl-control",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2202.08938v2",
  "pdf_url": "https://arxiv.org/pdf/2202.08938v2",
  "html_url": "https://arxiv.org/html/2202.08938v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.41
 },
 {
  "id": "2202.05306",
  "slug": "characterizing-and-overcoming-the-greedy-nature-of-learning-in-multi-m",
  "title": "Characterizing and overcoming the greedy nature of learning in multi-modal deep neural networks",
  "abstract": "We hypothesize that due to the greedy nature of learning in multi-modal deep neural networks, these models tend to rely on just one modality while under-fitting the other modalities. Such behavior is counter-intuitive and hurts the models' generalization, as we observe empirically. To estimate the model's dependence on each modality, we compute the gain on the accuracy when the model has access to it in addition to another modality. We refer to this gain as the conditional utilization rate. In the experiments, we consistently observe an imbalance in conditional utilization rates between modalities, across multiple tasks and architectures. Since conditional utilization rate cannot be computed efficiently during training, we introduce a proxy for it based on the pace at which the model learns from each modality, which we refer to as the conditional learning speed. We propose an algorithm to balance the conditional learning speeds between modalities during training and demonstrate that it indeed addresses the issue of greedy learning. The proposed algorithm improves the model's generalization on three datasets: Colored MNIST, ModelNet40, and NVIDIA Dynamic Hand Gesture.",
  "published": "2022-02-10",
  "updated": "2022-09-16",
  "year": "2022",
  "authors": [
   "Nan Wu",
   "Stanis\u0142aw Jastrz\u0119bski",
   "Kyunghyun Cho",
   "Krzysztof J. Geras"
  ],
  "author_count": 4,
  "categories": [
   "cs.LG",
   "cs.CV"
  ],
  "primary_category": "cs.LG",
  "venue": "ICML",
  "venue_source": "semantic-scholar",
  "citations": 152,
  "influential_citations": 24,
  "tldr": "An algorithm is proposed to balance the conditional learning speeds between modalities during training and it is demonstrated that it indeed addresses the issue of greedy learning.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Nan Wu",
    "id": "2068343106",
    "h_index": 12,
    "papers": 20
   },
   {
    "name": "Stanislaw Jastrzebski",
    "id": "40569328",
    "h_index": 26,
    "papers": 47
   },
   {
    "name": "Kyunghyun Cho",
    "id": "1979489",
    "h_index": 96,
    "papers": 329
   },
   {
    "name": "Krzysztof J. Geras",
    "id": "2376144",
    "h_index": 30,
    "papers": 88
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [
   "NVIDIA"
  ],
  "abs_url": "https://arxiv.org/abs/2202.05306v3",
  "pdf_url": "https://arxiv.org/pdf/2202.05306v3",
  "html_url": "https://arxiv.org/html/2202.05306v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.18
 },
 {
  "id": "2202.04200",
  "slug": "maskgit-masked-generative-image-transformer",
  "title": "MaskGIT: Masked Generative Image Transformer",
  "abstract": "Generative transformers have experienced rapid popularity growth in the computer vision community in synthesizing high-fidelity and high-resolution images. The best generative transformer models so far, however, still treat an image naively as a sequence of tokens, and decode an image sequentially following the raster scan ordering (i.e. line-by-line). We find this strategy neither optimal nor efficient. This paper proposes a novel image synthesis paradigm using a bidirectional transformer decoder, which we term MaskGIT. During training, MaskGIT learns to predict randomly masked tokens by attending to tokens in all directions. At inference time, the model begins with generating all tokens of an image simultaneously, and then refines the image iteratively conditioned on the previous generation. Our experiments demonstrate that MaskGIT significantly outperforms the state-of-the-art transformer model on the ImageNet dataset, and accelerates autoregressive decoding by up to 64x. Besides, we illustrate that MaskGIT can be easily extended to various image editing tasks, such as inpainting, extrapolation, and image manipulation.",
  "published": "2022-02-08",
  "updated": "2022-02-08",
  "year": "2022",
  "authors": [
   "Huiwen Chang",
   "Han Zhang",
   "Lu Jiang",
   "Ce Liu",
   "William T. Freeman"
  ],
  "author_count": 5,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 1270,
  "influential_citations": 230,
  "tldr": "The proposed MaskGIT is a novel image synthesis paradigm using a bidirectional transformer decoder that significantly outperforms the state-of-the-art transformer model on the ImageNet dataset, and accelerates autoregressive decoding by up to 48x.",
  "doi": "10.1109/CVPR52688.2022.01103",
  "oa_pdf": "https://arxiv.org/pdf/2202.04200",
  "s2_authors": [
   {
    "name": "Huiwen Chang",
    "id": "2914394",
    "h_index": 28,
    "papers": 38
   },
   {
    "name": "Han Zhang",
    "id": "2119079641",
    "h_index": 14,
    "papers": 20
   },
   {
    "name": "Lu Jiang",
    "id": "39978626",
    "h_index": 48,
    "papers": 78
   },
   {
    "name": "Ce Liu",
    "id": "2107890439",
    "h_index": 28,
    "papers": 34
   },
   {
    "name": "W. Freeman",
    "id": "1768236",
    "h_index": 119,
    "papers": 343
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2202.04200v1",
  "pdf_url": "https://arxiv.org/pdf/2202.04200v1",
  "html_url": "https://arxiv.org/html/2202.04200v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "2202.02005",
  "slug": "bc-z-zero-shot-task-generalization-with-robotic-imitation-learning",
  "title": "BC-Z: Zero-Shot Task Generalization with Robotic Imitation Learning",
  "abstract": "In this paper, we study the problem of enabling a vision-based robotic manipulation system to generalize to novel tasks, a long-standing challenge in robot learning. We approach the challenge from an imitation learning perspective, aiming to study how scaling and broadening the data collected can facilitate such generalization. To that end, we develop an interactive and flexible imitation learning system that can learn from both demonstrations and interventions and can be conditioned on different forms of information that convey the task, including pre-trained embeddings of natural language or videos of humans performing the task. When scaling data collection on a real robot to more than 100 distinct tasks, we find that this system can perform 24 unseen manipulation tasks with an average success rate of 44%, without any robot demonstrations for those tasks.",
  "published": "2022-02-04",
  "updated": "2022-02-04",
  "year": "2022",
  "authors": [
   "Eric Jang",
   "Alex Irpan",
   "Mohi Khansari",
   "Daniel Kappler",
   "Frederik Ebert",
   "Corey Lynch",
   "Sergey Levine",
   "Chelsea Finn"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 835,
  "influential_citations": 65,
  "tldr": "An interactive and flexible imitation learning system that can learn from both demonstrations and interventions and can be conditioned on different forms of information that convey the task, including pre-trained embeddings of natural language or videos of humans performing the task.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Eric Jang",
    "id": "145116380",
    "h_index": 20,
    "papers": 30
   },
   {
    "name": "A. Irpan",
    "id": "17818078",
    "h_index": 22,
    "papers": 32
   },
   {
    "name": "Mohi Khansari",
    "id": "30559411",
    "h_index": 14,
    "papers": 22
   },
   {
    "name": "Daniel Kappler",
    "id": "2435984",
    "h_index": 21,
    "papers": 37
   },
   {
    "name": "F. Ebert",
    "id": "27535721",
    "h_index": 15,
    "papers": 22
   },
   {
    "name": "Corey Lynch",
    "id": "32245472",
    "h_index": 20,
    "papers": 27
   },
   {
    "name": "S. Levine",
    "id": "1736651",
    "h_index": 167,
    "papers": 551
   },
   {
    "name": "Chelsea Finn",
    "id": "46881670",
    "h_index": 94,
    "papers": 240
   }
  ],
  "comment": "CoRL 2021, 23 pages",
  "topics": [
   "imitation-diffusion",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2202.02005v1",
  "pdf_url": "https://arxiv.org/pdf/2202.02005v1",
  "html_url": "https://arxiv.org/html/2202.02005v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.42
 },
 {
  "id": "2202.01771",
  "slug": "pre-trained-language-models-for-interactive-decision-making",
  "title": "Pre-Trained Language Models for Interactive Decision-Making",
  "abstract": "Language model (LM) pre-training is useful in many language processing tasks. But can pre-trained LMs be further leveraged for more general machine learning problems? We propose an approach for using LMs to scaffold learning and generalization in general sequential decision-making problems. In this approach, goals and observations are represented as a sequence of embeddings, and a policy network initialized with a pre-trained LM predicts the next action. We demonstrate that this framework enables effective combinatorial generalization across different environments and supervisory modalities. We begin by assuming access to a set of expert demonstrations, and show that initializing policies with LMs and fine-tuning them via behavior cloning improves task completion rates by 43.6% in the VirtualHome environment. Next, we integrate an active data gathering procedure in which agents iteratively interact with the environment, relabel past \"failed\" experiences with new goals, and update their policies in a self-supervised loop. Active data gathering further improves combinatorial generalization, outperforming the best baseline by 25.1%. Finally, we explain these results by investigating three possible factors underlying the effectiveness of the LM-based policy. We find that sequential input representations (vs. fixed-dimensional feature vectors) and LM-based weight initialization are both important for generalization. Surprisingly, however, the format of the policy inputs encoding (e.g. as a natural language string vs. an arbitrary sequential encoding) has little influence. Together, these results suggest that language modeling induces representations that are useful for modeling not just language, but also goals and plans; these representations can aid learning and generalization even outside of language processing.",
  "published": "2022-02-03",
  "updated": "2022-10-29",
  "year": "2022",
  "authors": [
   "Shuang Li",
   "Xavier Puig",
   "Chris Paxton",
   "Yilun Du",
   "Clinton Wang",
   "Linxi Fan",
   "Tao Chen",
   "De-An Huang",
   "Ekin Aky\u00fcrek",
   "Anima Anandkumar",
   "Jacob Andreas",
   "Igor Mordatch",
   "Antonio Torralba",
   "Yuke Zhu"
  ],
  "author_count": 14,
  "categories": [
   "cs.LG",
   "cs.CL"
  ],
  "primary_category": "cs.LG",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 339,
  "influential_citations": 19,
  "tldr": "This work proposes an approach for using LMs to scaffold learning and generalization in general sequential decision-making problems, and shows that this framework enables effective combinatorial generalization across different environments and supervisory modalities.",
  "doi": "10.52202/068431-2262",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Shuang Li",
    "id": "145015904",
    "h_index": 25,
    "papers": 41
   },
   {
    "name": "Xavier Puig",
    "id": "143872936",
    "h_index": 10,
    "papers": 14
   },
   {
    "name": "Yilun Du",
    "id": "15394275",
    "h_index": 48,
    "papers": 86
   },
   {
    "name": "Clinton Jia Wang",
    "id": "144276007",
    "h_index": 7,
    "papers": 15
   },
   {
    "name": "Ekin Aky\u00fcrek",
    "id": "1992708068",
    "h_index": 16,
    "papers": 21
   },
   {
    "name": "A. Torralba",
    "id": "143805211",
    "h_index": 142,
    "papers": 372
   },
   {
    "name": "Jacob Andreas",
    "id": "2112400",
    "h_index": 53,
    "papers": 93
   },
   {
    "name": "Igor Mordatch",
    "id": "2316241372",
    "h_index": 16,
    "papers": 19
   }
  ],
  "comment": "",
  "topics": [
   "imitation-diffusion",
   "foundation-pretraining",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2202.01771v4",
  "pdf_url": "https://arxiv.org/pdf/2202.01771v4",
  "html_url": "https://arxiv.org/html/2202.01771v4",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.03
 },
 {
  "id": "2202.00164",
  "slug": "dexvip-learning-dexterous-grasping-with-human-hand-pose-priors-from-vi",
  "title": "DexVIP: Learning Dexterous Grasping with Human Hand Pose Priors from Video",
  "abstract": "Dexterous multi-fingered robotic hands have a formidable action space, yet their morphological similarity to the human hand holds immense potential to accelerate robot learning. We propose DexVIP, an approach to learn dexterous robotic grasping from human-object interactions present in in-the-wild YouTube videos. We do this by curating grasp images from human-object interaction videos and imposing a prior over the agent's hand pose when learning to grasp with deep reinforcement learning. A key advantage of our method is that the learned policy is able to leverage free-form in-the-wild visual data. As a result, it can easily scale to new objects, and it sidesteps the standard practice of collecting human demonstrations in a lab -- a much more expensive and indirect way to capture human expertise. Through experiments on 27 objects with a 30-DoF simulated robot hand, we demonstrate that DexVIP compares favorably to existing approaches that lack a hand pose prior or rely on specialized tele-operation equipment to obtain human demonstrations, while also being faster to train. Project page: https://vision.cs.utexas.edu/projects/dexvip-dexterous-grasp-pose-prior",
  "published": "2022-02-01",
  "updated": "2022-02-01",
  "year": "2022",
  "authors": [
   "Priyanka Mandikal",
   "Kristen Grauman"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 147,
  "influential_citations": 5,
  "tldr": "Through experiments on 27 objects with a 30-DoF simulated robot hand, it is demonstrated that DexVIP compares favorably to existing approaches that lack a hand pose prior or rely on specialized tele-operation equipment to obtain human demonstrations, while also being faster to train.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Priyanka Mandikal",
    "id": "51126291",
    "h_index": 10,
    "papers": 13
   },
   {
    "name": "K. Grauman",
    "id": "1794409",
    "h_index": 99,
    "papers": 295
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "egocentric-data",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2202.00164v1",
  "pdf_url": "https://arxiv.org/pdf/2202.00164v1",
  "html_url": "https://arxiv.org/html/2202.00164v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.67
 },
 {
  "id": "2202.00161",
  "slug": "cic-contrastive-intrinsic-control-for-unsupervised-skill-discovery",
  "title": "CIC: Contrastive Intrinsic Control for Unsupervised Skill Discovery",
  "abstract": "We introduce Contrastive Intrinsic Control (CIC), an algorithm for unsupervised skill discovery that maximizes the mutual information between state-transitions and latent skill vectors. CIC utilizes contrastive learning between state-transitions and skills to learn behavior embeddings and maximizes the entropy of these embeddings as an intrinsic reward to encourage behavioral diversity. We evaluate our algorithm on the Unsupervised Reinforcement Learning Benchmark, which consists of a long reward-free pre-training phase followed by a short adaptation phase to downstream tasks with extrinsic rewards. CIC substantially improves over prior methods in terms of adaptation efficiency, outperforming prior unsupervised skill discovery methods by 1.79x and the next leading overall exploration algorithm by 1.18x.",
  "published": "2022-02-01",
  "updated": "2022-03-30",
  "year": "2022",
  "authors": [
   "Michael Laskin",
   "Hao Liu",
   "Xue Bin Peng",
   "Denis Yarats",
   "Aravind Rajeswaran",
   "Pieter Abbeel"
  ],
  "author_count": 6,
  "categories": [
   "cs.LG",
   "cs.AI"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 89,
  "influential_citations": 10,
  "tldr": "Contrastive Intrinsic Control (CIC), an algorithm for unsupervised skill discovery that maximizes the mutual information between state-transitions and latent skill vectors, substantially improves over prior methods in terms of adaptation efficiency.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "M. Laskin",
    "id": "51093256",
    "h_index": 22,
    "papers": 39
   },
   {
    "name": "Hao Liu",
    "id": "2143855835",
    "h_index": 21,
    "papers": 27
   },
   {
    "name": "Xue Bin Peng",
    "id": "2375236722",
    "h_index": 34,
    "papers": 41
   },
   {
    "name": "Denis Yarats",
    "id": "13759615",
    "h_index": 21,
    "papers": 31
   },
   {
    "name": "A. Rajeswaran",
    "id": "19275599",
    "h_index": 34,
    "papers": 58
   },
   {
    "name": "P. Abbeel",
    "id": "1689992",
    "h_index": 165,
    "papers": 596
   }
  ],
  "comment": "Project website: https://sites.google.com/view/cicrl/",
  "topics": [
   "rl-control",
   "navigation",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2202.00161v3",
  "pdf_url": "https://arxiv.org/pdf/2202.00161v3",
  "html_url": "https://arxiv.org/html/2202.00161v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.95
 },
 {
  "id": "2201.11903",
  "slug": "chain-of-thought-prompting-elicits-reasoning-in-large-language-models",
  "title": "Chain-of-Thought Prompting Elicits Reasoning in Large Language Models",
  "abstract": "We explore how generating a chain of thought -- a series of intermediate reasoning steps -- significantly improves the ability of large language models to perform complex reasoning. In particular, we show how such reasoning abilities emerge naturally in sufficiently large language models via a simple method called chain of thought prompting, where a few chain of thought demonstrations are provided as exemplars in prompting. Experiments on three large language models show that chain of thought prompting improves performance on a range of arithmetic, commonsense, and symbolic reasoning tasks. The empirical gains can be striking. For instance, prompting a 540B-parameter language model with just eight chain of thought exemplars achieves state of the art accuracy on the GSM8K benchmark of math word problems, surpassing even finetuned GPT-3 with a verifier.",
  "published": "2022-01-28",
  "updated": "2023-01-10",
  "year": "2022",
  "authors": [
   "Jason Wei",
   "Xuezhi Wang",
   "Dale Schuurmans",
   "Maarten Bosma",
   "Brian Ichter",
   "Fei Xia",
   "Ed Chi",
   "Quoc Le",
   "Denny Zhou"
  ],
  "author_count": 9,
  "categories": [
   "cs.CL",
   "cs.AI"
  ],
  "primary_category": "cs.CL",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 21316,
  "influential_citations": 1388,
  "tldr": "Experiments on three large language models show that chain of thought prompting improves performance on a range of arithmetic, commonsense, and symbolic reasoning tasks.",
  "doi": "10.52202/068431-1800",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jason Wei",
    "id": "119640649",
    "h_index": 27,
    "papers": 37
   },
   {
    "name": "Xuezhi Wang",
    "id": "2275277634",
    "h_index": 19,
    "papers": 49
   },
   {
    "name": "Dale Schuurmans",
    "id": "1714772",
    "h_index": 65,
    "papers": 264
   },
   {
    "name": "Maarten Bosma",
    "id": "40377863",
    "h_index": 14,
    "papers": 56
   },
   {
    "name": "Ed H. Chi",
    "id": "2226805",
    "h_index": 79,
    "papers": 284
   },
   {
    "name": "F. Xia",
    "id": "144956443",
    "h_index": 25,
    "papers": 31
   },
   {
    "name": "Quoc Le",
    "id": "1998340269",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Denny Zhou",
    "id": "65855107",
    "h_index": 42,
    "papers": 60
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2201.11903v6",
  "pdf_url": "https://arxiv.org/pdf/2201.11903v6",
  "html_url": "https://arxiv.org/html/2201.11903v6",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "2201.08117",
  "slug": "learning-robust-perceptive-locomotion-for-quadrupedal-robots-in-the-wi",
  "title": "Learning robust perceptive locomotion for quadrupedal robots in the wild",
  "abstract": "Legged robots that can operate autonomously in remote and hazardous environments will greatly increase opportunities for exploration into under-explored areas. Exteroceptive perception is crucial for fast and energy-efficient locomotion: perceiving the terrain before making contact with it enables planning and adaptation of the gait ahead of time to maintain speed and stability. However, utilizing exteroceptive perception robustly for locomotion has remained a grand challenge in robotics. Snow, vegetation, and water visually appear as obstacles on which the robot cannot step~-- or are missing altogether due to high reflectance. Additionally, depth perception can degrade due to difficult lighting, dust, fog, reflective or transparent surfaces, sensor occlusion, and more. For this reason, the most robust and general solutions to legged locomotion to date rely solely on proprioception. This severely limits locomotion speed, because the robot has to physically feel out the terrain before adapting its gait accordingly. Here we present a robust and general solution to integrating exteroceptive and proprioceptive perception for legged locomotion. We leverage an attention-based recurrent encoder that integrates proprioceptive and exteroceptive input. The encoder is trained end-to-end and learns to seamlessly combine the different perception modalities without resorting to heuristics. The result is a legged locomotion controller with high robustness and speed. The controller was tested in a variety of challenging natural and urban environments over multiple seasons and completed an hour-long hike in the Alps in the time recommended for human hikers.",
  "published": "2022-01-20",
  "updated": "2022-01-20",
  "year": "2022",
  "authors": [
   "Takahiro Miki",
   "Joonho Lee",
   "Jemin Hwangbo",
   "Lorenz Wellhausen",
   "Vladlen Koltun",
   "Marco Hutter"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "Science Robotics",
  "venue_source": "semantic-scholar",
  "citations": 1154,
  "influential_citations": 65,
  "tldr": "An attention-based recurrent encoder is leverage that integrates proprioceptive and exteroceptive input and learns to seamlessly combine the different perception modalities without resorting to heuristics to create a legged locomotion controller with high robustness and speed.",
  "doi": "10.1126/scirobotics.abk2822",
  "oa_pdf": "https://arxiv.org/pdf/2201.08117",
  "s2_authors": [
   {
    "name": "Takahiro Miki",
    "id": "1825794582",
    "h_index": 21,
    "papers": 31
   },
   {
    "name": "Joonho Lee",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Jemin Hwangbo",
    "id": "1707297",
    "h_index": 25,
    "papers": 49
   },
   {
    "name": "Lorenz Wellhausen",
    "id": "7153704",
    "h_index": 22,
    "papers": 26
   },
   {
    "name": "V. Koltun",
    "id": "145231047",
    "h_index": 114,
    "papers": 239
   },
   {
    "name": "Marco Hutter",
    "id": "14349870",
    "h_index": 80,
    "papers": 279
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2201.08117v1",
  "pdf_url": "https://arxiv.org/pdf/2201.08117v1",
  "html_url": "https://arxiv.org/html/2201.08117v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "2201.07207",
  "slug": "language-models-as-zero-shot-planners-extracting-actionable-knowledge",
  "title": "Language Models as Zero-Shot Planners: Extracting Actionable Knowledge for Embodied Agents",
  "abstract": "Can world knowledge learned by large language models (LLMs) be used to act in interactive environments? In this paper, we investigate the possibility of grounding high-level tasks, expressed in natural language (e.g. \"make breakfast\"), to a chosen set of actionable steps (e.g. \"open fridge\"). While prior work focused on learning from explicit step-by-step examples of how to act, we surprisingly find that if pre-trained LMs are large enough and prompted appropriately, they can effectively decompose high-level tasks into mid-level plans without any further training. However, the plans produced naively by LLMs often cannot map precisely to admissible actions. We propose a procedure that conditions on existing demonstrations and semantically translates the plans to admissible actions. Our evaluation in the recent VirtualHome environment shows that the resulting method substantially improves executability over the LLM baseline. The conducted human evaluation reveals a trade-off between executability and correctness but shows a promising sign towards extracting actionable knowledge from language models. Website at https://huangwl18.github.io/language-planner",
  "published": "2022-01-18",
  "updated": "2022-03-08",
  "year": "2022",
  "authors": [
   "Wenlong Huang",
   "Pieter Abbeel",
   "Deepak Pathak",
   "Igor Mordatch"
  ],
  "author_count": 4,
  "categories": [
   "cs.LG",
   "cs.AI",
   "cs.CL",
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.LG",
  "venue": "ICML",
  "venue_source": "semantic-scholar",
  "citations": 1686,
  "influential_citations": 99,
  "tldr": "This paper investigates the possibility of grounding high-level tasks, expressed in natural language, to a chosen set of actionable steps and proposes a procedure that conditions on existing demonstrations and semantically translates the plans to admissible actions.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Wenlong Huang",
    "id": "2158105356",
    "h_index": 10,
    "papers": 11
   },
   {
    "name": "P. Abbeel",
    "id": "1689992",
    "h_index": 165,
    "papers": 596
   },
   {
    "name": "Deepak Pathak",
    "id": "2004879394",
    "h_index": 24,
    "papers": 32
   },
   {
    "name": "Igor Mordatch",
    "id": "2080746",
    "h_index": 34,
    "papers": 49
   }
  ],
  "comment": "Project website at https://huangwl18.github.io/language-planner",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2201.07207v2",
  "pdf_url": "https://arxiv.org/pdf/2201.07207v2",
  "html_url": "https://arxiv.org/html/2201.07207v2",
  "code_url": "https://huangwl18.github.io/language-planner",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.5
 },
 {
  "id": "2112.10752",
  "slug": "high-resolution-image-synthesis-with-latent-diffusion-models",
  "title": "High-Resolution Image Synthesis with Latent Diffusion Models",
  "abstract": "By decomposing the image formation process into a sequential application of denoising autoencoders, diffusion models (DMs) achieve state-of-the-art synthesis results on image data and beyond. Additionally, their formulation allows for a guiding mechanism to control the image generation process without retraining. However, since these models typically operate directly in pixel space, optimization of powerful DMs often consumes hundreds of GPU days and inference is expensive due to sequential evaluations. To enable DM training on limited computational resources while retaining their quality and flexibility, we apply them in the latent space of powerful pretrained autoencoders. In contrast to previous work, training diffusion models on such a representation allows for the first time to reach a near-optimal point between complexity reduction and detail preservation, greatly boosting visual fidelity. By introducing cross-attention layers into the model architecture, we turn diffusion models into powerful and flexible generators for general conditioning inputs such as text or bounding boxes and high-resolution synthesis becomes possible in a convolutional manner. Our latent diffusion models (LDMs) achieve a new state of the art for image inpainting and highly competitive performance on various tasks, including unconditional image generation, semantic scene synthesis, and super-resolution, while significantly reducing computational requirements compared to pixel-based DMs. Code is available at https://github.com/CompVis/latent-diffusion .",
  "published": "2021-12-20",
  "updated": "2022-04-13",
  "year": "2021",
  "authors": [
   "Robin Rombach",
   "Andreas Blattmann",
   "Dominik Lorenz",
   "Patrick Esser",
   "Bj\u00f6rn Ommer"
  ],
  "author_count": 5,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 26964,
  "influential_citations": 5678,
  "tldr": "These latent diffusion models achieve new state of the art scores for image inpainting and class-conditional image synthesis and highly competitive performance on various tasks, including unconditional image generation, text-to-image synthesis, and super-resolution, while significantly reducing computational requirements compared to pixel-based DMs.",
  "doi": "10.1109/CVPR52688.2022.01042",
  "oa_pdf": "https://arxiv.org/pdf/2112.10752",
  "s2_authors": [
   {
    "name": "Robin Rombach",
    "id": "1660819540",
    "h_index": 22,
    "papers": 29
   },
   {
    "name": "A. Blattmann",
    "id": "119843260",
    "h_index": 17,
    "papers": 25
   },
   {
    "name": "Dominik Lorenz",
    "id": "2053482699",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "Patrick Esser",
    "id": "35175531",
    "h_index": 17,
    "papers": 24
   },
   {
    "name": "B. Ommer",
    "id": "1796707",
    "h_index": 42,
    "papers": 122
   }
  ],
  "comment": "CVPR 2022",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2112.10752v2",
  "pdf_url": "https://arxiv.org/pdf/2112.10752v2",
  "html_url": "https://arxiv.org/html/2112.10752v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "2112.06374",
  "slug": "learning-generalizable-vision-tactile-robotic-grasping-strategy-for-de",
  "title": "Learning Generalizable Vision-Tactile Robotic Grasping Strategy for Deformable Objects via Transformer",
  "abstract": "Reliable robotic grasping, especially with deformable objects such as fruits, remains a challenging task due to underactuated contact interactions with a gripper, unknown object dynamics and geometries. In this study, we propose a Transformer-based robotic grasping framework for rigid grippers that leverage tactile and visual information for safe object grasping. Specifically, the Transformer models learn physical feature embeddings with sensor feedback through performing two pre-defined explorative actions (pinching and sliding) and predict a grasping outcome through a multilayer perceptron (MLP) with a given grasping strength. Using these predictions, the gripper predicts a safe grasping strength via inference. Compared with convolutional recurrent networks, the Transformer models can capture the long-term dependencies across the image sequences and process spatial-temporal features simultaneously. We first benchmark the Transformer models on a public dataset for slip detection. Following that, we show that the Transformer models outperform a CNN+LSTM model in terms of grasping accuracy and computational efficiency. We also collect a new fruit grasping dataset and conduct online grasping experiments using the proposed framework for both seen and unseen fruits. {In addition, we extend our model to objects with different shapes and demonstrate the effectiveness of our pre-trained model trained on our large-scale fruit dataset. Our codes and dataset are public on GitHub.",
  "published": "2021-12-13",
  "updated": "2023-07-23",
  "year": "2021",
  "authors": [
   "Yunhai Han",
   "Kelin Yu",
   "Rahul Batra",
   "Nathan Boyd",
   "Chaitanya Mehta",
   "Tuo Zhao",
   "Yu She",
   "Seth Hutchinson",
   "Ye Zhao"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 105,
  "influential_citations": 4,
  "tldr": "This study proposes a transformer-based robotic grasping framework for rigid grippers that leverage tactile and visual information for safe object grasping and shows that the transformer models outperform a CNN + LSTM model in terms of grasping accuracy and computational efficiency.",
  "doi": "10.1109/tmech.2024.3400789",
  "oa_pdf": "https://arxiv.org/pdf/2112.06374",
  "s2_authors": [
   {
    "name": "Yunhai Han",
    "id": "1995513527",
    "h_index": 8,
    "papers": 14
   },
   {
    "name": "Kelin Yu",
    "id": "2261909068",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Rahul Batra",
    "id": "48603932",
    "h_index": 3,
    "papers": 20
   },
   {
    "name": "Nathan Boyd",
    "id": "2066984448",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Chaitanya Mehta",
    "id": "2346121471",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "T. Zhao",
    "id": "36345161",
    "h_index": 41,
    "papers": 122
   },
   {
    "name": "Y. She",
    "id": "2392034",
    "h_index": 17,
    "papers": 43
   },
   {
    "name": "S. Hutchinson",
    "id": "144193296",
    "h_index": 45,
    "papers": 271
   },
   {
    "name": "Ye Zhao",
    "id": "97522088",
    "h_index": 19,
    "papers": 77
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2112.06374v6",
  "pdf_url": "https://arxiv.org/pdf/2112.06374v6",
  "html_url": "https://arxiv.org/html/2112.06374v6",
  "code_url": null,
  "presented": [
   {
    "club": "saturday-robotics",
    "club_name": "Saturday Robotics",
    "session_number": 25,
    "session_title": "Saturday Robotics x Trossen @ Mission Robotics & World Models Reading Club 25: Contact-Rich Robot Learning Human Videos & Tactile. SF 8/22",
    "date_text": "Saturday, August 22, 2026 | 2:00 PM \u2013 5:00 PM",
    "city": "San Francisco, CA",
    "url": "https://lu.ma/yl76re1b",
    "listed_as": ""
   }
  ],
  "club_note": "",
  "featured": true,
  "signal": 5.03
 },
 {
  "id": "2112.05062",
  "slug": "learning-transferable-motor-skills-with-hierarchical-latent-mixture-po",
  "title": "Learning Transferable Motor Skills with Hierarchical Latent Mixture Policies",
  "abstract": "For robots operating in the real world, it is desirable to learn reusable behaviours that can effectively be transferred and adapted to numerous tasks and scenarios. We propose an approach to learn abstract motor skills from data using a hierarchical mixture latent variable model. In contrast to existing work, our method exploits a three-level hierarchy of both discrete and continuous latent variables, to capture a set of high-level behaviours while allowing for variance in how they are executed. We demonstrate in manipulation domains that the method can effectively cluster offline data into distinct, executable behaviours, while retaining the flexibility of a continuous latent variable model. The resulting skills can be transferred and fine-tuned on new tasks, unseen objects, and from state to vision-based policies, yielding better sample efficiency and asymptotic performance compared to existing skill- and imitation-based methods. We further analyse how and when the skills are most beneficial: they encourage directed exploration to cover large regions of the state space relevant to the task, making them most effective in challenging sparse-reward settings.",
  "published": "2021-12-09",
  "updated": "2022-03-14",
  "year": "2021",
  "authors": [
   "Dushyant Rao",
   "Fereshteh Sadeghi",
   "Leonard Hasenclever",
   "Markus Wulfmeier",
   "Martina Zambelli",
   "Giulia Vezzani",
   "Dhruva Tirumala",
   "Yusuf Aytar",
   "Josh Merel",
   "Nicolas Heess",
   "Raia Hadsell"
  ],
  "author_count": 11,
  "categories": [
   "cs.LG",
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.LG",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 34,
  "influential_citations": 1,
  "tldr": "This work proposes an approach to learn abstract motor skills from data using a hierarchical mixture latent variable model, which exploits a three-level hierarchy of both discrete and continuous latent variables, to capture a set of high-level behaviours while allowing for variance in how they are executed.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Dushyant Rao",
    "id": "143668237",
    "h_index": 18,
    "papers": 35
   },
   {
    "name": "Fereshteh Sadeghi",
    "id": "3253737",
    "h_index": 15,
    "papers": 31
   },
   {
    "name": "Leonard Hasenclever",
    "id": "40401956",
    "h_index": 28,
    "papers": 50
   },
   {
    "name": "Markus Wulfmeier",
    "id": "3331786",
    "h_index": 25,
    "papers": 64
   },
   {
    "name": "Martina Zambelli",
    "id": "7455600",
    "h_index": 11,
    "papers": 17
   },
   {
    "name": "G. Vezzani",
    "id": "3433312",
    "h_index": 12,
    "papers": 24
   },
   {
    "name": "Dhruva Tirumala",
    "id": "7794353",
    "h_index": 13,
    "papers": 21
   },
   {
    "name": "Y. Aytar",
    "id": "3152281",
    "h_index": 30,
    "papers": 57
   },
   {
    "name": "J. Merel",
    "id": "1879232",
    "h_index": 28,
    "papers": 51
   },
   {
    "name": "N. Heess",
    "id": "2801204",
    "h_index": 73,
    "papers": 192
   },
   {
    "name": "R. Hadsell",
    "id": "2315504",
    "h_index": 51,
    "papers": 111
   }
  ],
  "comment": "",
  "topics": [
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2112.05062v2",
  "pdf_url": "https://arxiv.org/pdf/2112.05062v2",
  "html_url": "https://arxiv.org/html/2112.05062v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.04
 },
 {
  "id": "2112.01511",
  "slug": "the-surprising-effectiveness-of-representation-learning-for-visual-imi",
  "title": "The Surprising Effectiveness of Representation Learning for Visual Imitation",
  "abstract": "While visual imitation learning offers one of the most effective ways of learning from visual demonstrations, generalizing from them requires either hundreds of diverse demonstrations, task specific priors, or large, hard-to-train parametric models. One reason such complexities arise is because standard visual imitation frameworks try to solve two coupled problems at once: learning a succinct but good representation from the diverse visual data, while simultaneously learning to associate the demonstrated actions with such representations. Such joint learning causes an interdependence between these two problems, which often results in needing large amounts of demonstrations for learning. To address this challenge, we instead propose to decouple representation learning from behavior learning for visual imitation. First, we learn a visual representation encoder from offline data using standard supervised and self-supervised learning methods. Once the representations are trained, we use non-parametric Locally Weighted Regression to predict the actions. We experimentally show that this simple decoupling improves the performance of visual imitation models on both offline demonstration datasets and real-robot door opening compared to prior work in visual imitation. All of our generated data, code, and robot videos are publicly available at https://jyopari.github.io/VINN/.",
  "published": "2021-12-02",
  "updated": "2021-12-06",
  "year": "2021",
  "authors": [
   "Jyothish Pari",
   "Nur Muhammad Shafiullah",
   "Sridhar Pandian Arunachalam",
   "Lerrel Pinto"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "RSS",
  "venue_source": "semantic-scholar",
  "citations": 226,
  "influential_citations": 25,
  "tldr": "This work proposes to decouple representation learning from behavior learning for visual imitation, and experimentally shows that this simple decoupling improves the performance of visual imitation models on both offline demonstration datasets and real-robot door opening.",
  "doi": "10.15607/rss.2022.xviii.010",
  "oa_pdf": "https://doi.org/10.15607/rss.2022.xviii.010",
  "s2_authors": [
   {
    "name": "Jyothish Pari",
    "id": "1518270974",
    "h_index": 9,
    "papers": 14
   },
   {
    "name": "Nur Muhammad (Mahi) Shafiullah",
    "id": "84146411",
    "h_index": 9,
    "papers": 11
   },
   {
    "name": "Sridhar Pandian Arunachalam",
    "id": "2144825236",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Lerrel Pinto",
    "id": "34026610",
    "h_index": 41,
    "papers": 70
   }
  ],
  "comment": "The first two authors contributed equally",
  "topics": [
   "imitation-diffusion"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2112.01511v2",
  "pdf_url": "https://arxiv.org/pdf/2112.01511v2",
  "html_url": "https://arxiv.org/html/2112.01511v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.86
 },
 {
  "id": "2111.06377",
  "slug": "masked-autoencoders-are-scalable-vision-learners",
  "title": "Masked Autoencoders Are Scalable Vision Learners",
  "abstract": "This paper shows that masked autoencoders (MAE) are scalable self-supervised learners for computer vision. Our MAE approach is simple: we mask random patches of the input image and reconstruct the missing pixels. It is based on two core designs. First, we develop an asymmetric encoder-decoder architecture, with an encoder that operates only on the visible subset of patches (without mask tokens), along with a lightweight decoder that reconstructs the original image from the latent representation and mask tokens. Second, we find that masking a high proportion of the input image, e.g., 75%, yields a nontrivial and meaningful self-supervisory task. Coupling these two designs enables us to train large models efficiently and effectively: we accelerate training (by 3x or more) and improve accuracy. Our scalable approach allows for learning high-capacity models that generalize well: e.g., a vanilla ViT-Huge model achieves the best accuracy (87.8%) among methods that use only ImageNet-1K data. Transfer performance in downstream tasks outperforms supervised pre-training and shows promising scaling behavior.",
  "published": "2021-11-11",
  "updated": "2021-12-19",
  "year": "2021",
  "authors": [
   "Kaiming He",
   "Xinlei Chen",
   "Saining Xie",
   "Yanghao Li",
   "Piotr Doll\u00e1r",
   "Ross Girshick"
  ],
  "author_count": 6,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 12583,
  "influential_citations": 2077,
  "tldr": "This paper develops an asymmetric encoder-decoder architecture, with an encoder that operates only on the visible subset of patches (without mask tokens), along with a lightweight decoder that reconstructs the original image from the latent representation and mask tokens.",
  "doi": "10.1109/CVPR52688.2022.01553",
  "oa_pdf": "https://doi.org/10.57702/hevhzb1p",
  "s2_authors": [
   {
    "name": "Kaiming He",
    "id": "2058350112",
    "h_index": 9,
    "papers": 11
   },
   {
    "name": "Xinlei Chen",
    "id": "39717886",
    "h_index": 39,
    "papers": 53
   },
   {
    "name": "Saining Xie",
    "id": "1817030",
    "h_index": 32,
    "papers": 46
   },
   {
    "name": "Yanghao Li",
    "id": "2366569281",
    "h_index": 7,
    "papers": 9
   },
   {
    "name": "Piotr Doll'ar",
    "id": "2065731243",
    "h_index": 7,
    "papers": 7
   },
   {
    "name": "Ross B. Girshick",
    "id": "2983898",
    "h_index": 80,
    "papers": 113
   }
  ],
  "comment": "Tech report. arXiv v2: add more transfer learning results; v3: add robustness evaluation",
  "topics": [
   "foundation-pretraining",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2111.06377v3",
  "pdf_url": "https://arxiv.org/pdf/2111.06377v3",
  "html_url": "https://arxiv.org/html/2111.06377v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "2111.03941",
  "slug": "time-discretization-invariant-safe-action-repetition-for-policy-gradie",
  "title": "Time Discretization-Invariant Safe Action Repetition for Policy Gradient Methods",
  "abstract": "In reinforcement learning, continuous time is often discretized by a time scale $\u03b4$, to which the resulting performance is known to be highly sensitive. In this work, we seek to find a $\u03b4$-invariant algorithm for policy gradient (PG) methods, which performs well regardless of the value of $\u03b4$. We first identify the underlying reasons that cause PG methods to fail as $\u03b4\\to 0$, proving that the variance of the PG estimator can diverge to infinity in stochastic environments under a certain assumption of stochasticity. While durative actions or action repetition can be employed to have $\u03b4$-invariance, previous action repetition methods cannot immediately react to unexpected situations in stochastic environments. We thus propose a novel $\u03b4$-invariant method named Safe Action Repetition (SAR) applicable to any existing PG algorithm. SAR can handle the stochasticity of environments by adaptively reacting to changes in states during action repetition. We empirically show that our method is not only $\u03b4$-invariant but also robust to stochasticity, outperforming previous $\u03b4$-invariant approaches on eight MuJoCo environments with both deterministic and stochastic settings. Our code is available at https://vision.snu.ac.kr/projects/sar.",
  "published": "2021-11-06",
  "updated": "2022-01-27",
  "year": "2021",
  "authors": [
   "Seohong Park",
   "Jaekyeom Kim",
   "Gunhee Kim"
  ],
  "author_count": 3,
  "categories": [
   "cs.LG",
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.LG",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 33,
  "influential_citations": 3,
  "tldr": "This work identifies the underlying reasons that cause PG methods to fail as $\\delta \\to 0$, proving that the variance of the PG estimator can diverge to infinity in stochastic environments under a certain assumption of Stochasticity.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Seohong Park",
    "id": "2118885320",
    "h_index": 16,
    "papers": 26
   },
   {
    "name": "Jaekyeom Kim",
    "id": "65924935",
    "h_index": 12,
    "papers": 22
   },
   {
    "name": "Gunhee Kim",
    "id": "70308241",
    "h_index": 22,
    "papers": 36
   }
  ],
  "comment": "Accepted to NeurIPS 2021",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2111.03941v6",
  "pdf_url": "https://arxiv.org/pdf/2111.03941v6",
  "html_url": "https://arxiv.org/html/2111.03941v6",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.03
 },
 {
  "id": "2111.00210",
  "slug": "mastering-atari-games-with-limited-data",
  "title": "Mastering Atari Games with Limited Data",
  "abstract": "Reinforcement learning has achieved great success in many applications. However, sample efficiency remains a key challenge, with prominent methods requiring millions (or even billions) of environment steps to train. Recently, there has been significant progress in sample efficient image-based RL algorithms; however, consistent human-level performance on the Atari game benchmark remains an elusive goal. We propose a sample efficient model-based visual RL algorithm built on MuZero, which we name EfficientZero. Our method achieves 194.3% mean human performance and 109.0% median performance on the Atari 100k benchmark with only two hours of real-time game experience and outperforms the state SAC in some tasks on the DMControl 100k benchmark. This is the first time an algorithm achieves super-human performance on Atari games with such little data. EfficientZero's performance is also close to DQN's performance at 200 million frames while we consume 500 times less data. EfficientZero's low sample complexity and high performance can bring RL closer to real-world applicability. We implement our algorithm in an easy-to-understand manner and it is available at https://github.com/YeWR/EfficientZero. We hope it will accelerate the research of MCTS-based RL algorithms in the wider community.",
  "published": "2021-10-30",
  "updated": "2021-12-12",
  "year": "2021",
  "authors": [
   "Weirui Ye",
   "Shaohuai Liu",
   "Thanard Kurutach",
   "Pieter Abbeel",
   "Yang Gao"
  ],
  "author_count": 5,
  "categories": [
   "cs.LG",
   "cs.AI",
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.LG",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 340,
  "influential_citations": 37,
  "tldr": "This work proposes a sample efficient model-based visual RL algorithm built on MuZero, which it is hoped will accelerate the research of MCTS-based RL algorithms in the wider community and achieves super-human performance on Atari games with such little data.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Weirui Ye",
    "id": "83546634",
    "h_index": 12,
    "papers": 22
   },
   {
    "name": "Shao-Wei Liu",
    "id": "48641958",
    "h_index": 13,
    "papers": 52
   },
   {
    "name": "Thanard Kurutach",
    "id": "2765564",
    "h_index": 9,
    "papers": 11
   },
   {
    "name": "P. Abbeel",
    "id": "1689992",
    "h_index": 165,
    "papers": 596
   },
   {
    "name": "Yang Gao",
    "id": "2145972631",
    "h_index": 10,
    "papers": 14
   }
  ],
  "comment": "Published at NeurIPS 2021; Homepage: https://yewr.github.io/projects/efficientzero/",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2111.00210v2",
  "pdf_url": "https://arxiv.org/pdf/2111.00210v2",
  "html_url": "https://arxiv.org/html/2111.00210v2",
  "code_url": "https://yewr.github.io/projects/efficientzero/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.03
 },
 {
  "id": "2110.14565",
  "slug": "dreamerpro-reconstruction-free-model-based-reinforcement-learning-with",
  "title": "DreamerPro: Reconstruction-Free Model-Based Reinforcement Learning with Prototypical Representations",
  "abstract": "Top-performing Model-Based Reinforcement Learning (MBRL) agents, such as Dreamer, learn the world model by reconstructing the image observations. Hence, they often fail to discard task-irrelevant details and struggle to handle visual distractions. To address this issue, previous work has proposed to contrastively learn the world model, but the performance tends to be inferior in the absence of distractions. In this paper, we seek to enhance robustness to distractions for MBRL agents. Specifically, we consider incorporating prototypical representations, which have yielded more accurate and robust results than contrastive approaches in computer vision. However, it remains elusive how prototypical representations can benefit temporal dynamics learning in MBRL, since they treat each image independently without capturing temporal structures. To this end, we propose to learn the prototypes from the recurrent states of the world model, thereby distilling temporal structures from past observations and actions into the prototypes. The resulting model, DreamerPro, successfully combines Dreamer with prototypes, making large performance gains on the DeepMind Control suite both in the standard setting and when there are complex background distractions. Code available at https://github.com/fdeng18/dreamer-pro .",
  "published": "2021-10-27",
  "updated": "2021-10-27",
  "year": "2021",
  "authors": [
   "Fei Deng",
   "Ingook Jang",
   "Sungjin Ahn"
  ],
  "author_count": 3,
  "categories": [
   "cs.LG",
   "cs.AI"
  ],
  "primary_category": "cs.LG",
  "venue": "ICML",
  "venue_source": "semantic-scholar",
  "citations": 96,
  "influential_citations": 17,
  "tldr": "This paper proposes to learn the prototypes from the recurrent states of the world model, thereby distilling temporal structures from past observations and actions into the prototypes, and proposes the resulting model, DreamerPro, which successfully combines Dreamer with prototypes, making large performance gains on the DeepMind Control suite both in the standard setting and when there are complex background distractions.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Fei Deng",
    "id": "1379946625",
    "h_index": 9,
    "papers": 14
   },
   {
    "name": "Ingook Jang",
    "id": "2941497",
    "h_index": 5,
    "papers": 27
   },
   {
    "name": "Sungjin Ahn",
    "id": "1882851",
    "h_index": 17,
    "papers": 28
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "rl-control",
   "safety-eval"
  ],
  "orgs": [
   "Google DeepMind"
  ],
  "abs_url": "https://arxiv.org/abs/2110.14565v1",
  "pdf_url": "https://arxiv.org/pdf/2110.14565v1",
  "html_url": "https://arxiv.org/html/2110.14565v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.99
 },
 {
  "id": "2110.10809",
  "slug": "hierarchical-skills-for-efficient-exploration",
  "title": "Hierarchical Skills for Efficient Exploration",
  "abstract": "In reinforcement learning, pre-trained low-level skills have the potential to greatly facilitate exploration. However, prior knowledge of the downstream task is required to strike the right balance between generality (fine-grained control) and specificity (faster learning) in skill design. In previous work on continuous control, the sensitivity of methods to this trade-off has not been addressed explicitly, as locomotion provides a suitable prior for navigation tasks, which have been of foremost interest. In this work, we analyze this trade-off for low-level policy pre-training with a new benchmark suite of diverse, sparse-reward tasks for bipedal robots. We alleviate the need for prior knowledge by proposing a hierarchical skill learning framework that acquires skills of varying complexity in an unsupervised manner. For utilization on downstream tasks, we present a three-layered hierarchical learning algorithm to automatically trade off between general and specific skills as required by the respective task. In our experiments, we show that our approach performs this trade-off effectively and achieves better results than current state-of-the-art methods for end- to-end hierarchical reinforcement learning and unsupervised skill discovery. Code and videos are available at https://facebookresearch.github.io/hsd3 .",
  "published": "2021-10-20",
  "updated": "2021-10-20",
  "year": "2021",
  "authors": [
   "Jonas Gehring",
   "Gabriel Synnaeve",
   "Andreas Krause",
   "Nicolas Usunier"
  ],
  "author_count": 4,
  "categories": [
   "cs.LG",
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.LG",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 52,
  "influential_citations": 2,
  "tldr": "This work alleviates the need for prior knowledge by proposing a hierarchical skill learning framework that acquires skills of varying complexity in an unsupervised manner, and presents a three-layered hierarchical learning algorithm to automatically trade off between general and specific skills as required by the respective task.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jonas Gehring",
    "id": "2401865",
    "h_index": 12,
    "papers": 20
   },
   {
    "name": "Gabriel Synnaeve",
    "id": "2282478",
    "h_index": 58,
    "papers": 209
   },
   {
    "name": "Andreas Krause",
    "id": "145343838",
    "h_index": 92,
    "papers": 288
   },
   {
    "name": "Nicolas Usunier",
    "id": "1746841",
    "h_index": 46,
    "papers": 131
   }
  ],
  "comment": "To appear in 35th Conference on Neural Information Processing Systems (NeurIPS 2021)",
  "topics": [
   "humanoids",
   "rl-control",
   "navigation",
   "foundation-pretraining"
  ],
  "orgs": [
   "Meta FAIR"
  ],
  "abs_url": "https://arxiv.org/abs/2110.10809v1",
  "pdf_url": "https://arxiv.org/pdf/2110.10809v1",
  "html_url": "https://arxiv.org/html/2110.10809v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.72
 },
 {
  "id": "2110.09514",
  "slug": "discovering-and-achieving-goals-via-world-models",
  "title": "Discovering and Achieving Goals via World Models",
  "abstract": "How can artificial agents learn to solve many diverse tasks in complex visual environments in the absence of any supervision? We decompose this question into two problems: discovering new goals and learning to reliably achieve them. We introduce Latent Explorer Achiever (LEXA), a unified solution to these that learns a world model from image inputs and uses it to train an explorer and an achiever policy from imagined rollouts. Unlike prior methods that explore by reaching previously visited states, the explorer plans to discover unseen surprising states through foresight, which are then used as diverse targets for the achiever to practice. After the unsupervised phase, LEXA solves tasks specified as goal images zero-shot without any additional learning. LEXA substantially outperforms previous approaches to unsupervised goal-reaching, both on prior benchmarks and on a new challenging benchmark with a total of 40 test tasks spanning across four standard robotic manipulation and locomotion domains. LEXA further achieves goals that require interacting with multiple objects in sequence. Finally, to demonstrate the scalability and generality of LEXA, we train a single general agent across four distinct environments. Code and videos at https://orybkin.github.io/lexa/",
  "published": "2021-10-18",
  "updated": "2021-10-18",
  "year": "2021",
  "authors": [
   "Russell Mendonca",
   "Oleh Rybkin",
   "Kostas Daniilidis",
   "Danijar Hafner",
   "Deepak Pathak"
  ],
  "author_count": 5,
  "categories": [
   "cs.LG",
   "cs.AI",
   "cs.CV",
   "cs.RO",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 172,
  "influential_citations": 23,
  "tldr": "The Latent Explorer Achiever (LEXA), a unified solution to these that learns a world model from image inputs and uses it to train an explorer and an achiever policy from imagined rollouts, substantially outperforms previous approaches to unsupervised goal-reaching.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "R. Mendonca",
    "id": "35509365",
    "h_index": 15,
    "papers": 24
   },
   {
    "name": "Oleh Rybkin",
    "id": "40900227",
    "h_index": 14,
    "papers": 27
   },
   {
    "name": "Kostas Daniilidis",
    "id": "2065557091",
    "h_index": 25,
    "papers": 89
   },
   {
    "name": "Danijar Hafner",
    "id": "35006479",
    "h_index": 25,
    "papers": 47
   },
   {
    "name": "Deepak Pathak",
    "id": "2004879394",
    "h_index": 24,
    "papers": 32
   }
  ],
  "comment": "NeurIPS 2021. First two authors contributed equally. Website at https://orybkin.github.io/lexa/",
  "topics": [
   "world-models",
   "humanoids"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2110.09514v1",
  "pdf_url": "https://arxiv.org/pdf/2110.09514v1",
  "html_url": "https://arxiv.org/html/2110.09514v1",
  "code_url": "https://orybkin.github.io/lexa/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.74
 },
 {
  "id": "2110.07058",
  "slug": "ego4d-around-the-world-in-3-000-hours-of-egocentric-video",
  "title": "Ego4D: Around the World in 3,000 Hours of Egocentric Video",
  "abstract": "We introduce Ego4D, a massive-scale egocentric video dataset and benchmark suite. It offers 3,670 hours of daily-life activity video spanning hundreds of scenarios (household, outdoor, workplace, leisure, etc.) captured by 931 unique camera wearers from 74 worldwide locations and 9 different countries. The approach to collection is designed to uphold rigorous privacy and ethics standards with consenting participants and robust de-identification procedures where relevant. Ego4D dramatically expands the volume of diverse egocentric video footage publicly available to the research community. Portions of the video are accompanied by audio, 3D meshes of the environment, eye gaze, stereo, and/or synchronized videos from multiple egocentric cameras at the same event. Furthermore, we present a host of new benchmark challenges centered around understanding the first-person visual experience in the past (querying an episodic memory), present (analyzing hand-object manipulation, audio-visual conversation, and social interactions), and future (forecasting activities). By publicly sharing this massive annotated dataset and benchmark suite, we aim to push the frontier of first-person perception. Project page: https://ego4d-data.org/",
  "published": "2021-10-13",
  "updated": "2022-03-11",
  "year": "2021",
  "authors": [
   "Kristen Grauman",
   "Andrew Westbury",
   "Eugene Byrne",
   "Zachary Chavis",
   "Antonino Furnari",
   "Rohit Girdhar",
   "Jackson Hamburger",
   "Hao Jiang",
   "Miao Liu",
   "Xingyu Liu",
   "Miguel Martin",
   "Tushar Nagarajan",
   "Ilija Radosavovic",
   "Santhosh Kumar Ramakrishnan",
   "Fiona Ryan",
   "Jayant Sharma",
   "Michael Wray",
   "Mengmeng Xu",
   "Eric Zhongcong Xu",
   "Chen Zhao",
   "Siddhant Bansal",
   "Dhruv Batra",
   "Vincent Cartillier",
   "Sean Crane",
   "Tien Do",
   "Morrie Doulaty",
   "Akshay Erapalli",
   "Christoph Feichtenhofer",
   "Adriano Fragomeni",
   "Qichen Fu",
   "Abrham Gebreselasie",
   "Cristina Gonzalez",
   "James Hillis",
   "Xuhua Huang",
   "Yifei Huang",
   "Wenqi Jia",
   "Weslie Khoo",
   "Jachym Kolar",
   "Satwik Kottur",
   "Anurag Kumar",
   "Federico Landini",
   "Chao Li",
   "Yanghao Li",
   "Zhenqiang Li",
   "Karttikeya Mangalam",
   "Raghava Modhugu",
   "Jonathan Munro",
   "Tullie Murrell",
   "Takumi Nishiyasu",
   "Will Price",
   "Paola Ruiz Puentes",
   "Merey Ramazanova",
   "Leda Sari",
   "Kiran Somasundaram",
   "Audrey Southerland",
   "Yusuke Sugano",
   "Ruijie Tao",
   "Minh Vo",
   "Yuchen Wang",
   "Xindi Wu",
   "Takuma Yagi",
   "Ziwei Zhao",
   "Yunyi Zhu",
   "Pablo Arbelaez",
   "David Crandall",
   "Dima Damen",
   "Giovanni Maria Farinella",
   "Christian Fuegen",
   "Bernard Ghanem",
   "Vamsi Krishna Ithapu",
   "C. V. Jawahar",
   "Hanbyul Joo",
   "Kris Kitani",
   "Haizhou Li",
   "Richard Newcombe",
   "Aude Oliva",
   "Hyun Soo Park",
   "James M. Rehg",
   "Yoichi Sato",
   "Jianbo Shi",
   "Mike Zheng Shou",
   "Antonio Torralba",
   "Lorenzo Torresani",
   "Mingfei Yan",
   "Jitendra Malik"
  ],
  "author_count": 85,
  "categories": [
   "cs.CV",
   "cs.AI"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 1994,
  "influential_citations": 286,
  "tldr": "Ego4D, a massive-scale egocentric video dataset and benchmark suite, is introduced and a host of new benchmark challenges centered around understanding the first-person visual experience in the past, present, and future are presented.",
  "doi": "10.1109/CVPR52688.2022.01842",
  "oa_pdf": "https://arxiv.org/pdf/2110.07058",
  "s2_authors": [
   {
    "name": "K. Grauman",
    "id": "1794409",
    "h_index": 99,
    "papers": 295
   },
   {
    "name": "Andrew Westbury",
    "id": "2127379149",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Eugene Byrne",
    "id": "2132203390",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Zachary Chavis",
    "id": "2312205852",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Antonino Furnari",
    "id": "1792681",
    "h_index": 29,
    "papers": 152
   },
   {
    "name": "Rohit Girdhar",
    "id": "3102850",
    "h_index": 31,
    "papers": 95
   },
   {
    "name": "Jackson Hamburger",
    "id": "2132855455",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Hao Jiang",
    "id": "143891655",
    "h_index": 20,
    "papers": 57
   },
   {
    "name": "Miao Liu",
    "id": "2108511234",
    "h_index": 16,
    "papers": 28
   },
   {
    "name": "Xingyu Liu",
    "id": "2146036705",
    "h_index": 10,
    "papers": 17
   },
   {
    "name": "Miguel Martin",
    "id": "2110608350",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Tushar Nagarajan",
    "id": "38661780",
    "h_index": 14,
    "papers": 33
   },
   {
    "name": "Ilija Radosavovic",
    "id": "30407997",
    "h_index": 20,
    "papers": 22
   },
   {
    "name": "Santhosh K. Ramakrishnan",
    "id": "21810992",
    "h_index": 18,
    "papers": 30
   },
   {
    "name": "Fiona Ryan",
    "id": "119797486",
    "h_index": 12,
    "papers": 17
   },
   {
    "name": "J. Sharma",
    "id": "50857765",
    "h_index": 15,
    "papers": 32
   },
   {
    "name": "Michael Wray",
    "id": "145032628",
    "h_index": 15,
    "papers": 51
   },
   {
    "name": "Mengmeng Xu",
    "id": "97375393",
    "h_index": 16,
    "papers": 49
   },
   {
    "name": "Eric Z. Xu",
    "id": "2065722815",
    "h_index": 8,
    "papers": 15
   },
   {
    "name": "Chen Zhao",
    "id": "2109529407",
    "h_index": 15,
    "papers": 30
   },
   {
    "name": "Siddhant Bansal",
    "id": "16936840",
    "h_index": 8,
    "papers": 17
   },
   {
    "name": "Dhruv Batra",
    "id": "1746610",
    "h_index": 87,
    "papers": 327
   },
   {
    "name": "Vincent Cartillier",
    "id": "51002409",
    "h_index": 8,
    "papers": 15
   },
   {
    "name": "S. Crane",
    "id": "36856979",
    "h_index": 6,
    "papers": 6
   },
   {
    "name": "Tien Do",
    "id": "145756059",
    "h_index": 11,
    "papers": 29
   },
   {
    "name": "Morrie Doulaty",
    "id": "2132839352",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Akshay Erapalli",
    "id": "2132829391",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Christoph Feichtenhofer",
    "id": "2322150",
    "h_index": 36,
    "papers": 57
   },
   {
    "name": "Adriano Fragomeni",
    "id": "2078661958",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Qichen Fu",
    "id": "2153259742",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Christian Fuegen",
    "id": "39547770",
    "h_index": 21,
    "papers": 50
   },
   {
    "name": "A. Gebreselasie",
    "id": "150013809",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Cristina Gonz\u00e1lez",
    "id": "2086769306",
    "h_index": 8,
    "papers": 13
   },
   {
    "name": "James M. Hillis",
    "id": "3325232",
    "h_index": 17,
    "papers": 52
   },
   {
    "name": "Xuhua Huang",
    "id": "2118520784",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Yifei Huang",
    "id": "48355651",
    "h_index": 24,
    "papers": 70
   },
   {
    "name": "Wenqi Jia",
    "id": "2072957749",
    "h_index": 8,
    "papers": 10
   },
   {
    "name": "Weslie Khoo",
    "id": "15626408",
    "h_index": 8,
    "papers": 14
   },
   {
    "name": "J\u00e1chym Kol\u00e1r",
    "id": "30380885",
    "h_index": 10,
    "papers": 28
   },
   {
    "name": "Satwik Kottur",
    "id": "2150275",
    "h_index": 17,
    "papers": 44
   },
   {
    "name": "Anurag Kumar",
    "id": "47311290",
    "h_index": 23,
    "papers": 68
   },
   {
    "name": "F. Landini",
    "id": "102333508",
    "h_index": 18,
    "papers": 129
   },
   {
    "name": "Chao Li",
    "id": "2150358103",
    "h_index": 7,
    "papers": 9
   },
   {
    "name": "Yanghao Li",
    "id": "2359205979",
    "h_index": 12,
    "papers": 14
   },
   {
    "name": "Zhenqiang Li",
    "id": "2109633863",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "K. Mangalam",
    "id": "11379939",
    "h_index": 23,
    "papers": 45
   },
   {
    "name": "Raghava Modhugu",
    "id": "2037870106",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Jonathan Munro",
    "id": "47077615",
    "h_index": 8,
    "papers": 19
   },
   {
    "name": "Tullie Murrell",
    "id": "1388373582",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "Takumi Nishiyasu",
    "id": "2047039599",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Will Price",
    "id": "50065546",
    "h_index": 8,
    "papers": 17
   },
   {
    "name": "Paola Ruiz Puentes",
    "id": "1946687752",
    "h_index": 8,
    "papers": 16
   },
   {
    "name": "Merey Ramazanova",
    "id": "4042496",
    "h_index": 9,
    "papers": 20
   },
   {
    "name": "Leda Sari",
    "id": "2769735",
    "h_index": 12,
    "papers": 36
   },
   {
    "name": "K. Somasundaram",
    "id": "31604945",
    "h_index": 13,
    "papers": 78
   },
   {
    "name": "A. Southerland",
    "id": "7824981",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "Yusuke Sugano",
    "id": "1751242",
    "h_index": 34,
    "papers": 83
   },
   {
    "name": "Ruijie Tao",
    "id": "1866636284",
    "h_index": 14,
    "papers": 33
   },
   {
    "name": "Minh Vo",
    "id": "1500674963",
    "h_index": 7,
    "papers": 12
   },
   {
    "name": "Yuchen Wang",
    "id": "2108898718",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Xindi Wu",
    "id": "2155226697",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Takuma Yagi",
    "id": "2052544968",
    "h_index": 9,
    "papers": 21
   },
   {
    "name": "Yunyi Zhu",
    "id": "2882203",
    "h_index": 7,
    "papers": 14
   },
   {
    "name": "P. Arbel\u00e1ez",
    "id": "9739979",
    "h_index": 21,
    "papers": 50
   },
   {
    "name": "David J. Crandall",
    "id": "2821130",
    "h_index": 54,
    "papers": 216
   },
   {
    "name": "D. Damen",
    "id": "145089978",
    "h_index": 45,
    "papers": 201
   },
   {
    "name": "G. Farinella",
    "id": "1729739",
    "h_index": 37,
    "papers": 320
   },
   {
    "name": "Bernard Ghanem",
    "id": "2931652",
    "h_index": 75,
    "papers": 329
   },
   {
    "name": "V. Ithapu",
    "id": "2736958",
    "h_index": 20,
    "papers": 72
   },
   {
    "name": "C. V. Jawahar",
    "id": "1694502",
    "h_index": 58,
    "papers": 450
   },
   {
    "name": "H. Joo",
    "id": "7996087",
    "h_index": 23,
    "papers": 43
   },
   {
    "name": "Kris Kitani",
    "id": "144040368",
    "h_index": 46,
    "papers": 116
   },
   {
    "name": "Haizhou Li",
    "id": "2108493029",
    "h_index": 2,
    "papers": 10
   },
   {
    "name": "Richard A. Newcombe",
    "id": "50366818",
    "h_index": 31,
    "papers": 57
   },
   {
    "name": "A. Oliva",
    "id": "143868587",
    "h_index": 73,
    "papers": 302
   },
   {
    "name": "H. Park",
    "id": "2110440192",
    "h_index": 30,
    "papers": 73
   },
   {
    "name": "James M. Rehg",
    "id": "144177248",
    "h_index": 84,
    "papers": 325
   },
   {
    "name": "Yoichi Sato",
    "id": "2110962740",
    "h_index": 3,
    "papers": 15
   },
   {
    "name": "Jianbo Shi",
    "id": "46865129",
    "h_index": 57,
    "papers": 168
   },
   {
    "name": "M. Shou",
    "id": "2047358650",
    "h_index": 49,
    "papers": 278
   },
   {
    "name": "A. Torralba",
    "id": "143805211",
    "h_index": 142,
    "papers": 372
   },
   {
    "name": "L. Torresani",
    "id": "1732879",
    "h_index": 57,
    "papers": 143
   },
   {
    "name": "Mingfei Yan",
    "id": "47901986",
    "h_index": 10,
    "papers": 37
   },
   {
    "name": "J. Malik",
    "id": "153652147",
    "h_index": 64,
    "papers": 99
   }
  ],
  "comment": "To appear in the Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR), 2022. This version updates the baseline result numbers for the Hands and Objects benchmark (appendix)",
  "topics": [
   "egocentric-data"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2110.07058v3",
  "pdf_url": "https://arxiv.org/pdf/2110.07058v3",
  "html_url": "https://arxiv.org/html/2110.07058v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "2110.05457",
  "slug": "legged-robots-that-keep-on-learning-fine-tuning-locomotion-policies-in",
  "title": "Legged Robots that Keep on Learning: Fine-Tuning Locomotion Policies in the Real World",
  "abstract": "Legged robots are physically capable of traversing a wide range of challenging environments, but designing controllers that are sufficiently robust to handle this diversity has been a long-standing challenge in robotics. Reinforcement learning presents an appealing approach for automating the controller design process and has been able to produce remarkably robust controllers when trained in a suitable range of environments. However, it is difficult to predict all likely conditions the robot will encounter during deployment and enumerate them at training-time. What if instead of training controllers that are robust enough to handle any eventuality, we enable the robot to continually learn in any setting it finds itself in? This kind of real-world reinforcement learning poses a number of challenges, including efficiency, safety, and autonomy. To address these challenges, we propose a practical robot reinforcement learning system for fine-tuning locomotion policies in the real world. We demonstrate that a modest amount of real-world training can substantially improve performance during deployment, and this enables a real A1 quadrupedal robot to autonomously fine-tune multiple locomotion skills in a range of environments, including an outdoor lawn and a variety of indoor terrains.",
  "published": "2021-10-11",
  "updated": "2021-10-11",
  "year": "2021",
  "authors": [
   "Laura Smith",
   "J. Chase Kew",
   "Xue Bin Peng",
   "Sehoon Ha",
   "Jie Tan",
   "Sergey Levine"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA",
  "venue_source": "semantic-scholar",
  "citations": 154,
  "influential_citations": 7,
  "tldr": "It is demonstrated that a modest amount of real-world training can substantially improve performance during deployment, and this enables a real A1 quadrupedal robot to autonomously fine-tune multiple locomotion skills in a range of environments, including an outdoor lawn and a variety of indoor terrains.",
  "doi": "10.1109/icra46639.2022.9812166",
  "oa_pdf": "https://arxiv.org/pdf/2110.05457",
  "s2_authors": [
   {
    "name": "Laura M. Smith",
    "id": "152447364",
    "h_index": 15,
    "papers": 16
   },
   {
    "name": "J. Kew",
    "id": "46181163",
    "h_index": 8,
    "papers": 18
   },
   {
    "name": "Xue Bin Peng",
    "id": "2375236722",
    "h_index": 34,
    "papers": 41
   },
   {
    "name": "Sehoon Ha",
    "id": "2248552",
    "h_index": 26,
    "papers": 69
   },
   {
    "name": "Jie Tan",
    "id": "1739176520",
    "h_index": 37,
    "papers": 68
   },
   {
    "name": "S. Levine",
    "id": "1736651",
    "h_index": 167,
    "papers": 551
   }
  ],
  "comment": "Project website: https://sites.google.com/berkeley.edu/fine-tuning-locomotion",
  "topics": [
   "humanoids",
   "rl-control",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2110.05457v1",
  "pdf_url": "https://arxiv.org/pdf/2110.05457v1",
  "html_url": "https://arxiv.org/html/2110.05457v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.69
 },
 {
  "id": "2110.04627",
  "slug": "vector-quantized-image-modeling-with-improved-vqgan",
  "title": "Vector-quantized Image Modeling with Improved VQGAN",
  "abstract": "Pretraining language models with next-token prediction on massive text corpora has delivered phenomenal zero-shot, few-shot, transfer learning and multi-tasking capabilities on both generative and discriminative language tasks. Motivated by this success, we explore a Vector-quantized Image Modeling (VIM) approach that involves pretraining a Transformer to predict rasterized image tokens autoregressively. The discrete image tokens are encoded from a learned Vision-Transformer-based VQGAN (ViT-VQGAN). We first propose multiple improvements over vanilla VQGAN from architecture to codebook learning, yielding better efficiency and reconstruction fidelity. The improved ViT-VQGAN further improves vector-quantized image modeling tasks, including unconditional, class-conditioned image generation and unsupervised representation learning. When trained on ImageNet at \\(256\\times256\\) resolution, we achieve Inception Score (IS) of 175.1 and Fr'echet Inception Distance (FID) of 4.17, a dramatic improvement over the vanilla VQGAN, which obtains 70.6 and 17.04 for IS and FID, respectively. Based on ViT-VQGAN and unsupervised pretraining, we further evaluate the pretrained Transformer by averaging intermediate features, similar to Image GPT (iGPT). This ImageNet-pretrained VIM-L significantly beats iGPT-L on linear-probe accuracy from 60.3% to 73.2% for a similar model size. VIM-L also outperforms iGPT-XL which is trained with extra web image data and larger model size.",
  "published": "2021-10-09",
  "updated": "2022-06-05",
  "year": "2021",
  "authors": [
   "Jiahui Yu",
   "Xin Li",
   "Jing Yu Koh",
   "Han Zhang",
   "Ruoming Pang",
   "James Qin",
   "Alexander Ku",
   "Yuanzhong Xu",
   "Jason Baldridge",
   "Yonghui Wu"
  ],
  "author_count": 10,
  "categories": [
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.CV",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 818,
  "influential_citations": 72,
  "tldr": "This work introduces a Vector-quantized Image Modeling (VIM) approach that involves pretraining a Transformer to predict rasterized image tokens autoregressively, and proposes multiple improvements over vanilla VQGAN from architecture to codebook learning, yielding better efficiency and reconstruction fidelity.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jiahui Yu",
    "id": "2338016295",
    "h_index": 7,
    "papers": 9
   },
   {
    "name": "Xin Li",
    "id": "2153897589",
    "h_index": 10,
    "papers": 17
   },
   {
    "name": "Jing Yu Koh",
    "id": "23978705",
    "h_index": 17,
    "papers": 24
   },
   {
    "name": "Han Zhang",
    "id": "2119079641",
    "h_index": 14,
    "papers": 20
   },
   {
    "name": "Ruoming Pang",
    "id": "34320634",
    "h_index": 46,
    "papers": 77
   },
   {
    "name": "James Qin",
    "id": "47901308",
    "h_index": 17,
    "papers": 24
   },
   {
    "name": "Alexander Ku",
    "id": "31702389",
    "h_index": 13,
    "papers": 18
   },
   {
    "name": "Yuanzhong Xu",
    "id": "2145139570",
    "h_index": 18,
    "papers": 29
   },
   {
    "name": "Jason Baldridge",
    "id": "1387994164",
    "h_index": 49,
    "papers": 128
   },
   {
    "name": "Yonghui Wu",
    "id": "48607963",
    "h_index": 65,
    "papers": 89
   }
  ],
  "comment": "Accepted in ICLR 2022",
  "topics": [
   "foundation-pretraining",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2110.04627v3",
  "pdf_url": "https://arxiv.org/pdf/2110.04627v3",
  "html_url": "https://arxiv.org/html/2110.04627v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.41
 },
 {
  "id": "2110.03555",
  "slug": "active-extrinsic-contact-sensing-application-to-general-peg-in-hole-in",
  "title": "Active Extrinsic Contact Sensing: Application to General Peg-in-Hole Insertion",
  "abstract": "We propose a method that actively estimates contact location between a grasped rigid object and its environment and uses this as input to a peg-in-hole insertion policy. An estimation model and an active tactile feedback controller work collaboratively to estimate the external contacts accurately. The controller helps the estimation model get a better estimate by regulating a consistent contact mode. The better estimation makes it easier for the controller to regulate the contact. We then train an object-agnostic insertion policy that learns to use the series of contact estimates to guide the insertion of an unseen peg into a hole. In contrast with previous works that learn a policy directly from tactile signals, since this policy is in contact configuration space, it can be learned directly in simulation. Lastly, we demonstrate and evaluate the active extrinsic contact line estimation and the trained insertion policy together in a real experiment. We show that the proposed method inserts various-shaped test objects with higher success rates and fewer insertion attempts than previous work with end-to-end approaches. See supplementary video and results at https://sites.google.com/view/active-extrinsic-contact.",
  "published": "2021-10-07",
  "updated": "2022-03-28",
  "year": "2021",
  "authors": [
   "Sangwoon Kim",
   "Alberto Rodriguez"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA",
  "venue_source": "semantic-scholar",
  "citations": 91,
  "influential_citations": 4,
  "tldr": "A method that actively estimates contact location between a grasped rigid object and its environment and uses this as input to a peg-in-hole insertion policy and train an object-agnostic insertion policy that learns to use the series of contact estimates to guide the insertion of an unseen peg into a hole is proposed.",
  "doi": "10.1109/icra46639.2022.9812017",
  "oa_pdf": "https://hdl.handle.net/1721.1/155741",
  "s2_authors": [
   {
    "name": "Sangwoon Kim",
    "id": "2109704582",
    "h_index": 7,
    "papers": 14
   },
   {
    "name": "Alberto Rodriguez",
    "id": "152532021",
    "h_index": 49,
    "papers": 113
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2110.03555v2",
  "pdf_url": "https://arxiv.org/pdf/2110.03555v2",
  "html_url": "https://arxiv.org/html/2110.03555v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.46
 },
 {
  "id": "2110.02207",
  "slug": "waypoint-models-for-instruction-guided-navigation-in-continuous-enviro",
  "title": "Waypoint Models for Instruction-guided Navigation in Continuous Environments",
  "abstract": "Little inquiry has explicitly addressed the role of action spaces in language-guided visual navigation -- either in terms of its effect on navigation success or the efficiency with which a robotic agent could execute the resulting trajectory. Building on the recently released VLN-CE setting for instruction following in continuous environments, we develop a class of language-conditioned waypoint prediction networks to examine this question. We vary the expressivity of these models to explore a spectrum between low-level actions and continuous waypoint prediction. We measure task performance and estimated execution time on a profiled LoCoBot robot. We find more expressive models result in simpler, faster to execute trajectories, but lower-level actions can achieve better navigation metrics by approximating shortest paths better. Further, our models outperform prior work in VLN-CE and set a new state-of-the-art on the public leaderboard -- increasing success rate by 4% with our best model on this challenging task.",
  "published": "2021-10-05",
  "updated": "2021-10-05",
  "year": "2021",
  "authors": [
   "Jacob Krantz",
   "Aaron Gokaslan",
   "Dhruv Batra",
   "Stefan Lee",
   "Oleksandr Maksymets"
  ],
  "author_count": 5,
  "categories": [
   "cs.CV",
   "cs.CL",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "ICCV",
  "venue_source": "semantic-scholar",
  "citations": 187,
  "influential_citations": 10,
  "tldr": "A class of language-conditioned waypoint prediction networks is developed to examine the role of action spaces in language-guided visual navigation and finds more expressive models result in simpler, faster to execute trajectories, but lower-level actions can achieve better navigation metrics by approximating shortest paths better.",
  "doi": "10.1109/iccv48922.2021.01488",
  "oa_pdf": "https://arxiv.org/pdf/2110.02207",
  "s2_authors": [
   {
    "name": "Jacob Krantz",
    "id": "51050450",
    "h_index": 11,
    "papers": 14
   },
   {
    "name": "Aaron Gokaslan",
    "id": "8278433",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "Dhruv Batra",
    "id": "1746610",
    "h_index": 87,
    "papers": 327
   },
   {
    "name": "Stefan Lee",
    "id": "1607486000",
    "h_index": 18,
    "papers": 26
   },
   {
    "name": "Oleksandr Maksymets",
    "id": "90536527",
    "h_index": 14,
    "papers": 20
   }
  ],
  "comment": "ICCV 2021",
  "topics": [
   "navigation",
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2110.02207v1",
  "pdf_url": "https://arxiv.org/pdf/2110.02207v1",
  "html_url": "https://arxiv.org/html/2110.02207v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.77
 },
 {
  "id": "2110.01517",
  "slug": "skill-induction-and-planning-with-latent-language",
  "title": "Skill Induction and Planning with Latent Language",
  "abstract": "We present a framework for learning hierarchical policies from demonstrations, using sparse natural language annotations to guide the discovery of reusable skills for autonomous decision-making. We formulate a generative model of action sequences in which goals generate sequences of high-level subtask descriptions, and these descriptions generate sequences of low-level actions. We describe how to train this model using primarily unannotated demonstrations by parsing demonstrations into sequences of named high-level subtasks, using only a small number of seed annotations to ground language in action. In trained models, natural language commands index a combinatorial library of skills; agents can use these skills to plan by generating high-level instruction sequences tailored to novel goals. We evaluate this approach in the ALFRED household simulation environment, providing natural language annotations for only 10% of demonstrations. It achieves task completion rates comparable to state-of-the-art models (outperforming several recent methods with access to ground-truth plans during training and evaluation) while providing structured and human-readable high-level plans.",
  "published": "2021-10-04",
  "updated": "2022-05-02",
  "year": "2021",
  "authors": [
   "Pratyusha Sharma",
   "Antonio Torralba",
   "Jacob Andreas"
  ],
  "author_count": 3,
  "categories": [
   "cs.LG",
   "cs.AI",
   "cs.CL",
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 132,
  "influential_citations": 4,
  "tldr": "A framework for learning hierarchical policies from demonstrations, using sparse natural language annotations to guide the discovery of reusable skills for autonomous decision-making, achieves performance comparable state-of-the-art models on ALFRED success rate and outperforming several recent methods with access to ground-truth plans.",
  "doi": "10.18653/v1/2022.acl-long.120",
  "oa_pdf": "https://aclanthology.org/2022.acl-long.120.pdf",
  "s2_authors": [
   {
    "name": "Pratyusha Sharma",
    "id": "50465425",
    "h_index": 16,
    "papers": 25
   },
   {
    "name": "A. Torralba",
    "id": "143805211",
    "h_index": 142,
    "papers": 372
   },
   {
    "name": "Jacob Andreas",
    "id": "2112400",
    "h_index": 53,
    "papers": 93
   }
  ],
  "comment": "14 pages, 7 figures",
  "topics": [
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2110.01517v2",
  "pdf_url": "https://arxiv.org/pdf/2110.01517v2",
  "html_url": "https://arxiv.org/html/2110.01517v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.12
 },
 {
  "id": "2110.00641",
  "slug": "batch-size-invariance-for-policy-optimization",
  "title": "Batch size-invariance for policy optimization",
  "abstract": "We say an algorithm is batch size-invariant if changes to the batch size can largely be compensated for by changes to other hyperparameters. Stochastic gradient descent is well-known to have this property at small batch sizes, via the learning rate. However, some policy optimization algorithms (such as PPO) do not have this property, because of how they control the size of policy updates. In this work we show how to make these algorithms batch size-invariant. Our key insight is to decouple the proximal policy (used for controlling policy updates) from the behavior policy (used for off-policy corrections). Our experiments help explain why these algorithms work, and additionally show how they can make more efficient use of stale data.",
  "published": "2021-10-01",
  "updated": "2022-09-24",
  "year": "2021",
  "authors": [
   "Jacob Hilton",
   "Karl Cobbe",
   "John Schulman"
  ],
  "author_count": 3,
  "categories": [
   "cs.LG",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 38,
  "influential_citations": 5,
  "tldr": "This work shows how to decouple the proximal policy ( used for controlling policy updates) from the behavior policy (used for off-policy corrections) to make these algorithms batch size-invariant.",
  "doi": "10.52202/068431-1243",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jacob Hilton",
    "id": "2052366271",
    "h_index": 9,
    "papers": 14
   },
   {
    "name": "K. Cobbe",
    "id": "6062736",
    "h_index": 11,
    "papers": 53
   },
   {
    "name": "John Schulman",
    "id": "47971768",
    "h_index": 45,
    "papers": 69
   }
  ],
  "comment": "32 pages. Code is available at https://github.com/openai/ppo-ewma",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2110.00641v3",
  "pdf_url": "https://arxiv.org/pdf/2110.00641v3",
  "html_url": "https://arxiv.org/html/2110.00641v3",
  "code_url": "https://github.com/openai/ppo-ewma",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.09
 },
 {
  "id": "2110.00534",
  "slug": "teach-task-driven-embodied-agents-that-chat",
  "title": "TEACh: Task-driven Embodied Agents that Chat",
  "abstract": "Robots operating in human spaces must be able to engage in natural language interaction with people, both understanding and executing instructions, and using conversation to resolve ambiguity and recover from mistakes. To study this, we introduce TEACh, a dataset of over 3,000 human--human, interactive dialogues to complete household tasks in simulation. A Commander with access to oracle information about a task communicates in natural language with a Follower. The Follower navigates through and interacts with the environment to complete tasks varying in complexity from \"Make Coffee\" to \"Prepare Breakfast\", asking questions and getting additional information from the Commander. We propose three benchmarks using TEACh to study embodied intelligence challenges, and we evaluate initial models' abilities in dialogue understanding, language grounding, and task execution.",
  "published": "2021-10-01",
  "updated": "2021-12-29",
  "year": "2021",
  "authors": [
   "Aishwarya Padmakumar",
   "Jesse Thomason",
   "Ayush Shrivastava",
   "Patrick Lange",
   "Anjali Narayan-Chen",
   "Spandana Gella",
   "Robinson Piramuthu",
   "Gokhan Tur",
   "Dilek Hakkani-Tur"
  ],
  "author_count": 9,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.CL",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "AAAI 2022",
  "venue_source": "arxiv-comment",
  "citations": 289,
  "influential_citations": 24,
  "tldr": "TEACh, a dataset of over 3,000 human-human, interactive dialogues to complete household tasks in simulation, is introduced and initial models' abilities in dialogue understanding, language grounding, and task execution are evaluated.",
  "doi": "10.1609/aaai.v36i2.20097",
  "oa_pdf": "https://doi.org/10.1609/aaai.v36i2.20097",
  "s2_authors": [
   {
    "name": "Aishwarya Padmakumar",
    "id": "2110665",
    "h_index": 14,
    "papers": 33
   },
   {
    "name": "Jesse Thomason",
    "id": "2665873",
    "h_index": 27,
    "papers": 63
   },
   {
    "name": "Ayush Shrivastava",
    "id": "3445289",
    "h_index": 9,
    "papers": 18
   },
   {
    "name": "P. Lange",
    "id": "26882347",
    "h_index": 13,
    "papers": 42
   },
   {
    "name": "Anjali Narayan-Chen",
    "id": "35526556",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Spandana Gella",
    "id": "2921001",
    "h_index": 20,
    "papers": 71
   },
   {
    "name": "Robinson Piramithu",
    "id": "2130465355",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Gokhan Tur",
    "id": "5108268",
    "h_index": 23,
    "papers": 93
   },
   {
    "name": "Dilek Z. Hakkani-T\u00fcr",
    "id": "1395813836",
    "h_index": 61,
    "papers": 357
   }
  ],
  "comment": "Accepted at AAAI 2022; 7 pages main, 28 pages total, 29 figures; Version 3 uses a new test set for EDH instances that restrict evaluation to state changes only on task-relevant objects",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2110.00534v3",
  "pdf_url": "https://arxiv.org/pdf/2110.00534v3",
  "html_url": "https://arxiv.org/html/2110.00534v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.96
 },
 {
  "id": "2109.13772",
  "slug": "nimbro-avatar-interactive-immersive-telepresence-with-force-feedback-t",
  "title": "NimbRo Avatar: Interactive Immersive Telepresence with Force-Feedback Telemanipulation",
  "abstract": "Robotic avatars promise immersive teleoperation with human-like manipulation and communication capabilities. We present such an avatar system, based on the key components of immersive 3D visualization and transparent force-feedback telemanipulation. Our avatar robot features an anthropomorphic bimanual arm configuration with dexterous hands. The remote human operator drives the arms and fingers through an exoskeleton-based operator station, which provides force feedback both at the wrist and for each finger. The robot torso is mounted on a holonomic base, providing locomotion capability in typical indoor scenarios, controlled using a 3D rudder device. Finally, the robot features a 6D movable head with stereo cameras, which stream images to a VR HMD worn by the operator. Movement latency is hidden using spherical rendering. The head also carries a telepresence screen displaying a synthesized image of the operator with facial animation, which enables direct interaction with remote persons. We evaluate our system successfully both in a user study with untrained operators as well as a longer and more complex integrated mission. We discuss lessons learned from the trials and possible improvements.",
  "published": "2021-09-28",
  "updated": "2021-09-28",
  "year": "2021",
  "authors": [
   "Max Schwarz",
   "Christian Lenz",
   "Andre Rochow",
   "Michael Schreiber",
   "Sven Behnke"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS",
  "venue_source": "semantic-scholar",
  "citations": 73,
  "influential_citations": 3,
  "tldr": "This avatar robot features an anthropomorphic bimanual arm configuration with dexterous hands, which drives the arms and fingers through an exoskeleton-based operator station, which provides force feedback both at the wrist and for each finger.",
  "doi": "10.1109/IROS51168.2021.9636191",
  "oa_pdf": "http://arxiv.org/pdf/2109.13772",
  "s2_authors": [
   {
    "name": "Max Schwarz",
    "id": "39754718",
    "h_index": 26,
    "papers": 71
   },
   {
    "name": "C. Lenz",
    "id": "2081140190",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Andre Rochow",
    "id": "2004014733",
    "h_index": 8,
    "papers": 15
   },
   {
    "name": "M. Schreiber",
    "id": "35012312",
    "h_index": 20,
    "papers": 91
   },
   {
    "name": "Sven Behnke",
    "id": "1699019",
    "h_index": 57,
    "papers": 527
   }
  ],
  "comment": "Accepted final version. IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS), Prague, Czech Republic, September 2021",
  "topics": [
   "dexterous-manipulation",
   "humanoids",
   "tactile",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2109.13772v1",
  "pdf_url": "https://arxiv.org/pdf/2109.13772v1",
  "html_url": "https://arxiv.org/html/2109.13772v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.37
 },
 {
  "id": "2109.13396",
  "slug": "bridge-data-boosting-generalization-of-robotic-skills-with-cross-domai",
  "title": "Bridge Data: Boosting Generalization of Robotic Skills with Cross-Domain Datasets",
  "abstract": "Robot learning holds the promise of learning policies that generalize broadly. However, such generalization requires sufficiently diverse datasets of the task of interest, which can be prohibitively expensive to collect. In other fields, such as computer vision, it is common to utilize shared, reusable datasets, such as ImageNet, to overcome this challenge, but this has proven difficult in robotics. In this paper, we ask: what would it take to enable practical data reuse in robotics for end-to-end skill learning? We hypothesize that the key is to use datasets with multiple tasks and multiple domains, such that a new user that wants to train their robot to perform a new task in a new domain can include this dataset in their training process and benefit from cross-task and cross-domain generalization. To evaluate this hypothesis, we collect a large multi-domain and multi-task dataset, with 7,200 demonstrations constituting 71 tasks across 10 environments, and empirically study how this data can improve the learning of new tasks in new environments. We find that jointly training with the proposed dataset and 50 demonstrations of a never-before-seen task in a new domain on average leads to a 2x improvement in success rate compared to using target domain data alone. We also find that data for only a few tasks in a new domain can bridge the domain gap and make it possible for a robot to perform a variety of prior tasks that were only seen in other domains. These results suggest that reusing diverse multi-task and multi-domain datasets, including our open-source dataset, may pave the way for broader robot generalization, eliminating the need to re-collect data for each new robot learning project.",
  "published": "2021-09-27",
  "updated": "2021-09-27",
  "year": "2021",
  "authors": [
   "Frederik Ebert",
   "Yanlai Yang",
   "Karl Schmeckpeper",
   "Bernadette Bucher",
   "Georgios Georgakis",
   "Kostas Daniilidis",
   "Chelsea Finn",
   "Sergey Levine"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "RSS",
  "venue_source": "semantic-scholar",
  "citations": 372,
  "influential_citations": 32,
  "tldr": "It is suggested that reusing diverse multi-task and multi-domain datasets, including the open-source dataset, may pave the way for broader robot generalization, eliminating the need to re-collect data for each new robot learning project.",
  "doi": "10.15607/rss.2022.xviii.063",
  "oa_pdf": "https://doi.org/10.15607/rss.2022.xviii.063",
  "s2_authors": [
   {
    "name": "F. Ebert",
    "id": "27535721",
    "h_index": 15,
    "papers": 22
   },
   {
    "name": "Yanlai Yang",
    "id": "3800238",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Karl Schmeckpeper",
    "id": "88726258",
    "h_index": 14,
    "papers": 35
   },
   {
    "name": "Bernadette Bucher",
    "id": "47015098",
    "h_index": 8,
    "papers": 27
   },
   {
    "name": "G. Georgakis",
    "id": "40379773",
    "h_index": 11,
    "papers": 25
   },
   {
    "name": "Kostas Daniilidis",
    "id": "2065557091",
    "h_index": 25,
    "papers": 89
   },
   {
    "name": "Chelsea Finn",
    "id": "46881670",
    "h_index": 94,
    "papers": 240
   },
   {
    "name": "S. Levine",
    "id": "1736651",
    "h_index": 167,
    "papers": 551
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2109.13396v1",
  "pdf_url": "https://arxiv.org/pdf/2109.13396v1",
  "html_url": "https://arxiv.org/html/2109.13396v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.07
 },
 {
  "id": "2109.12098",
  "slug": "cliport-what-and-where-pathways-for-robotic-manipulation",
  "title": "CLIPort: What and Where Pathways for Robotic Manipulation",
  "abstract": "How can we imbue robots with the ability to manipulate objects precisely but also to reason about them in terms of abstract concepts? Recent works in manipulation have shown that end-to-end networks can learn dexterous skills that require precise spatial reasoning, but these methods often fail to generalize to new goals or quickly learn transferable concepts across tasks. In parallel, there has been great progress in learning generalizable semantic representations for vision and language by training on large-scale internet data, however these representations lack the spatial understanding necessary for fine-grained manipulation. To this end, we propose a framework that combines the best of both worlds: a two-stream architecture with semantic and spatial pathways for vision-based manipulation. Specifically, we present CLIPort, a language-conditioned imitation-learning agent that combines the broad semantic understanding (what) of CLIP [1] with the spatial precision (where) of Transporter [2]. Our end-to-end framework is capable of solving a variety of language-specified tabletop tasks from packing unseen objects to folding cloths, all without any explicit representations of object poses, instance segmentations, memory, symbolic states, or syntactic structures. Experiments in simulated and real-world settings show that our approach is data efficient in few-shot settings and generalizes effectively to seen and unseen semantic concepts. We even learn one multi-task policy for 10 simulated and 9 real-world tasks that is better or comparable to single-task policies.",
  "published": "2021-09-24",
  "updated": "2021-09-24",
  "year": "2021",
  "authors": [
   "Mohit Shridhar",
   "Lucas Manuelli",
   "Dieter Fox"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.CL",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 992,
  "influential_citations": 102,
  "tldr": "CLIPort is presented, a language-conditioned imitation-learning agent that combines the broad semantic understanding of CLIP with the spatial precision of Transporter and is capable of solving a variety of language-specified tabletop tasks without any explicit representations of object poses, instance segmentations, memory, symbolic states, or syntactic structures.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Mohit Shridhar",
    "id": "33516562",
    "h_index": 15,
    "papers": 23
   },
   {
    "name": "Lucas Manuelli",
    "id": "2033958",
    "h_index": 11,
    "papers": 21
   },
   {
    "name": "D. Fox",
    "id": "145197953",
    "h_index": 133,
    "papers": 428
   }
  ],
  "comment": "CoRL 2021. Project Website: https://cliport.github.io/",
  "topics": [
   "dexterous-manipulation",
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2109.12098v1",
  "pdf_url": "https://arxiv.org/pdf/2109.12098v1",
  "html_url": "https://arxiv.org/html/2109.12098v1",
  "code_url": "https://cliport.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.5
 },
 {
  "id": "2109.11978",
  "slug": "learning-to-walk-in-minutes-using-massively-parallel-deep-reinforcemen",
  "title": "Learning to Walk in Minutes Using Massively Parallel Deep Reinforcement Learning",
  "abstract": "In this work, we present and study a training set-up that achieves fast policy generation for real-world robotic tasks by using massive parallelism on a single workstation GPU. We analyze and discuss the impact of different training algorithm components in the massively parallel regime on the final policy performance and training times. In addition, we present a novel game-inspired curriculum that is well suited for training with thousands of simulated robots in parallel. We evaluate the approach by training the quadrupedal robot ANYmal to walk on challenging terrain. The parallel approach allows training policies for flat terrain in under four minutes, and in twenty minutes for uneven terrain. This represents a speedup of multiple orders of magnitude compared to previous work. Finally, we transfer the policies to the real robot to validate the approach. We open-source our training code to help accelerate further research in the field of learned legged locomotion.",
  "published": "2021-09-24",
  "updated": "2022-08-19",
  "year": "2021",
  "authors": [
   "Nikita Rudin",
   "David Hoeller",
   "Philipp Reist",
   "Marco Hutter"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 1051,
  "influential_citations": 127,
  "tldr": "A training set-up that achieves fast policy generation for real-world robotic tasks by using massive parallelism on a single workstation GPU is presented and a novel game-inspired curriculum is presented that is well suited for training with thousands of simulated robots in parallel.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "N. Rudin",
    "id": "2113243810",
    "h_index": 17,
    "papers": 17
   },
   {
    "name": "David Hoeller",
    "id": "71054073",
    "h_index": 18,
    "papers": 19
   },
   {
    "name": "Philipp Reist",
    "id": "2938001",
    "h_index": 8,
    "papers": 13
   },
   {
    "name": "Marco Hutter",
    "id": "14349870",
    "h_index": 80,
    "papers": 279
   }
  ],
  "comment": "CoRL 2021 Project website: : https://leggedrobotics.github.io/legged_gym/ Video: https://youtu.be/8sO7VS3q8d0",
  "topics": [
   "humanoids",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2109.11978v3",
  "pdf_url": "https://arxiv.org/pdf/2109.11978v3",
  "html_url": "https://arxiv.org/html/2109.11978v3",
  "code_url": "https://leggedrobotics.github.io/legged_gym/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.5
 },
 {
  "id": "2109.11052",
  "slug": "on-bonus-based-exploration-methods-in-the-arcade-learning-environment",
  "title": "On Bonus-Based Exploration Methods in the Arcade Learning Environment",
  "abstract": "Research on exploration in reinforcement learning, as applied to Atari 2600 game-playing, has emphasized tackling difficult exploration problems such as Montezuma's Revenge (Bellemare et al., 2016). Recently, bonus-based exploration methods, which explore by augmenting the environment reward, have reached above-human average performance on such domains. In this paper we reassess popular bonus-based exploration methods within a common evaluation framework. We combine Rainbow (Hessel et al., 2018) with different exploration bonuses and evaluate its performance on Montezuma's Revenge, Bellemare et al.'s set of hard of exploration games with sparse rewards, and the whole Atari 2600 suite. We find that while exploration bonuses lead to higher score on Montezuma's Revenge they do not provide meaningful gains over the simpler $\u03b5$-greedy scheme. In fact, we find that methods that perform best on that game often underperform $\u03b5$-greedy on easy exploration Atari 2600 games. We find that our conclusions remain valid even when hyperparameters are tuned for these easy-exploration games. Finally, we find that none of the methods surveyed benefit from additional training samples (1 billion frames, versus Rainbow's 200 million) on Bellemare et al.'s hard exploration games. Our results suggest that recent gains in Montezuma's Revenge may be better attributed to architecture change, rather than better exploration schemes; and that the real pace of progress in exploration research for Atari 2600 games may have been obfuscated by good results on a single domain.",
  "published": "2021-09-22",
  "updated": "2021-09-22",
  "year": "2021",
  "authors": [
   "Adrien Ali Ta\u00efga",
   "William Fedus",
   "Marlos C. Machado",
   "Aaron Courville",
   "Marc G. Bellemare"
  ],
  "author_count": 5,
  "categories": [
   "cs.LG"
  ],
  "primary_category": "cs.LG",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 75,
  "influential_citations": 6,
  "tldr": "It is found that while exploration bonuses lead to higher score on Montezuma's Revenge they do not provide meaningful gains over the simpler epsilon-greedy scheme, and that methods that perform best on that game often underperform epsilon-greedy on easy exploration Atari 2600 games.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Adrien Ali Taiga",
    "id": "1583061741",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "W. Fedus",
    "id": "26958176",
    "h_index": 31,
    "papers": 51
   },
   {
    "name": "Marlos C. Machado",
    "id": "40066857",
    "h_index": 24,
    "papers": 64
   },
   {
    "name": "Aaron C. Courville",
    "id": "1760871",
    "h_index": 94,
    "papers": 241
   },
   {
    "name": "Marc G. Bellemare",
    "id": "1792298",
    "h_index": 44,
    "papers": 92
   }
  ],
  "comment": "Full version of arXiv:1908.02388",
  "topics": [
   "rl-control",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2109.11052v1",
  "pdf_url": "https://arxiv.org/pdf/2109.11052v1",
  "html_url": "https://arxiv.org/html/2109.11052v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.38
 },
 {
  "id": "2109.07627",
  "slug": "adversarially-regularized-policy-learning-guided-by-trajectory-optimiz",
  "title": "Adversarially Regularized Policy Learning Guided by Trajectory Optimization",
  "abstract": "Recent advancement in combining trajectory optimization with function approximation (especially neural networks) shows promise in learning complex control policies for diverse tasks in robot systems. Despite their great flexibility, the large neural networks for parameterizing control policies impose significant challenges. The learned neural control policies are often overcomplex and non-smooth, which can easily cause unexpected or diverging robot motions. Therefore, they often yield poor generalization performance in practice. To address this issue, we propose adVErsarially Regularized pOlicy learNIng guided by trajeCtory optimizAtion (VERONICA) for learning smooth control policies. Specifically, our proposed approach controls the smoothness (local Lipschitz continuity) of the neural control policies by stabilizing the output control with respect to the worst-case perturbation to the input state. Our experiments on robot manipulation show that our proposed approach not only improves the sample efficiency of neural policy learning but also enhances the robustness of the policy against various types of disturbances, including sensor noise, environmental uncertainty, and model mismatch.",
  "published": "2021-09-16",
  "updated": "2022-04-05",
  "year": "2021",
  "authors": [
   "Zhigen Zhao",
   "Simiao Zuo",
   "Tuo Zhao",
   "Ye Zhao"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 12,
  "influential_citations": 0,
  "tldr": "The proposed approach controls the smoothness (local Lipschitz continuity) of the neural control policies by stabilizing the output control with respect to the worst-case perturbation to the input state, which improves the sample efficiency of neural policy learning and enhances the robustness of the policy against various types of disturbances.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhigen Zhao",
    "id": "2941525",
    "h_index": 14,
    "papers": 41
   },
   {
    "name": "Simiao Zuo",
    "id": "52194893",
    "h_index": 15,
    "papers": 32
   },
   {
    "name": "T. Zhao",
    "id": "36345161",
    "h_index": 41,
    "papers": 122
   },
   {
    "name": "Ye Zhao",
    "id": "97522088",
    "h_index": 19,
    "papers": 77
   }
  ],
  "comment": "Accepted at L4DC 2022",
  "topics": [
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2109.07627v3",
  "pdf_url": "https://arxiv.org/pdf/2109.07627v3",
  "html_url": "https://arxiv.org/html/2109.07627v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.11
 },
 {
  "id": "2109.06780",
  "slug": "benchmarking-the-spectrum-of-agent-capabilities",
  "title": "Benchmarking the Spectrum of Agent Capabilities",
  "abstract": "Evaluating the general abilities of intelligent agents requires complex simulation environments. Existing benchmarks typically evaluate only one narrow task per environment, requiring researchers to perform expensive training runs on many different environments. We introduce Crafter, an open world survival game with visual inputs that evaluates a wide range of general abilities within a single environment. Agents either learn from the provided reward signal or through intrinsic objectives and are evaluated by semantically meaningful achievements that can be unlocked during each episode, such as discovering resources and crafting tools. Consistently unlocking all achievements requires strong generalization, deep exploration, and long-term reasoning. We experimentally verify that Crafter is of appropriate difficulty to drive future research and provide baselines scores of reward agents and unsupervised agents. Furthermore, we observe sophisticated behaviors emerging from maximizing the reward signal, such as building tunnel systems, bridges, houses, and plantations. We hope that Crafter will accelerate research progress by quickly evaluating a wide spectrum of abilities.",
  "published": "2021-09-14",
  "updated": "2022-02-12",
  "year": "2021",
  "authors": [
   "Danijar Hafner"
  ],
  "author_count": 1,
  "categories": [
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.AI",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 226,
  "influential_citations": 43,
  "tldr": "Crafter is introduced, an open world survival game with visual inputs that evaluates a wide range of general abilities within a single environment that will accelerate research progress by quickly evaluating a wide spectrum of abilities.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Danijar Hafner",
    "id": "35006479",
    "h_index": 25,
    "papers": 47
   }
  ],
  "comment": "Published at ICLR 2022. Website: https://danijar.com/crafter",
  "topics": [
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2109.06780v2",
  "pdf_url": "https://arxiv.org/pdf/2109.06780v2",
  "html_url": "https://arxiv.org/html/2109.06780v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.86
 },
 {
  "id": "2109.06275",
  "slug": "mindcraft-theory-of-mind-modeling-for-situated-dialogue-in-collaborati",
  "title": "MindCraft: Theory of Mind Modeling for Situated Dialogue in Collaborative Tasks",
  "abstract": "An ideal integration of autonomous agents in a human world implies that they are able to collaborate on human terms. In particular, theory of mind plays an important role in maintaining common ground during human collaboration and communication. To enable theory of mind modeling in situated interactions, we introduce a fine-grained dataset of collaborative tasks performed by pairs of human subjects in the 3D virtual blocks world of Minecraft. It provides information that captures partners' beliefs of the world and of each other as an interaction unfolds, bringing abundant opportunities to study human collaborative behaviors in situated language communication. As a first step towards our goal of developing embodied AI agents able to infer belief states of collaborative partners in situ, we build and present results on computational models for several theory of mind tasks.",
  "published": "2021-09-13",
  "updated": "2021-09-13",
  "year": "2021",
  "authors": [
   "Cristian-Paul Bara",
   "Sky CH-Wang",
   "Joyce Chai"
  ],
  "author_count": 3,
  "categories": [
   "cs.AI",
   "cs.CL",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.AI",
  "venue": "",
  "venue_source": "",
  "citations": 88,
  "influential_citations": 5,
  "tldr": "A fine-grained dataset of collaborative tasks performed by pairs of human subjects in the 3D virtual blocks world of Minecraft is introduced to enable theory of mind modeling in situated interactions.",
  "doi": "10.18653/v1/2021.emnlp-main.85",
  "oa_pdf": "https://aclanthology.org/2021.emnlp-main.85.pdf",
  "s2_authors": [
   {
    "name": "Cristian-Paul Bara",
    "id": "89392298",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Sky CH-Wang",
    "id": "12782016",
    "h_index": 8,
    "papers": 15
   },
   {
    "name": "J. Chai",
    "id": "1707259",
    "h_index": 35,
    "papers": 144
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2109.06275v1",
  "pdf_url": "https://arxiv.org/pdf/2109.06275v1",
  "html_url": "https://arxiv.org/html/2109.06275v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.95
 },
 {
  "id": "2109.01115",
  "slug": "learning-language-conditioned-robot-behavior-from-offline-data-and-cro",
  "title": "Learning Language-Conditioned Robot Behavior from Offline Data and Crowd-Sourced Annotation",
  "abstract": "We study the problem of learning a range of vision-based manipulation tasks from a large offline dataset of robot interaction. In order to accomplish this, humans need easy and effective ways of specifying tasks to the robot. Goal images are one popular form of task specification, as they are already grounded in the robot's observation space. However, goal images also have a number of drawbacks: they are inconvenient for humans to provide, they can over-specify the desired behavior leading to a sparse reward signal, or under-specify task information in the case of non-goal reaching tasks. Natural language provides a convenient and flexible alternative for task specification, but comes with the challenge of grounding language in the robot's observation space. To scalably learn this grounding we propose to leverage offline robot datasets (including highly sub-optimal, autonomously collected data) with crowd-sourced natural language labels. With this data, we learn a simple classifier which predicts if a change in state completes a language instruction. This provides a language-conditioned reward function that can then be used for offline multi-task RL. In our experiments, we find that on language-conditioned manipulation tasks our approach outperforms both goal-image specifications and language conditioned imitation techniques by more than 25%, and is able to perform visuomotor tasks from natural language, such as \"open the right drawer\" and \"move the stapler\", on a Franka Emika Panda robot.",
  "published": "2021-09-02",
  "updated": "2021-10-31",
  "year": "2021",
  "authors": [
   "Suraj Nair",
   "Eric Mitchell",
   "Kevin Chen",
   "Brian Ichter",
   "Silvio Savarese",
   "Chelsea Finn"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 193,
  "influential_citations": 12,
  "tldr": "This approach outperforms both goal-image specifications and language conditioned imitation techniques by more than 25%, and is able to perform visuomotor tasks from natural language, such as\"open the right drawer\" and\"move the stapler\", on a Franka Emika Panda robot.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Suraj Nair",
    "id": "4734949",
    "h_index": 20,
    "papers": 36
   },
   {
    "name": "E. Mitchell",
    "id": "49688913",
    "h_index": 22,
    "papers": 36
   },
   {
    "name": "Kevin Chen",
    "id": "143887468",
    "h_index": 14,
    "papers": 15
   },
   {
    "name": "Brian Ichter",
    "id": "2704814",
    "h_index": 37,
    "papers": 60
   },
   {
    "name": "S. Savarese",
    "id": "1702137",
    "h_index": 115,
    "papers": 346
   },
   {
    "name": "Chelsea Finn",
    "id": "46881670",
    "h_index": 94,
    "papers": 240
   }
  ],
  "comment": "Conference on Robot Learning (CoRL) 2021. 24 Pages, 18 Figures",
  "topics": [
   "vla",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2109.01115v2",
  "pdf_url": "https://arxiv.org/pdf/2109.01115v2",
  "html_url": "https://arxiv.org/html/2109.01115v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.79
 },
 {
  "id": "2108.10470",
  "slug": "isaac-gym-high-performance-gpu-based-physics-simulation-for-robot-lear",
  "title": "Isaac Gym: High Performance GPU-Based Physics Simulation For Robot Learning",
  "abstract": "Isaac Gym offers a high performance learning platform to train policies for wide variety of robotics tasks directly on GPU. Both physics simulation and the neural network policy training reside on GPU and communicate by directly passing data from physics buffers to PyTorch tensors without ever going through any CPU bottlenecks. This leads to blazing fast training times for complex robotics tasks on a single GPU with 2-3 orders of magnitude improvements compared to conventional RL training that uses a CPU based simulator and GPU for neural networks. We host the results and videos at \\url{https://sites.google.com/view/isaacgym-nvidia} and isaac gym can be downloaded at \\url{https://developer.nvidia.com/isaac-gym}.",
  "published": "2021-08-24",
  "updated": "2021-08-25",
  "year": "2021",
  "authors": [
   "Viktor Makoviychuk",
   "Lukasz Wawrzyniak",
   "Yunrong Guo",
   "Michelle Lu",
   "Kier Storey",
   "Miles Macklin",
   "David Hoeller",
   "Nikita Rudin",
   "Arthur Allshire",
   "Ankur Handa",
   "Gavriel State"
  ],
  "author_count": 11,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 1910,
  "influential_citations": 139,
  "tldr": "Isaac Gym offers a high performance learning platform to train policies for wide variety of robotics tasks directly on GPU with 2-3 orders of magnitude improvements compared to conventional RL training that uses a CPU based simulator and GPU for neural networks.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Viktor Makoviychuk",
    "id": "79875630",
    "h_index": 18,
    "papers": 21
   },
   {
    "name": "Lukasz Wawrzyniak",
    "id": "144275374",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Yunrong Guo",
    "id": "2029890590",
    "h_index": 13,
    "papers": 14
   },
   {
    "name": "Michelle Lu",
    "id": "2149495306",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Kier Storey",
    "id": "1899226",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "M. Macklin",
    "id": "46637939",
    "h_index": 30,
    "papers": 53
   },
   {
    "name": "David Hoeller",
    "id": "71054073",
    "h_index": 18,
    "papers": 19
   },
   {
    "name": "N. Rudin",
    "id": "2113243810",
    "h_index": 17,
    "papers": 17
   },
   {
    "name": "Arthur Allshire",
    "id": "2061149217",
    "h_index": 9,
    "papers": 12
   },
   {
    "name": "Ankur Handa",
    "id": "34653454",
    "h_index": 33,
    "papers": 55
   },
   {
    "name": "Gavriel State",
    "id": "82261827",
    "h_index": 10,
    "papers": 14
   }
  ],
  "comment": "tech report on isaac-gym",
  "topics": [
   "sim2real"
  ],
  "orgs": [
   "NVIDIA"
  ],
  "abs_url": "https://arxiv.org/abs/2108.10470v2",
  "pdf_url": "https://arxiv.org/pdf/2108.10470v2",
  "html_url": "https://arxiv.org/html/2108.10470v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.5
 },
 {
  "id": "2108.05877",
  "slug": "dexmv-imitation-learning-for-dexterous-manipulation-from-human-videos",
  "title": "DexMV: Imitation Learning for Dexterous Manipulation from Human Videos",
  "abstract": "While significant progress has been made on understanding hand-object interactions in computer vision, it is still very challenging for robots to perform complex dexterous manipulation. In this paper, we propose a new platform and pipeline DexMV (Dexterous Manipulation from Videos) for imitation learning. We design a platform with: (i) a simulation system for complex dexterous manipulation tasks with a multi-finger robot hand and (ii) a computer vision system to record large-scale demonstrations of a human hand conducting the same tasks. In our novel pipeline, we extract 3D hand and object poses from videos, and propose a novel demonstration translation method to convert human motion to robot demonstrations. We then apply and benchmark multiple imitation learning algorithms with the demonstrations. We show that the demonstrations can indeed improve robot learning by a large margin and solve the complex tasks which reinforcement learning alone cannot solve. More details can be found in the project page: https://yzqin.github.io/dexmv",
  "published": "2021-08-12",
  "updated": "2022-07-06",
  "year": "2021",
  "authors": [
   "Yuzhe Qin",
   "Yueh-Hua Wu",
   "Shaowei Liu",
   "Hanwen Jiang",
   "Ruihan Yang",
   "Yang Fu",
   "Xiaolong Wang"
  ],
  "author_count": 7,
  "categories": [
   "cs.LG",
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.LG",
  "venue": "ECCV",
  "venue_source": "semantic-scholar",
  "citations": 344,
  "influential_citations": 17,
  "tldr": "This paper proposes a new platform and pipeline DexMV for imitation learning, and proposes a novel demonstration translation method to convert human motion to robot demonstrations that can improve robot learning by a large margin and solve the complex tasks which reinforcement learning alone cannot solve.",
  "doi": "10.1007/978-3-031-19842-7_33",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yuzhe Qin",
    "id": "12701031",
    "h_index": 24,
    "papers": 34
   },
   {
    "name": "Yueh-Hua Wu",
    "id": "31609618",
    "h_index": 10,
    "papers": 13
   },
   {
    "name": "Shaowei Liu",
    "id": "2119051505",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Hanwen Jiang",
    "id": "2152630535",
    "h_index": 14,
    "papers": 27
   },
   {
    "name": "Ruihan Yang",
    "id": "143955842",
    "h_index": 18,
    "papers": 22
   },
   {
    "name": "Yang Fu",
    "id": "46956309",
    "h_index": 9,
    "papers": 14
   },
   {
    "name": "Xiaolong Wang",
    "id": "122024152",
    "h_index": 42,
    "papers": 63
   }
  ],
  "comment": "https://yzqin.github.io/dexmv",
  "topics": [
   "dexterous-manipulation",
   "egocentric-data",
   "imitation-diffusion",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2108.05877v5",
  "pdf_url": "https://arxiv.org/pdf/2108.05877v5",
  "html_url": "https://arxiv.org/html/2108.05877v5",
  "code_url": "https://yzqin.github.io/dexmv",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.04
 },
 {
  "id": "2108.03298",
  "slug": "what-matters-in-learning-from-offline-human-demonstrations-for-robot-m",
  "title": "What Matters in Learning from Offline Human Demonstrations for Robot Manipulation",
  "abstract": "Imitating human demonstrations is a promising approach to endow robots with various manipulation capabilities. While recent advances have been made in imitation learning and batch (offline) reinforcement learning, a lack of open-source human datasets and reproducible learning methods make assessing the state of the field difficult. In this paper, we conduct an extensive study of six offline learning algorithms for robot manipulation on five simulated and three real-world multi-stage manipulation tasks of varying complexity, and with datasets of varying quality. Our study analyzes the most critical challenges when learning from offline human data for manipulation. Based on the study, we derive a series of lessons including the sensitivity to different algorithmic design choices, the dependence on the quality of the demonstrations, and the variability based on the stopping criteria due to the different objectives in training and evaluation. We also highlight opportunities for learning from human datasets, such as the ability to learn proficient policies on challenging, multi-stage tasks beyond the scope of current reinforcement learning methods, and the ability to easily scale to natural, real-world manipulation scenarios where only raw sensory signals are available. We have open-sourced our datasets and all algorithm implementations to facilitate future research and fair comparisons in learning from human demonstration data. Codebase, datasets, trained models, and more available at https://arise-initiative.github.io/robomimic-web/",
  "published": "2021-08-06",
  "updated": "2021-09-25",
  "year": "2021",
  "authors": [
   "Ajay Mandlekar",
   "Danfei Xu",
   "Josiah Wong",
   "Soroush Nasiriany",
   "Chen Wang",
   "Rohun Kulkarni",
   "Li Fei-Fei",
   "Silvio Savarese",
   "Yuke Zhu",
   "Roberto Mart\u00edn-Mart\u00edn"
  ],
  "author_count": 10,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 996,
  "influential_citations": 121,
  "tldr": "This study analyzes the most critical challenges when learning from offline human data for manipulation and highlights opportunities for learning from human datasets, such as the ability to learn proficient policies on challenging, multi-stage tasks beyond the scope of current reinforcement learning methods.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "A. Mandlekar",
    "id": "49686756",
    "h_index": 36,
    "papers": 67
   },
   {
    "name": "Danfei Xu",
    "id": "2068265",
    "h_index": 33,
    "papers": 45
   },
   {
    "name": "Josiah Wong",
    "id": "33808086",
    "h_index": 11,
    "papers": 13
   },
   {
    "name": "Soroush Nasiriany",
    "id": "3457048",
    "h_index": 18,
    "papers": 24
   },
   {
    "name": "Chen Wang",
    "id": "2109119431",
    "h_index": 18,
    "papers": 28
   },
   {
    "name": "Rohun Kulkarni",
    "id": "2072231304",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Li Fei-Fei",
    "id": "48004138",
    "h_index": 143,
    "papers": 606
   },
   {
    "name": "S. Savarese",
    "id": "1702137",
    "h_index": 115,
    "papers": 346
   },
   {
    "name": "Yuke Zhu",
    "id": "2117748",
    "h_index": 57,
    "papers": 130
   },
   {
    "name": "Roberto Mart'in-Mart'in",
    "id": "2065917078",
    "h_index": 20,
    "papers": 32
   }
  ],
  "comment": "CoRL 2021 (Oral)",
  "topics": [
   "egocentric-data",
   "imitation-diffusion",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2108.03298v2",
  "pdf_url": "https://arxiv.org/pdf/2108.03298v2",
  "html_url": "https://arxiv.org/html/2108.03298v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "2108.00385",
  "slug": "transformer-based-deep-imitation-learning-for-dual-arm-robot-manipulat",
  "title": "Transformer-based deep imitation learning for dual-arm robot manipulation",
  "abstract": "Deep imitation learning is promising for solving dexterous manipulation tasks because it does not require an environment model and pre-programmed robot behavior. However, its application to dual-arm manipulation tasks remains challenging. In a dual-arm manipulation setup, the increased number of state dimensions caused by the additional robot manipulators causes distractions and results in poor performance of the neural networks. We address this issue using a self-attention mechanism that computes dependencies between elements in a sequential input and focuses on important elements. A Transformer, a variant of self-attention architecture, is applied to deep imitation learning to solve dual-arm manipulation tasks in the real world. The proposed method has been tested on dual-arm manipulation tasks using a real robot. The experimental results demonstrated that the Transformer-based deep imitation learning architecture can attend to the important features among the sensory inputs, therefore reducing distractions and improving manipulation performance when compared with the baseline architecture without the self-attention mechanisms. Data from this and related works are available at: https://sites.google.com/view/multi-task-fine.",
  "published": "2021-08-01",
  "updated": "2025-05-21",
  "year": "2021",
  "authors": [
   "Heecheol Kim",
   "Yoshiyuki Ohmura",
   "Yasuo Kuniyoshi"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS",
  "venue_source": "semantic-scholar",
  "citations": 71,
  "influential_citations": 1,
  "tldr": "The experimental results demonstrated that the Transformer-based deep imitation learning architecture can attend to the important features among the sensory inputs, therefore reducing distractions and improving manipulation performance when compared with the baseline architecture without the self-attention mechanisms.",
  "doi": "10.1109/IROS51168.2021.9636301",
  "oa_pdf": "https://arxiv.org/pdf/2108.00385",
  "s2_authors": [
   {
    "name": "Heecheol Kim",
    "id": "1742368876",
    "h_index": 8,
    "papers": 18
   },
   {
    "name": "Y. Ohmura",
    "id": "1771982",
    "h_index": 16,
    "papers": 77
   },
   {
    "name": "Y. Kuniyoshi",
    "id": "1744602",
    "h_index": 51,
    "papers": 444
   }
  ],
  "comment": "8 pages. Accepted in 2021 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS)",
  "topics": [
   "dexterous-manipulation",
   "imitation-diffusion"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2108.00385v3",
  "pdf_url": "https://arxiv.org/pdf/2108.00385v3",
  "html_url": "https://arxiv.org/html/2108.00385v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.36
 },
 {
  "id": "2107.14226",
  "slug": "learning-more-skills-through-optimistic-exploration",
  "title": "Learning more skills through optimistic exploration",
  "abstract": "Unsupervised skill learning objectives (Gregor et al., 2016, Eysenbach et al., 2018) allow agents to learn rich repertoires of behavior in the absence of extrinsic rewards. They work by simultaneously training a policy to produce distinguishable latent-conditioned trajectories, and a discriminator to evaluate distinguishability by trying to infer latents from trajectories. The hope is for the agent to explore and master the environment by encouraging each skill (latent) to reliably reach different states. However, an inherent exploration problem lingers: when a novel state is actually encountered, the discriminator will necessarily not have seen enough training data to produce accurate and confident skill classifications, leading to low intrinsic reward for the agent and effective penalization of the sort of exploration needed to actually maximize the objective. To combat this inherent pessimism towards exploration, we derive an information gain auxiliary objective that involves training an ensemble of discriminators and rewarding the policy for their disagreement. Our objective directly estimates the epistemic uncertainty that comes from the discriminator not having seen enough training examples, thus providing an intrinsic reward more tailored to the true objective compared to pseudocount-based methods (Burda et al., 2019). We call this exploration bonus discriminator disagreement intrinsic reward, or DISDAIN. We demonstrate empirically that DISDAIN improves skill learning both in a tabular grid world (Four Rooms) and the 57 games of the Atari Suite (from pixels). Thus, we encourage researchers to treat pessimism with DISDAIN.",
  "published": "2021-07-29",
  "updated": "2022-05-12",
  "year": "2021",
  "authors": [
   "DJ Strouse",
   "Kate Baumli",
   "David Warde-Farley",
   "Vlad Mnih",
   "Steven Hansen"
  ],
  "author_count": 5,
  "categories": [
   "cs.LG",
   "cs.AI",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 54,
  "influential_citations": 6,
  "tldr": "It is demonstrated empirically that DISDAIN improves skill learning both in a tabular grid world (Four Rooms) and the 57 games of the Atari Suite (from pixels) and is encouraged to treat pessimism with DIS DAIN.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "D. Strouse",
    "id": "69925460",
    "h_index": 14,
    "papers": 21
   },
   {
    "name": "Kate Baumli",
    "id": "1734809439",
    "h_index": 9,
    "papers": 13
   },
   {
    "name": "David Warde-Farley",
    "id": "1393680089",
    "h_index": 25,
    "papers": 36
   },
   {
    "name": "Vlad Mnih",
    "id": "123588356",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "S. Hansen",
    "id": "35231584",
    "h_index": 10,
    "papers": 23
   }
  ],
  "comment": "Accepted at ICLR 2022 (spotlight)",
  "topics": [
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2107.14226v6",
  "pdf_url": "https://arxiv.org/pdf/2107.14226v6",
  "html_url": "https://arxiv.org/html/2107.14226v6",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.24
 },
 {
  "id": "2107.13545",
  "slug": "fully-autonomous-real-world-reinforcement-learning-with-applications-t",
  "title": "Fully Autonomous Real-World Reinforcement Learning with Applications to Mobile Manipulation",
  "abstract": "We study how robots can autonomously learn skills that require a combination of navigation and grasping. While reinforcement learning in principle provides for automated robotic skill learning, in practice reinforcement learning in the real world is challenging and often requires extensive instrumentation and supervision. Our aim is to devise a robotic reinforcement learning system for learning navigation and manipulation together, in an autonomous way without human intervention, enabling continual learning under realistic assumptions. Our proposed system, ReLMM, can learn continuously on a real-world platform without any environment instrumentation, without human intervention, and without access to privileged information, such as maps, objects positions, or a global view of the environment. Our method employs a modularized policy with components for manipulation and navigation, where manipulation policy uncertainty drives exploration for the navigation controller, and the manipulation module provides rewards for navigation. We evaluate our method on a room cleanup task, where the robot must navigate to and pick up items scattered on the floor. After a grasp curriculum training phase, ReLMM can learn navigation and grasping together fully automatically, in around 40 hours of autonomous real-world training.",
  "published": "2021-07-28",
  "updated": "2021-12-07",
  "year": "2021",
  "authors": [
   "Charles Sun",
   "J\u0119drzej Orbik",
   "Coline Devin",
   "Brian Yang",
   "Abhishek Gupta",
   "Glen Berseth",
   "Sergey Levine"
  ],
  "author_count": 7,
  "categories": [
   "cs.LG",
   "cs.RO"
  ],
  "primary_category": "cs.LG",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 74,
  "influential_citations": 1,
  "tldr": "The aim is to devise a robotic reinforcement learning system for learning navigation and manipulation together, in an autonomous way without human intervention, enabling continual learning under realistic assumptions.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Charles Sun",
    "id": "2153033522",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Jkedrzej Orbik",
    "id": "2121326280",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Coline Devin",
    "id": "144373380",
    "h_index": 24,
    "papers": 39
   },
   {
    "name": "Brian Yang",
    "id": "10252437",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Abhishek Gupta",
    "id": "144150274",
    "h_index": 33,
    "papers": 287
   },
   {
    "name": "Glen Berseth",
    "id": "2312919053",
    "h_index": 15,
    "papers": 28
   },
   {
    "name": "S. Levine",
    "id": "1736651",
    "h_index": 167,
    "papers": 551
   }
  ],
  "comment": "16 pages, Published at CoRL 2021",
  "topics": [
   "dexterous-manipulation",
   "rl-control",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2107.13545v3",
  "pdf_url": "https://arxiv.org/pdf/2107.13545v3",
  "html_url": "https://arxiv.org/html/2107.13545v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.38
 },
 {
  "id": "2107.09645",
  "slug": "mastering-visual-continuous-control-improved-data-augmented-reinforcem",
  "title": "Mastering Visual Continuous Control: Improved Data-Augmented Reinforcement Learning",
  "abstract": "We present DrQ-v2, a model-free reinforcement learning (RL) algorithm for visual continuous control. DrQ-v2 builds on DrQ, an off-policy actor-critic approach that uses data augmentation to learn directly from pixels. We introduce several improvements that yield state-of-the-art results on the DeepMind Control Suite. Notably, DrQ-v2 is able to solve complex humanoid locomotion tasks directly from pixel observations, previously unattained by model-free RL. DrQ-v2 is conceptually simple, easy to implement, and provides significantly better computational footprint compared to prior work, with the majority of tasks taking just 8 hours to train on a single GPU. Finally, we publicly release DrQ-v2's implementation to provide RL practitioners with a strong and computationally efficient baseline.",
  "published": "2021-07-20",
  "updated": "2021-07-20",
  "year": "2021",
  "authors": [
   "Denis Yarats",
   "Rob Fergus",
   "Alessandro Lazaric",
   "Lerrel Pinto"
  ],
  "author_count": 4,
  "categories": [
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.AI",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 501,
  "influential_citations": 107,
  "tldr": "DrQ-v2 builds on DrQ, an off-policy actor-critic approach that uses data augmentation to learn directly from pixels and is able to solve complex humanoid locomotion tasks directly from pixel observations.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Denis Yarats",
    "id": "13759615",
    "h_index": 21,
    "papers": 31
   },
   {
    "name": "R. Fergus",
    "id": "2276554",
    "h_index": 78,
    "papers": 125
   },
   {
    "name": "A. Lazaric",
    "id": "3254390",
    "h_index": 43,
    "papers": 177
   },
   {
    "name": "Lerrel Pinto",
    "id": "34026610",
    "h_index": 41,
    "papers": 70
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "rl-control"
  ],
  "orgs": [
   "Google DeepMind"
  ],
  "abs_url": "https://arxiv.org/abs/2107.09645v1",
  "pdf_url": "https://arxiv.org/pdf/2107.09645v1",
  "html_url": "https://arxiv.org/html/2107.09645v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.7
 },
 {
  "id": "2107.08408",
  "slug": "pre-trained-language-models-as-prior-knowledge-for-playing-text-based",
  "title": "Pre-trained Language Models as Prior Knowledge for Playing Text-based Games",
  "abstract": "Recently, text world games have been proposed to enable artificial agents to understand and reason about real-world scenarios. These text-based games are challenging for artificial agents, as it requires an understanding of and interaction using natural language in a partially observable environment. Agents observe the environment via textual descriptions designed to be challenging enough for even human players. Past approaches have not paid enough attention to the language understanding capability of the proposed agents. Typically, these approaches train from scratch, an agent that learns both textual representations and the gameplay online during training using a temporal loss function. Given the sample-inefficiency of RL approaches, it is inefficient to learn rich enough textual representations to be able to understand and reason using the textual observation in such a complicated game environment setting. In this paper, we improve the semantic understanding of the agent by proposing a simple RL with LM framework where we use transformer-based language models with Deep RL models. We perform a detailed study of our framework to demonstrate how our model outperforms all existing agents on the popular game, Zork1, to achieve a score of 44.7, which is 1.6 higher than the state-of-the-art model. Overall, our proposed approach outperforms 4 games out of the 14 text-based games, while performing comparable to the state-of-the-art models on the remaining games.",
  "published": "2021-07-18",
  "updated": "2021-12-23",
  "year": "2021",
  "authors": [
   "Ishika Singh",
   "Gargi Singh",
   "Ashutosh Modi"
  ],
  "author_count": 3,
  "categories": [
   "cs.CL",
   "cs.AI",
   "cs.MA",
   "cs.RO"
  ],
  "primary_category": "cs.CL",
  "venue": "",
  "venue_source": "",
  "citations": 31,
  "influential_citations": 0,
  "tldr": "This paper proposes a simple RL with LM framework where it uses transformer-based language models with Deep RL models to improve the semantic understanding of the agent by proposing a simple RL with LM framework.",
  "doi": "10.5555/3535850.3536091",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ishika Singh",
    "id": "144399662",
    "h_index": 9,
    "papers": 21
   },
   {
    "name": "Gargi Singh",
    "id": "2148286347",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Ashutosh Modi",
    "id": "2477939",
    "h_index": 25,
    "papers": 68
   }
  ],
  "comment": "40 Pages (8 Pages main content + 1 Page references + 31 Pages Appendix). Some new results added",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2107.08408v2",
  "pdf_url": "https://arxiv.org/pdf/2107.08408v2",
  "html_url": "https://arxiv.org/html/2107.08408v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.51
 },
 {
  "id": "2107.08398",
  "slug": "unsupervised-skill-discovery-and-skill-learning-in-minecraft",
  "title": "Unsupervised Skill-Discovery and Skill-Learning in Minecraft",
  "abstract": "Pre-training Reinforcement Learning agents in a task-agnostic manner has shown promising results. However, previous works still struggle in learning and discovering meaningful skills in high-dimensional state-spaces, such as pixel-spaces. We approach the problem by leveraging unsupervised skill discovery and self-supervised learning of state representations. In our work, we learn a compact latent representation by making use of variational and contrastive techniques. We demonstrate that both enable RL agents to learn a set of basic navigation skills by maximizing an information theoretic objective. We assess our method in Minecraft 3D pixel maps with different complexities. Our results show that representations and conditioned policies learned from pixels are enough for toy examples, but do not scale to realistic and complex maps. To overcome these limitations, we explore alternative input observations such as the relative position of the agent along with the raw pixels.",
  "published": "2021-07-18",
  "updated": "2021-07-18",
  "year": "2021",
  "authors": [
   "Juan Jos\u00e9 Nieto",
   "Roger Creus",
   "Xavier Giro-i-Nieto"
  ],
  "author_count": 3,
  "categories": [
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.AI",
  "venue": "ICML",
  "venue_source": "arxiv-comment",
  "citations": 7,
  "influential_citations": 0,
  "tldr": "This work demonstrates that unsupervised skill discovery and self-supervised learning of state representations enable RL agents to learn a set of basic navigation skills by maximizing an information theoretic objective.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "J. J. Nieto",
    "id": "2065430983",
    "h_index": 11,
    "papers": 21
   },
   {
    "name": "Roger Creus",
    "id": "1720802717",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Xavier Gir\u00f3-i-Nieto",
    "id": "1398090762",
    "h_index": 29,
    "papers": 117
   }
  ],
  "comment": "Accepted at ICML Unsupervised RL Workshop, 8 pages",
  "topics": [
   "rl-control",
   "navigation",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2107.08398v1",
  "pdf_url": "https://arxiv.org/pdf/2107.08398v1",
  "html_url": "https://arxiv.org/html/2107.08398v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.4
 },
 {
  "id": "2107.04034",
  "slug": "rma-rapid-motor-adaptation-for-legged-robots",
  "title": "RMA: Rapid Motor Adaptation for Legged Robots",
  "abstract": "Successful real-world deployment of legged robots would require them to adapt in real-time to unseen scenarios like changing terrains, changing payloads, wear and tear. This paper presents Rapid Motor Adaptation (RMA) algorithm to solve this problem of real-time online adaptation in quadruped robots. RMA consists of two components: a base policy and an adaptation module. The combination of these components enables the robot to adapt to novel situations in fractions of a second. RMA is trained completely in simulation without using any domain knowledge like reference trajectories or predefined foot trajectory generators and is deployed on the A1 robot without any fine-tuning. We train RMA on a varied terrain generator using bioenergetics-inspired rewards and deploy it on a variety of difficult terrains including rocky, slippery, deformable surfaces in environments with grass, long vegetation, concrete, pebbles, stairs, sand, etc. RMA shows state-of-the-art performance across diverse real-world as well as simulation experiments. Video results at https://ashish-kmr.github.io/rma-legged-robots/",
  "published": "2021-07-08",
  "updated": "2021-07-08",
  "year": "2021",
  "authors": [
   "Ashish Kumar",
   "Zipeng Fu",
   "Deepak Pathak",
   "Jitendra Malik"
  ],
  "author_count": 4,
  "categories": [
   "cs.LG",
   "cs.AI",
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.LG",
  "venue": "RSS",
  "venue_source": "semantic-scholar",
  "citations": 981,
  "influential_citations": 93,
  "tldr": "Rapid Motor Adaptation algorithm is presented to solve the problem of real-time online adaptation in quadruped robots by trained completely in simulation without using any domain knowledge like reference trajectories or predefined foot trajectory generators and deployed on the A1 robot without any fine-tuning.",
  "doi": "10.15607/RSS.2021.XVII.011",
  "oa_pdf": "https://doi.org/10.15607/rss.2021.xvii.011",
  "s2_authors": [
   {
    "name": "Ashish Kumar",
    "id": "2118069675",
    "h_index": 15,
    "papers": 32
   },
   {
    "name": "Zipeng Fu",
    "id": "89704471",
    "h_index": 13,
    "papers": 18
   },
   {
    "name": "Deepak Pathak",
    "id": "2004879394",
    "h_index": 24,
    "papers": 32
   },
   {
    "name": "Jitendra Malik",
    "id": "143751119",
    "h_index": 139,
    "papers": 397
   }
  ],
  "comment": "RSS 2021. Webpage at https://ashish-kmr.github.io/rma-legged-robots/",
  "topics": [
   "humanoids"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2107.04034v1",
  "pdf_url": "https://arxiv.org/pdf/2107.04034v1",
  "html_url": "https://arxiv.org/html/2107.04034v1",
  "code_url": "https://ashish-kmr.github.io/rma-legged-robots/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.49
 },
 {
  "id": "2107.03996",
  "slug": "learning-vision-guided-quadrupedal-locomotion-end-to-end-with-cross-mo",
  "title": "Learning Vision-Guided Quadrupedal Locomotion End-to-End with Cross-Modal Transformers",
  "abstract": "We propose to address quadrupedal locomotion tasks using Reinforcement Learning (RL) with a Transformer-based model that learns to combine proprioceptive information and high-dimensional depth sensor inputs. While learning-based locomotion has made great advances using RL, most methods still rely on domain randomization for training blind agents that generalize to challenging terrains. Our key insight is that proprioceptive states only offer contact measurements for immediate reaction, whereas an agent equipped with visual sensory observations can learn to proactively maneuver environments with obstacles and uneven terrain by anticipating changes in the environment many steps ahead. In this paper, we introduce LocoTransformer, an end-to-end RL method that leverages both proprioceptive states and visual observations for locomotion control. We evaluate our method in challenging simulated environments with different obstacles and uneven terrain. We transfer our learned policy from simulation to a real robot by running it indoors and in the wild with unseen obstacles and terrain. Our method not only significantly improves over baselines, but also achieves far better generalization performance, especially when transferred to the real robot. Our project page with videos is at https://rchalyang.github.io/LocoTransformer/ .",
  "published": "2021-07-08",
  "updated": "2022-05-26",
  "year": "2021",
  "authors": [
   "Ruihan Yang",
   "Minghao Zhang",
   "Nicklas Hansen",
   "Huazhe Xu",
   "Xiaolong Wang"
  ],
  "author_count": 5,
  "categories": [
   "cs.LG",
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.LG",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 157,
  "influential_citations": 9,
  "tldr": "LocoTransformer is introduced, an end-to-end RL method that leverages both proprioceptive states and visual observations for locomotion control that significantly improves over baselines and achieves far better generalization performance, especially when transferred to the real robot.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ruihan Yang",
    "id": "143955842",
    "h_index": 18,
    "papers": 22
   },
   {
    "name": "Minghao Zhang",
    "id": "2112152893",
    "h_index": 7,
    "papers": 14
   },
   {
    "name": "Nicklas Hansen",
    "id": "1491707104",
    "h_index": 19,
    "papers": 39
   },
   {
    "name": "Huazhe Xu",
    "id": "2109369100",
    "h_index": 14,
    "papers": 17
   },
   {
    "name": "Xiaolong Wang",
    "id": "122024152",
    "h_index": 42,
    "papers": 63
   }
  ],
  "comment": "Our project page with videos is at https://RchalYang.github.io/LocoTransformer",
  "topics": [
   "humanoids",
   "sim2real",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2107.03996v3",
  "pdf_url": "https://arxiv.org/pdf/2107.03996v3",
  "html_url": "https://arxiv.org/html/2107.03996v3",
  "code_url": "https://RchalYang.github.io/LocoTransformer",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.7
 },
 {
  "id": "2107.01518",
  "slug": "hierarchical-policies-for-cluttered-scene-grasping-with-latent-plans",
  "title": "Hierarchical Policies for Cluttered-Scene Grasping with Latent Plans",
  "abstract": "6D grasping in cluttered scenes is a longstanding problem in robotic manipulation. Open-loop manipulation pipelines may fail due to inaccurate state estimation, while most end-to-end grasping methods have not yet scaled to complex scenes with obstacles. In this work, we propose a new method for end-to-end learning of 6D grasping in cluttered scenes. Our hierarchical framework learns collision-free target-driven grasping based on partial point cloud observations. We learn an embedding space to encode expert grasping plans during training and a variational autoencoder to sample diverse grasping trajectories at test time. Furthermore, we train a critic network for plan selection and an option classifier for switching to an instance grasping policy through hierarchical reinforcement learning. We evaluate our method and compare against several baselines in simulation, as well as demonstrate that our latent planning can generalize to real-world cluttered-scene grasping tasks. Our videos and code can be found at https://sites.google.com/view/latent-grasping .",
  "published": "2021-07-04",
  "updated": "2022-01-11",
  "year": "2021",
  "authors": [
   "Lirui Wang",
   "Xiangyun Meng",
   "Yu Xiang",
   "Dieter Fox"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "RA-L",
  "venue_source": "semantic-scholar",
  "citations": 29,
  "influential_citations": 1,
  "tldr": "A hierarchical framework learns collision-free target-driven grasping based on partial point cloud observations and an embedding space to encode expert grasping plans during training and a variational autoencoder to sample diverse grasping trajectories at test time.",
  "doi": "10.1109/LRA.2022.3143198",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Lirui Wang",
    "id": "2108537099",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Yu Xiang",
    "id": "144863550",
    "h_index": 31,
    "papers": 53
   },
   {
    "name": "D. Fox",
    "id": "145197953",
    "h_index": 133,
    "papers": 428
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "rl-control",
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2107.01518v3",
  "pdf_url": "https://arxiv.org/pdf/2107.01518v3",
  "html_url": "https://arxiv.org/html/2107.01518v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.98
 },
 {
  "id": "2107.00135",
  "slug": "attention-bottlenecks-for-multimodal-fusion",
  "title": "Attention Bottlenecks for Multimodal Fusion",
  "abstract": "Humans perceive the world by concurrently processing and fusing high-dimensional inputs from multiple modalities such as vision and audio. Machine perception models, in stark contrast, are typically modality-specific and optimised for unimodal benchmarks, and hence late-stage fusion of final representations or predictions from each modality (`late-fusion') is still a dominant paradigm for multimodal video classification. Instead, we introduce a novel transformer based architecture that uses `fusion bottlenecks' for modality fusion at multiple layers. Compared to traditional pairwise self-attention, our model forces information between different modalities to pass through a small number of bottleneck latents, requiring the model to collate and condense the most relevant information in each modality and only share what is necessary. We find that such a strategy improves fusion performance, at the same time reducing computational cost. We conduct thorough ablation studies, and achieve state-of-the-art results on multiple audio-visual classification benchmarks including Audioset, Epic-Kitchens and VGGSound. All code and models will be released.",
  "published": "2021-06-30",
  "updated": "2022-11-30",
  "year": "2021",
  "authors": [
   "Arsha Nagrani",
   "Shan Yang",
   "Anurag Arnab",
   "Aren Jansen",
   "Cordelia Schmid",
   "Chen Sun"
  ],
  "author_count": 6,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 849,
  "influential_citations": 73,
  "tldr": "This work introduces a novel transformer based architecture that uses `fusion bottlenecks' for modality fusion at multiple layers and finds that such a strategy improves fusion performance, at the same time reducing computational cost.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Arsha Nagrani",
    "id": "19263506",
    "h_index": 36,
    "papers": 62
   },
   {
    "name": "Shan Yang",
    "id": "2115298670",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Anurag Arnab",
    "id": "31638576",
    "h_index": 27,
    "papers": 45
   },
   {
    "name": "A. Jansen",
    "id": "35996413",
    "h_index": 40,
    "papers": 102
   },
   {
    "name": "Cordelia Schmid",
    "id": "2462253",
    "h_index": 153,
    "papers": 466
   },
   {
    "name": "Chen Sun",
    "id": "1491624845",
    "h_index": 45,
    "papers": 87
   }
  ],
  "comment": "Published at NeurIPS 2021. Note this version updates numbers due to a bug in the AudioSet mAP calculation in Table 1 (last row)",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2107.00135v3",
  "pdf_url": "https://arxiv.org/pdf/2107.00135v3",
  "html_url": "https://arxiv.org/html/2107.00135v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.43
 },
 {
  "id": "2106.14876",
  "slug": "multi-task-curriculum-learning-in-a-complex-visual-hard-exploration-do",
  "title": "Multi-task curriculum learning in a complex, visual, hard-exploration domain: Minecraft",
  "abstract": "An important challenge in reinforcement learning is training agents that can solve a wide variety of tasks. If tasks depend on each other (e.g. needing to learn to walk before learning to run), curriculum learning can speed up learning by focusing on the next best task to learn. We explore curriculum learning in a complex, visual domain with many hard exploration challenges: Minecraft. We find that learning progress (defined as a change in success probability of a task) is a reliable measure of learnability for automatically constructing an effective curriculum. We introduce a learning-progress based curriculum and test it on a complex reinforcement learning problem (called \"Simon Says\") where an agent is instructed to obtain a desired goal item. Many of the required skills depend on each other. Experiments demonstrate that: (1) a within-episode exploration bonus for obtaining new items improves performance, (2) dynamically adjusting this bonus across training such that it only applies to items the agent cannot reliably obtain yet further increases performance, (3) the learning-progress based curriculum elegantly follows the learning curve of the agent, and (4) when the learning-progress based curriculum is combined with the dynamic exploration bonus it learns much more efficiently and obtains far higher performance than uniform baselines. These results suggest that combining intra-episode and across-training exploration bonuses with learning progress creates a promising method for automated curriculum generation, which may substantially increase our ability to train more capable, generally intelligent agents.",
  "published": "2021-06-28",
  "updated": "2021-06-28",
  "year": "2021",
  "authors": [
   "Ingmar Kanitscheider",
   "Joost Huizinga",
   "David Farhi",
   "William Hebgen Guss",
   "Brandon Houghton",
   "Raul Sampedro",
   "Peter Zhokhov",
   "Bowen Baker",
   "Adrien Ecoffet",
   "Jie Tang",
   "Oleg Klimov",
   "Jeff Clune"
  ],
  "author_count": 12,
  "categories": [
   "cs.LG",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 38,
  "influential_citations": 3,
  "tldr": "The results suggest that combining intra-episode and across-training exploration bonuses with learning progress creates a promising method for automated curriculum generation, which may substantially increase the ability to train more capable, generally intelligent agents.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "I. Kanitscheider",
    "id": "3151440",
    "h_index": 15,
    "papers": 45
   },
   {
    "name": "Joost Huizinga",
    "id": "39378983",
    "h_index": 18,
    "papers": 36
   },
   {
    "name": "David Farhi",
    "id": "2065430571",
    "h_index": 11,
    "papers": 32
   },
   {
    "name": "William H. Guss",
    "id": "39121861",
    "h_index": 12,
    "papers": 20
   },
   {
    "name": "Brandon Houghton",
    "id": "103681415",
    "h_index": 9,
    "papers": 16
   },
   {
    "name": "Raul Sampedro",
    "id": "2076161792",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Peter Zhokhov",
    "id": "2115450657",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Bowen Baker",
    "id": "40566201",
    "h_index": 12,
    "papers": 18
   },
   {
    "name": "Adrien Ecoffet",
    "id": "66821245",
    "h_index": 10,
    "papers": 19
   },
   {
    "name": "Jie Tang",
    "id": "2109541439",
    "h_index": 31,
    "papers": 51
   },
   {
    "name": "Oleg Klimov",
    "id": "2067138712",
    "h_index": 6,
    "papers": 6
   },
   {
    "name": "J. Clune",
    "id": "2552141",
    "h_index": 53,
    "papers": 119
   }
  ],
  "comment": "first submission",
  "topics": [
   "rl-control",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2106.14876v1",
  "pdf_url": "https://arxiv.org/pdf/2106.14876v1",
  "html_url": "https://arxiv.org/html/2106.14876v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.59
 },
 {
  "id": "2106.12534",
  "slug": "coarse-to-fine-q-attention-efficient-learning-for-visual-robotic-manip",
  "title": "Coarse-to-Fine Q-attention: Efficient Learning for Visual Robotic Manipulation via Discretisation",
  "abstract": "We present a coarse-to-fine discretisation method that enables the use of discrete reinforcement learning approaches in place of unstable and data-inefficient actor-critic methods in continuous robotics domains. This approach builds on the recently released ARM algorithm, which replaces the continuous next-best pose agent with a discrete one, with coarse-to-fine Q-attention. Given a voxelised scene, coarse-to-fine Q-attention learns what part of the scene to 'zoom' into. When this 'zooming' behaviour is applied iteratively, it results in a near-lossless discretisation of the translation space, and allows the use of a discrete action, deep Q-learning method. We show that our new coarse-to-fine algorithm achieves state-of-the-art performance on several difficult sparsely rewarded RLBench vision-based robotics tasks, and can train real-world policies, tabula rasa, in a matter of minutes, with as little as 3 demonstrations.",
  "published": "2021-06-23",
  "updated": "2022-03-15",
  "year": "2021",
  "authors": [
   "Stephen James",
   "Kentaro Wada",
   "Tristan Laidlow",
   "Andrew J. Davison"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 201,
  "influential_citations": 23,
  "tldr": "A coarse-to-fine discretisation method that enables the use of discrete reinforcement learning approaches in place of unstable and data-inefficient actorcritic methods in continuous robotics domains, and achieves state-of-the-art performance on several difficult sparsely rewarded RLBench vision-based robotics tasks.",
  "doi": "10.1109/CVPR52688.2022.01337",
  "oa_pdf": "https://arxiv.org/pdf/2106.12534",
  "s2_authors": [
   {
    "name": "Stephen James",
    "id": "2055291154",
    "h_index": 22,
    "papers": 37
   },
   {
    "name": "Kentaro Wada",
    "id": "2112787966",
    "h_index": 12,
    "papers": 26
   },
   {
    "name": "Tristan Laidlow",
    "id": "3422926",
    "h_index": 11,
    "papers": 15
   },
   {
    "name": "A. Davison",
    "id": "2052135690",
    "h_index": 77,
    "papers": 188
   }
  ],
  "comment": "Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR 2022). Videos and code: https://sites.google.com/view/c2f-q-attention",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2106.12534v2",
  "pdf_url": "https://arxiv.org/pdf/2106.12534v2",
  "html_url": "https://arxiv.org/html/2106.12534v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.81
 },
 {
  "id": "2106.08851",
  "slug": "gelsight-wedge-measuring-high-resolution-3d-contact-geometry-with-a-co",
  "title": "GelSight Wedge: Measuring High-Resolution 3D Contact Geometry with a Compact Robot Finger",
  "abstract": "Vision-based tactile sensors have the potential to provide important contact geometry to localize the objective with visual occlusion. However, it is challenging to measure high-resolution 3D contact geometry for a compact robot finger, to simultaneously meet optical and mechanical constraints. In this work, we present the GelSight Wedge sensor, which is optimized to have a compact shape for robot fingers, while achieving high-resolution 3D reconstruction. We evaluate the 3D reconstruction under different lighting configurations, and extend the method from 3 lights to 1 or 2 lights. We demonstrate the flexibility of the design by shrinking the sensor to the size of a human finger for fine manipulation tasks. We also show the effectiveness and potential of the reconstructed 3D geometry for pose tracking in the 3D space.",
  "published": "2021-06-16",
  "updated": "2021-06-16",
  "year": "2021",
  "authors": [
   "Shaoxiong Wang",
   "Yu She",
   "Branden Romero",
   "Edward Adelson"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA",
  "venue_source": "semantic-scholar",
  "citations": 188,
  "influential_citations": 31,
  "tldr": "The GelSight Wedge sensor is presented, which is optimized to have a compact shape for robot fingers, while achieving high-resolution 3D reconstruction, and the effectiveness and potential of the reconstructed 3D geometry for pose tracking in the 3D space is shown.",
  "doi": "10.1109/ICRA48506.2021.9560783",
  "oa_pdf": "https://arxiv.org/pdf/2106.08851",
  "s2_authors": [
   {
    "name": "Shaoxiong Wang",
    "id": "7488549",
    "h_index": 14,
    "papers": 16
   },
   {
    "name": "Y. She",
    "id": "2392034",
    "h_index": 17,
    "papers": 43
   },
   {
    "name": "Branden Romero",
    "id": "21201572",
    "h_index": 9,
    "papers": 10
   },
   {
    "name": "E. Adelson",
    "id": "145358192",
    "h_index": 87,
    "papers": 272
   }
  ],
  "comment": "ICRA 2021",
  "topics": [
   "tactile",
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2106.08851v1",
  "pdf_url": "https://arxiv.org/pdf/2106.08851v1",
  "html_url": "https://arxiv.org/html/2106.08851v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.78
 },
 {
  "id": "2106.08254",
  "slug": "beit-bert-pre-training-of-image-transformers",
  "title": "BEiT: BERT Pre-Training of Image Transformers",
  "abstract": "We introduce a self-supervised vision representation model BEiT, which stands for Bidirectional Encoder representation from Image Transformers. Following BERT developed in the natural language processing area, we propose a masked image modeling task to pretrain vision Transformers. Specifically, each image has two views in our pre-training, i.e, image patches (such as 16x16 pixels), and visual tokens (i.e., discrete tokens). We first \"tokenize\" the original image into visual tokens. Then we randomly mask some image patches and fed them into the backbone Transformer. The pre-training objective is to recover the original visual tokens based on the corrupted image patches. After pre-training BEiT, we directly fine-tune the model parameters on downstream tasks by appending task layers upon the pretrained encoder. Experimental results on image classification and semantic segmentation show that our model achieves competitive results with previous pre-training methods. For example, base-size BEiT achieves 83.2% top-1 accuracy on ImageNet-1K, significantly outperforming from-scratch DeiT training (81.8%) with the same setup. Moreover, large-size BEiT obtains 86.3% only using ImageNet-1K, even outperforming ViT-L with supervised pre-training on ImageNet-22K (85.2%). The code and pretrained models are available at https://aka.ms/beit.",
  "published": "2021-06-15",
  "updated": "2022-09-03",
  "year": "2021",
  "authors": [
   "Hangbo Bao",
   "Li Dong",
   "Songhao Piao",
   "Furu Wei"
  ],
  "author_count": 4,
  "categories": [
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.CV",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 3857,
  "influential_citations": 484,
  "tldr": "A self-supervised vision representation model BEiT, which stands for Bidirectional Encoder representation from Image Transformers, is introduced, and results on image classification and semantic segmentation show that the model achieves competitive results with previous pre-training methods.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Hangbo Bao",
    "id": "10699417",
    "h_index": 17,
    "papers": 28
   },
   {
    "name": "Li Dong",
    "id": "145307652",
    "h_index": 63,
    "papers": 118
   },
   {
    "name": "Furu Wei",
    "id": "49807919",
    "h_index": 105,
    "papers": 326
   }
  ],
  "comment": "A Path to the BERT Moment of CV",
  "topics": [
   "foundation-pretraining",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2106.08254v2",
  "pdf_url": "https://arxiv.org/pdf/2106.08254v2",
  "html_url": "https://arxiv.org/html/2106.08254v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "2106.02036",
  "slug": "anticipative-video-transformer",
  "title": "Anticipative Video Transformer",
  "abstract": "We propose Anticipative Video Transformer (AVT), an end-to-end attention-based video modeling architecture that attends to the previously observed video in order to anticipate future actions. We train the model jointly to predict the next action in a video sequence, while also learning frame feature encoders that are predictive of successive future frames' features. Compared to existing temporal aggregation strategies, AVT has the advantage of both maintaining the sequential progression of observed actions while still capturing long-range dependencies--both critical for the anticipation task. Through extensive experiments, we show that AVT obtains the best reported performance on four popular action anticipation benchmarks: EpicKitchens-55, EpicKitchens-100, EGTEA Gaze+, and 50-Salads; and it wins first place in the EpicKitchens-100 CVPR'21 challenge.",
  "published": "2021-06-03",
  "updated": "2021-09-22",
  "year": "2021",
  "authors": [
   "Rohit Girdhar",
   "Kristen Grauman"
  ],
  "author_count": 2,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.LG",
   "cs.MM"
  ],
  "primary_category": "cs.CV",
  "venue": "ICCV",
  "venue_source": "semantic-scholar",
  "citations": 289,
  "influential_citations": 41,
  "tldr": "Anticipative Video Transformer has the advantage of both maintaining the sequential progression of observed actions while still capturing long-range dependencies\u2014both critical for the anticipation task.",
  "doi": "10.1109/ICCV48922.2021.01325",
  "oa_pdf": "https://arxiv.org/pdf/2106.02036",
  "s2_authors": [
   {
    "name": "Rohit Girdhar",
    "id": "3102850",
    "h_index": 31,
    "papers": 95
   },
   {
    "name": "K. Grauman",
    "id": "1794409",
    "h_index": 99,
    "papers": 295
   }
  ],
  "comment": "ICCV 2021. Ranked #1 in CVPR'21 EPIC-Kitchens-100 Action Anticipation challenge. Webpage/code/models: http://facebookresearch.github.io/AVT",
  "topics": [],
  "orgs": [
   "Meta FAIR"
  ],
  "abs_url": "https://arxiv.org/abs/2106.02036v2",
  "pdf_url": "https://arxiv.org/pdf/2106.02036v2",
  "html_url": "https://arxiv.org/html/2106.02036v2",
  "code_url": "https://facebookresearch.github.io/AVT",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.46
 },
 {
  "id": "2105.14829",
  "slug": "q-attention-enabling-efficient-learning-for-vision-based-robotic-manip",
  "title": "Q-attention: Enabling Efficient Learning for Vision-based Robotic Manipulation",
  "abstract": "Despite the success of reinforcement learning methods, they have yet to have their breakthrough moment when applied to a broad range of robotic manipulation tasks. This is partly due to the fact that reinforcement learning algorithms are notoriously difficult and time consuming to train, which is exacerbated when training from images rather than full-state inputs. As humans perform manipulation tasks, our eyes closely monitor every step of the process with our gaze focusing sequentially on the objects being manipulated. With this in mind, we present our Attention-driven Robotic Manipulation (ARM) algorithm, which is a general manipulation algorithm that can be applied to a range of sparse-rewarded tasks, given only a small number of demonstrations. ARM splits the complex task of manipulation into a 3 stage pipeline: (1) a Q-attention agent extracts relevant pixel locations from RGB and point cloud inputs, (2) a next-best pose agent that accepts crops from the Q-attention agent and outputs poses, and (3) a control agent that takes the goal pose and outputs joint actions. We show that current learning algorithms fail on a range of RLBench tasks, whilst ARM is successful.",
  "published": "2021-05-31",
  "updated": "2022-02-04",
  "year": "2021",
  "authors": [
   "Stephen James",
   "Andrew J. Davison"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "RA-L",
  "venue_source": "semantic-scholar",
  "citations": 154,
  "influential_citations": 18,
  "tldr": "The Attention-driven Robotic Manipulation (ARM) algorithm is presented, which is a general manipulation algorithm that can be applied to a range of sparse-rewarded tasks, given only a small number of demonstrations.",
  "doi": "10.1109/lra.2022.3140817",
  "oa_pdf": "https://arxiv.org/pdf/2105.14829",
  "s2_authors": [
   {
    "name": "Stephen James",
    "id": "2055291154",
    "h_index": 22,
    "papers": 37
   },
   {
    "name": "A. Davison",
    "id": "2052135690",
    "h_index": 77,
    "papers": 188
   }
  ],
  "comment": "IEEE Robotics and Automation Letters, 2022 (+ presentation at ICRA 2022). Videos and code found at: https://sites.google.com/view/q-attention",
  "topics": [
   "rl-control",
   "spatial-3d",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2105.14829v2",
  "pdf_url": "https://arxiv.org/pdf/2105.14829v2",
  "html_url": "https://arxiv.org/html/2105.14829v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.69
 },
 {
  "id": "2105.09371",
  "slug": "voila-visual-observation-only-imitation-learning-for-autonomous-naviga",
  "title": "VOILA: Visual-Observation-Only Imitation Learning for Autonomous Navigation",
  "abstract": "While imitation learning for vision based autonomous mobile robot navigation has recently received a great deal of attention in the research community, existing approaches typically require state action demonstrations that were gathered using the deployment platform. However, what if one cannot easily outfit their platform to record these demonstration signals or worse yet the demonstrator does not have access to the platform at all? Is imitation learning for vision based autonomous navigation even possible in such scenarios? In this work, we hypothesize that the answer is yes and that recent ideas from the Imitation from Observation (IfO) literature can be brought to bear such that a robot can learn to navigate using only ego centric video collected by a demonstrator, even in the presence of viewpoint mismatch. To this end, we introduce a new algorithm, Visual Observation only Imitation Learning for Autonomous navigation (VOILA), that can successfully learn navigation policies from a single video demonstration collected from a physically different agent. We evaluate VOILA in the photorealistic AirSim simulator and show that VOILA not only successfully imitates the expert, but that it also learns navigation policies that can generalize to novel environments. Further, we demonstrate the effectiveness of VOILA in a real world setting by showing that it allows a wheeled Jackal robot to successfully imitate a human walking in an environment using a video recorded using a mobile phone camera.",
  "published": "2021-05-19",
  "updated": "2021-10-08",
  "year": "2021",
  "authors": [
   "Haresh Karnan",
   "Garrett Warnell",
   "Xuesu Xiao",
   "Peter Stone"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA",
  "venue_source": "semantic-scholar",
  "citations": 77,
  "influential_citations": 0,
  "tldr": "A new algorithm, Visual-Observation-only Imitation Learning for Autonomous navigation (VOILA), that can successfully learn navigation policies from a single video demonstration collected from a physically different agent is introduced.",
  "doi": "10.1109/icra46639.2022.9812316",
  "oa_pdf": "https://arxiv.org/pdf/2105.09371",
  "s2_authors": [
   {
    "name": "Haresh Karnan",
    "id": "27636362",
    "h_index": 14,
    "papers": 25
   },
   {
    "name": "Garrett Warnell",
    "id": "1938253",
    "h_index": 30,
    "papers": 85
   },
   {
    "name": "Xuesu Xiao",
    "id": "2118724825",
    "h_index": 21,
    "papers": 45
   },
   {
    "name": "P. Stone",
    "id": "144848112",
    "h_index": 95,
    "papers": 704
   }
  ],
  "comment": "Under Submission to ICRA+RAL 2022",
  "topics": [
   "sim2real",
   "imitation-diffusion",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2105.09371v2",
  "pdf_url": "https://arxiv.org/pdf/2105.09371v2",
  "html_url": "https://arxiv.org/html/2105.09371v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.39
 },
 {
  "id": "2105.08328",
  "slug": "blind-bipedal-stair-traversal-via-sim-to-real-reinforcement-learning",
  "title": "Blind Bipedal Stair Traversal via Sim-to-Real Reinforcement Learning",
  "abstract": "Accurate and precise terrain estimation is a difficult problem for robot locomotion in real-world environments. Thus, it is useful to have systems that do not depend on accurate estimation to the point of fragility. In this paper, we explore the limits of such an approach by investigating the problem of traversing stair-like terrain without any external perception or terrain models on a bipedal robot. For such blind bipedal platforms, the problem appears difficult (even for humans) due to the surprise elevation changes. Our main contribution is to show that sim-to-real reinforcement learning (RL) can achieve robust locomotion over stair-like terrain on the bipedal robot Cassie using only proprioceptive feedback. Importantly, this only requires modifying an existing flat-terrain training RL framework to include stair-like terrain randomization, without any changes in reward function. To our knowledge, this is the first controller for a bipedal, human-scale robot capable of reliably traversing a variety of real-world stairs and other stair-like disturbances using only proprioception.",
  "published": "2021-05-18",
  "updated": "2021-05-18",
  "year": "2021",
  "authors": [
   "Jonah Siekmann",
   "Kevin Green",
   "John Warila",
   "Alan Fern",
   "Jonathan Hurst"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "RSS",
  "venue_source": "semantic-scholar",
  "citations": 254,
  "influential_citations": 4,
  "tldr": "This paper shows that sim-to-real reinforcement learning (RL) can achieve robust locomotion over stair-like terrain on the bipedal robot Cassie using only proprioceptive feedback, and only requires modifying an existing flat-terrain training RL framework to include stair- like terrain randomization, without any changes in reward function.",
  "doi": "10.15607/RSS.2021.XVII.061",
  "oa_pdf": "https://doi.org/10.15607/rss.2021.xvii.061",
  "s2_authors": [
   {
    "name": "Jonah Siekmann",
    "id": "1736548025",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "Kevin R. Green",
    "id": "1491175164",
    "h_index": 11,
    "papers": 28
   },
   {
    "name": "John Warila",
    "id": "2093911322",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Alan Fern",
    "id": "145841336",
    "h_index": 46,
    "papers": 223
   },
   {
    "name": "J. Hurst",
    "id": "48660029",
    "h_index": 33,
    "papers": 85
   }
  ],
  "comment": "Accepted to RSS 2021. Submission video available at https://youtu.be/MPhEmC6b6XU and video of a supplemental robustness test at https://youtu.be/nuhHiKEtaZQ",
  "topics": [
   "humanoids",
   "sim2real",
   "rl-control",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2105.08328v1",
  "pdf_url": "https://arxiv.org/pdf/2105.08328v1",
  "html_url": "https://arxiv.org/html/2105.08328v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.91
 },
 {
  "id": "2105.05233",
  "slug": "diffusion-models-beat-gans-on-image-synthesis",
  "title": "Diffusion Models Beat GANs on Image Synthesis",
  "abstract": "We show that diffusion models can achieve image sample quality superior to the current state-of-the-art generative models. We achieve this on unconditional image synthesis by finding a better architecture through a series of ablations. For conditional image synthesis, we further improve sample quality with classifier guidance: a simple, compute-efficient method for trading off diversity for fidelity using gradients from a classifier. We achieve an FID of 2.97 on ImageNet 128$\\times$128, 4.59 on ImageNet 256$\\times$256, and 7.72 on ImageNet 512$\\times$512, and we match BigGAN-deep even with as few as 25 forward passes per sample, all while maintaining better coverage of the distribution. Finally, we find that classifier guidance combines well with upsampling diffusion models, further improving FID to 3.94 on ImageNet 256$\\times$256 and 3.85 on ImageNet 512$\\times$512. We release our code at https://github.com/openai/guided-diffusion",
  "published": "2021-05-11",
  "updated": "2021-06-01",
  "year": "2021",
  "authors": [
   "Prafulla Dhariwal",
   "Alex Nichol"
  ],
  "author_count": 2,
  "categories": [
   "cs.LG",
   "cs.AI",
   "cs.CV",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 12798,
  "influential_citations": 1180,
  "tldr": "It is shown that diffusion models can achieve image sample quality superior to the current state-of-the-art generative models, and classifier guidance combines well with upsampling diffusion models, further improving FID to 3.94 on ImageNet 256$\\times$256 and 3.85 on imageNet 512$\\ times$512.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Prafulla Dhariwal",
    "id": "6515819",
    "h_index": 20,
    "papers": 43
   },
   {
    "name": "Alex Nichol",
    "id": "38967461",
    "h_index": 15,
    "papers": 33
   }
  ],
  "comment": "Added compute requirements, ImageNet 256$\\times$256 upsampling FID and samples, DDIM guided sampler, fixed typos",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2105.05233v4",
  "pdf_url": "https://arxiv.org/pdf/2105.05233v4",
  "html_url": "https://arxiv.org/html/2105.05233v4",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "2105.03655",
  "slug": "flingbot-the-unreasonable-effectiveness-of-dynamic-manipulation-for-cl",
  "title": "FlingBot: The Unreasonable Effectiveness of Dynamic Manipulation for Cloth Unfolding",
  "abstract": "High-velocity dynamic actions (e.g., fling or throw) play a crucial role in our everyday interaction with deformable objects by improving our efficiency and effectively expanding our physical reach range. Yet, most prior works have tackled cloth manipulation using exclusively single-arm quasi-static actions, which requires a large number of interactions for challenging initial cloth configurations and strictly limits the maximum cloth size by the robot's reach range. In this work, we demonstrate the effectiveness of dynamic flinging actions for cloth unfolding with our proposed self-supervised learning framework, FlingBot. Our approach learns how to unfold a piece of fabric from arbitrary initial configurations using a pick, stretch, and fling primitive for a dual-arm setup from visual observations. The final system achieves over 80% coverage within 3 actions on novel cloths, can unfold cloths larger than the system's reach range, and generalizes to T-shirts despite being trained on only rectangular cloths. We also finetuned FlingBot on a real-world dual-arm robot platform, where it increased the cloth coverage over 4 times more than the quasi-static baseline did. The simplicity of FlingBot combined with its superior performance over quasi-static baselines demonstrates the effectiveness of dynamic actions for deformable object manipulation.",
  "published": "2021-05-08",
  "updated": "2021-10-18",
  "year": "2021",
  "authors": [
   "Huy Ha",
   "Shuran Song"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 214,
  "influential_citations": 30,
  "tldr": "The simplicity of FlingBot combined with its superior performance over quasi-static baselines demonstrates the effectiveness of dynamic actions for deformable object manipulation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Huy Ha",
    "id": "9091886",
    "h_index": 8,
    "papers": 14
   },
   {
    "name": "Shuran Song",
    "id": "3340170",
    "h_index": 59,
    "papers": 90
   }
  ],
  "comment": "11 pages, 6 figures. Code, data, and simulation environment publicly available at https://flingbot.cs.columbia.edu",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2105.03655v3",
  "pdf_url": "https://arxiv.org/pdf/2105.03655v3",
  "html_url": "https://arxiv.org/html/2105.03655v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.83
 },
 {
  "id": "2104.12229",
  "slug": "vector-neurons-a-general-framework-for-so-3-equivariant-networks",
  "title": "Vector Neurons: A General Framework for SO(3)-Equivariant Networks",
  "abstract": "Invariance and equivariance to the rotation group have been widely discussed in the 3D deep learning community for pointclouds. Yet most proposed methods either use complex mathematical tools that may limit their accessibility, or are tied to specific input data types and network architectures. In this paper, we introduce a general framework built on top of what we call Vector Neuron representations for creating SO(3)-equivariant neural networks for pointcloud processing. Extending neurons from 1D scalars to 3D vectors, our vector neurons enable a simple mapping of SO(3) actions to latent spaces thereby providing a framework for building equivariance in common neural operations -- including linear layers, non-linearities, pooling, and normalizations. Due to their simplicity, vector neurons are versatile and, as we demonstrate, can be incorporated into diverse network architecture backbones, allowing them to process geometry inputs in arbitrary poses. Despite its simplicity, our method performs comparably well in accuracy and generalization with other more complex and specialized state-of-the-art methods on classification and segmentation tasks. We also show for the first time a rotation equivariant reconstruction network.",
  "published": "2021-04-25",
  "updated": "2021-04-25",
  "year": "2021",
  "authors": [
   "Congyue Deng",
   "Or Litany",
   "Yueqi Duan",
   "Adrien Poulenard",
   "Andrea Tagliasacchi",
   "Leonidas Guibas"
  ],
  "author_count": 6,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "ICCV",
  "venue_source": "semantic-scholar",
  "citations": 444,
  "influential_citations": 95,
  "tldr": "",
  "doi": "10.1109/ICCV48922.2021.01198",
  "oa_pdf": "https://arxiv.org/pdf/2104.12229",
  "s2_authors": [
   {
    "name": "Congyue Deng",
    "id": "1445309116",
    "h_index": 15,
    "papers": 35
   },
   {
    "name": "O. Litany",
    "id": "2528439",
    "h_index": 38,
    "papers": 105
   },
   {
    "name": "Yueqi Duan",
    "id": "2178958",
    "h_index": 27,
    "papers": 63
   },
   {
    "name": "A. Poulenard",
    "id": "13022326",
    "h_index": 12,
    "papers": 14
   },
   {
    "name": "Andrea Tagliasacchi",
    "id": "1796480",
    "h_index": 46,
    "papers": 107
   },
   {
    "name": "L. Guibas",
    "id": "2231847421",
    "h_index": 111,
    "papers": 487
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2104.12229v1",
  "pdf_url": "https://arxiv.org/pdf/2104.12229v1",
  "html_url": "https://arxiv.org/html/2104.12229v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.15
 },
 {
  "id": "2104.10157",
  "slug": "videogpt-video-generation-using-vq-vae-and-transformers",
  "title": "VideoGPT: Video Generation using VQ-VAE and Transformers",
  "abstract": "We present VideoGPT: a conceptually simple architecture for scaling likelihood based generative modeling to natural videos. VideoGPT uses VQ-VAE that learns downsampled discrete latent representations of a raw video by employing 3D convolutions and axial self-attention. A simple GPT-like architecture is then used to autoregressively model the discrete latents using spatio-temporal position encodings. Despite the simplicity in formulation and ease of training, our architecture is able to generate samples competitive with state-of-the-art GAN models for video generation on the BAIR Robot dataset, and generate high fidelity natural videos from UCF-101 and Tumbler GIF Dataset (TGIF). We hope our proposed architecture serves as a reproducible reference for a minimalistic implementation of transformer based video generation models. Samples and code are available at https://wilson1yan.github.io/videogpt/index.html",
  "published": "2021-04-20",
  "updated": "2021-09-14",
  "year": "2021",
  "authors": [
   "Wilson Yan",
   "Yunzhi Zhang",
   "Pieter Abbeel",
   "Aravind Srinivas"
  ],
  "author_count": 4,
  "categories": [
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 787,
  "influential_citations": 70,
  "tldr": "VideoGPT uses VQ-VAE that learns downsampled discrete latent representations of a raw video by employing 3D convolutions and axial self-attention to autoregressively model the discrete latents using spatio-temporal position encodings.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Wilson Yan",
    "id": "46705049",
    "h_index": 12,
    "papers": 18
   },
   {
    "name": "Yunzhi Zhang",
    "id": "2369217398",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "P. Abbeel",
    "id": "1689992",
    "h_index": 165,
    "papers": 596
   },
   {
    "name": "A. Srinivas",
    "id": "41207614",
    "h_index": 20,
    "papers": 35
   }
  ],
  "comment": "Project website: https://wilson1yan.github.io/videogpt/index.html",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2104.10157v2",
  "pdf_url": "https://arxiv.org/pdf/2104.10157v2",
  "html_url": "https://arxiv.org/html/2104.10157v2",
  "code_url": "https://wilson1yan.github.io/videogpt/index.html",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.9
 },
 {
  "id": "2104.09864",
  "slug": "roformer-enhanced-transformer-with-rotary-position-embedding",
  "title": "RoFormer: Enhanced Transformer with Rotary Position Embedding",
  "abstract": "Position encoding recently has shown effective in the transformer architecture. It enables valuable supervision for dependency modeling between elements at different positions of the sequence. In this paper, we first investigate various methods to integrate positional information into the learning process of transformer-based language models. Then, we propose a novel method named Rotary Position Embedding(RoPE) to effectively leverage the positional information. Specifically, the proposed RoPE encodes the absolute position with a rotation matrix and meanwhile incorporates the explicit relative position dependency in self-attention formulation. Notably, RoPE enables valuable properties, including the flexibility of sequence length, decaying inter-token dependency with increasing relative distances, and the capability of equipping the linear self-attention with relative position encoding. Finally, we evaluate the enhanced transformer with rotary position embedding, also called RoFormer, on various long text classification benchmark datasets. Our experiments show that it consistently overcomes its alternatives. Furthermore, we provide a theoretical analysis to explain some experimental results. RoFormer is already integrated into Huggingface: \\url{https://huggingface.co/docs/transformers/model_doc/roformer}.",
  "published": "2021-04-20",
  "updated": "2023-11-08",
  "year": "2021",
  "authors": [
   "Jianlin Su",
   "Yu Lu",
   "Shengfeng Pan",
   "Ahmed Murtadha",
   "Bo Wen",
   "Yunfeng Liu"
  ],
  "author_count": 6,
  "categories": [
   "cs.CL",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.CL",
  "venue": "Neurocomputing",
  "venue_source": "semantic-scholar",
  "citations": 6215,
  "influential_citations": 498,
  "tldr": "A novel method named Rotary Position Embedding(RoPE) is proposed to effectively leverage the positional information in transformer-based language models and enables valuable properties, including the flexibility of sequence length, decaying inter-token dependency with increasing relative distances, and the capability of equipping the linear self-attention with relative position encoding.",
  "doi": "10.1016/j.neucom.2023.127063",
  "oa_pdf": "https://arxiv.org/pdf/2104.09864",
  "s2_authors": [
   {
    "name": "Jianlin Su",
    "id": "51111230",
    "h_index": 18,
    "papers": 31
   },
   {
    "name": "Yu Lu",
    "id": "2140045110",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Shengfeng Pan",
    "id": "1382633722",
    "h_index": 10,
    "papers": 16
   },
   {
    "name": "Bo Wen",
    "id": "2079396269",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Yunfeng Liu",
    "id": "1807486863",
    "h_index": 10,
    "papers": 13
   }
  ],
  "comment": "fixed some typos",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2104.09864v5",
  "pdf_url": "https://arxiv.org/pdf/2104.09864v5",
  "html_url": "https://arxiv.org/html/2104.09864v5",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "2104.08212",
  "slug": "mt-opt-continuous-multi-task-robotic-reinforcement-learning-at-scale",
  "title": "MT-Opt: Continuous Multi-Task Robotic Reinforcement Learning at Scale",
  "abstract": "General-purpose robotic systems must master a large repertoire of diverse skills to be useful in a range of daily tasks. While reinforcement learning provides a powerful framework for acquiring individual behaviors, the time needed to acquire each skill makes the prospect of a generalist robot trained with RL daunting. In this paper, we study how a large-scale collective robotic learning system can acquire a repertoire of behaviors simultaneously, sharing exploration, experience, and representations across tasks. In this framework new tasks can be continuously instantiated from previously learned tasks improving overall performance and capabilities of the system. To instantiate this system, we develop a scalable and intuitive framework for specifying new tasks through user-provided examples of desired outcomes, devise a multi-robot collective learning system for data collection that simultaneously collects experience for multiple tasks, and develop a scalable and generalizable multi-task deep reinforcement learning method, which we call MT-Opt. We demonstrate how MT-Opt can learn a wide range of skills, including semantic picking (i.e., picking an object from a particular category), placing into various fixtures (e.g., placing a food item onto a plate), covering, aligning, and rearranging. We train and evaluate our system on a set of 12 real-world tasks with data collected from 7 robots, and demonstrate the performance of our system both in terms of its ability to generalize to structurally similar new tasks, and acquire distinct new tasks more quickly by leveraging past experience. We recommend viewing the videos at https://karolhausman.github.io/mt-opt/",
  "published": "2021-04-16",
  "updated": "2021-04-27",
  "year": "2021",
  "authors": [
   "Dmitry Kalashnikov",
   "Jacob Varley",
   "Yevgen Chebotar",
   "Benjamin Swanson",
   "Rico Jonschkowski",
   "Chelsea Finn",
   "Sergey Levine",
   "Karol Hausman"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 335,
  "influential_citations": 10,
  "tldr": "A large-scale collective robotic learning system that can acquire a repertoire of behaviors simultaneously, sharing exploration, experience, and representations across tasks, and develop a scalable and generalizable multi-task deep reinforcement learning method, which is called MT-Opt.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Dmitry Kalashnikov",
    "id": "48313860",
    "h_index": 21,
    "papers": 34
   },
   {
    "name": "Jacob Varley",
    "id": "31568090",
    "h_index": 16,
    "papers": 28
   },
   {
    "name": "Yevgen Chebotar",
    "id": "2527420",
    "h_index": 33,
    "papers": 57
   },
   {
    "name": "Benjamin Swanson",
    "id": "2065016304",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Rico Jonschkowski",
    "id": "2699042",
    "h_index": 20,
    "papers": 40
   },
   {
    "name": "Chelsea Finn",
    "id": "46881670",
    "h_index": 94,
    "papers": 240
   },
   {
    "name": "S. Levine",
    "id": "1736651",
    "h_index": 167,
    "papers": 551
   },
   {
    "name": "Karol Hausman",
    "id": "1944801",
    "h_index": 47,
    "papers": 122
   }
  ],
  "comment": "",
  "topics": [
   "rl-control",
   "navigation",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2104.08212v2",
  "pdf_url": "https://arxiv.org/pdf/2104.08212v2",
  "html_url": "https://arxiv.org/html/2104.08212v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.53
 },
 {
  "id": "2104.06159",
  "slug": "muesli-combining-improvements-in-policy-optimization",
  "title": "Muesli: Combining Improvements in Policy Optimization",
  "abstract": "We propose a novel policy update that combines regularized policy optimization with model learning as an auxiliary loss. The update (henceforth Muesli) matches MuZero's state-of-the-art performance on Atari. Notably, Muesli does so without using deep search: it acts directly with a policy network and has computation speed comparable to model-free baselines. The Atari results are complemented by extensive ablations, and by additional results on continuous control and 9x9 Go.",
  "published": "2021-04-13",
  "updated": "2022-03-31",
  "year": "2021",
  "authors": [
   "Matteo Hessel",
   "Ivo Danihelka",
   "Fabio Viola",
   "Arthur Guez",
   "Simon Schmitt",
   "Laurent Sifre",
   "Theophane Weber",
   "David Silver",
   "Hado van Hasselt"
  ],
  "author_count": 9,
  "categories": [
   "cs.LG",
   "cs.AI"
  ],
  "primary_category": "cs.LG",
  "venue": "ICML",
  "venue_source": "semantic-scholar",
  "citations": 69,
  "influential_citations": 10,
  "tldr": "A novel policy update that combines regularized policy optimization with model learning as an auxiliary loss and does so without using deep search: it acts directly with a policy network and has computation speed comparable to model-free baselines.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Matteo Hessel",
    "id": "39357484",
    "h_index": 25,
    "papers": 39
   },
   {
    "name": "Ivo Danihelka",
    "id": "1841008",
    "h_index": 21,
    "papers": 30
   },
   {
    "name": "Fabio Viola",
    "id": "47963165",
    "h_index": 19,
    "papers": 32
   },
   {
    "name": "A. Guez",
    "id": "35099444",
    "h_index": 27,
    "papers": 50
   },
   {
    "name": "Simon Schmitt",
    "id": "152380508",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "L. Sifre",
    "id": "2175946",
    "h_index": 28,
    "papers": 42
   },
   {
    "name": "T. Weber",
    "id": "143947744",
    "h_index": 33,
    "papers": 67
   },
   {
    "name": "David Silver",
    "id": "145824029",
    "h_index": 80,
    "papers": 120
   },
   {
    "name": "H. V. Hasselt",
    "id": "7634925",
    "h_index": 40,
    "papers": 67
   }
  ],
  "comment": "",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2104.06159v2",
  "pdf_url": "https://arxiv.org/pdf/2104.06159v2",
  "html_url": "https://arxiv.org/html/2104.06159v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.35
 },
 {
  "id": "2104.05859",
  "slug": "rapid-exploration-for-open-world-navigation-with-latent-goal-models",
  "title": "Rapid Exploration for Open-World Navigation with Latent Goal Models",
  "abstract": "We describe a robotic learning system for autonomous exploration and navigation in diverse, open-world environments. At the core of our method is a learned latent variable model of distances and actions, along with a non-parametric topological memory of images. We use an information bottleneck to regularize the learned policy, giving us (i) a compact visual representation of goals, (ii) improved generalization capabilities, and (iii) a mechanism for sampling feasible goals for exploration. Trained on a large offline dataset of prior experience, the model acquires a representation of visual goals that is robust to task-irrelevant distractors. We demonstrate our method on a mobile ground robot in open-world exploration scenarios. Given an image of a goal that is up to 80 meters away, our method leverages its representation to explore and discover the goal in under 20 minutes, even amidst previously-unseen obstacles and weather conditions. Please check out the project website for videos of our experiments and information about the real-world dataset used at https://sites.google.com/view/recon-robot.",
  "published": "2021-04-12",
  "updated": "2023-10-11",
  "year": "2021",
  "authors": [
   "Dhruv Shah",
   "Benjamin Eysenbach",
   "Gregory Kahn",
   "Nicholas Rhinehart",
   "Sergey Levine"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 160,
  "influential_citations": 10,
  "tldr": "A robotic learning system for autonomous exploration and navigation in diverse, open-world environments using a learned latent variable model of distances and actions, along with a non-parametric topological memory of images to create a compact visual representation of goals.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Dhruv Shah",
    "id": "2322628540",
    "h_index": 29,
    "papers": 63
   },
   {
    "name": "Benjamin Eysenbach",
    "id": "8140754",
    "h_index": 34,
    "papers": 75
   },
   {
    "name": "G. Kahn",
    "id": "46292812",
    "h_index": 20,
    "papers": 29
   },
   {
    "name": "Nicholas Rhinehart",
    "id": "1974383",
    "h_index": 20,
    "papers": 37
   },
   {
    "name": "S. Levine",
    "id": "1736651",
    "h_index": 167,
    "papers": 551
   }
  ],
  "comment": "Presented at 5th Annual Conference on Robot Learning (CoRL 2021), London, UK as an Oral Talk. Project page and dataset release at https://sites.google.com/view/recon-robot",
  "topics": [
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2104.05859v5",
  "pdf_url": "https://arxiv.org/pdf/2104.05859v5",
  "html_url": "https://arxiv.org/html/2104.05859v5",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.71
 },
 {
  "id": "2104.04644",
  "slug": "fast-and-efficient-locomotion-via-learned-gait-transitions",
  "title": "Fast and Efficient Locomotion via Learned Gait Transitions",
  "abstract": "We focus on the problem of developing energy efficient controllers for quadrupedal robots. Animals can actively switch gaits at different speeds to lower their energy consumption. In this paper, we devise a hierarchical learning framework, in which distinctive locomotion gaits and natural gait transitions emerge automatically with a simple reward of energy minimization. We use evolutionary strategies (ES) to train a high-level gait policy that specifies gait patterns of each foot, while the low-level convex MPC controller optimizes the motor commands so that the robot can walk at a desired velocity using that gait pattern. We test our learning framework on a quadruped robot and demonstrate automatic gait transitions, from walking to trotting and to fly-trotting, as the robot increases its speed. We show that the learned hierarchical controller consumes much less energy across a wide range of locomotion speed than baseline controllers.",
  "published": "2021-04-09",
  "updated": "2021-11-22",
  "year": "2021",
  "authors": [
   "Yuxiang Yang",
   "Tingnan Zhang",
   "Erwin Coumans",
   "Jie Tan",
   "Byron Boots"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 132,
  "influential_citations": 6,
  "tldr": "A hierarchical learning framework is devised, in which distinctive locomotion gaits and natural gait transitions emerge automatically with a simple reward of energy minimization, which shows that the learned hierarchical controller consumes much less energy across a wide range of locomotion speed than baseline controllers.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yuxiang Yang",
    "id": "2108795581",
    "h_index": 18,
    "papers": 50
   },
   {
    "name": "Tingnan Zhang",
    "id": "28292148",
    "h_index": 26,
    "papers": 53
   },
   {
    "name": "Erwin Coumans",
    "id": "1716551",
    "h_index": 15,
    "papers": 24
   },
   {
    "name": "Jie Tan",
    "id": "1739176520",
    "h_index": 37,
    "papers": 68
   },
   {
    "name": "Byron Boots",
    "id": "3288815",
    "h_index": 49,
    "papers": 182
   }
  ],
  "comment": "Published in CoRL 2021. Website: Website: https://sites.google.com/view/fast-and-efficient Code: https://github.com/yxyang/fast_and_efficient",
  "topics": [
   "humanoids"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2104.04644v3",
  "pdf_url": "https://arxiv.org/pdf/2104.04644v3",
  "html_url": "https://arxiv.org/html/2104.04644v3",
  "code_url": "https://github.com/yxyang/fast_and_efficient",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.62
 },
 {
  "id": "2104.04631",
  "slug": "dexycb-a-benchmark-for-capturing-hand-grasping-of-objects",
  "title": "DexYCB: A Benchmark for Capturing Hand Grasping of Objects",
  "abstract": "We introduce DexYCB, a new dataset for capturing hand grasping of objects. We first compare DexYCB with a related one through cross-dataset evaluation. We then present a thorough benchmark of state-of-the-art approaches on three relevant tasks: 2D object and keypoint detection, 6D object pose estimation, and 3D hand pose estimation. Finally, we evaluate a new robotics-relevant task: generating safe robot grasps in human-to-robot object handover. Dataset and code are available at https://dex-ycb.github.io.",
  "published": "2021-04-09",
  "updated": "2021-04-09",
  "year": "2021",
  "authors": [
   "Yu-Wei Chao",
   "Wei Yang",
   "Yu Xiang",
   "Pavlo Molchanov",
   "Ankur Handa",
   "Jonathan Tremblay",
   "Yashraj S. Narang",
   "Karl Van Wyk",
   "Umar Iqbal",
   "Stan Birchfield",
   "Jan Kautz",
   "Dieter Fox"
  ],
  "author_count": 12,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 480,
  "influential_citations": 90,
  "tldr": "This work introduces DexYCB, a new dataset for capturing hand grasping of objects, and presents a thorough benchmark of state-of-the-art approaches on three relevant tasks: 2D object and keypoint detection, 6D object pose estimation, and 3D hand pose estimation.",
  "doi": "10.1109/CVPR46437.2021.00893",
  "oa_pdf": "https://arxiv.org/pdf/2104.04631",
  "s2_authors": [
   {
    "name": "Yu-Wei Chao",
    "id": "2820136",
    "h_index": 25,
    "papers": 38
   },
   {
    "name": "Wei Yang",
    "id": "2150080732",
    "h_index": 24,
    "papers": 36
   },
   {
    "name": "Yu Xiang",
    "id": "144863550",
    "h_index": 31,
    "papers": 53
   },
   {
    "name": "Pavlo Molchanov",
    "id": "2824500",
    "h_index": 50,
    "papers": 147
   },
   {
    "name": "Ankur Handa",
    "id": "34653454",
    "h_index": 33,
    "papers": 55
   },
   {
    "name": "Jonathan Tremblay",
    "id": "31943350",
    "h_index": 27,
    "papers": 56
   },
   {
    "name": "Yashraj S. Narang",
    "id": "5046361",
    "h_index": 20,
    "papers": 36
   },
   {
    "name": "Karl Van Wyk",
    "id": "2423933",
    "h_index": 20,
    "papers": 54
   },
   {
    "name": "Umar Iqbal",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Stan Birchfield",
    "id": "2238841",
    "h_index": 52,
    "papers": 152
   },
   {
    "name": "Jan Kautz",
    "id": "2376331447",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "D. Fox",
    "id": "145197953",
    "h_index": 133,
    "papers": 428
   }
  ],
  "comment": "Accepted to CVPR 2021",
  "topics": [
   "dexterous-manipulation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2104.04631v1",
  "pdf_url": "https://arxiv.org/pdf/2104.04631v1",
  "html_url": "https://arxiv.org/html/2104.04631v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.18
 },
 {
  "id": "2104.03304",
  "slug": "hand-object-contact-consistency-reasoning-for-human-grasps-generation",
  "title": "Hand-Object Contact Consistency Reasoning for Human Grasps Generation",
  "abstract": "While predicting robot grasps with parallel jaw grippers have been well studied and widely applied in robot manipulation tasks, the study on natural human grasp generation with a multi-finger hand remains a very challenging problem. In this paper, we propose to generate human grasps given a 3D object in the world. Our key observation is that it is crucial to model the consistency between the hand contact points and object contact regions. That is, we encourage the prior hand contact points to be close to the object surface and the object common contact regions to be touched by the hand at the same time. Based on the hand-object contact consistency, we design novel objectives in training the human grasp generation model and also a new self-supervised task which allows the grasp generation network to be adjusted even during test time. Our experiments show significant improvement in human grasp generation over state-of-the-art approaches by a large margin. More interestingly, by optimizing the model during test time with the self-supervised task, it helps achieve larger gain on unseen and out-of-domain objects. Project page: https://hwjiang1510.github.io/GraspTTA/",
  "published": "2021-04-07",
  "updated": "2021-04-07",
  "year": "2021",
  "authors": [
   "Hanwen Jiang",
   "Shaowei Liu",
   "Jiashun Wang",
   "Xiaolong Wang"
  ],
  "author_count": 4,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "ICCV",
  "venue_source": "semantic-scholar",
  "citations": 260,
  "influential_citations": 44,
  "tldr": "This paper proposes to generate human grasps given a 3D object in the world and designs novel objectives in training the human grasp generation model and also a new self-supervised task which allows the grasp generation network to be adjusted even during test time.",
  "doi": "10.1109/ICCV48922.2021.01092",
  "oa_pdf": "https://arxiv.org/pdf/2104.03304",
  "s2_authors": [
   {
    "name": "Hanwen Jiang",
    "id": "2152630535",
    "h_index": 14,
    "papers": 27
   },
   {
    "name": "Shaowei Liu",
    "id": "2119051505",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Jiashun Wang",
    "id": "2110144663",
    "h_index": 10,
    "papers": 13
   },
   {
    "name": "Xiaolong Wang",
    "id": "122024152",
    "h_index": 42,
    "papers": 63
   }
  ],
  "comment": "Project page: https://hwjiang1510.github.io/GraspTTA/",
  "topics": [
   "dexterous-manipulation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2104.03304v1",
  "pdf_url": "https://arxiv.org/pdf/2104.03304v1",
  "html_url": "https://arxiv.org/html/2104.03304v1",
  "code_url": "https://hwjiang1510.github.io/GraspTTA/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.92
 },
 {
  "id": "2103.16817",
  "slug": "learning-generalizable-robotic-reward-functions-from-in-the-wild-human",
  "title": "Learning Generalizable Robotic Reward Functions from \"In-The-Wild\" Human Videos",
  "abstract": "We are motivated by the goal of generalist robots that can complete a wide range of tasks across many environments. Critical to this is the robot's ability to acquire some metric of task success or reward, which is necessary for reinforcement learning, planning, or knowing when to ask for help. For a general-purpose robot operating in the real world, this reward function must also be able to generalize broadly across environments, tasks, and objects, while depending only on on-board sensor observations (e.g. RGB images). While deep learning on large and diverse datasets has shown promise as a path towards such generalization in computer vision and natural language, collecting high quality datasets of robotic interaction at scale remains an open challenge. In contrast, \"in-the-wild\" videos of humans (e.g. YouTube) contain an extensive collection of people doing interesting tasks across a diverse range of settings. In this work, we propose a simple approach, Domain-agnostic Video Discriminator (DVD), that learns multitask reward functions by training a discriminator to classify whether two videos are performing the same task, and can generalize by virtue of learning from a small amount of robot data with a broad dataset of human videos. We find that by leveraging diverse human datasets, this reward function (a) can generalize zero shot to unseen environments, (b) generalize zero shot to unseen tasks, and (c) can be combined with visual model predictive control to solve robotic manipulation tasks on a real WidowX200 robot in an unseen environment from a single human demo.",
  "published": "2021-03-31",
  "updated": "2021-03-31",
  "year": "2021",
  "authors": [
   "Annie S. Chen",
   "Suraj Nair",
   "Chelsea Finn"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "RSS",
  "venue_source": "semantic-scholar",
  "citations": 191,
  "influential_citations": 10,
  "tldr": "This work proposes a simple approach, Domain-agnostic Video Discriminator (DVD), that learns multitask reward functions by training a discriminator to classify whether two videos are performing the same task, and can generalize by virtue of learning from a small amount of robot data with a broad dataset of human videos.",
  "doi": "10.15607/RSS.2021.XVII.012",
  "oa_pdf": "https://doi.org/10.15607/rss.2021.xvii.012",
  "s2_authors": [
   {
    "name": "Annie S. Chen",
    "id": "2111073657",
    "h_index": 12,
    "papers": 23
   },
   {
    "name": "Suraj Nair",
    "id": "4734949",
    "h_index": 20,
    "papers": 36
   },
   {
    "name": "Chelsea Finn",
    "id": "46881670",
    "h_index": 94,
    "papers": 240
   }
  ],
  "comment": "https://sites.google.com/view/dvd-human-videos",
  "topics": [
   "egocentric-data",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2103.16817v1",
  "pdf_url": "https://arxiv.org/pdf/2103.16817v1",
  "html_url": "https://arxiv.org/html/2103.16817v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.78
 },
 {
  "id": "2103.15691",
  "slug": "vivit-a-video-vision-transformer",
  "title": "ViViT: A Video Vision Transformer",
  "abstract": "We present pure-transformer based models for video classification, drawing upon the recent success of such models in image classification. Our model extracts spatio-temporal tokens from the input video, which are then encoded by a series of transformer layers. In order to handle the long sequences of tokens encountered in video, we propose several, efficient variants of our model which factorise the spatial- and temporal-dimensions of the input. Although transformer-based models are known to only be effective when large training datasets are available, we show how we can effectively regularise the model during training and leverage pretrained image models to be able to train on comparatively small datasets. We conduct thorough ablation studies, and achieve state-of-the-art results on multiple video classification benchmarks including Kinetics 400 and 600, Epic Kitchens, Something-Something v2 and Moments in Time, outperforming prior methods based on deep 3D convolutional networks. To facilitate further research, we release code at https://github.com/google-research/scenic/tree/main/scenic/projects/vivit",
  "published": "2021-03-29",
  "updated": "2021-11-01",
  "year": "2021",
  "authors": [
   "Anurag Arnab",
   "Mostafa Dehghani",
   "Georg Heigold",
   "Chen Sun",
   "Mario Lu\u010di\u0107",
   "Cordelia Schmid"
  ],
  "author_count": 6,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "ICCV",
  "venue_source": "semantic-scholar",
  "citations": 3197,
  "influential_citations": 296,
  "tldr": "This work shows how to effectively regularise the model during training and leverage pretrained image models to be able to train on comparatively small datasets, and achieves state-of-the-art results on multiple video classification benchmarks.",
  "doi": "10.1109/ICCV48922.2021.00676",
  "oa_pdf": "https://arxiv.org/pdf/2103.15691",
  "s2_authors": [
   {
    "name": "Anurag Arnab",
    "id": "31638576",
    "h_index": 27,
    "papers": 45
   },
   {
    "name": "Mostafa Dehghani",
    "id": "3226635",
    "h_index": 42,
    "papers": 94
   },
   {
    "name": "G. Heigold",
    "id": "2280399",
    "h_index": 31,
    "papers": 84
   },
   {
    "name": "Chen Sun",
    "id": "1491624845",
    "h_index": 45,
    "papers": 87
   },
   {
    "name": "M. Lu\u010di\u0107",
    "id": "34302129",
    "h_index": 38,
    "papers": 67
   },
   {
    "name": "Cordelia Schmid",
    "id": "2462253",
    "h_index": 153,
    "papers": 466
   }
  ],
  "comment": "ICCV 2021. Code at https://github.com/google-research/scenic/tree/main/scenic/projects/vivit",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2103.15691v2",
  "pdf_url": "https://arxiv.org/pdf/2103.15691v2",
  "html_url": "https://arxiv.org/html/2103.15691v2",
  "code_url": "https://github.com/google-research/scenic",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.5
 },
 {
  "id": "2103.14295",
  "slug": "reinforcement-learning-for-robust-parameterized-locomotion-control-of",
  "title": "Reinforcement Learning for Robust Parameterized Locomotion Control of Bipedal Robots",
  "abstract": "Developing robust walking controllers for bipedal robots is a challenging endeavor. Traditional model-based locomotion controllers require simplifying assumptions and careful modelling; any small errors can result in unstable control. To address these challenges for bipedal locomotion, we present a model-free reinforcement learning framework for training robust locomotion policies in simulation, which can then be transferred to a real bipedal Cassie robot. To facilitate sim-to-real transfer, domain randomization is used to encourage the policies to learn behaviors that are robust across variations in system dynamics. The learned policies enable Cassie to perform a set of diverse and dynamic behaviors, while also being more robust than traditional controllers and prior learning-based methods that use residual control. We demonstrate this on versatile walking behaviors such as tracking a target walking velocity, walking height, and turning yaw.",
  "published": "2021-03-26",
  "updated": "2021-03-26",
  "year": "2021",
  "authors": [
   "Zhongyu Li",
   "Xuxin Cheng",
   "Xue Bin Peng",
   "Pieter Abbeel",
   "Sergey Levine",
   "Glen Berseth",
   "Koushil Sreenath"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA",
  "venue_source": "semantic-scholar",
  "citations": 301,
  "influential_citations": 7,
  "tldr": "A model-free reinforcement learning framework for training robust locomotion policies in simulation, which can be transferred to a real bipedal Cassie robot, and domain randomization is used to encourage the policies to learn behaviors that are robust across variations in system dynamics.",
  "doi": "10.1109/ICRA48506.2021.9560769",
  "oa_pdf": "https://arxiv.org/pdf/2103.14295",
  "s2_authors": [
   {
    "name": "Zhongyu Li",
    "id": "1491078398",
    "h_index": 24,
    "papers": 50
   },
   {
    "name": "Xuxin Cheng",
    "id": "90080090",
    "h_index": 13,
    "papers": 13
   },
   {
    "name": "Xue Bin Peng",
    "id": "2375236722",
    "h_index": 34,
    "papers": 41
   },
   {
    "name": "P. Abbeel",
    "id": "1689992",
    "h_index": 165,
    "papers": 596
   },
   {
    "name": "S. Levine",
    "id": "1736651",
    "h_index": 167,
    "papers": 551
   },
   {
    "name": "Glen Berseth",
    "id": "2312919053",
    "h_index": 15,
    "papers": 28
   },
   {
    "name": "K. Sreenath",
    "id": "144116765",
    "h_index": 55,
    "papers": 231
   }
  ],
  "comment": "To appear on 2021 International Conference on Robotics and Automation (ICRA 2021)",
  "topics": [
   "humanoids",
   "sim2real",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2103.14295v1",
  "pdf_url": "https://arxiv.org/pdf/2103.14295v1",
  "html_url": "https://arxiv.org/html/2103.14295v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.98
 },
 {
  "id": "2103.14127",
  "slug": "contact-graspnet-efficient-6-dof-grasp-generation-in-cluttered-scenes",
  "title": "Contact-GraspNet: Efficient 6-DoF Grasp Generation in Cluttered Scenes",
  "abstract": "Grasping unseen objects in unconstrained, cluttered environments is an essential skill for autonomous robotic manipulation. Despite recent progress in full 6-DoF grasp learning, existing approaches often consist of complex sequential pipelines that possess several potential failure points and run-times unsuitable for closed-loop grasping. Therefore, we propose an end-to-end network that efficiently generates a distribution of 6-DoF parallel-jaw grasps directly from a depth recording of a scene. Our novel grasp representation treats 3D points of the recorded point cloud as potential grasp contacts. By rooting the full 6-DoF grasp pose and width in the observed point cloud, we can reduce the dimensionality of our grasp representation to 4-DoF which greatly facilitates the learning process. Our class-agnostic approach is trained on 17 million simulated grasps and generalizes well to real world sensor data. In a robotic grasping study of unseen objects in structured clutter we achieve over 90% success rate, cutting the failure rate in half compared to a recent state-of-the-art method.",
  "published": "2021-03-25",
  "updated": "2021-03-25",
  "year": "2021",
  "authors": [
   "Martin Sundermeyer",
   "Arsalan Mousavian",
   "Rudolph Triebel",
   "Dieter Fox"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA",
  "venue_source": "semantic-scholar",
  "citations": 547,
  "influential_citations": 65,
  "tldr": "This work proposes an end-to-end network that efficiently generates a distribution of 6-DoF parallel-jaw grasps directly from a depth recording of a scene and treats 3D points of the recorded point cloud as potential grasp contacts, and reduces the dimensionality of the grasp representation to 4- doF which greatly facilitates the learning process.",
  "doi": "10.1109/ICRA48506.2021.9561877",
  "oa_pdf": "https://elib.dlr.de/145798/1/Contact-GraspNet.pdf",
  "s2_authors": [
   {
    "name": "M. Sundermeyer",
    "id": "2748591",
    "h_index": 25,
    "papers": 51
   },
   {
    "name": "A. Mousavian",
    "id": "3040583",
    "h_index": 36,
    "papers": 57
   },
   {
    "name": "Rudolph Triebel",
    "id": "2750689",
    "h_index": 39,
    "papers": 161
   },
   {
    "name": "D. Fox",
    "id": "145197953",
    "h_index": 133,
    "papers": 428
   }
  ],
  "comment": "ICRA 2021. Video of the real world experiments and code are available at https://research.nvidia.com/publication/2021-03_Contact-GraspNet%3A--Efficient",
  "topics": [
   "dexterous-manipulation",
   "spatial-3d"
  ],
  "orgs": [
   "NVIDIA"
  ],
  "abs_url": "https://arxiv.org/abs/2103.14127v1",
  "pdf_url": "https://arxiv.org/pdf/2103.14127v1",
  "html_url": "https://arxiv.org/html/2103.14127v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.74
 },
 {
  "id": "2103.09016",
  "slug": "manipulator-independent-representations-for-visual-imitation",
  "title": "Manipulator-Independent Representations for Visual Imitation",
  "abstract": "Imitation learning is an effective tool for robotic learning tasks where specifying a reinforcement learning (RL) reward is not feasible or where the exploration problem is particularly difficult. Imitation, typically behavior cloning or inverse RL, derive a policy from a collection of first-person action-state trajectories. This is contrary to how humans and other animals imitate: we observe a behavior, even from other species, understand its perceived effect on the state of the environment, and figure out what actions our body can perform to reach a similar outcome. In this work, we explore the possibility of third-person visual imitation of manipulation trajectories, only from vision and without access to actions, demonstrated by embodiments different to the ones of our imitating agent. Specifically, we investigate what would be an appropriate representation method with which an RL agent can visually track trajectories of complex manipulation behavior -- non-planar with multiple-object interactions -- demonstrated by experts with different embodiments. We present a way to train manipulator-independent representations (MIR) that primarily focus on the change in the environment and have all the characteristics that make them suitable for cross-embodiment visual imitation with RL: cross-domain alignment, temporal smoothness, and being actionable. We show that with our proposed method our agents are able to imitate, with complex robot control, trajectories from a variety of embodiments and with significant visual and dynamics differences, e.g. simulation-to-reality gap.",
  "published": "2021-03-16",
  "updated": "2021-03-18",
  "year": "2021",
  "authors": [
   "Yuxiang Zhou",
   "Yusuf Aytar",
   "Konstantinos Bousmalis"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "RSS",
  "venue_source": "semantic-scholar",
  "citations": 25,
  "influential_citations": 0,
  "tldr": "A way to train manipulator-independent representations (MIR) that primarily focus on the change in the environment and have all the characteristics that make them suitable for cross-embodiment visual imitation with RL: cross-domain alignment, temporal smoothness, and being actionable is presented.",
  "doi": "10.15607/RSS.2021.XVII.002",
  "oa_pdf": "https://doi.org/10.15607/rss.2021.xvii.002",
  "s2_authors": [
   {
    "name": "Yuxiang Zhou",
    "id": "2145925767",
    "h_index": 11,
    "papers": 14
   },
   {
    "name": "Y. Aytar",
    "id": "3152281",
    "h_index": 30,
    "papers": 57
   },
   {
    "name": "Konstantinos Bousmalis",
    "id": "2732737",
    "h_index": 21,
    "papers": 37
   }
  ],
  "comment": "",
  "topics": [
   "egocentric-data",
   "sim2real",
   "imitation-diffusion",
   "rl-control",
   "navigation",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2103.09016v2",
  "pdf_url": "https://arxiv.org/pdf/2103.09016v2",
  "html_url": "https://arxiv.org/html/2103.09016v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.91
 },
 {
  "id": "2103.05825",
  "slug": "ella-exploration-through-learned-language-abstraction",
  "title": "ELLA: Exploration through Learned Language Abstraction",
  "abstract": "Building agents capable of understanding language instructions is critical to effective and robust human-AI collaboration. Recent work focuses on training these agents via reinforcement learning in environments with synthetic language; however, instructions often define long-horizon, sparse-reward tasks, and learning policies requires many episodes of experience. We introduce ELLA: Exploration through Learned Language Abstraction, a reward shaping approach geared towards boosting sample efficiency in sparse reward environments by correlating high-level instructions with simpler low-level constituents. ELLA has two key elements: 1) A termination classifier that identifies when agents complete low-level instructions, and 2) A relevance classifier that correlates low-level instructions with success on high-level tasks. We learn the termination classifier offline from pairs of instructions and terminal states. Notably, in departure from prior work in language and abstraction, we learn the relevance classifier online, without relying on an explicit decomposition of high-level instructions to low-level instructions. On a suite of complex BabyAI environments with varying instruction complexities and reward sparsity, ELLA shows gains in sample efficiency relative to language-based shaping and traditional RL methods.",
  "published": "2021-03-10",
  "updated": "2021-10-30",
  "year": "2021",
  "authors": [
   "Suvir Mirchandani",
   "Siddharth Karamcheti",
   "Dorsa Sadigh"
  ],
  "author_count": 3,
  "categories": [
   "cs.CL",
   "cs.AI",
   "cs.LG",
   "cs.RO"
  ],
  "primary_category": "cs.CL",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 64,
  "influential_citations": 6,
  "tldr": "ELLA: Exploration through Learned Language Abstraction, a reward shaping approach geared towards boosting sample efficiency in sparse reward environments by correlating high-level instructions with simpler low-level constituents, shows gains in sample efficiency relative to language-based shaping and traditional RL methods.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Suvir Mirchandani",
    "id": "2247302",
    "h_index": 13,
    "papers": 21
   },
   {
    "name": "Siddharth Karamcheti",
    "id": "10737060",
    "h_index": 21,
    "papers": 38
   },
   {
    "name": "Dorsa Sadigh",
    "id": "1779671",
    "h_index": 69,
    "papers": 226
   }
  ],
  "comment": "19 pages, 9 figures. Published in Conference on Neural Information Processing Systems (NeurIPS) 2021. Appendix includes supplementary experiments",
  "topics": [
   "rl-control",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2103.05825v2",
  "pdf_url": "https://arxiv.org/pdf/2103.05825v2",
  "html_url": "https://arxiv.org/html/2103.05825v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.31
 },
 {
  "id": "2103.02201",
  "slug": "semantic-constraints-to-represent-common-sense-required-in-household-a",
  "title": "Semantic constraints to represent common sense required in household actions for multi-modal Learning-from-observation robot",
  "abstract": "The paradigm of learning-from-observation (LfO) enables a robot to learn how to perform actions by observing human-demonstrated actions. Previous research in LfO have mainly focused on the industrial domain which only consist of the observable physical constraints between a manipulating tool and the robot's working environment. In order to extend this paradigm to the household domain which consists non-observable constraints derived from a human's common sense; we introduce the idea of semantic constraints. The semantic constraints are represented similar to the physical constraints by defining a contact with an imaginary semantic environment. We thoroughly investigate the necessary and sufficient set of contact state and state transitions to understand the different types of physical and semantic constraints. We then apply our constraint representation to analyze various actions in top hit household YouTube videos and real home cooking recordings. We further categorize the frequently appearing constraint patterns into physical, semantic, and multistage task groups and verify that these groups are not only necessary but a sufficient set for covering standard household actions. Finally, we conduct a preliminary experiment using textual input to explore the possibilities of combining verbal and visual input for recognizing the task groups. Our results provide promising directions for incorporating common sense in the literature of robot teaching.",
  "published": "2021-03-03",
  "updated": "2021-03-03",
  "year": "2021",
  "authors": [
   "Katsushi Ikeuchi",
   "Naoki Wake",
   "Riku Arakawa",
   "Kazuhiro Sasabuchi",
   "Jun Takamatsu"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "Int. J. Robotics Res.",
  "venue_source": "semantic-scholar",
  "citations": 18,
  "influential_citations": 0,
  "tldr": "This work introduces the idea of semantic constraints, which are represented similarly to the physical constraints by defining an imaginary contact with an imaginary environment, and provides promising directions for incorporating common sense into the robot teaching literature.",
  "doi": "10.1177/02783649231212929",
  "oa_pdf": "https://journals.sagepub.com/doi/pdf/10.1177/02783649231212929",
  "s2_authors": [
   {
    "name": "K. Ikeuchi",
    "id": "1739452",
    "h_index": 79,
    "papers": 885
   },
   {
    "name": "Naoki Wake",
    "id": "1789820",
    "h_index": 14,
    "papers": 58
   },
   {
    "name": "Riku Arakawa",
    "id": "83976651",
    "h_index": 18,
    "papers": 45
   },
   {
    "name": "Kazuhiro Sasabuchi",
    "id": "1882605",
    "h_index": 9,
    "papers": 40
   },
   {
    "name": "J. Takamatsu",
    "id": "21825576",
    "h_index": 23,
    "papers": 295
   }
  ],
  "comment": "18 pages, 31 figures",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2103.02201v1",
  "pdf_url": "https://arxiv.org/pdf/2103.02201v1",
  "html_url": "https://arxiv.org/html/2103.02201v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.78
 },
 {
  "id": "2103.00020",
  "slug": "learning-transferable-visual-models-from-natural-language-supervision",
  "title": "Learning Transferable Visual Models From Natural Language Supervision",
  "abstract": "State-of-the-art computer vision systems are trained to predict a fixed set of predetermined object categories. This restricted form of supervision limits their generality and usability since additional labeled data is needed to specify any other visual concept. Learning directly from raw text about images is a promising alternative which leverages a much broader source of supervision. We demonstrate that the simple pre-training task of predicting which caption goes with which image is an efficient and scalable way to learn SOTA image representations from scratch on a dataset of 400 million (image, text) pairs collected from the internet. After pre-training, natural language is used to reference learned visual concepts (or describe new ones) enabling zero-shot transfer of the model to downstream tasks. We study the performance of this approach by benchmarking on over 30 different existing computer vision datasets, spanning tasks such as OCR, action recognition in videos, geo-localization, and many types of fine-grained object classification. The model transfers non-trivially to most tasks and is often competitive with a fully supervised baseline without the need for any dataset specific training. For instance, we match the accuracy of the original ResNet-50 on ImageNet zero-shot without needing to use any of the 1.28 million training examples it was trained on. We release our code and pre-trained model weights at https://github.com/OpenAI/CLIP.",
  "published": "2021-02-26",
  "updated": "2021-02-26",
  "year": "2021",
  "authors": [
   "Alec Radford",
   "Jong Wook Kim",
   "Chris Hallacy",
   "Aditya Ramesh",
   "Gabriel Goh",
   "Sandhini Agarwal",
   "Girish Sastry",
   "Amanda Askell",
   "Pamela Mishkin",
   "Jack Clark",
   "Gretchen Krueger",
   "Ilya Sutskever"
  ],
  "author_count": 12,
  "categories": [
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.CV",
  "venue": "ICML",
  "venue_source": "semantic-scholar",
  "citations": 54583,
  "influential_citations": 10216,
  "tldr": "It is demonstrated that the simple pre-training task of predicting which caption goes with which image is an efficient and scalable way to learn SOTA image representations from scratch on a dataset of 400 million (image, text) pairs collected from the internet.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Alec Radford",
    "id": "38909097",
    "h_index": 33,
    "papers": 153
   },
   {
    "name": "Jong Wook Kim",
    "id": "2110935237",
    "h_index": 10,
    "papers": 12
   },
   {
    "name": "Chris Hallacy",
    "id": "2004021329",
    "h_index": 9,
    "papers": 35
   },
   {
    "name": "A. Ramesh",
    "id": "1992922591",
    "h_index": 15,
    "papers": 26
   },
   {
    "name": "Gabriel Goh",
    "id": "40087786",
    "h_index": 14,
    "papers": 17
   },
   {
    "name": "S. Agarwal",
    "id": "144517868",
    "h_index": 20,
    "papers": 72
   },
   {
    "name": "G. Sastry",
    "id": "144864359",
    "h_index": 17,
    "papers": 89
   },
   {
    "name": "Amanda Askell",
    "id": "119609682",
    "h_index": 18,
    "papers": 29
   },
   {
    "name": "Pamela Mishkin",
    "id": "2051714782",
    "h_index": 17,
    "papers": 66
   },
   {
    "name": "Jack Clark",
    "id": "2115193883",
    "h_index": 21,
    "papers": 24
   },
   {
    "name": "Gretchen Krueger",
    "id": "2064404342",
    "h_index": 13,
    "papers": 70
   },
   {
    "name": "I. Sutskever",
    "id": "1701686",
    "h_index": 75,
    "papers": 166
   }
  ],
  "comment": "",
  "topics": [
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2103.00020v1",
  "pdf_url": "https://arxiv.org/pdf/2103.00020v1",
  "html_url": "https://arxiv.org/html/2103.00020v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "2102.12092",
  "slug": "zero-shot-text-to-image-generation",
  "title": "Zero-Shot Text-to-Image Generation",
  "abstract": "Text-to-image generation has traditionally focused on finding better modeling assumptions for training on a fixed dataset. These assumptions might involve complex architectures, auxiliary losses, or side information such as object part labels or segmentation masks supplied during training. We describe a simple approach for this task based on a transformer that autoregressively models the text and image tokens as a single stream of data. With sufficient data and scale, our approach is competitive with previous domain-specific models when evaluated in a zero-shot fashion.",
  "published": "2021-02-24",
  "updated": "2021-02-26",
  "year": "2021",
  "authors": [
   "Aditya Ramesh",
   "Mikhail Pavlov",
   "Gabriel Goh",
   "Scott Gray",
   "Chelsea Voss",
   "Alec Radford",
   "Mark Chen",
   "Ilya Sutskever"
  ],
  "author_count": 8,
  "categories": [
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.CV",
  "venue": "ICML",
  "venue_source": "semantic-scholar",
  "citations": 6686,
  "influential_citations": 403,
  "tldr": "This work describes a simple approach based on a transformer that autoregressively models the text and image tokens as a single stream of data that is competitive with previous domain-specific models when evaluated in a zero-shot fashion.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "A. Ramesh",
    "id": "1992922591",
    "h_index": 15,
    "papers": 26
   },
   {
    "name": "Mikhail Pavlov",
    "id": "2068123790",
    "h_index": 10,
    "papers": 29
   },
   {
    "name": "Gabriel Goh",
    "id": "40087786",
    "h_index": 14,
    "papers": 17
   },
   {
    "name": "Scott Gray",
    "id": "145565184",
    "h_index": 14,
    "papers": 57
   },
   {
    "name": "Chelsea Voss",
    "id": "153387869",
    "h_index": 13,
    "papers": 18
   },
   {
    "name": "Alec Radford",
    "id": "38909097",
    "h_index": 33,
    "papers": 153
   },
   {
    "name": "Mark Chen",
    "id": "2108828435",
    "h_index": 16,
    "papers": 82
   },
   {
    "name": "I. Sutskever",
    "id": "1701686",
    "h_index": 75,
    "papers": 166
   }
  ],
  "comment": "",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2102.12092v2",
  "pdf_url": "https://arxiv.org/pdf/2102.12092v2",
  "html_url": "https://arxiv.org/html/2102.12092v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "2102.10407",
  "slug": "visualgpt-data-efficient-adaptation-of-pretrained-language-models-for",
  "title": "VisualGPT: Data-efficient Adaptation of Pretrained Language Models for Image Captioning",
  "abstract": "The ability to quickly learn from a small quantity oftraining data widens the range of machine learning applications. In this paper, we propose a data-efficient image captioning model, VisualGPT, which leverages the linguistic knowledge from a large pretrained language model(LM). A crucial challenge is to balance between the use of visual information in the image and prior linguistic knowledge acquired from pretraining. We designed a novel self-resurrecting encoder-decoder attention mechanism to quickly adapt the pretrained LM as the language decoder ona small amount of in-domain training data. The proposed self-resurrecting activation unit produces sparse activations but has reduced susceptibility to zero gradients. We train the proposed model, VisualGPT, on 0.1%, 0.5% and 1% of MSCOCO and Conceptual Captions training data. Under these conditions, we outperform the best baseline model by up to 10.8% CIDEr on MS COCO and upto 5.4% CIDEr on Conceptual Captions. Further, Visual-GPT achieves the state-of-the-art result on IU X-ray, a medical report generation dataset. To the best of our knowledge, this is the first work that improves data efficiency of image captioning by utilizing LM pretrained on unimodal data. Our code is available at: https://github.com/Vision-CAIR/VisualGPT.",
  "published": "2021-02-20",
  "updated": "2022-03-30",
  "year": "2021",
  "authors": [
   "Jun Chen",
   "Han Guo",
   "Kai Yi",
   "Boyang Li",
   "Mohamed Elhoseiny"
  ],
  "author_count": 5,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.CL",
   "cs.MM"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 302,
  "influential_citations": 14,
  "tldr": "This work proposes VisualGPT, which employs a novel self-resurrecting encoder-decoder attention mechanism to quickly adapt the PLM with a small amount of in-domain image-text data and achieves the state-of-the-art result on IU X-ray, a medical report generation dataset.",
  "doi": "10.1109/CVPR52688.2022.01750",
  "oa_pdf": "https://arxiv.org/pdf/2102.10407",
  "s2_authors": [
   {
    "name": "Jun Chen",
    "id": "2153417252",
    "h_index": 11,
    "papers": 18
   },
   {
    "name": "Han Guo",
    "id": "1960405436",
    "h_index": 4,
    "papers": 19
   },
   {
    "name": "Kai Yi",
    "id": "100516201",
    "h_index": 8,
    "papers": 20
   },
   {
    "name": "Boyang Albert Li",
    "id": "1728712",
    "h_index": 27,
    "papers": 121
   },
   {
    "name": "Mohamed Elhoseiny",
    "id": "1712479",
    "h_index": 27,
    "papers": 91
   }
  ],
  "comment": "",
  "topics": [
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2102.10407v5",
  "pdf_url": "https://arxiv.org/pdf/2102.10407v5",
  "html_url": "https://arxiv.org/html/2102.10407v5",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.98
 },
 {
  "id": "2102.08363",
  "slug": "combo-conservative-offline-model-based-policy-optimization",
  "title": "COMBO: Conservative Offline Model-Based Policy Optimization",
  "abstract": "Model-based algorithms, which learn a dynamics model from logged experience and perform some sort of pessimistic planning under the learned model, have emerged as a promising paradigm for offline reinforcement learning (offline RL). However, practical variants of such model-based algorithms rely on explicit uncertainty quantification for incorporating pessimism. Uncertainty estimation with complex models, such as deep neural networks, can be difficult and unreliable. We overcome this limitation by developing a new model-based offline RL algorithm, COMBO, that regularizes the value function on out-of-support state-action tuples generated via rollouts under the learned model. This results in a conservative estimate of the value function for out-of-support state-action tuples, without requiring explicit uncertainty estimation. We theoretically show that our method optimizes a lower bound on the true policy value, that this bound is tighter than that of prior methods, and our approach satisfies a policy improvement guarantee in the offline setting. Through experiments, we find that COMBO consistently performs as well or better as compared to prior offline model-free and model-based methods on widely studied offline RL benchmarks, including image-based tasks.",
  "published": "2021-02-16",
  "updated": "2022-01-27",
  "year": "2021",
  "authors": [
   "Tianhe Yu",
   "Aviral Kumar",
   "Rafael Rafailov",
   "Aravind Rajeswaran",
   "Sergey Levine",
   "Chelsea Finn"
  ],
  "author_count": 6,
  "categories": [
   "cs.LG",
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.LG",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 542,
  "influential_citations": 63,
  "tldr": "This work develops a new model-based offline RL algorithm, COMBO, that regularizes the value function on out-of-support state-action tuples generated via rollouts under the learned model, and finds that it consistently performs as well or better as compared to prior offline model-free and model- based methods on widely studied offline RL benchmarks, including image-based tasks.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Tianhe Yu",
    "id": "10909315",
    "h_index": 31,
    "papers": 48
   },
   {
    "name": "Aviral Kumar",
    "id": "1488785534",
    "h_index": 46,
    "papers": 83
   },
   {
    "name": "Rafael Rafailov",
    "id": "102801230",
    "h_index": 25,
    "papers": 44
   },
   {
    "name": "A. Rajeswaran",
    "id": "19275599",
    "h_index": 34,
    "papers": 58
   },
   {
    "name": "S. Levine",
    "id": "1736651",
    "h_index": 167,
    "papers": 551
   },
   {
    "name": "Chelsea Finn",
    "id": "46881670",
    "h_index": 94,
    "papers": 240
   }
  ],
  "comment": "NeurIPS 2021",
  "topics": [
   "rl-control",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2102.08363v2",
  "pdf_url": "https://arxiv.org/pdf/2102.08363v2",
  "html_url": "https://arxiv.org/html/2102.08363v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.23
 },
 {
  "id": "2102.06741",
  "slug": "discovery-of-options-via-meta-learned-subgoals",
  "title": "Discovery of Options via Meta-Learned Subgoals",
  "abstract": "Temporal abstractions in the form of options have been shown to help reinforcement learning (RL) agents learn faster. However, despite prior work on this topic, the problem of discovering options through interaction with an environment remains a challenge. In this paper, we introduce a novel meta-gradient approach for discovering useful options in multi-task RL environments. Our approach is based on a manager-worker decomposition of the RL agent, in which a manager maximises rewards from the environment by learning a task-dependent policy over both a set of task-independent discovered-options and primitive actions. The option-reward and termination functions that define a subgoal for each option are parameterised as neural networks and trained via meta-gradients to maximise their usefulness. Empirical analysis on gridworld and DeepMind Lab tasks show that: (1) our approach can discover meaningful and diverse temporally-extended options in multi-task RL domains, (2) the discovered options are frequently used by the agent while learning to solve the training tasks, and (3) that the discovered options help a randomly initialised manager learn faster in completely new tasks.",
  "published": "2021-02-12",
  "updated": "2021-02-12",
  "year": "2021",
  "authors": [
   "Vivek Veeriah",
   "Tom Zahavy",
   "Matteo Hessel",
   "Zhongwen Xu",
   "Junhyuk Oh",
   "Iurii Kemaev",
   "Hado van Hasselt",
   "David Silver",
   "Satinder Singh"
  ],
  "author_count": 9,
  "categories": [
   "cs.LG",
   "cs.AI"
  ],
  "primary_category": "cs.LG",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 38,
  "influential_citations": 0,
  "tldr": "A novel meta-gradient approach for discovering useful options in multi-task RL environments based on a manager-worker decomposition of the RL agent, in which a manager maximises rewards from the environment by learning a task-dependent policy over both a set of task-independent discovered-options and primitive actions.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Vivek Veeriah",
    "id": "2300921",
    "h_index": 13,
    "papers": 29
   },
   {
    "name": "Tom Zahavy",
    "id": "3331540",
    "h_index": 23,
    "papers": 70
   },
   {
    "name": "Matteo Hessel",
    "id": "39357484",
    "h_index": 25,
    "papers": 39
   },
   {
    "name": "Zhongwen Xu",
    "id": "2351434",
    "h_index": 27,
    "papers": 62
   },
   {
    "name": "Junhyuk Oh",
    "id": "2894414",
    "h_index": 25,
    "papers": 36
   },
   {
    "name": "Iurii Kemaev",
    "id": "51883910",
    "h_index": 7,
    "papers": 9
   },
   {
    "name": "H. V. Hasselt",
    "id": "7634925",
    "h_index": 40,
    "papers": 67
   },
   {
    "name": "David Silver",
    "id": "145824029",
    "h_index": 80,
    "papers": 120
   },
   {
    "name": "Satinder Singh",
    "id": "2108384183",
    "h_index": 28,
    "papers": 67
   }
  ],
  "comment": "",
  "topics": [
   "rl-control"
  ],
  "orgs": [
   "Google DeepMind"
  ],
  "abs_url": "https://arxiv.org/abs/2102.06741v1",
  "pdf_url": "https://arxiv.org/pdf/2102.06741v1",
  "html_url": "https://arxiv.org/html/2102.06741v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.59
 },
 {
  "id": "2102.06171",
  "slug": "high-performance-large-scale-image-recognition-without-normalization",
  "title": "High-Performance Large-Scale Image Recognition Without Normalization",
  "abstract": "Batch normalization is a key component of most image classification models, but it has many undesirable properties stemming from its dependence on the batch size and interactions between examples. Although recent work has succeeded in training deep ResNets without normalization layers, these models do not match the test accuracies of the best batch-normalized networks, and are often unstable for large learning rates or strong data augmentations. In this work, we develop an adaptive gradient clipping technique which overcomes these instabilities, and design a significantly improved class of Normalizer-Free ResNets. Our smaller models match the test accuracy of an EfficientNet-B7 on ImageNet while being up to 8.7x faster to train, and our largest models attain a new state-of-the-art top-1 accuracy of 86.5%. In addition, Normalizer-Free models attain significantly better performance than their batch-normalized counterparts when finetuning on ImageNet after large-scale pre-training on a dataset of 300 million labeled images, with our best models obtaining an accuracy of 89.2%. Our code is available at https://github.com/deepmind/ deepmind-research/tree/master/nfnets",
  "published": "2021-02-11",
  "updated": "2021-02-11",
  "year": "2021",
  "authors": [
   "Andrew Brock",
   "Soham De",
   "Samuel L. Smith",
   "Karen Simonyan"
  ],
  "author_count": 4,
  "categories": [
   "cs.CV",
   "cs.LG",
   "stat.ML"
  ],
  "primary_category": "cs.CV",
  "venue": "ICML",
  "venue_source": "semantic-scholar",
  "citations": 645,
  "influential_citations": 61,
  "tldr": "An adaptive gradient clipping technique is developed which overcomes instabilities in batch normalization, and a significantly improved class of Normalizer-Free ResNets is designed, achieving significantly better performance when finetuning on ImageNet.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Andrew Brock",
    "id": "2065040247",
    "h_index": 16,
    "papers": 19
   },
   {
    "name": "Soham De",
    "id": "3252772",
    "h_index": 26,
    "papers": 39
   },
   {
    "name": "Samuel L. Smith",
    "id": "2157770601",
    "h_index": 20,
    "papers": 27
   },
   {
    "name": "K. Simonyan",
    "id": "34838386",
    "h_index": 65,
    "papers": 108
   }
  ],
  "comment": "",
  "topics": [
   "foundation-pretraining"
  ],
  "orgs": [
   "Google DeepMind"
  ],
  "abs_url": "https://arxiv.org/abs/2102.06171v1",
  "pdf_url": "https://arxiv.org/pdf/2102.06171v1",
  "html_url": "https://arxiv.org/html/2102.06171v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.81
 },
 {
  "id": "2102.05095",
  "slug": "is-space-time-attention-all-you-need-for-video-understanding",
  "title": "Is Space-Time Attention All You Need for Video Understanding?",
  "abstract": "We present a convolution-free approach to video classification built exclusively on self-attention over space and time. Our method, named \"TimeSformer,\" adapts the standard Transformer architecture to video by enabling spatiotemporal feature learning directly from a sequence of frame-level patches. Our experimental study compares different self-attention schemes and suggests that \"divided attention,\" where temporal attention and spatial attention are separately applied within each block, leads to the best video classification accuracy among the design choices considered. Despite the radically new design, TimeSformer achieves state-of-the-art results on several action recognition benchmarks, including the best reported accuracy on Kinetics-400 and Kinetics-600. Finally, compared to 3D convolutional networks, our model is faster to train, it can achieve dramatically higher test efficiency (at a small drop in accuracy), and it can also be applied to much longer video clips (over one minute long). Code and models are available at: https://github.com/facebookresearch/TimeSformer.",
  "published": "2021-02-09",
  "updated": "2021-06-09",
  "year": "2021",
  "authors": [
   "Gedas Bertasius",
   "Heng Wang",
   "Lorenzo Torresani"
  ],
  "author_count": 3,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "ICML",
  "venue_source": "semantic-scholar",
  "citations": 3275,
  "influential_citations": 410,
  "tldr": "This work presents a convolution-free approach to video classification built exclusively on self-attention over space and time, which adapts the standard Transformer architecture to video by enabling spatiotemporal feature learning directly from a sequence of frame-level patches.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Gedas Bertasius",
    "id": "3313330",
    "h_index": 29,
    "papers": 76
   },
   {
    "name": "Heng Wang",
    "id": "46506697",
    "h_index": 24,
    "papers": 33
   },
   {
    "name": "L. Torresani",
    "id": "1732879",
    "h_index": 57,
    "papers": 143
   }
  ],
  "comment": "Accepted to ICML 2021",
  "topics": [],
  "orgs": [
   "Meta FAIR"
  ],
  "abs_url": "https://arxiv.org/abs/2102.05095v4",
  "pdf_url": "https://arxiv.org/pdf/2102.05095v4",
  "html_url": "https://arxiv.org/html/2102.05095v4",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.0
 },
 {
  "id": "2101.11812",
  "slug": "swingbot-learning-physical-features-from-in-hand-tactile-exploration-f",
  "title": "SwingBot: Learning Physical Features from In-hand Tactile Exploration for Dynamic Swing-up Manipulation",
  "abstract": "Several robot manipulation tasks are extremely sensitive to variations of the physical properties of the manipulated objects. One such task is manipulating objects by using gravity or arm accelerations, increasing the importance of mass, center of mass, and friction information. We present SwingBot, a robot that is able to learn the physical features of a held object through tactile exploration. Two exploration actions (tilting and shaking) provide the tactile information used to create a physical feature embedding space. With this embedding, SwingBot is able to predict the swing angle achieved by a robot performing dynamic swing-up manipulations on a previously unseen object. Using these predictions, it is able to search for the optimal control parameters for a desired swing-up angle. We show that with the learned physical features our end-to-end self-supervised learning pipeline is able to substantially improve the accuracy of swinging up unseen objects. We also show that objects with similar dynamics are closer to each other on the embedding space and that the embedding can be disentangled into values of specific physical properties.",
  "published": "2021-01-28",
  "updated": "2021-01-28",
  "year": "2021",
  "authors": [
   "Chen Wang",
   "Shaoxiong Wang",
   "Branden Romero",
   "Filipe Veiga",
   "Edward Adelson"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS",
  "venue_source": "semantic-scholar",
  "citations": 125,
  "influential_citations": 3,
  "tldr": "S SwingBot is a robot that is able to learn the physical features of an held object through tactile exploration, and with the learned physical features its end-to-end self-supervised learning pipeline is ability to substantially improve the accuracy of swinging up unseen objects.",
  "doi": "10.1109/IROS45743.2020.9341006",
  "oa_pdf": "https://arxiv.org/pdf/2101.11812",
  "s2_authors": [
   {
    "name": "Chen Wang",
    "id": "2109119431",
    "h_index": 18,
    "papers": 28
   },
   {
    "name": "Shaoxiong Wang",
    "id": "7488549",
    "h_index": 14,
    "papers": 16
   },
   {
    "name": "Branden Romero",
    "id": "21201572",
    "h_index": 9,
    "papers": 10
   },
   {
    "name": "Filipe Veiga",
    "id": "144539107",
    "h_index": 12,
    "papers": 24
   },
   {
    "name": "E. Adelson",
    "id": "145358192",
    "h_index": 87,
    "papers": 272
   }
  ],
  "comment": "IROS 2020",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2101.11812v1",
  "pdf_url": "https://arxiv.org/pdf/2101.11812v1",
  "html_url": "https://arxiv.org/html/2101.11812v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.6
 },
 {
  "id": "2101.07393",
  "slug": "grounding-language-to-entities-and-dynamics-for-generalization-in-rein",
  "title": "Grounding Language to Entities and Dynamics for Generalization in Reinforcement Learning",
  "abstract": "We investigate the use of natural language to drive the generalization of control policies and introduce the new multi-task environment Messenger with free-form text manuals describing the environment dynamics. Unlike previous work, Messenger does not assume prior knowledge connecting text and state observations $-$ the control policy must simultaneously ground the game manual to entity symbols and dynamics in the environment. We develop a new model, EMMA (Entity Mapper with Multi-modal Attention) which uses an entity-conditioned attention module that allows for selective focus over relevant descriptions in the manual for each entity in the environment. EMMA is end-to-end differentiable and learns a latent grounding of entities and dynamics from text to observations using only environment rewards. EMMA achieves successful zero-shot generalization to unseen games with new dynamics, obtaining a 40% higher win rate compared to multiple baselines. However, win rate on the hardest stage of Messenger remains low (10%), demonstrating the need for additional work in this direction.",
  "published": "2021-01-19",
  "updated": "2021-06-11",
  "year": "2021",
  "authors": [
   "Austin W. Hanjie",
   "Victor Zhong",
   "Karthik Narasimhan"
  ],
  "author_count": 3,
  "categories": [
   "cs.CL",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.CL",
  "venue": "ICML",
  "venue_source": "semantic-scholar",
  "citations": 63,
  "influential_citations": 12,
  "tldr": "A new model, EMMA (Entity Mapper with Multi-modal Attention) which uses an entity-conditioned attention module that allows for selective focus over relevant descriptions in the manual for each entity in the environment, achieves successful zero-shot generalization to unseen games with new dynamics.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "H. Wang",
    "id": "39698929",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Karthik Narasimhan",
    "id": "144958935",
    "h_index": 34,
    "papers": 70
   }
  ],
  "comment": "Accepted to ICML 2021. Note author list and name changes from previous version",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2101.07393v2",
  "pdf_url": "https://arxiv.org/pdf/2101.07393v2",
  "html_url": "https://arxiv.org/html/2101.07393v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.31
 },
 {
  "id": "2101.07241",
  "slug": "learning-by-watching-physical-imitation-of-manipulation-skills-from-hu",
  "title": "Learning by Watching: Physical Imitation of Manipulation Skills from Human Videos",
  "abstract": "Learning from visual data opens the potential to accrue a large range of manipulation behaviors by leveraging human demonstrations without specifying each of them mathematically, but rather through natural task specification. In this paper, we present Learning by Watching (LbW), an algorithmic framework for policy learning through imitation from a single video specifying the task. The key insights of our method are two-fold. First, since the human arms may not have the same morphology as robot arms, our framework learns unsupervised human to robot translation to overcome the morphology mismatch issue. Second, to capture the details in salient regions that are crucial for learning state representations, our model performs unsupervised keypoint detection on the translated robot videos. The detected keypoints form a structured representation that contains semantically meaningful information and can be used directly for computing reward and policy learning. We evaluate the effectiveness of our LbW framework on five robot manipulation tasks, including reaching, pushing, sliding, coffee making, and drawer closing. Extensive experimental evaluations demonstrate that our method performs favorably against the state-of-the-art approaches.",
  "published": "2021-01-18",
  "updated": "2021-11-14",
  "year": "2021",
  "authors": [
   "Haoyu Xiong",
   "Quanzhou Li",
   "Yun-Chun Chen",
   "Homanga Bharadhwaj",
   "Samarth Sinha",
   "Animesh Garg"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS",
  "venue_source": "semantic-scholar",
  "citations": 139,
  "influential_citations": 3,
  "tldr": "This paper presents Learning by Watching (LbW), an algorithmic framework for policy learning through imitation from a single video specifying the task, and learns unsupervised human to robot translation to overcome the morphology mis-match issue.",
  "doi": "10.1109/IROS51168.2021.9636080",
  "oa_pdf": "https://arxiv.org/pdf/2101.07241",
  "s2_authors": [
   {
    "name": "Haoyu Xiong",
    "id": "2054472895",
    "h_index": 2,
    "papers": 5
   },
   {
    "name": "Quanzhou Li",
    "id": "122927114",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Yun-Chun Chen",
    "id": "1965873910",
    "h_index": 13,
    "papers": 17
   },
   {
    "name": "Homanga Bharadhwaj",
    "id": "51113848",
    "h_index": 23,
    "papers": 59
   },
   {
    "name": "Samarth Sinha",
    "id": "89140693",
    "h_index": 19,
    "papers": 34
   },
   {
    "name": "Animesh Garg",
    "id": "1873736",
    "h_index": 60,
    "papers": 163
   }
  ],
  "comment": "Project Website: https://www.pair.toronto.edu/lbw-kp/",
  "topics": [
   "egocentric-data",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2101.07241v2",
  "pdf_url": "https://arxiv.org/pdf/2101.07241v2",
  "html_url": "https://arxiv.org/html/2101.07241v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.65
 },
 {
  "id": "2012.09841",
  "slug": "taming-transformers-for-high-resolution-image-synthesis",
  "title": "Taming Transformers for High-Resolution Image Synthesis",
  "abstract": "Designed to learn long-range interactions on sequential data, transformers continue to show state-of-the-art results on a wide variety of tasks. In contrast to CNNs, they contain no inductive bias that prioritizes local interactions. This makes them expressive, but also computationally infeasible for long sequences, such as high-resolution images. We demonstrate how combining the effectiveness of the inductive bias of CNNs with the expressivity of transformers enables them to model and thereby synthesize high-resolution images. We show how to (i) use CNNs to learn a context-rich vocabulary of image constituents, and in turn (ii) utilize transformers to efficiently model their composition within high-resolution images. Our approach is readily applied to conditional synthesis tasks, where both non-spatial information, such as object classes, and spatial information, such as segmentations, can control the generated image. In particular, we present the first results on semantically-guided synthesis of megapixel images with transformers and obtain the state of the art among autoregressive models on class-conditional ImageNet. Code and pretrained models can be found at https://github.com/CompVis/taming-transformers .",
  "published": "2020-12-17",
  "updated": "2021-06-23",
  "year": "2020",
  "authors": [
   "Patrick Esser",
   "Robin Rombach",
   "Bj\u00f6rn Ommer"
  ],
  "author_count": 3,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 4543,
  "influential_citations": 693,
  "tldr": "It is demonstrated how combining the effectiveness of the inductive bias of CNNs with the expressivity of transformers enables them to model and thereby synthesize high-resolution images.",
  "doi": "10.1109/CVPR46437.2021.01268",
  "oa_pdf": "https://arxiv.org/pdf/2012.09841",
  "s2_authors": [
   {
    "name": "Patrick Esser",
    "id": "35175531",
    "h_index": 17,
    "papers": 24
   },
   {
    "name": "Robin Rombach",
    "id": "1660819540",
    "h_index": 22,
    "papers": 29
   },
   {
    "name": "B. Ommer",
    "id": "1796707",
    "h_index": 42,
    "papers": 122
   }
  ],
  "comment": "Changelog can be found in the supplementary",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2012.09841v3",
  "pdf_url": "https://arxiv.org/pdf/2012.09841v3",
  "html_url": "https://arxiv.org/html/2012.09841v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "2012.05672",
  "slug": "imitating-interactive-intelligence",
  "title": "Imitating Interactive Intelligence",
  "abstract": "A common vision from science fiction is that robots will one day inhabit our physical spaces, sense the world as we do, assist our physical labours, and communicate with us through natural language. Here we study how to design artificial agents that can interact naturally with humans using the simplification of a virtual environment. This setting nevertheless integrates a number of the central challenges of artificial intelligence (AI) research: complex visual perception and goal-directed physical control, grounded language comprehension and production, and multi-agent social interaction. To build agents that can robustly interact with humans, we would ideally train them while they interact with humans. However, this is presently impractical. Therefore, we approximate the role of the human with another learned agent, and use ideas from inverse reinforcement learning to reduce the disparities between human-human and agent-agent interactive behaviour. Rigorously evaluating our agents poses a great challenge, so we develop a variety of behavioural tests, including evaluation by humans who watch videos of agents or interact directly with them. These evaluations convincingly demonstrate that interactive training and auxiliary losses improve agent behaviour beyond what is achieved by supervised learning of actions alone. Further, we demonstrate that agent capabilities generalise beyond literal experiences in the dataset. Finally, we train evaluation models whose ratings of agents agree well with human judgement, thus permitting the evaluation of new agent models without additional effort. Taken together, our results in this virtual environment provide evidence that large-scale human behavioural imitation is a promising tool to create intelligent, interactive agents, and the challenge of reliably evaluating such agents is possible to surmount.",
  "published": "2020-12-10",
  "updated": "2021-01-21",
  "year": "2020",
  "authors": [
   "Josh Abramson",
   "Arun Ahuja",
   "Iain Barr",
   "Arthur Brussee",
   "Federico Carnevale",
   "Mary Cassin",
   "Rachita Chhaparia",
   "Stephen Clark",
   "Bogdan Damoc",
   "Andrew Dudzik",
   "Petko Georgiev",
   "Aurelia Guy",
   "Tim Harley",
   "Felix Hill",
   "Alden Hung",
   "Zachary Kenton",
   "Jessica Landon",
   "Timothy Lillicrap",
   "Kory Mathewson",
   "So\u0148a Mokr\u00e1",
   "Alistair Muldal",
   "Adam Santoro",
   "Nikolay Savinov",
   "Vikrant Varma",
   "Greg Wayne",
   "Duncan Williams",
   "Nathaniel Wong",
   "Chen Yan",
   "Rui Zhu"
  ],
  "author_count": 29,
  "categories": [
   "cs.LG",
   "cs.AI",
   "cs.MA"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 74,
  "influential_citations": 3,
  "tldr": "The results in this virtual environment provide evidence that large-scale human behavioural imitation is a promising tool to create intelligent, interactive agents, and the challenge of reliably evaluating such agents is possible to surmount.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Josh Abramson",
    "id": "3041463",
    "h_index": 7,
    "papers": 8
   },
   {
    "name": "Arun Ahuja",
    "id": "37968006",
    "h_index": 22,
    "papers": 49
   },
   {
    "name": "Arthur Brussee",
    "id": "104251960",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Federico Carnevale",
    "id": "32561676",
    "h_index": 9,
    "papers": 14
   },
   {
    "name": "Mary Cassin",
    "id": "147433059",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "S. Clark",
    "id": "144523372",
    "h_index": 53,
    "papers": 154
   },
   {
    "name": "A. Dudzik",
    "id": "145585375",
    "h_index": 8,
    "papers": 14
   },
   {
    "name": "Petko Georgiev",
    "id": "1737522",
    "h_index": 22,
    "papers": 46
   },
   {
    "name": "Aurelia Guy",
    "id": "40895205",
    "h_index": 10,
    "papers": 17
   },
   {
    "name": "Tim Harley",
    "id": "3367786",
    "h_index": 13,
    "papers": 17
   },
   {
    "name": "Felix Hill",
    "id": "145783676",
    "h_index": 42,
    "papers": 77
   },
   {
    "name": "Alden Hung",
    "id": "1572095637",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Zachary Kenton",
    "id": "40947466",
    "h_index": 20,
    "papers": 31
   },
   {
    "name": "Jessica Landon",
    "id": "2065404873",
    "h_index": 7,
    "papers": 12
   },
   {
    "name": "T. Lillicrap",
    "id": "2542999",
    "h_index": 69,
    "papers": 154
   },
   {
    "name": "K. Mathewson",
    "id": "3422828",
    "h_index": 15,
    "papers": 46
   },
   {
    "name": "Alistair Muldal",
    "id": "50654556",
    "h_index": 12,
    "papers": 15
   },
   {
    "name": "Adam Santoro",
    "id": "35030998",
    "h_index": 38,
    "papers": 53
   },
   {
    "name": "N. Savinov",
    "id": "2417003",
    "h_index": 17,
    "papers": 27
   },
   {
    "name": "Vikrant Varma",
    "id": "144711236",
    "h_index": 10,
    "papers": 15
   },
   {
    "name": "Greg Wayne",
    "id": "89504302",
    "h_index": 33,
    "papers": 47
   },
   {
    "name": "Nathaniel Wong",
    "id": "1571809179",
    "h_index": 7,
    "papers": 8
   },
   {
    "name": "Chen Yan",
    "id": "2116550367",
    "h_index": 8,
    "papers": 15
   },
   {
    "name": "Rui Zhu",
    "id": "2070271342",
    "h_index": 7,
    "papers": 12
   }
  ],
  "comment": "",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2012.05672v2",
  "pdf_url": "https://arxiv.org/pdf/2012.05672v2",
  "html_url": "https://arxiv.org/html/2012.05672v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.88
 },
 {
  "id": "2012.01859",
  "slug": "goal-driven-robotic-pushing-using-tactile-and-proprioceptive-feedback",
  "title": "Goal-Driven Robotic Pushing Using Tactile and Proprioceptive Feedback",
  "abstract": "In robots, nonprehensile manipulation operations such as pushing are a useful way of moving large, heavy or unwieldy objects, moving multiple objects at once, or reducing uncertainty in the location or pose of objects. In this study, we propose a reactive and adaptive method for robotic pushing that uses rich feedback from a high-resolution optical tactile sensor to control push movements instead of relying on analytical or data-driven models of push interactions. Specifically, we use goal-driven tactile exploration to actively search for stable pushing configurations that cause the object to maintain its pose relative to the pusher while incrementally moving the pusher and object towards the target. We evaluate our method by pushing objects across planar and curved surfaces. For planar surfaces, we show that the method is accurate and robust to variations in initial contact position/angle, object shape and start position; for curved surfaces, the performance is degraded slightly. An immediate consequence of our work is that it shows that explicit models of push interactions might be sufficient but are not necessary for this type of task. It also raises the interesting question of which aspects of the system should be modelled to achieve the best performance and generalization across a wide range of scenarios. Finally, it highlights the importance of testing on non-planar surfaces and in other more complex environments when developing new methods for robotic pushing.",
  "published": "2020-12-03",
  "updated": "2021-08-02",
  "year": "2020",
  "authors": [
   "John Lloyd",
   "Nathan F. Lepora"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "T-RO",
  "venue_source": "semantic-scholar",
  "citations": 71,
  "influential_citations": 0,
  "tldr": "This study proposes a reactive and adaptive method for robotic pushing that uses rich feedback from a high-resolution optical tactile sensor to control push movements instead of relying on analytical or data-driven models of push interactions.",
  "doi": "10.1109/tro.2021.3104471",
  "oa_pdf": "https://research-information.bris.ac.uk/en/publications/59da35d2-4814-4637-9f89-7b6de66bc8db",
  "s2_authors": [
   {
    "name": "John Lloyd",
    "id": "2054582550",
    "h_index": 16,
    "papers": 23
   },
   {
    "name": "N. Lepora",
    "id": "2467565",
    "h_index": 37,
    "papers": 214
   }
  ],
  "comment": "Accepted in IEEE Transactions on Robotics. A video demonstrating the approach can be found at https://youtu.be/6fAlHWfLP7I",
  "topics": [
   "tactile",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2012.01859v3",
  "pdf_url": "https://arxiv.org/pdf/2012.01859v3",
  "html_url": "https://arxiv.org/html/2012.01859v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.36
 },
 {
  "id": "2012.00924",
  "slug": "cpf-learning-a-contact-potential-field-to-model-the-hand-object-intera",
  "title": "CPF: Learning a Contact Potential Field to Model the Hand-Object Interaction",
  "abstract": "Modeling the hand-object (HO) interaction not only requires estimation of the HO pose, but also pays attention to the contact due to their interaction. Significant progress has been made in estimating hand and object separately with deep learning methods, simultaneous HO pose estimation and contact modeling has not yet been fully explored. In this paper, we present an explicit contact representation namely Contact Potential Field (CPF), and a learning-fitting hybrid framework namely MIHO to Modeling the Interaction of Hand and Object. In CPF, we treat each contacting HO vertex pair as a spring-mass system. Hence the whole system forms a potential field with minimal elastic energy at the grasp position. Extensive experiments on the two commonly used benchmarks have demonstrated that our method can achieve state-of-the-art in several reconstruction metrics, and allow us to produce more physically plausible HO pose even when the ground-truth exhibits severe interpenetration or disjointedness. Our code is available at https://github.com/lixiny/CPF.",
  "published": "2020-12-02",
  "updated": "2021-09-11",
  "year": "2020",
  "authors": [
   "Lixin Yang",
   "Xinyu Zhan",
   "Kailin Li",
   "Wenqiang Xu",
   "Jiefeng Li",
   "Cewu Lu"
  ],
  "author_count": 6,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "ICCV",
  "venue_source": "semantic-scholar",
  "citations": 178,
  "influential_citations": 14,
  "tldr": "An explicit contact representation namely Contact Potential Field (CPF), and a learning-fitting hybrid framework namely MIHO to Modeling the Interaction of Hand and Object are presented.",
  "doi": "10.1109/ICCV48922.2021.01091",
  "oa_pdf": "https://arxiv.org/pdf/2012.00924",
  "s2_authors": [
   {
    "name": "Lixin Yang",
    "id": "2111809756",
    "h_index": 16,
    "papers": 41
   },
   {
    "name": "Xinyu Zhan",
    "id": "2114716656",
    "h_index": 10,
    "papers": 18
   },
   {
    "name": "Kailin Li",
    "id": "2023790905",
    "h_index": 9,
    "papers": 16
   },
   {
    "name": "Wenqiang Xu",
    "id": "2000358050",
    "h_index": 20,
    "papers": 53
   },
   {
    "name": "Jiefeng Li",
    "id": "49299169",
    "h_index": 14,
    "papers": 19
   },
   {
    "name": "Cewu Lu",
    "id": "2293350006",
    "h_index": 19,
    "papers": 34
   }
  ],
  "comment": "ICCV 2021, (reduce PDF file size)",
  "topics": [
   "dexterous-manipulation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2012.00924v4",
  "pdf_url": "https://arxiv.org/pdf/2012.00924v4",
  "html_url": "https://arxiv.org/html/2012.00924v4",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.75
 },
 {
  "id": "2012.00759",
  "slug": "max-deeplab-end-to-end-panoptic-segmentation-with-mask-transformers",
  "title": "MaX-DeepLab: End-to-End Panoptic Segmentation with Mask Transformers",
  "abstract": "We present MaX-DeepLab, the first end-to-end model for panoptic segmentation. Our approach simplifies the current pipeline that depends heavily on surrogate sub-tasks and hand-designed components, such as box detection, non-maximum suppression, thing-stuff merging, etc. Although these sub-tasks are tackled by area experts, they fail to comprehensively solve the target task. By contrast, our MaX-DeepLab directly predicts class-labeled masks with a mask transformer, and is trained with a panoptic quality inspired loss via bipartite matching. Our mask transformer employs a dual-path architecture that introduces a global memory path in addition to a CNN path, allowing direct communication with any CNN layers. As a result, MaX-DeepLab shows a significant 7.1% PQ gain in the box-free regime on the challenging COCO dataset, closing the gap between box-based and box-free methods for the first time. A small variant of MaX-DeepLab improves 3.0% PQ over DETR with similar parameters and M-Adds. Furthermore, MaX-DeepLab, without test time augmentation, achieves new state-of-the-art 51.3% PQ on COCO test-dev set. Code is available at https://github.com/google-research/deeplab2.",
  "published": "2020-12-01",
  "updated": "2021-07-12",
  "year": "2020",
  "authors": [
   "Huiyu Wang",
   "Yukun Zhu",
   "Hartwig Adam",
   "Alan Yuille",
   "Liang-Chieh Chen"
  ],
  "author_count": 5,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 615,
  "influential_citations": 34,
  "tldr": "MaX-DeepLab, the first end-to-end model for panoptic segmentation, is presented, and shows a significant 7.1% PQ gain in the box-free regime on the challenging COCO dataset, closing the gap between box-based and box- free methods for the first time.",
  "doi": "10.1109/CVPR46437.2021.00542",
  "oa_pdf": "https://arxiv.org/pdf/2012.00759",
  "s2_authors": [
   {
    "name": "Huiyu Wang",
    "id": "1587922010",
    "h_index": 10,
    "papers": 13
   },
   {
    "name": "Yukun Zhu",
    "id": "1844940337",
    "h_index": 18,
    "papers": 22
   },
   {
    "name": "Hartwig Adam",
    "id": "2595180",
    "h_index": 44,
    "papers": 70
   },
   {
    "name": "A. Yuille",
    "id": "145081362",
    "h_index": 137,
    "papers": 757
   },
   {
    "name": "Liang-Chieh Chen",
    "id": "34192119",
    "h_index": 41,
    "papers": 57
   }
  ],
  "comment": "CVPR 2021",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2012.00759v3",
  "pdf_url": "https://arxiv.org/pdf/2012.00759v3",
  "html_url": "https://arxiv.org/html/2012.00759v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.29
 },
 {
  "id": "2011.10650",
  "slug": "very-deep-vaes-generalize-autoregressive-models-and-can-outperform-the",
  "title": "Very Deep VAEs Generalize Autoregressive Models and Can Outperform Them on Images",
  "abstract": "We present a hierarchical VAE that, for the first time, generates samples quickly while outperforming the PixelCNN in log-likelihood on all natural image benchmarks. We begin by observing that, in theory, VAEs can actually represent autoregressive models, as well as faster, better models if they exist, when made sufficiently deep. Despite this, autoregressive models have historically outperformed VAEs in log-likelihood. We test if insufficient depth explains why by scaling a VAE to greater stochastic depth than previously explored and evaluating it CIFAR-10, ImageNet, and FFHQ. In comparison to the PixelCNN, these very deep VAEs achieve higher likelihoods, use fewer parameters, generate samples thousands of times faster, and are more easily applied to high-resolution images. Qualitative studies suggest this is because the VAE learns efficient hierarchical visual representations. We release our source code and models at https://github.com/openai/vdvae.",
  "published": "2020-11-20",
  "updated": "2021-03-16",
  "year": "2020",
  "authors": [
   "Rewon Child"
  ],
  "author_count": 1,
  "categories": [
   "cs.LG",
   "cs.CV"
  ],
  "primary_category": "cs.LG",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 405,
  "influential_citations": 70,
  "tldr": "This work presents a hierarchical VAE that, for the first time, outperforms the PixelCNN in log-likelihood on all natural image benchmarks and visualize the generative process and show the VAEs learn efficient hierarchical visual representations.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "R. Child",
    "id": "48422824",
    "h_index": 14,
    "papers": 96
   }
  ],
  "comment": "17 pages, 14 figures",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2011.10650v2",
  "pdf_url": "https://arxiv.org/pdf/2011.10650v2",
  "html_url": "https://arxiv.org/html/2011.10650v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.11
 },
 {
  "id": "2010.11929",
  "slug": "an-image-is-worth-16x16-words-transformers-for-image-recognition-at-sc",
  "title": "An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale",
  "abstract": "While the Transformer architecture has become the de-facto standard for natural language processing tasks, its applications to computer vision remain limited. In vision, attention is either applied in conjunction with convolutional networks, or used to replace certain components of convolutional networks while keeping their overall structure in place. We show that this reliance on CNNs is not necessary and a pure transformer applied directly to sequences of image patches can perform very well on image classification tasks. When pre-trained on large amounts of data and transferred to multiple mid-sized or small image recognition benchmarks (ImageNet, CIFAR-100, VTAB, etc.), Vision Transformer (ViT) attains excellent results compared to state-of-the-art convolutional networks while requiring substantially fewer computational resources to train.",
  "published": "2020-10-22",
  "updated": "2021-06-03",
  "year": "2020",
  "authors": [
   "Alexey Dosovitskiy",
   "Lucas Beyer",
   "Alexander Kolesnikov",
   "Dirk Weissenborn",
   "Xiaohua Zhai",
   "Thomas Unterthiner",
   "Mostafa Dehghani",
   "Matthias Minderer",
   "Georg Heigold",
   "Sylvain Gelly",
   "Jakob Uszkoreit",
   "Neil Houlsby"
  ],
  "author_count": 12,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.CV",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 68574,
  "influential_citations": 7396,
  "tldr": "Vision Transformer (ViT) attains excellent results compared to state-of-the-art convolutional networks while requiring substantially fewer computational resources to train.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Alexey Dosovitskiy",
    "id": "2841331",
    "h_index": 55,
    "papers": 82
   },
   {
    "name": "Lucas Beyer",
    "id": "39611591",
    "h_index": 39,
    "papers": 59
   },
   {
    "name": "Alexander Kolesnikov",
    "id": "144629422",
    "h_index": 30,
    "papers": 50
   },
   {
    "name": "Dirk Weissenborn",
    "id": "3319373",
    "h_index": 19,
    "papers": 39
   },
   {
    "name": "Xiaohua Zhai",
    "id": "2743563",
    "h_index": 40,
    "papers": 71
   },
   {
    "name": "Thomas Unterthiner",
    "id": "2465270",
    "h_index": 30,
    "papers": 60
   },
   {
    "name": "Mostafa Dehghani",
    "id": "2274215058",
    "h_index": 2,
    "papers": 12
   },
   {
    "name": "M. Minderer",
    "id": "46352821",
    "h_index": 22,
    "papers": 29
   },
   {
    "name": "G. Heigold",
    "id": "2280399",
    "h_index": 31,
    "papers": 84
   },
   {
    "name": "S. Gelly",
    "id": "1802148",
    "h_index": 42,
    "papers": 105
   },
   {
    "name": "Jakob Uszkoreit",
    "id": "39328010",
    "h_index": 32,
    "papers": 55
   },
   {
    "name": "N. Houlsby",
    "id": "2815290",
    "h_index": 52,
    "papers": 102
   }
  ],
  "comment": "Fine-tuning code and pre-trained models are available at https://github.com/google-research/vision_transformer. ICLR camera-ready version with 2 small modifications: 1) Added a discussion of CLS vs GAP classifier in the appendix, 2) Fixed an error in exaFLOPs computation in Figure 5 and Table 6 (relative performance of models is basically not affected)",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2010.11929v2",
  "pdf_url": "https://arxiv.org/pdf/2010.11929v2",
  "html_url": "https://arxiv.org/html/2010.11929v2",
  "code_url": "https://github.com/google-research/vision_transformer.",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.5
 },
 {
  "id": "2010.11251",
  "slug": "learning-quadrupedal-locomotion-over-challenging-terrain",
  "title": "Learning Quadrupedal Locomotion over Challenging Terrain",
  "abstract": "Some of the most challenging environments on our planet are accessible to quadrupedal animals but remain out of reach for autonomous machines. Legged locomotion can dramatically expand the operational domains of robotics. However, conventional controllers for legged locomotion are based on elaborate state machines that explicitly trigger the execution of motion primitives and reflexes. These designs have escalated in complexity while falling short of the generality and robustness of animal locomotion. Here we present a radically robust controller for legged locomotion in challenging natural environments. We present a novel solution to incorporating proprioceptive feedback in locomotion control and demonstrate remarkable zero-shot generalization from simulation to natural environments. The controller is trained by reinforcement learning in simulation. It is based on a neural network that acts on a stream of proprioceptive signals. The trained controller has taken two generations of quadrupedal ANYmal robots to a variety of natural environments that are beyond the reach of prior published work in legged locomotion. The controller retains its robustness under conditions that have never been encountered during training: deformable terrain such as mud and snow, dynamic footholds such as rubble, and overground impediments such as thick vegetation and gushing water. The presented work opens new frontiers for robotics and indicates that radical robustness in natural environments can be achieved by training in much simpler domains.",
  "published": "2020-10-21",
  "updated": "2020-10-21",
  "year": "2020",
  "authors": [
   "Joonho Lee",
   "Jemin Hwangbo",
   "Lorenz Wellhausen",
   "Vladlen Koltun",
   "Marco Hutter"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.LG",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "Science Robotics",
  "venue_source": "semantic-scholar",
  "citations": 1740,
  "influential_citations": 101,
  "tldr": "The presented work indicates that robust locomotion in natural environments can be achieved by training in simple domains.",
  "doi": "10.1126/scirobotics.abc5986",
  "oa_pdf": "https://arxiv.org/pdf/2010.11251",
  "s2_authors": [
   {
    "name": "Joonho Lee",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Jemin Hwangbo",
    "id": "1707297",
    "h_index": 25,
    "papers": 49
   },
   {
    "name": "Lorenz Wellhausen",
    "id": "7153704",
    "h_index": 22,
    "papers": 26
   },
   {
    "name": "V. Koltun",
    "id": "145231047",
    "h_index": 114,
    "papers": 239
   },
   {
    "name": "Marco Hutter",
    "id": "14349870",
    "h_index": 80,
    "papers": 279
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "rl-control",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2010.11251v1",
  "pdf_url": "https://arxiv.org/pdf/2010.11251v1",
  "html_url": "https://arxiv.org/html/2010.11251v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "2010.07954",
  "slug": "room-across-room-multilingual-vision-and-language-navigation-with-dens",
  "title": "Room-Across-Room: Multilingual Vision-and-Language Navigation with Dense Spatiotemporal Grounding",
  "abstract": "We introduce Room-Across-Room (RxR), a new Vision-and-Language Navigation (VLN) dataset. RxR is multilingual (English, Hindi, and Telugu) and larger (more paths and instructions) than other VLN datasets. It emphasizes the role of language in VLN by addressing known biases in paths and eliciting more references to visible entities. Furthermore, each word in an instruction is time-aligned to the virtual poses of instruction creators and validators. We establish baseline scores for monolingual and multilingual settings and multitask learning when including Room-to-Room annotations. We also provide results for a model that learns from synchronized pose traces by focusing only on portions of the panorama attended to in human demonstrations. The size, scope and detail of RxR dramatically expands the frontier for research on embodied language agents in simulated, photo-realistic environments.",
  "published": "2020-10-15",
  "updated": "2020-10-15",
  "year": "2020",
  "authors": [
   "Alexander Ku",
   "Peter Anderson",
   "Roma Patel",
   "Eugene Ie",
   "Jason Baldridge"
  ],
  "author_count": 5,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.CL"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 570,
  "influential_citations": 79,
  "tldr": "The size, scope and detail of Room-Across-Room (RxR) dramatically expands the frontier for research on embodied language agents in simulated, photo-realistic environments.",
  "doi": "10.18653/v1/2020.emnlp-main.356",
  "oa_pdf": "https://www.aclweb.org/anthology/2020.emnlp-main.356.pdf",
  "s2_authors": [
   {
    "name": "Alexander Ku",
    "id": "31702389",
    "h_index": 13,
    "papers": 18
   },
   {
    "name": "Peter Anderson",
    "id": "2141916997",
    "h_index": 14,
    "papers": 20
   },
   {
    "name": "Roma Patel",
    "id": "2087020178",
    "h_index": 8,
    "papers": 13
   },
   {
    "name": "Eugene Ie",
    "id": "2042413",
    "h_index": 23,
    "papers": 38
   },
   {
    "name": "Jason Baldridge",
    "id": "1387994164",
    "h_index": 49,
    "papers": 128
   }
  ],
  "comment": "EMNLP 2020",
  "topics": [
   "egocentric-data",
   "navigation",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2010.07954v1",
  "pdf_url": "https://arxiv.org/pdf/2010.07954v1",
  "html_url": "https://arxiv.org/html/2010.07954v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.76
 },
 {
  "id": "2010.04159",
  "slug": "deformable-detr-deformable-transformers-for-end-to-end-object-detectio",
  "title": "Deformable DETR: Deformable Transformers for End-to-End Object Detection",
  "abstract": "DETR has been recently proposed to eliminate the need for many hand-designed components in object detection while demonstrating good performance. However, it suffers from slow convergence and limited feature spatial resolution, due to the limitation of Transformer attention modules in processing image feature maps. To mitigate these issues, we proposed Deformable DETR, whose attention modules only attend to a small set of key sampling points around a reference. Deformable DETR can achieve better performance than DETR (especially on small objects) with 10 times less training epochs. Extensive experiments on the COCO benchmark demonstrate the effectiveness of our approach. Code is released at https://github.com/fundamentalvision/Deformable-DETR.",
  "published": "2020-10-08",
  "updated": "2021-03-18",
  "year": "2020",
  "authors": [
   "Xizhou Zhu",
   "Weijie Su",
   "Lewei Lu",
   "Bin Li",
   "Xiaogang Wang",
   "Jifeng Dai"
  ],
  "author_count": 6,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 8178,
  "influential_citations": 988,
  "tldr": "Deformable DETR, whose attention modules only attend to a small set of key sampling points around a reference, can achieve better performance than DETR (especially on small objects) with 10$\\times less training epochs.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Xizhou Zhu",
    "id": "2578924",
    "h_index": 48,
    "papers": 82
   },
   {
    "name": "Weijie Su",
    "id": "145499378",
    "h_index": 11,
    "papers": 12
   },
   {
    "name": "Lewei Lu",
    "id": "152309485",
    "h_index": 32,
    "papers": 53
   },
   {
    "name": "Bin Li",
    "id": "2183101614",
    "h_index": 21,
    "papers": 80
   },
   {
    "name": "Xiaogang Wang",
    "id": "93768810",
    "h_index": 50,
    "papers": 82
   },
   {
    "name": "Jifeng Dai",
    "id": "3304536",
    "h_index": 66,
    "papers": 100
   }
  ],
  "comment": "ICLR 2021 Oral",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2010.04159v4",
  "pdf_url": "https://arxiv.org/pdf/2010.04159v4",
  "html_url": "https://arxiv.org/html/2010.04159v4",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "2010.03768",
  "slug": "alfworld-aligning-text-and-embodied-environments-for-interactive-learn",
  "title": "ALFWorld: Aligning Text and Embodied Environments for Interactive Learning",
  "abstract": "Given a simple request like Put a washed apple in the kitchen fridge, humans can reason in purely abstract terms by imagining action sequences and scoring their likelihood of success, prototypicality, and efficiency, all without moving a muscle. Once we see the kitchen in question, we can update our abstract plans to fit the scene. Embodied agents require the same abilities, but existing work does not yet provide the infrastructure necessary for both reasoning abstractly and executing concretely. We address this limitation by introducing ALFWorld, a simulator that enables agents to learn abstract, text based policies in TextWorld (C\u00f4t\u00e9 et al., 2018) and then execute goals from the ALFRED benchmark (Shridhar et al., 2020) in a rich visual environment. ALFWorld enables the creation of a new BUTLER agent whose abstract knowledge, learned in TextWorld, corresponds directly to concrete, visually grounded actions. In turn, as we demonstrate empirically, this fosters better agent generalization than training only in the visually grounded environment. BUTLER's simple, modular design factors the problem to allow researchers to focus on models for improving every piece of the pipeline (language understanding, planning, navigation, and visual scene understanding).",
  "published": "2020-10-08",
  "updated": "2021-03-14",
  "year": "2020",
  "authors": [
   "Mohit Shridhar",
   "Xingdi Yuan",
   "Marc-Alexandre C\u00f4t\u00e9",
   "Yonatan Bisk",
   "Adam Trischler",
   "Matthew Hausknecht"
  ],
  "author_count": 6,
  "categories": [
   "cs.CL",
   "cs.AI",
   "cs.CV",
   "cs.LG",
   "cs.RO"
  ],
  "primary_category": "cs.CL",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 1184,
  "influential_citations": 255,
  "tldr": "ALFWorld, a simulator that enables agents to learn abstract, text-based policies in TextWorld and then execute goals from the ALFRED benchmark in a rich visual environment, enables the creation of a new BUTLER agent whose abstract knowledge corresponds directly to concrete, visually grounded actions.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Mohit Shridhar",
    "id": "33516562",
    "h_index": 15,
    "papers": 23
   },
   {
    "name": "Xingdi Yuan",
    "id": "2854297",
    "h_index": 26,
    "papers": 41
   },
   {
    "name": "Marc-Alexandre C\u00f4t\u00e9",
    "id": "40638665",
    "h_index": 26,
    "papers": 68
   },
   {
    "name": "Yonatan Bisk",
    "id": "3312309",
    "h_index": 46,
    "papers": 144
   },
   {
    "name": "Adam Trischler",
    "id": "3382568",
    "h_index": 34,
    "papers": 65
   },
   {
    "name": "Matthew J. Hausknecht",
    "id": "3308897",
    "h_index": 27,
    "papers": 50
   }
  ],
  "comment": "ICLR 2021; Data, code, and videos are available at alfworld.github.io",
  "topics": [
   "sim2real",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2010.03768v2",
  "pdf_url": "https://arxiv.org/pdf/2010.03768v2",
  "html_url": "https://arxiv.org/html/2010.03768v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "2010.02193",
  "slug": "mastering-atari-with-discrete-world-models",
  "title": "Mastering Atari with Discrete World Models",
  "abstract": "Intelligent agents need to generalize from past experience to achieve goals in complex environments. World models facilitate such generalization and allow learning behaviors from imagined outcomes to increase sample-efficiency. While learning world models from image inputs has recently become feasible for some tasks, modeling Atari games accurately enough to derive successful behaviors has remained an open challenge for many years. We introduce DreamerV2, a reinforcement learning agent that learns behaviors purely from predictions in the compact latent space of a powerful world model. The world model uses discrete representations and is trained separately from the policy. DreamerV2 constitutes the first agent that achieves human-level performance on the Atari benchmark of 55 tasks by learning behaviors inside a separately trained world model. With the same computational budget and wall-clock time, Dreamer V2 reaches 200M frames and surpasses the final performance of the top single-GPU agents IQN and Rainbow. DreamerV2 is also applicable to tasks with continuous actions, where it learns an accurate world model of a complex humanoid robot and solves stand-up and walking from only pixel inputs.",
  "published": "2020-10-05",
  "updated": "2022-02-12",
  "year": "2020",
  "authors": [
   "Danijar Hafner",
   "Timothy Lillicrap",
   "Mohammad Norouzi",
   "Jimmy Ba"
  ],
  "author_count": 4,
  "categories": [
   "cs.LG",
   "cs.AI",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 1321,
  "influential_citations": 173,
  "tldr": "DreamerV2 constitutes the first agent that achieves human-level performance on the Atari benchmark of 55 tasks by learning behaviors inside a separately trained world model, and exceeds the final performance of the top single-GPU agents IQN and Rainbow.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Danijar Hafner",
    "id": "35006479",
    "h_index": 25,
    "papers": 47
   },
   {
    "name": "T. Lillicrap",
    "id": "2542999",
    "h_index": 69,
    "papers": 154
   },
   {
    "name": "Mohammad Norouzi",
    "id": "144739074",
    "h_index": 71,
    "papers": 179
   },
   {
    "name": "Jimmy Ba",
    "id": "2503659",
    "h_index": 46,
    "papers": 84
   }
  ],
  "comment": "Published at ICLR 2021. Website: https://danijar.com/dreamerv2",
  "topics": [
   "world-models",
   "humanoids",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2010.02193v4",
  "pdf_url": "https://arxiv.org/pdf/2010.02193v4",
  "html_url": "https://arxiv.org/html/2010.02193v4",
  "code_url": null,
  "presented": [
   {
    "club": "saturday-robotics",
    "club_name": "Saturday Robotics",
    "session_number": 7,
    "session_title": "Robotics & World Models Reading Club 07: Learning to Dream: World Models, Imagination, Path to Foundation Models for Control \u2014 Los Altos",
    "date_text": "Saturday, May 9, 2026 | 2:00 PM \u2013 5:00 PM",
    "city": "",
    "url": "https://lu.ma/srhe0vuo",
    "listed_as": "Mastering Atari with Discrete World Models (2020)"
   }
  ],
  "club_note": "Introduces discrete latent states (categorical RSSM)",
  "featured": true,
  "signal": 7.5
 },
 {
  "id": "2009.04416",
  "slug": "phasic-policy-gradient",
  "title": "Phasic Policy Gradient",
  "abstract": "We introduce Phasic Policy Gradient (PPG), a reinforcement learning framework which modifies traditional on-policy actor-critic methods by separating policy and value function training into distinct phases. In prior methods, one must choose between using a shared network or separate networks to represent the policy and value function. Using separate networks avoids interference between objectives, while using a shared network allows useful features to be shared. PPG is able to achieve the best of both worlds by splitting optimization into two phases, one that advances training and one that distills features. PPG also enables the value function to be more aggressively optimized with a higher level of sample reuse. Compared to PPO, we find that PPG significantly improves sample efficiency on the challenging Procgen Benchmark.",
  "published": "2020-09-09",
  "updated": "2020-09-09",
  "year": "2020",
  "authors": [
   "Karl Cobbe",
   "Jacob Hilton",
   "Oleg Klimov",
   "John Schulman"
  ],
  "author_count": 4,
  "categories": [
   "cs.LG",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "ICML",
  "venue_source": "semantic-scholar",
  "citations": 206,
  "influential_citations": 32,
  "tldr": "Phasic Policy Gradient, a reinforcement learning framework which modifies traditional on-policy actor-critic methods by separating policy and value function training into distinct phases, significantly improves sample efficiency on the challenging Procgen Benchmark.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "K. Cobbe",
    "id": "6062736",
    "h_index": 11,
    "papers": 53
   },
   {
    "name": "Jacob Hilton",
    "id": "144890163",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "Oleg Klimov",
    "id": "2067138712",
    "h_index": 6,
    "papers": 6
   },
   {
    "name": "John Schulman",
    "id": "47971768",
    "h_index": 45,
    "papers": 69
   }
  ],
  "comment": "",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2009.04416v1",
  "pdf_url": "https://arxiv.org/pdf/2009.04416v1",
  "html_url": "https://arxiv.org/html/2009.04416v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.82
 },
 {
  "id": "2009.01791",
  "slug": "action-and-perception-as-divergence-minimization",
  "title": "Action and Perception as Divergence Minimization",
  "abstract": "To learn directed behaviors in complex environments, intelligent agents need to optimize objective functions. Various objectives are known for designing artificial agents, including task rewards and intrinsic motivation. However, it is unclear how the known objectives relate to each other, which objectives remain yet to be discovered, and which objectives better describe the behavior of humans. We introduce the Action Perception Divergence (APD), an approach for categorizing the space of possible objective functions for embodied agents. We show a spectrum that reaches from narrow to general objectives. While the narrow objectives correspond to domain-specific rewards as typical in reinforcement learning, the general objectives maximize information with the environment through latent variable models of input sequences. Intuitively, these agents use perception to align their beliefs with the world and use actions to align the world with their beliefs. They infer representations that are informative of past inputs, explore future inputs that are informative of their representations, and select actions or skills that maximally influence future inputs. This explains a wide range of unsupervised objectives from a single principle, including representation learning, information gain, empowerment, and skill discovery. Our findings suggest leveraging powerful world models for unsupervised exploration as a path toward highly adaptive agents that seek out large niches in their environments, rendering task rewards optional.",
  "published": "2020-09-03",
  "updated": "2022-02-13",
  "year": "2020",
  "authors": [
   "Danijar Hafner",
   "Pedro A. Ortega",
   "Jimmy Ba",
   "Thomas Parr",
   "Karl Friston",
   "Nicolas Heess"
  ],
  "author_count": 6,
  "categories": [
   "cs.AI",
   "cs.IT",
   "cs.LG",
   "stat.ML"
  ],
  "primary_category": "cs.AI",
  "venue": "",
  "venue_source": "",
  "citations": 60,
  "influential_citations": 3,
  "tldr": "A unified objective for action and perception of intelligent agents is introduced, and interpreting the target distribution as a latent variable model suggests powerful world models as a path toward highly adaptive agents that seek large niches in their environments, rendering task rewards optional.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Danijar Hafner",
    "id": "35006479",
    "h_index": 25,
    "papers": 47
   },
   {
    "name": "Pedro A. Ortega",
    "id": "145981974",
    "h_index": 24,
    "papers": 69
   },
   {
    "name": "Jimmy Ba",
    "id": "2503659",
    "h_index": 46,
    "papers": 84
   },
   {
    "name": "Thomas Parr",
    "id": "47363526",
    "h_index": 47,
    "papers": 117
   },
   {
    "name": "Karl J. Friston",
    "id": "1737497",
    "h_index": 255,
    "papers": 1450
   },
   {
    "name": "N. Heess",
    "id": "2801204",
    "h_index": 73,
    "papers": 192
   }
  ],
  "comment": "Website: https://danijar.com/apd",
  "topics": [
   "world-models",
   "rl-control",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2009.01791v3",
  "pdf_url": "https://arxiv.org/pdf/2009.01791v3",
  "html_url": "https://arxiv.org/html/2009.01791v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.79
 },
 {
  "id": "2009.01439",
  "slug": "learning-dexterous-grasping-with-object-centric-visual-affordances",
  "title": "Learning Dexterous Grasping with Object-Centric Visual Affordances",
  "abstract": "Dexterous robotic hands are appealing for their agility and human-like morphology, yet their high degree of freedom makes learning to manipulate challenging. We introduce an approach for learning dexterous grasping. Our key idea is to embed an object-centric visual affordance model within a deep reinforcement learning loop to learn grasping policies that favor the same object regions favored by people. Unlike traditional approaches that learn from human demonstration trajectories (e.g., hand joint sequences captured with a glove), the proposed prior is object-centric and image-based, allowing the agent to anticipate useful affordance regions for objects unseen during policy learning. We demonstrate our idea with a 30-DoF five-fingered robotic hand simulator on 40 objects from two datasets, where it successfully and efficiently learns policies for stable functional grasps. Our affordance-guided policies are significantly more effective, generalize better to novel objects, train 3 X faster than the baselines, and are more robust to noisy sensor readings and actuation. Our work offers a step towards manipulation agents that learn by watching how people use objects, without requiring state and action information about the human body. Project website: http://vision.cs.utexas.edu/projects/graff-dexterous-affordance-grasp",
  "published": "2020-09-03",
  "updated": "2021-06-16",
  "year": "2020",
  "authors": [
   "Priyanka Mandikal",
   "Kristen Grauman"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA",
  "venue_source": "semantic-scholar",
  "citations": 170,
  "influential_citations": 3,
  "tldr": "This work proposes an approach for learning dexterous grasping that embeds an object-centric visual affordance model within a deep reinforcement learning loop to learn grasping policies that favor the same object regions favored by people.",
  "doi": "10.1109/ICRA48506.2021.9561802",
  "oa_pdf": "https://arxiv.org/pdf/2009.01439",
  "s2_authors": [
   {
    "name": "Priyanka Mandikal",
    "id": "51126291",
    "h_index": 10,
    "papers": 13
   },
   {
    "name": "K. Grauman",
    "id": "1794409",
    "h_index": 99,
    "papers": 295
   }
  ],
  "comment": "Accepted at ICRA 2021",
  "topics": [
   "dexterous-manipulation",
   "egocentric-data",
   "sim2real",
   "rl-control",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2009.01439v2",
  "pdf_url": "https://arxiv.org/pdf/2009.01439v2",
  "html_url": "https://arxiv.org/html/2009.01439v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.73
 },
 {
  "id": "2008.04899",
  "slug": "visual-imitation-made-easy",
  "title": "Visual Imitation Made Easy",
  "abstract": "Visual imitation learning provides a framework for learning complex manipulation behaviors by leveraging human demonstrations. However, current interfaces for imitation such as kinesthetic teaching or teleoperation prohibitively restrict our ability to efficiently collect large-scale data in the wild. Obtaining such diverse demonstration data is paramount for the generalization of learned skills to novel scenarios. In this work, we present an alternate interface for imitation that simplifies the data collection process while allowing for easy transfer to robots. We use commercially available reacher-grabber assistive tools both as a data collection device and as the robot's end-effector. To extract action information from these visual demonstrations, we use off-the-shelf Structure from Motion (SfM) techniques in addition to training a finger detection network. We experimentally evaluate on two challenging tasks: non-prehensile pushing and prehensile stacking, with 1000 diverse demonstrations for each task. For both tasks, we use standard behavior cloning to learn executable policies from the previously collected offline demonstrations. To improve learning performance, we employ a variety of data augmentations and provide an extensive analysis of its effects. Finally, we demonstrate the utility of our interface by evaluating on real robotic scenarios with previously unseen objects and achieve a 87% success rate on pushing and a 62% success rate on stacking. Robot videos are available at https://dhiraj100892.github.io/Visual-Imitation-Made-Easy.",
  "published": "2020-08-11",
  "updated": "2020-08-11",
  "year": "2020",
  "authors": [
   "Sarah Young",
   "Dhiraj Gandhi",
   "Shubham Tulsiani",
   "Abhinav Gupta",
   "Pieter Abbeel",
   "Lerrel Pinto"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 164,
  "influential_citations": 3,
  "tldr": "This work presents an alternate interface for imitation that simplifies the data collection process while allowing for easy transfer to robots and uses commercially available reacher-grabber assistive tools both as a data collection device and as the robot's end-effector.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "S. Young",
    "id": "145137801",
    "h_index": 7,
    "papers": 141
   },
   {
    "name": "Dhiraj Gandhi",
    "id": "3393217",
    "h_index": 13,
    "papers": 18
   },
   {
    "name": "Shubham Tulsiani",
    "id": "2757335",
    "h_index": 45,
    "papers": 98
   },
   {
    "name": "A. Gupta",
    "id": "1726095131",
    "h_index": 96,
    "papers": 211
   },
   {
    "name": "P. Abbeel",
    "id": "1689992",
    "h_index": 165,
    "papers": 596
   },
   {
    "name": "Lerrel Pinto",
    "id": "34026610",
    "h_index": 41,
    "papers": 70
   }
  ],
  "comment": "",
  "topics": [
   "egocentric-data",
   "imitation-diffusion",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2008.04899v1",
  "pdf_url": "https://arxiv.org/pdf/2008.04899v1",
  "html_url": "https://arxiv.org/html/2008.04899v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.72
 },
 {
  "id": "2008.04442",
  "slug": "spatio-temporal-attention-model-for-tactile-texture-recognition",
  "title": "Spatio-temporal Attention Model for Tactile Texture Recognition",
  "abstract": "Recently, tactile sensing has attracted great interest in robotics, especially for facilitating exploration of unstructured environments and effective manipulation. A detailed understanding of the surface textures via tactile sensing is essential for many of these tasks. Previous works on texture recognition using camera based tactile sensors have been limited to treating all regions in one tactile image or all samples in one tactile sequence equally, which includes much irrelevant or redundant information. In this paper, we propose a novel Spatio-Temporal Attention Model (STAM) for tactile texture recognition, which is the very first of its kind to our best knowledge. The proposed STAM pays attention to both spatial focus of each single tactile texture and the temporal correlation of a tactile sequence. In the experiments to discriminate 100 different fabric textures, the spatially and temporally selective attention has resulted in a significant improvement of the recognition accuracy, by up to 18.8%, compared to the non-attention based models. Specifically, after introducing noisy data that is collected before the contact happens, our proposed STAM can learn the salient features efficiently and the accuracy can increase by 15.23% on average compared with the CNN based baseline approach. The improved tactile texture perception can be applied to facilitate robot tasks like grasping and manipulation.",
  "published": "2020-08-10",
  "updated": "2020-08-10",
  "year": "2020",
  "authors": [
   "Guanqun Cao",
   "Yi Zhou",
   "Danushka Bollegala",
   "Shan Luo"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS",
  "venue_source": "semantic-scholar",
  "citations": 60,
  "influential_citations": 4,
  "tldr": "A novel Spatio-Temporal Attention Model (STAM) for tactile texture recognition, which is the very first of its kind to the best knowledge, is proposed and can be applied to facilitate robot tasks like grasping and manipulation.",
  "doi": "10.1109/IROS45743.2020.9341333",
  "oa_pdf": "https://arxiv.org/pdf/2008.04442",
  "s2_authors": [
   {
    "name": "Guanqun Cao",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Yi Zhou",
    "id": "32066669",
    "h_index": 7,
    "papers": 21
   },
   {
    "name": "Danushka Bollegala",
    "id": "2720656",
    "h_index": 37,
    "papers": 182
   },
   {
    "name": "Shan Luo",
    "id": "145524951",
    "h_index": 26,
    "papers": 65
   }
  ],
  "comment": "7 pages, accepted by International Conference on Intelligent Robots and Systems 2020",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2008.04442v1",
  "pdf_url": "https://arxiv.org/pdf/2008.04442v1",
  "html_url": "https://arxiv.org/html/2008.04442v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.29
 },
 {
  "id": "2007.14535",
  "slug": "dreaming-model-based-reinforcement-learning-by-latent-imagination-with",
  "title": "Dreaming: Model-based Reinforcement Learning by Latent Imagination without Reconstruction",
  "abstract": "In the present paper, we propose a decoder-free extension of Dreamer, a leading model-based reinforcement learning (MBRL) method from pixels. Dreamer is a sample- and cost-efficient solution to robot learning, as it is used to train latent state-space models based on a variational autoencoder and to conduct policy optimization by latent trajectory imagination. However, this autoencoding based approach often causes object vanishing, in which the autoencoder fails to perceives key objects for solving control tasks, and thus significantly limiting Dreamer's potential. This work aims to relieve this Dreamer's bottleneck and enhance its performance by means of removing the decoder. For this purpose, we firstly derive a likelihood-free and InfoMax objective of contrastive learning from the evidence lower bound of Dreamer. Secondly, we incorporate two components, (i) independent linear dynamics and (ii) the random crop data augmentation, to the learning scheme so as to improve the training performance. In comparison to Dreamer and other recent model-free reinforcement learning methods, our newly devised Dreamer with InfoMax and without generative decoder (Dreaming) achieves the best scores on 5 difficult simulated robotics tasks, in which Dreamer suffers from object vanishing.",
  "published": "2020-07-29",
  "updated": "2021-03-12",
  "year": "2020",
  "authors": [
   "Masashi Okada",
   "Tadahiro Taniguchi"
  ],
  "author_count": 2,
  "categories": [
   "cs.LG",
   "cs.AI",
   "eess.SY",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "ICRA",
  "venue_source": "semantic-scholar",
  "citations": 108,
  "influential_citations": 3,
  "tldr": "This work aims to relieve this Dreamer's bottleneck and enhance its performance by means of removing the decoder, and derives a likelihood- free and InfoMax objective of contrastive learning from the evidence lower bound of Dreamer.",
  "doi": "10.1109/ICRA48506.2021.9560734",
  "oa_pdf": "https://arxiv.org/pdf/2007.14535",
  "s2_authors": [
   {
    "name": "Masashi Okada",
    "id": "2114306014",
    "h_index": 9,
    "papers": 22
   },
   {
    "name": "T. Taniguchi",
    "id": "1684099",
    "h_index": 30,
    "papers": 259
   }
  ],
  "comment": "Accepted to ICRA2021. Camera ready version",
  "topics": [
   "world-models",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2007.14535v2",
  "pdf_url": "https://arxiv.org/pdf/2007.14535v2",
  "html_url": "https://arxiv.org/html/2007.14535v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.54
 },
 {
  "id": "2007.05929",
  "slug": "data-efficient-reinforcement-learning-with-self-predictive-representat",
  "title": "Data-Efficient Reinforcement Learning with Self-Predictive Representations",
  "abstract": "While deep reinforcement learning excels at solving tasks where large amounts of data can be collected through virtually unlimited interaction with the environment, learning from limited interaction remains a key challenge. We posit that an agent can learn more efficiently if we augment reward maximization with self-supervised objectives based on structure in its visual input and sequential interaction with the environment. Our method, Self-Predictive Representations(SPR), trains an agent to predict its own latent state representations multiple steps into the future. We compute target representations for future states using an encoder which is an exponential moving average of the agent's parameters and we make predictions using a learned transition model. On its own, this future prediction objective outperforms prior methods for sample-efficient deep RL from pixels. We further improve performance by adding data augmentation to the future prediction loss, which forces the agent's representations to be consistent across multiple views of an observation. Our full self-supervised objective, which combines future prediction and data augmentation, achieves a median human-normalized score of 0.415 on Atari in a setting limited to 100k steps of environment interaction, which represents a 55% relative improvement over the previous state-of-the-art. Notably, even in this limited data regime, SPR exceeds expert human scores on 7 out of 26 games. The code associated with this work is available at https://github.com/mila-iqia/spr",
  "published": "2020-07-12",
  "updated": "2021-05-20",
  "year": "2020",
  "authors": [
   "Max Schwarzer",
   "Ankesh Anand",
   "Rishab Goel",
   "R Devon Hjelm",
   "Aaron Courville",
   "Philip Bachman"
  ],
  "author_count": 6,
  "categories": [
   "cs.LG",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 436,
  "influential_citations": 66,
  "tldr": "This work trains an agent to predict its own latent state representations multiple steps into the future using an encoder which is an exponential moving average of the agent's parameters and a learned transition model, and improves performance by adding data augmentation to the future prediction loss.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Max Schwarzer",
    "id": "51881243",
    "h_index": 12,
    "papers": 21
   },
   {
    "name": "Ankesh Anand",
    "id": "12679121",
    "h_index": 14,
    "papers": 28
   },
   {
    "name": "Rishab Goel",
    "id": "46186660",
    "h_index": 10,
    "papers": 20
   },
   {
    "name": "R. Devon Hjelm",
    "id": "40482726",
    "h_index": 28,
    "papers": 59
   },
   {
    "name": "Aaron C. Courville",
    "id": "1760871",
    "h_index": 94,
    "papers": 241
   },
   {
    "name": "Philip Bachman",
    "id": "143902541",
    "h_index": 20,
    "papers": 40
   }
  ],
  "comment": "The first two authors contributed equally to this work. v4 includes new ablations and reformatting for ICLR camera ready",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2007.05929v4",
  "pdf_url": "https://arxiv.org/pdf/2007.05929v4",
  "html_url": "https://arxiv.org/html/2007.05929v4",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.14
 },
 {
  "id": "2007.04976",
  "slug": "one-policy-to-control-them-all-shared-modular-policies-for-agent-agnos",
  "title": "One Policy to Control Them All: Shared Modular Policies for Agent-Agnostic Control",
  "abstract": "Reinforcement learning is typically concerned with learning control policies tailored to a particular agent. We investigate whether there exists a single global policy that can generalize to control a wide variety of agent morphologies -- ones in which even dimensionality of state and action spaces changes. We propose to express this global policy as a collection of identical modular neural networks, dubbed as Shared Modular Policies (SMP), that correspond to each of the agent's actuators. Every module is only responsible for controlling its corresponding actuator and receives information from only its local sensors. In addition, messages are passed between modules, propagating information between distant modules. We show that a single modular policy can successfully generate locomotion behaviors for several planar agents with different skeletal structures such as monopod hoppers, quadrupeds, bipeds, and generalize to variants not seen during training -- a process that would normally require training and manual hyperparameter tuning for each morphology. We observe that a wide variety of drastically diverse locomotion styles across morphologies as well as centralized coordination emerges via message passing between decentralized modules purely from the reinforcement learning objective. Videos and code at https://huangwl18.github.io/modular-rl/",
  "published": "2020-07-09",
  "updated": "2020-07-09",
  "year": "2020",
  "authors": [
   "Wenlong Huang",
   "Igor Mordatch",
   "Deepak Pathak"
  ],
  "author_count": 3,
  "categories": [
   "cs.LG",
   "cs.CV",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "ICML",
  "venue_source": "semantic-scholar",
  "citations": 230,
  "influential_citations": 18,
  "tldr": "It is shown that a single modular policy can successfully generate locomotion behaviors for several planar agents with different skeletal structures such as monopod hoppers, quadrupeds, bipeds, and generalize to variants not seen during training -- a process that would normally require training and manual hyperparameter tuning for each morphology.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Wenlong Huang",
    "id": "2158105356",
    "h_index": 10,
    "papers": 11
   },
   {
    "name": "Igor Mordatch",
    "id": "2080746",
    "h_index": 34,
    "papers": 49
   },
   {
    "name": "Deepak Pathak",
    "id": "38236002",
    "h_index": 34,
    "papers": 78
   }
  ],
  "comment": "Accepted at ICML 2020. Videos and code at https://huangwl18.github.io/modular-rl/",
  "topics": [
   "humanoids",
   "rl-control",
   "hardware-codesign"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2007.04976v1",
  "pdf_url": "https://arxiv.org/pdf/2007.04976v1",
  "html_url": "https://arxiv.org/html/2007.04976v1",
  "code_url": "https://huangwl18.github.io/modular-rl/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.86
 },
 {
  "id": "2007.03898",
  "slug": "nvae-a-deep-hierarchical-variational-autoencoder",
  "title": "NVAE: A Deep Hierarchical Variational Autoencoder",
  "abstract": "Normalizing flows, autoregressive models, variational autoencoders (VAEs), and deep energy-based models are among competing likelihood-based frameworks for deep generative learning. Among them, VAEs have the advantage of fast and tractable sampling and easy-to-access encoding networks. However, they are currently outperformed by other models such as normalizing flows and autoregressive models. While the majority of the research in VAEs is focused on the statistical challenges, we explore the orthogonal direction of carefully designing neural architectures for hierarchical VAEs. We propose Nouveau VAE (NVAE), a deep hierarchical VAE built for image generation using depth-wise separable convolutions and batch normalization. NVAE is equipped with a residual parameterization of Normal distributions and its training is stabilized by spectral regularization. We show that NVAE achieves state-of-the-art results among non-autoregressive likelihood-based models on the MNIST, CIFAR-10, CelebA 64, and CelebA HQ datasets and it provides a strong baseline on FFHQ. For example, on CIFAR-10, NVAE pushes the state-of-the-art from 2.98 to 2.91 bits per dimension, and it produces high-quality images on CelebA HQ. To the best of our knowledge, NVAE is the first successful VAE applied to natural images as large as 256$\\times$256 pixels. The source code is available at https://github.com/NVlabs/NVAE .",
  "published": "2020-07-08",
  "updated": "2021-01-08",
  "year": "2020",
  "authors": [
   "Arash Vahdat",
   "Jan Kautz"
  ],
  "author_count": 2,
  "categories": [
   "stat.ML",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "stat.ML",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 1131,
  "influential_citations": 125,
  "tldr": "NVAE is the first successful VAE applied to natural images as large as 256$\\times$256 pixels and achieves state-of-the-art results among non-autoregressive likelihood-based models on the MNIST, CIFAR-10, CelebA 64, and CelebA HQ datasets and it provides a strong baseline on FFHQ.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Arash Vahdat",
    "id": "3214848",
    "h_index": 30,
    "papers": 57
   },
   {
    "name": "Jan Kautz",
    "id": "2376331464",
    "h_index": 15,
    "papers": 18
   }
  ],
  "comment": "Neural Information Processing Systems (NeurIPS) 2020 (spotlight)",
  "topics": [
   "video-generation"
  ],
  "orgs": [
   "NVIDIA"
  ],
  "abs_url": "https://arxiv.org/abs/2007.03898v3",
  "pdf_url": "https://arxiv.org/pdf/2007.03898v3",
  "html_url": "https://arxiv.org/html/2007.03898v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.0
 },
 {
  "id": "2006.13256",
  "slug": "rescaling-egocentric-vision",
  "title": "Rescaling Egocentric Vision",
  "abstract": "This paper introduces the pipeline to extend the largest dataset in egocentric vision, EPIC-KITCHENS. The effort culminates in EPIC-KITCHENS-100, a collection of 100 hours, 20M frames, 90K actions in 700 variable-length videos, capturing long-term unscripted activities in 45 environments, using head-mounted cameras. Compared to its previous version, EPIC-KITCHENS-100 has been annotated using a novel pipeline that allows denser (54% more actions per minute) and more complete annotations of fine-grained actions (+128% more action segments). This collection enables new challenges such as action detection and evaluating the \"test of time\" - i.e. whether models trained on data collected in 2018 can generalise to new footage collected two years later. The dataset is aligned with 6 challenges: action recognition (full and weak supervision), action detection, action anticipation, cross-modal retrieval (from captions), as well as unsupervised domain adaptation for action recognition. For each challenge, we define the task, provide baselines and evaluation metrics",
  "published": "2020-06-23",
  "updated": "2021-09-17",
  "year": "2020",
  "authors": [
   "Dima Damen",
   "Hazel Doughty",
   "Giovanni Maria Farinella",
   "Antonino Furnari",
   "Evangelos Kazakos",
   "Jian Ma",
   "Davide Moltisanti",
   "Jonathan Munro",
   "Toby Perrett",
   "Will Price",
   "Michael Wray"
  ],
  "author_count": 11,
  "categories": [
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 756,
  "influential_citations": 105,
  "tldr": "The pipeline to extend the largest dataset in egocentric vision, EPIC-KITCHENS, using a novel pipeline that allows denser and more complete annotations of fine-grained actions and enables new challenges such as action detection and evaluating the \u201ctest of time\u201d\u2014i.e. whether models trained on data collected in 2018 can generalise to new footage collected two years later.",
  "doi": "10.5523/bris.2g1n6qdydwa9u22shpxqzp0t8m",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "D. Damen",
    "id": "145089978",
    "h_index": 45,
    "papers": 201
   },
   {
    "name": "Hazel Doughty",
    "id": "28798386",
    "h_index": 17,
    "papers": 52
   },
   {
    "name": "G. Farinella",
    "id": "1729739",
    "h_index": 37,
    "papers": 320
   },
   {
    "name": "Antonino Furnari",
    "id": "1792681",
    "h_index": 29,
    "papers": 152
   },
   {
    "name": "E. Kazakos",
    "id": "12387007",
    "h_index": 19,
    "papers": 80
   },
   {
    "name": "Jian Ma",
    "id": "153155900",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "D. Moltisanti",
    "id": "3420479",
    "h_index": 10,
    "papers": 24
   },
   {
    "name": "Jonathan Munro",
    "id": "47077615",
    "h_index": 8,
    "papers": 19
   },
   {
    "name": "Toby Perrett",
    "id": "2682004",
    "h_index": 12,
    "papers": 33
   },
   {
    "name": "Will Price",
    "id": "50065546",
    "h_index": 8,
    "papers": 17
   },
   {
    "name": "Michael Wray",
    "id": "145032628",
    "h_index": 15,
    "papers": 51
   }
  ],
  "comment": "Accepted at the International Journal of Computer Vision (IJCV). Dataset available from: http://epic-kitchens.github.io/",
  "topics": [
   "egocentric-data",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2006.13256v4",
  "pdf_url": "https://arxiv.org/pdf/2006.13256v4",
  "html_url": "https://arxiv.org/html/2006.13256v4",
  "code_url": "https://epic-kitchens.github.io/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.88
 },
 {
  "id": "2006.11239",
  "slug": "denoising-diffusion-probabilistic-models",
  "title": "Denoising Diffusion Probabilistic Models",
  "abstract": "We present high quality image synthesis results using diffusion probabilistic models, a class of latent variable models inspired by considerations from nonequilibrium thermodynamics. Our best results are obtained by training on a weighted variational bound designed according to a novel connection between diffusion probabilistic models and denoising score matching with Langevin dynamics, and our models naturally admit a progressive lossy decompression scheme that can be interpreted as a generalization of autoregressive decoding. On the unconditional CIFAR10 dataset, we obtain an Inception score of 9.46 and a state-of-the-art FID score of 3.17. On 256x256 LSUN, we obtain sample quality similar to ProgressiveGAN. Our implementation is available at https://github.com/hojonathanho/diffusion",
  "published": "2020-06-19",
  "updated": "2020-12-16",
  "year": "2020",
  "authors": [
   "Jonathan Ho",
   "Ajay Jain",
   "Pieter Abbeel"
  ],
  "author_count": 3,
  "categories": [
   "cs.LG",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 34349,
  "influential_citations": 4855,
  "tldr": "High quality image synthesis results are presented using diffusion probabilistic models, a class of latent variable models inspired by considerations from nonequilibrium thermodynamics, which naturally admit a progressive lossy decompression scheme that can be interpreted as a generalization of autoregressive decoding.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jonathan Ho",
    "id": "2126278",
    "h_index": 25,
    "papers": 32
   },
   {
    "name": "Ajay Jain",
    "id": "1623995772",
    "h_index": 11,
    "papers": 18
   },
   {
    "name": "P. Abbeel",
    "id": "1689992",
    "h_index": 165,
    "papers": 596
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2006.11239v2",
  "pdf_url": "https://arxiv.org/pdf/2006.11239v2",
  "html_url": "https://arxiv.org/html/2006.11239v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "2006.10178",
  "slug": "variational-state-space-models-for-localisation-and-dense-3d-mapping-i",
  "title": "Variational State-Space Models for Localisation and Dense 3D Mapping in 6 DoF",
  "abstract": "We solve the problem of 6-DoF localisation and 3D dense reconstruction in spatial environments as approximate Bayesian inference in a deep state-space model. Our approach leverages both learning and domain knowledge from multiple-view geometry and rigid-body dynamics. This results in an expressive predictive model of the world, often missing in current state-of-the-art visual SLAM solutions. The combination of variational inference, neural networks and a differentiable raycaster ensures that our model is amenable to end-to-end gradient-based optimisation. We evaluate our approach on realistic unmanned aerial vehicle flight data, nearing the performance of state-of-the-art visual-inertial odometry systems. We demonstrate the applicability of the model to generative prediction and planning.",
  "published": "2020-06-17",
  "updated": "2021-03-15",
  "year": "2020",
  "authors": [
   "Atanas Mirchev",
   "Baris Kayalibay",
   "Patrick van der Smagt",
   "Justin Bayer"
  ],
  "author_count": 4,
  "categories": [
   "stat.ML",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "stat.ML",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 11,
  "influential_citations": 0,
  "tldr": "This principled treatment of uncertainty and probabilistic inference overcomes the shortcoming of current state-of-the-art solutions to rely on heavily engineered, heterogeneous pipelines and enables the use of neural networks for system identification and a differentiable raycaster for the emission model.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Atanas Mirchev",
    "id": "3293265",
    "h_index": 7,
    "papers": 15
   },
   {
    "name": "Baris Kayalibay",
    "id": "8794101",
    "h_index": 5,
    "papers": 10
   },
   {
    "name": "Patrick van der Smagt",
    "id": "1715782",
    "h_index": 44,
    "papers": 211
   },
   {
    "name": "Justin Bayer",
    "id": "145040409",
    "h_index": 18,
    "papers": 51
   }
  ],
  "comment": "Update for ICLR2021",
  "topics": [
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2006.10178v3",
  "pdf_url": "https://arxiv.org/pdf/2006.10178v3",
  "html_url": "https://arxiv.org/html/2006.10178v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.58
 },
 {
  "id": "2006.07733",
  "slug": "bootstrap-your-own-latent-a-new-approach-to-self-supervised-learning",
  "title": "Bootstrap your own latent: A new approach to self-supervised Learning",
  "abstract": "We introduce Bootstrap Your Own Latent (BYOL), a new approach to self-supervised image representation learning. BYOL relies on two neural networks, referred to as online and target networks, that interact and learn from each other. From an augmented view of an image, we train the online network to predict the target network representation of the same image under a different augmented view. At the same time, we update the target network with a slow-moving average of the online network. While state-of-the art methods rely on negative pairs, BYOL achieves a new state of the art without them. BYOL reaches $74.3\\%$ top-1 classification accuracy on ImageNet using a linear evaluation with a ResNet-50 architecture and $79.6\\%$ with a larger ResNet. We show that BYOL performs on par or better than the current state of the art on both transfer and semi-supervised benchmarks. Our implementation and pretrained models are given on GitHub.",
  "published": "2020-06-13",
  "updated": "2020-09-10",
  "year": "2020",
  "authors": [
   "Jean-Bastien Grill",
   "Florian Strub",
   "Florent Altch\u00e9",
   "Corentin Tallec",
   "Pierre H. Richemond",
   "Elena Buchatskaya",
   "Carl Doersch",
   "Bernardo Avila Pires",
   "Zhaohan Daniel Guo",
   "Mohammad Gheshlaghi Azar",
   "Bilal Piot",
   "Koray Kavukcuoglu",
   "R\u00e9mi Munos",
   "Michal Valko"
  ],
  "author_count": 14,
  "categories": [
   "cs.LG",
   "cs.CV",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 8952,
  "influential_citations": 1332,
  "tldr": "This work introduces Bootstrap Your Own Latent (BYOL), a new approach to self-supervised image representation learning that performs on par or better than the current state of the art on both transfer and semi- supervised benchmarks.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jean-Bastien Grill",
    "id": "145840757",
    "h_index": 15,
    "papers": 34
   },
   {
    "name": "Florian Strub",
    "id": "3367628",
    "h_index": 25,
    "papers": 43
   },
   {
    "name": "Florent Altch'e",
    "id": "2064347514",
    "h_index": 7,
    "papers": 8
   },
   {
    "name": "Corentin Tallec",
    "id": "31803582",
    "h_index": 19,
    "papers": 25
   },
   {
    "name": "Pierre H. Richemond",
    "id": "16326904",
    "h_index": 14,
    "papers": 25
   },
   {
    "name": "Elena Buchatskaya",
    "id": "118801223",
    "h_index": 12,
    "papers": 36
   },
   {
    "name": "Carl Doersch",
    "id": "2786693",
    "h_index": 33,
    "papers": 57
   },
   {
    "name": "B. '. Pires",
    "id": "3429927",
    "h_index": 18,
    "papers": 32
   },
   {
    "name": "Z. Guo",
    "id": "3407143",
    "h_index": 13,
    "papers": 20
   },
   {
    "name": "M. G. Azar",
    "id": "37666967",
    "h_index": 32,
    "papers": 52
   },
   {
    "name": "Bilal Piot",
    "id": "1808897",
    "h_index": 44,
    "papers": 80
   },
   {
    "name": "K. Kavukcuoglu",
    "id": "2645384",
    "h_index": 76,
    "papers": 124
   },
   {
    "name": "R. Munos",
    "id": "1708654",
    "h_index": 90,
    "papers": 246
   },
   {
    "name": "Michal Valko",
    "id": "1806291",
    "h_index": 44,
    "papers": 182
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2006.07733v3",
  "pdf_url": "https://arxiv.org/pdf/2006.07733v3",
  "html_url": "https://arxiv.org/html/2006.07733v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "2006.06669",
  "slug": "understanding-human-hands-in-contact-at-internet-scale",
  "title": "Understanding Human Hands in Contact at Internet Scale",
  "abstract": "Hands are the central means by which humans manipulate their world and being able to reliably extract hand state information from Internet videos of humans engaged in their hands has the potential to pave the way to systems that can learn from petabytes of video data. This paper proposes steps towards this by inferring a rich representation of hands engaged in interaction method that includes: hand location, side, contact state, and a box around the object in contact. To support this effort, we gather a large-scale dataset of hands in contact with objects consisting of 131 days of footage as well as a 100K annotated hand-contact video frame dataset. The learned model on this dataset can serve as a foundation for hand-contact understanding in videos. We quantitatively evaluate it both on its own and in service of predicting and learning from 3D meshes of human hands.",
  "published": "2020-06-11",
  "updated": "2020-06-11",
  "year": "2020",
  "authors": [
   "Dandan Shan",
   "Jiaqi Geng",
   "Michelle Shu",
   "David F. Fouhey"
  ],
  "author_count": 4,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 414,
  "influential_citations": 78,
  "tldr": "A rich representation of hands engaged in interaction method that includes: hand location, side, contact state, and a box around the object in contact is inferred by inferring a large-scale dataset of hands in contact with objects.",
  "doi": "10.1109/cvpr42600.2020.00989",
  "oa_pdf": "https://arxiv.org/pdf/2006.06669",
  "s2_authors": [
   {
    "name": "Dandan Shan",
    "id": "2058873326",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Jiaqi Geng",
    "id": "2052582769",
    "h_index": 2,
    "papers": 6
   },
   {
    "name": "Michelle Shu",
    "id": "38826848",
    "h_index": 7,
    "papers": 9
   },
   {
    "name": "David F. Fouhey",
    "id": "2287942790",
    "h_index": 10,
    "papers": 20
   }
  ],
  "comment": "To appear at CVPR 2020 (Oral). Project and dataset webpage: http://fouheylab.eecs.umich.edu/~dandans/projects/100DOH/",
  "topics": [
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2006.06669v1",
  "pdf_url": "https://arxiv.org/pdf/2006.06669v1",
  "html_url": "https://arxiv.org/html/2006.06669v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.12
 },
 {
  "id": "2006.05990",
  "slug": "what-matters-in-on-policy-reinforcement-learning-a-large-scale-empiric",
  "title": "What Matters In On-Policy Reinforcement Learning? A Large-Scale Empirical Study",
  "abstract": "In recent years, on-policy reinforcement learning (RL) has been successfully applied to many different continuous control tasks. While RL algorithms are often conceptually simple, their state-of-the-art implementations take numerous low- and high-level design decisions that strongly affect the performance of the resulting agents. Those choices are usually not extensively discussed in the literature, leading to discrepancy between published descriptions of algorithms and their implementations. This makes it hard to attribute progress in RL and slows down overall progress [Engstrom'20]. As a step towards filling that gap, we implement >50 such ``choices'' in a unified on-policy RL framework, allowing us to investigate their impact in a large-scale empirical study. We train over 250'000 agents in five continuous control environments of different complexity and provide insights and practical recommendations for on-policy training of RL agents.",
  "published": "2020-06-10",
  "updated": "2020-06-10",
  "year": "2020",
  "authors": [
   "Marcin Andrychowicz",
   "Anton Raichuk",
   "Piotr Sta\u0144czyk",
   "Manu Orsini",
   "Sertan Girgin",
   "Raphael Marinier",
   "L\u00e9onard Hussenot",
   "Matthieu Geist",
   "Olivier Pietquin",
   "Marcin Michalski",
   "Sylvain Gelly",
   "Olivier Bachem"
  ],
  "author_count": 12,
  "categories": [
   "cs.LG",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 288,
  "influential_citations": 20,
  "tldr": "This work implements >50 such ``choices'' in a unified on-policy RL framework, allowing them to investigate their impact in a large-scale empirical study and provides insights and practical recommendations for on- policy training of RL agents.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Marcin Andrychowicz",
    "id": "2206490",
    "h_index": 24,
    "papers": 35
   },
   {
    "name": "Anton Raichuk",
    "id": "150918315",
    "h_index": 10,
    "papers": 12
   },
   {
    "name": "Piotr Sta'nczyk",
    "id": "2067024592",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Manu Orsini",
    "id": "1741487247",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Sertan Girgin",
    "id": "35022714",
    "h_index": 23,
    "papers": 78
   },
   {
    "name": "Rapha\u00ebl Marinier",
    "id": "52153018",
    "h_index": 13,
    "papers": 18
   },
   {
    "name": "L'eonard Hussenot",
    "id": "122562941",
    "h_index": 16,
    "papers": 24
   },
   {
    "name": "M. Geist",
    "id": "1737555",
    "h_index": 42,
    "papers": 192
   },
   {
    "name": "O. Pietquin",
    "id": "1721354",
    "h_index": 51,
    "papers": 270
   },
   {
    "name": "Marcin Michalski",
    "id": "145605490",
    "h_index": 10,
    "papers": 12
   },
   {
    "name": "S. Gelly",
    "id": "1802148",
    "h_index": 42,
    "papers": 105
   },
   {
    "name": "Olivier Bachem",
    "id": "1936951",
    "h_index": 41,
    "papers": 85
   }
  ],
  "comment": "",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2006.05990v1",
  "pdf_url": "https://arxiv.org/pdf/2006.05990v1",
  "html_url": "https://arxiv.org/html/2006.05990v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.46
 },
 {
  "id": "2006.00979",
  "slug": "acme-a-research-framework-for-distributed-reinforcement-learning",
  "title": "Acme: A Research Framework for Distributed Reinforcement Learning",
  "abstract": "Deep reinforcement learning (RL) has led to many recent and groundbreaking advances. However, these advances have often come at the cost of both increased scale in the underlying architectures being trained as well as increased complexity of the RL algorithms used to train them. These increases have in turn made it more difficult for researchers to rapidly prototype new ideas or reproduce published RL algorithms. To address these concerns this work describes Acme, a framework for constructing novel RL algorithms that is specifically designed to enable agents that are built using simple, modular components that can be used at various scales of execution. While the primary goal of Acme is to provide a framework for algorithm development, a secondary goal is to provide simple reference implementations of important or state-of-the-art algorithms. These implementations serve both as a validation of our design decisions as well as an important contribution to reproducibility in RL research. In this work we describe the major design decisions made within Acme and give further details as to how its components can be used to implement various algorithms. Our experiments provide baselines for a number of common and state-of-the-art algorithms as well as showing how these algorithms can be scaled up for much larger and more complex environments. This highlights one of the primary advantages of Acme, namely that it can be used to implement large, distributed RL algorithms that can run at massive scales while still maintaining the inherent readability of that implementation. This work presents a second version of the paper which coincides with an increase in modularity, additional emphasis on offline, imitation and learning from demonstrations algorithms, as well as various new agents implemented as part of Acme.",
  "published": "2020-06-01",
  "updated": "2022-09-20",
  "year": "2020",
  "authors": [
   "Matthew W. Hoffman",
   "Bobak Shahriari",
   "John Aslanides",
   "Gabriel Barth-Maron",
   "Nikola Momchev",
   "Danila Sinopalnikov",
   "Piotr Sta\u0144czyk",
   "Sabela Ramos",
   "Anton Raichuk",
   "Damien Vincent",
   "L\u00e9onard Hussenot",
   "Robert Dadashi",
   "Gabriel Dulac-Arnold",
   "Manu Orsini",
   "Alexis Jacq",
   "Johan Ferret",
   "Nino Vieillard",
   "Seyed Kamyar Seyed Ghasemipour",
   "Sertan Girgin",
   "Olivier Pietquin",
   "Feryal Behbahani",
   "Tamara Norman",
   "Abbas Abdolmaleki",
   "Albin Cassirer",
   "Fan Yang",
   "Kate Baumli",
   "Sarah Henderson",
   "Abe Friesen",
   "Ruba Haroun",
   "Alex Novikov",
   "Sergio G\u00f3mez Colmenarejo",
   "Serkan Cabi",
   "Caglar Gulcehre",
   "Tom Le Paine",
   "Srivatsan Srinivasan",
   "Andrew Cowie",
   "Ziyu Wang",
   "Bilal Piot",
   "Nando de Freitas"
  ],
  "author_count": 39,
  "categories": [
   "cs.LG",
   "cs.AI"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 250,
  "influential_citations": 24,
  "tldr": "It is shown that the design decisions behind Acme lead to agents that can be scaled both up and down and that, for the most part, greater levels of parallelization result in agents with equivalent performance, just faster.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Matthew W. Hoffman",
    "id": "3243579",
    "h_index": 23,
    "papers": 34
   },
   {
    "name": "Bobak Shahriari",
    "id": "2067577",
    "h_index": 15,
    "papers": 42
   },
   {
    "name": "John Aslanides",
    "id": "9958912",
    "h_index": 14,
    "papers": 19
   },
   {
    "name": "Gabriel Barth-Maron",
    "id": "1403998955",
    "h_index": 15,
    "papers": 29
   },
   {
    "name": "Feryal M. P. Behbahani",
    "id": "145124447",
    "h_index": 14,
    "papers": 24
   },
   {
    "name": "Tamara Norman",
    "id": "1734809451",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "A. Abdolmaleki",
    "id": "2799799",
    "h_index": 29,
    "papers": 95
   },
   {
    "name": "Albin Cassirer",
    "id": "51042571",
    "h_index": 9,
    "papers": 18
   },
   {
    "name": "Fan Yang",
    "id": "145338228",
    "h_index": 9,
    "papers": 17
   },
   {
    "name": "Kate Baumli",
    "id": "1734809439",
    "h_index": 9,
    "papers": 13
   },
   {
    "name": "Sarah Henderson",
    "id": "2057025033",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Alexander Novikov",
    "id": "2050212830",
    "h_index": 21,
    "papers": 25
   },
   {
    "name": "Sergio Gomez Colmenarejo",
    "id": "2016840",
    "h_index": 21,
    "papers": 24
   },
   {
    "name": "Serkan Cabi",
    "id": "12159303",
    "h_index": 19,
    "papers": 29
   },
   {
    "name": "Caglar Gulcehre",
    "id": "146372255",
    "h_index": 15,
    "papers": 27
   },
   {
    "name": "T. Paine",
    "id": "40470211",
    "h_index": 25,
    "papers": 43
   },
   {
    "name": "A. Cowie",
    "id": "143964037",
    "h_index": 9,
    "papers": 23
   },
   {
    "name": "Ziyun Wang",
    "id": "2117966548",
    "h_index": 34,
    "papers": 56
   },
   {
    "name": "Bilal Piot",
    "id": "1808897",
    "h_index": 44,
    "papers": 80
   },
   {
    "name": "Nando de Freitas",
    "id": "1737568",
    "h_index": 79,
    "papers": 193
   }
  ],
  "comment": "This work presents a second version of the paper which coincides with an increase in modularity, additional emphasis on offline, imitation and learning from demonstrations algorithms, as well as various new agents implemented as part of Acme",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2006.00979v2",
  "pdf_url": "https://arxiv.org/pdf/2006.00979v2",
  "html_url": "https://arxiv.org/html/2006.00979v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.4
 },
 {
  "id": "2006.00906",
  "slug": "center-of-mass-based-robust-grasp-planning-for-unknown-objects-using-t",
  "title": "Center-of-Mass-based Robust Grasp Planning for Unknown Objects Using Tactile-Visual Sensors",
  "abstract": "An unstable grasp pose can lead to slip, thus an unstable grasp pose can be predicted by slip detection. A regrasp is required afterwards to correct the grasp pose in order to finish the task. In this work, we propose a novel regrasp planner with multi-sensor modules to plan grasp adjustments with the feedback from a slip detector. Then a regrasp planner is trained to estimate the location of center of mass, which helps robots find an optimal grasp pose. The dataset in this work consists of 1 025 slip experiments and 1 347 regrasps collected by one pair of tactile sensors, an RGB-D camera and one Franka Emika robot arm equipped with joint force/torque sensors. We show that our algorithm can successfully detect and classify the slip for 5 unknown test objects with an accuracy of 76.88% and a regrasp planner increases the grasp success rate by 31.0% compared to the state-of-the-art vision-based grasping algorithm.",
  "published": "2020-06-01",
  "updated": "2020-06-01",
  "year": "2020",
  "authors": [
   "Qian Feng",
   "Zhaopeng Chen",
   "Jun Deng",
   "Chunhui Gao",
   "Jianwei Zhang",
   "Alois Knoll"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA",
  "venue_source": "semantic-scholar",
  "citations": 24,
  "influential_citations": 0,
  "tldr": "A novel regrasp planner with multi-sensor modules to plan grasp adjustments with the feedback from a slip detector is proposed, which is trained to estimate the location of center of mass, which helps robots find an optimal grasp pose.",
  "doi": "10.1109/ICRA40945.2020.9196815",
  "oa_pdf": "https://mediatum.ub.tum.de/1586185",
  "s2_authors": [
   {
    "name": "Qian Feng",
    "id": "2068043370",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Zhaopeng Chen",
    "id": "1720777196",
    "h_index": 20,
    "papers": 48
   },
   {
    "name": "Jun Deng",
    "id": "2153368943",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Chunhui Gao",
    "id": "1477970830",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Jianwei Zhang",
    "id": "2108091177",
    "h_index": 9,
    "papers": 19
   },
   {
    "name": "A. Knoll",
    "id": "143873832",
    "h_index": 56,
    "papers": 879
   }
  ],
  "comment": "6 pages + references",
  "topics": [
   "dexterous-manipulation",
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2006.00906v1",
  "pdf_url": "https://arxiv.org/pdf/2006.00906v1",
  "html_url": "https://arxiv.org/html/2006.00906v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.9
 },
 {
  "id": "2005.13239",
  "slug": "mopo-model-based-offline-policy-optimization",
  "title": "MOPO: Model-based Offline Policy Optimization",
  "abstract": "Offline reinforcement learning (RL) refers to the problem of learning policies entirely from a large batch of previously collected data. This problem setting offers the promise of utilizing such datasets to acquire policies without any costly or dangerous active exploration. However, it is also challenging, due to the distributional shift between the offline training data and those states visited by the learned policy. Despite significant recent progress, the most successful prior methods are model-free and constrain the policy to the support of data, precluding generalization to unseen states. In this paper, we first observe that an existing model-based RL algorithm already produces significant gains in the offline setting compared to model-free approaches. However, standard model-based RL methods, designed for the online setting, do not provide an explicit mechanism to avoid the offline setting's distributional shift issue. Instead, we propose to modify the existing model-based RL methods by applying them with rewards artificially penalized by the uncertainty of the dynamics. We theoretically show that the algorithm maximizes a lower bound of the policy's return under the true MDP. We also characterize the trade-off between the gain and risk of leaving the support of the batch data. Our algorithm, Model-based Offline Policy Optimization (MOPO), outperforms standard model-based RL algorithms and prior state-of-the-art model-free offline RL algorithms on existing offline RL benchmarks and two challenging continuous control tasks that require generalizing from data collected for a different task. The code is available at https://github.com/tianheyu927/mopo.",
  "published": "2020-05-27",
  "updated": "2020-11-22",
  "year": "2020",
  "authors": [
   "Tianhe Yu",
   "Garrett Thomas",
   "Lantao Yu",
   "Stefano Ermon",
   "James Zou",
   "Sergey Levine",
   "Chelsea Finn",
   "Tengyu Ma"
  ],
  "author_count": 8,
  "categories": [
   "cs.LG",
   "cs.AI",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 962,
  "influential_citations": 156,
  "tldr": "A new model-based offline RL algorithm is proposed that applies the variance of a Lipschitz-regularized model as a penalty to the reward function, and it is found that this algorithm outperforms both standard model- based RL methods and existing state-of-the-art model-free offline RL approaches on existing offline RL benchmarks, as well as two challenging continuous control tasks.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Tianhe Yu",
    "id": "10909315",
    "h_index": 31,
    "papers": 48
   },
   {
    "name": "G. Thomas",
    "id": "8234443",
    "h_index": 9,
    "papers": 12
   },
   {
    "name": "Lantao Yu",
    "id": "3469209",
    "h_index": 19,
    "papers": 38
   },
   {
    "name": "S. Ermon",
    "id": "2490652",
    "h_index": 104,
    "papers": 479
   },
   {
    "name": "James Y. Zou",
    "id": "145085305",
    "h_index": 56,
    "papers": 142
   },
   {
    "name": "S. Levine",
    "id": "1736651",
    "h_index": 167,
    "papers": 551
   },
   {
    "name": "Chelsea Finn",
    "id": "46881670",
    "h_index": 94,
    "papers": 240
   },
   {
    "name": "Tengyu Ma",
    "id": "1901958",
    "h_index": 73,
    "papers": 678
   }
  ],
  "comment": "NeurIPS 2020. First two authors contributed equally. Last two authors advised equally",
  "topics": [
   "rl-control",
   "navigation",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2005.13239v6",
  "pdf_url": "https://arxiv.org/pdf/2005.13239v6",
  "html_url": "https://arxiv.org/html/2005.13239v6",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.48
 },
 {
  "id": "2005.12872",
  "slug": "end-to-end-object-detection-with-transformers",
  "title": "End-to-End Object Detection with Transformers",
  "abstract": "We present a new method that views object detection as a direct set prediction problem. Our approach streamlines the detection pipeline, effectively removing the need for many hand-designed components like a non-maximum suppression procedure or anchor generation that explicitly encode our prior knowledge about the task. The main ingredients of the new framework, called DEtection TRansformer or DETR, are a set-based global loss that forces unique predictions via bipartite matching, and a transformer encoder-decoder architecture. Given a fixed small set of learned object queries, DETR reasons about the relations of the objects and the global image context to directly output the final set of predictions in parallel. The new model is conceptually simple and does not require a specialized library, unlike many other modern detectors. DETR demonstrates accuracy and run-time performance on par with the well-established and highly-optimized Faster RCNN baseline on the challenging COCO object detection dataset. Moreover, DETR can be easily generalized to produce panoptic segmentation in a unified manner. We show that it significantly outperforms competitive baselines. Training code and pretrained models are available at https://github.com/facebookresearch/detr.",
  "published": "2020-05-26",
  "updated": "2020-05-28",
  "year": "2020",
  "authors": [
   "Nicolas Carion",
   "Francisco Massa",
   "Gabriel Synnaeve",
   "Nicolas Usunier",
   "Alexander Kirillov",
   "Sergey Zagoruyko"
  ],
  "author_count": 6,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "ECCV",
  "venue_source": "semantic-scholar",
  "citations": 19613,
  "influential_citations": 1933,
  "tldr": "This work presents a new method that views object detection as a direct set prediction problem, and demonstrates accuracy and run-time performance on par with the well-established and highly-optimized Faster RCNN baseline on the challenging COCO object detection dataset.",
  "doi": "10.1007/978-3-030-58452-8_13",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Nicolas Carion",
    "id": "3422899",
    "h_index": 11,
    "papers": 17
   },
   {
    "name": "Francisco Massa",
    "id": "1403239967",
    "h_index": 11,
    "papers": 14
   },
   {
    "name": "Gabriel Synnaeve",
    "id": "2282478",
    "h_index": 58,
    "papers": 209
   },
   {
    "name": "Nicolas Usunier",
    "id": "1746841",
    "h_index": 46,
    "papers": 131
   },
   {
    "name": "Alexander Kirillov",
    "id": "144843400",
    "h_index": 19,
    "papers": 44
   },
   {
    "name": "Sergey Zagoruyko",
    "id": "2134433",
    "h_index": 17,
    "papers": 21
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [
   "Meta FAIR"
  ],
  "abs_url": "https://arxiv.org/abs/2005.12872v3",
  "pdf_url": "https://arxiv.org/pdf/2005.12872v3",
  "html_url": "https://arxiv.org/html/2005.12872v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.0
 },
 {
  "id": "2005.07648",
  "slug": "language-conditioned-imitation-learning-over-unstructured-data",
  "title": "Language Conditioned Imitation Learning over Unstructured Data",
  "abstract": "Natural language is perhaps the most flexible and intuitive way for humans to communicate tasks to a robot. Prior work in imitation learning typically requires each task be specified with a task id or goal image -- something that is often impractical in open-world environments. On the other hand, previous approaches in instruction following allow agent behavior to be guided by language, but typically assume structure in the observations, actuators, or language that limit their applicability to complex settings like robotics. In this work, we present a method for incorporating free-form natural language conditioning into imitation learning. Our approach learns perception from pixels, natural language understanding, and multitask continuous control end-to-end as a single neural network. Unlike prior work in imitation learning, our method is able to incorporate unlabeled and unstructured demonstration data (i.e. no task or language labels). We show this dramatically improves language conditioned performance, while reducing the cost of language annotation to less than 1% of total data. At test time, a single language conditioned visuomotor policy trained with our method can perform a wide variety of robotic manipulation skills in a 3D environment, specified only with natural language descriptions of each task (e.g. \"open the drawer...now pick up the block...now press the green button...\"). To scale up the number of instructions an agent can follow, we propose combining text conditioned policies with large pretrained neural language models. We find this allows a policy to be robust to many out-of-distribution synonym instructions, without requiring new demonstrations. See videos of a human typing live text commands to our agent at language-play.github.io",
  "published": "2020-05-15",
  "updated": "2021-07-07",
  "year": "2020",
  "authors": [
   "Corey Lynch",
   "Pierre Sermanet"
  ],
  "author_count": 2,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CL",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "RSS",
  "venue_source": "semantic-scholar",
  "citations": 310,
  "influential_citations": 26,
  "tldr": "This work presents a method for incorporating free-form natural language conditioning into imitation learning, and proposes combining text conditioned policies with large pretrained neural language models to scale up the number of instructions an agent can follow.",
  "doi": "10.15607/RSS.2021.XVII.047",
  "oa_pdf": "https://doi.org/10.15607/rss.2021.xvii.047",
  "s2_authors": [
   {
    "name": "Corey Lynch",
    "id": "32245472",
    "h_index": 20,
    "papers": 27
   },
   {
    "name": "P. Sermanet",
    "id": "3142556",
    "h_index": 39,
    "papers": 77
   }
  ],
  "comment": "Published at RSS 2021",
  "topics": [
   "imitation-diffusion",
   "data-teleop",
   "hardware-codesign",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2005.07648v2",
  "pdf_url": "https://arxiv.org/pdf/2005.07648v2",
  "html_url": "https://arxiv.org/html/2005.07648v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.99
 },
 {
  "id": "2005.07513",
  "slug": "a-distributional-view-on-multi-objective-policy-optimization",
  "title": "A Distributional View on Multi-Objective Policy Optimization",
  "abstract": "Many real-world problems require trading off multiple competing objectives. However, these objectives are often in different units and/or scales, which can make it challenging for practitioners to express numerical preferences over objectives in their native units. In this paper we propose a novel algorithm for multi-objective reinforcement learning that enables setting desired preferences for objectives in a scale-invariant way. We propose to learn an action distribution for each objective, and we use supervised learning to fit a parametric policy to a combination of these distributions. We demonstrate the effectiveness of our approach on challenging high-dimensional real and simulated robotics tasks, and show that setting different preferences in our framework allows us to trace out the space of nondominated solutions.",
  "published": "2020-05-15",
  "updated": "2020-05-15",
  "year": "2020",
  "authors": [
   "Abbas Abdolmaleki",
   "Sandy H. Huang",
   "Leonard Hasenclever",
   "Michael Neunert",
   "H. Francis Song",
   "Martina Zambelli",
   "Murilo F. Martins",
   "Nicolas Heess",
   "Raia Hadsell",
   "Martin Riedmiller"
  ],
  "author_count": 10,
  "categories": [
   "cs.LG",
   "cs.AI",
   "cs.RO",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "ICML",
  "venue_source": "semantic-scholar",
  "citations": 96,
  "influential_citations": 3,
  "tldr": "This paper proposes a novel algorithm for multi-objective reinforcement learning that enables setting desired preferences for objectives in a scale-invariant way, and uses supervised learning to fit a parametric policy to a combination of these distributions.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "A. Abdolmaleki",
    "id": "2799799",
    "h_index": 29,
    "papers": 95
   },
   {
    "name": "Sandy H. Huang",
    "id": "2064588",
    "h_index": 15,
    "papers": 25
   },
   {
    "name": "Leonard Hasenclever",
    "id": "40401956",
    "h_index": 28,
    "papers": 50
   },
   {
    "name": "M. Neunert",
    "id": "2366050",
    "h_index": 29,
    "papers": 50
   },
   {
    "name": "H. F. Song",
    "id": "2107148568",
    "h_index": 13,
    "papers": 15
   },
   {
    "name": "Martina Zambelli",
    "id": "7455600",
    "h_index": 11,
    "papers": 17
   },
   {
    "name": "M. Martins",
    "id": "145279513",
    "h_index": 8,
    "papers": 19
   },
   {
    "name": "N. Heess",
    "id": "2801204",
    "h_index": 73,
    "papers": 192
   },
   {
    "name": "R. Hadsell",
    "id": "2315504",
    "h_index": 51,
    "papers": 111
   },
   {
    "name": "Martin A. Riedmiller",
    "id": "3137672",
    "h_index": 55,
    "papers": 219
   }
  ],
  "comment": "",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2005.07513v1",
  "pdf_url": "https://arxiv.org/pdf/2005.07513v1",
  "html_url": "https://arxiv.org/html/2005.07513v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.49
 },
 {
  "id": "2005.05960",
  "slug": "planning-to-explore-via-self-supervised-world-models",
  "title": "Planning to Explore via Self-Supervised World Models",
  "abstract": "Reinforcement learning allows solving complex tasks, however, the learning tends to be task-specific and the sample efficiency remains a challenge. We present Plan2Explore, a self-supervised reinforcement learning agent that tackles both these challenges through a new approach to self-supervised exploration and fast adaptation to new tasks, which need not be known during exploration. During exploration, unlike prior methods which retrospectively compute the novelty of observations after the agent has already reached them, our agent acts efficiently by leveraging planning to seek out expected future novelty. After exploration, the agent quickly adapts to multiple downstream tasks in a zero or a few-shot manner. We evaluate on challenging control tasks from high-dimensional image inputs. Without any training supervision or task-specific interaction, Plan2Explore outperforms prior self-supervised exploration methods, and in fact, almost matches the performances oracle which has access to rewards. Videos and code at https://ramanans1.github.io/plan2explore/",
  "published": "2020-05-12",
  "updated": "2020-06-30",
  "year": "2020",
  "authors": [
   "Ramanan Sekar",
   "Oleh Rybkin",
   "Kostas Daniilidis",
   "Pieter Abbeel",
   "Danijar Hafner",
   "Deepak Pathak"
  ],
  "author_count": 6,
  "categories": [
   "cs.LG",
   "cs.AI",
   "cs.CV",
   "cs.NE",
   "cs.RO",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "ICML",
  "venue_source": "semantic-scholar",
  "citations": 521,
  "influential_citations": 61,
  "tldr": "Without any training supervision or task-specific interaction, Plan2Explore outperforms prior self-supervised exploration methods, and in fact, almost matches the performances oracle which has access to rewards.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "R. Sekar",
    "id": "10735446",
    "h_index": 8,
    "papers": 19
   },
   {
    "name": "Oleh Rybkin",
    "id": "40900227",
    "h_index": 14,
    "papers": 27
   },
   {
    "name": "Kostas Daniilidis",
    "id": "1751586",
    "h_index": 78,
    "papers": 348
   },
   {
    "name": "P. Abbeel",
    "id": "1689992",
    "h_index": 165,
    "papers": 596
   },
   {
    "name": "Danijar Hafner",
    "id": "35006479",
    "h_index": 25,
    "papers": 47
   },
   {
    "name": "Deepak Pathak",
    "id": "38236002",
    "h_index": 34,
    "papers": 78
   }
  ],
  "comment": "Accepted at ICML 2020. Videos and code at https://ramanans1.github.io/plan2explore/",
  "topics": [
   "world-models",
   "rl-control",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2005.05960v2",
  "pdf_url": "https://arxiv.org/pdf/2005.05960v2",
  "html_url": "https://arxiv.org/html/2005.05960v2",
  "code_url": "https://ramanans1.github.io/plan2explore/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.22
 },
 {
  "id": "2005.00928",
  "slug": "quantifying-attention-flow-in-transformers",
  "title": "Quantifying Attention Flow in Transformers",
  "abstract": "In the Transformer model, \"self-attention\" combines information from attended embeddings into the representation of the focal embedding in the next layer. Thus, across layers of the Transformer, information originating from different tokens gets increasingly mixed. This makes attention weights unreliable as explanations probes. In this paper, we consider the problem of quantifying this flow of information through self-attention. We propose two methods for approximating the attention to input tokens given attention weights, attention rollout and attention flow, as post hoc methods when we use attention weights as the relative relevance of the input tokens. We show that these methods give complementary views on the flow of information, and compared to raw attention, both yield higher correlations with importance scores of input tokens obtained using an ablation method and input gradients.",
  "published": "2020-05-02",
  "updated": "2020-05-31",
  "year": "2020",
  "authors": [
   "Samira Abnar",
   "Willem Zuidema"
  ],
  "author_count": 2,
  "categories": [
   "cs.LG",
   "cs.AI",
   "cs.CL"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 1359,
  "influential_citations": 138,
  "tldr": "This paper proposes two methods for approximating the attention to input tokens given attention weights, attention rollout and attention flow, as post hoc methods when the authors use attention weights as the relative relevance of the input tokens.",
  "doi": "10.18653/v1/2020.acl-main.385",
  "oa_pdf": "https://www.aclweb.org/anthology/2020.acl-main.385.pdf",
  "s2_authors": [
   {
    "name": "Samira Abnar",
    "id": "2786352",
    "h_index": 15,
    "papers": 23
   },
   {
    "name": "W. Zuidema",
    "id": "83390207",
    "h_index": 8,
    "papers": 18
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2005.00928v2",
  "pdf_url": "https://arxiv.org/pdf/2005.00928v2",
  "html_url": "https://arxiv.org/html/2005.00928v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.0
 },
 {
  "id": "2004.14990",
  "slug": "reinforcement-learning-with-augmented-data",
  "title": "Reinforcement Learning with Augmented Data",
  "abstract": "Learning from visual observations is a fundamental yet challenging problem in Reinforcement Learning (RL). Although algorithmic advances combined with convolutional neural networks have proved to be a recipe for success, current methods are still lacking on two fronts: (a) data-efficiency of learning and (b) generalization to new environments. To this end, we present Reinforcement Learning with Augmented Data (RAD), a simple plug-and-play module that can enhance most RL algorithms. We perform the first extensive study of general data augmentations for RL on both pixel-based and state-based inputs, and introduce two new data augmentations - random translate and random amplitude scale. We show that augmentations such as random translate, crop, color jitter, patch cutout, random convolutions, and amplitude scale can enable simple RL algorithms to outperform complex state-of-the-art methods across common benchmarks. RAD sets a new state-of-the-art in terms of data-efficiency and final performance on the DeepMind Control Suite benchmark for pixel-based control as well as OpenAI Gym benchmark for state-based control. We further demonstrate that RAD significantly improves test-time generalization over existing methods on several OpenAI ProcGen benchmarks. Our RAD module and training code are available at https://www.github.com/MishaLaskin/rad.",
  "published": "2020-04-30",
  "updated": "2020-11-05",
  "year": "2020",
  "authors": [
   "Michael Laskin",
   "Kimin Lee",
   "Adam Stooke",
   "Lerrel Pinto",
   "Pieter Abbeel",
   "Aravind Srinivas"
  ],
  "author_count": 6,
  "categories": [
   "cs.LG",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 808,
  "influential_citations": 115,
  "tldr": "It is shown that augmentations such as random translate, crop, color jitter, patch cutout, random convolutions, and amplitude scale can enable simple RL algorithms to outperform complex state-of-the-art methods across common benchmarks.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "M. Laskin",
    "id": "51093256",
    "h_index": 22,
    "papers": 39
   },
   {
    "name": "Kimin Lee",
    "id": "3436470",
    "h_index": 37,
    "papers": 69
   },
   {
    "name": "Adam Stooke",
    "id": "47541311",
    "h_index": 8,
    "papers": 16
   },
   {
    "name": "Lerrel Pinto",
    "id": "34026610",
    "h_index": 41,
    "papers": 70
   },
   {
    "name": "P. Abbeel",
    "id": "1689992",
    "h_index": 165,
    "papers": 596
   },
   {
    "name": "A. Srinivas",
    "id": "41207614",
    "h_index": 20,
    "papers": 35
   }
  ],
  "comment": "NeurIPS 2020 camera-ready version. First two authors contributed equally, website: https://mishalaskin.github.io/rad code: https://github.com/MishaLaskin/rad and https://github.com/pokaxpoka/rad_procgen",
  "topics": [
   "rl-control"
  ],
  "orgs": [
   "Google DeepMind"
  ],
  "abs_url": "https://arxiv.org/abs/2004.14990v5",
  "pdf_url": "https://arxiv.org/pdf/2004.14990v5",
  "html_url": "https://arxiv.org/html/2004.14990v5",
  "code_url": "https://mishalaskin.github.io/rad",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.91
 },
 {
  "id": "2004.13649",
  "slug": "image-augmentation-is-all-you-need-regularizing-deep-reinforcement-lea",
  "title": "Image Augmentation Is All You Need: Regularizing Deep Reinforcement Learning from Pixels",
  "abstract": "We propose a simple data augmentation technique that can be applied to standard model-free reinforcement learning algorithms, enabling robust learning directly from pixels without the need for auxiliary losses or pre-training. The approach leverages input perturbations commonly used in computer vision tasks to regularize the value function. Existing model-free approaches, such as Soft Actor-Critic (SAC), are not able to train deep networks effectively from image pixels. However, the addition of our augmentation method dramatically improves SAC's performance, enabling it to reach state-of-the-art performance on the DeepMind control suite, surpassing model-based (Dreamer, PlaNet, and SLAC) methods and recently proposed contrastive learning (CURL). Our approach can be combined with any model-free reinforcement learning algorithm, requiring only minor modifications. An implementation can be found at https://sites.google.com/view/data-regularized-q.",
  "published": "2020-04-28",
  "updated": "2021-03-07",
  "year": "2020",
  "authors": [
   "Ilya Kostrikov",
   "Denis Yarats",
   "Rob Fergus"
  ],
  "author_count": 3,
  "categories": [
   "cs.LG",
   "cs.CV",
   "eess.IV",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 958,
  "influential_citations": 162,
  "tldr": "The addition of the augmentation method dramatically improves SAC's performance, enabling it to reach state-of-the-art performance on the DeepMind control suite, surpassing model-based methods and recently proposed contrastive learning (CURL).",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ilya Kostrikov",
    "id": "2000906",
    "h_index": 29,
    "papers": 43
   },
   {
    "name": "Denis Yarats",
    "id": "13759615",
    "h_index": 21,
    "papers": 31
   },
   {
    "name": "R. Fergus",
    "id": "2276554",
    "h_index": 78,
    "papers": 125
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "rl-control",
   "foundation-pretraining"
  ],
  "orgs": [
   "Google DeepMind"
  ],
  "abs_url": "https://arxiv.org/abs/2004.13649v4",
  "pdf_url": "https://arxiv.org/pdf/2004.13649v4",
  "html_url": "https://arxiv.org/html/2004.13649v4",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.98
 },
 {
  "id": "2004.10151",
  "slug": "experience-grounds-language",
  "title": "Experience Grounds Language",
  "abstract": "Language understanding research is held back by a failure to relate language to the physical world it describes and to the social interactions it facilitates. Despite the incredible effectiveness of language processing models to tackle tasks after being trained on text alone, successful linguistic communication relies on a shared experience of the world. It is this shared experience that makes utterances meaningful. Natural language processing is a diverse field, and progress throughout its development has come from new representational theories, modeling techniques, data collection paradigms, and tasks. We posit that the present success of representation learning approaches trained on large, text-only corpora requires the parallel tradition of research on the broader physical and social context of language to address the deeper questions of communication.",
  "published": "2020-04-21",
  "updated": "2020-11-02",
  "year": "2020",
  "authors": [
   "Yonatan Bisk",
   "Ari Holtzman",
   "Jesse Thomason",
   "Jacob Andreas",
   "Yoshua Bengio",
   "Joyce Chai",
   "Mirella Lapata",
   "Angeliki Lazaridou",
   "Jonathan May",
   "Aleksandr Nisnevich",
   "Nicolas Pinto",
   "Joseph Turian"
  ],
  "author_count": 12,
  "categories": [
   "cs.CL",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.CL",
  "venue": "",
  "venue_source": "",
  "citations": 443,
  "influential_citations": 18,
  "tldr": "It is posited that the present success of representation learning approaches trained on large text corpora can be deeply enriched from the parallel tradition of research on the contextual and social nature of language.",
  "doi": "10.18653/v1/2020.emnlp-main.703",
  "oa_pdf": "https://www.aclweb.org/anthology/2020.emnlp-main.703.pdf",
  "s2_authors": [
   {
    "name": "Yonatan Bisk",
    "id": "3312309",
    "h_index": 46,
    "papers": 144
   },
   {
    "name": "Ari Holtzman",
    "id": "14487640",
    "h_index": 24,
    "papers": 43
   },
   {
    "name": "Jesse Thomason",
    "id": "2665873",
    "h_index": 27,
    "papers": 63
   },
   {
    "name": "Jacob Andreas",
    "id": "2112400",
    "h_index": 53,
    "papers": 93
   },
   {
    "name": "Yoshua Bengio",
    "id": "1751762",
    "h_index": 212,
    "papers": 813
   },
   {
    "name": "J. Chai",
    "id": "1707259",
    "h_index": 35,
    "papers": 144
   },
   {
    "name": "Mirella Lapata",
    "id": "1747893",
    "h_index": 87,
    "papers": 391
   },
   {
    "name": "Angeliki Lazaridou",
    "id": "2672644",
    "h_index": 35,
    "papers": 88
   },
   {
    "name": "Jonathan May",
    "id": "143823227",
    "h_index": 33,
    "papers": 112
   },
   {
    "name": "Aleksandr Nisnevich",
    "id": "17109242",
    "h_index": 4,
    "papers": 7
   },
   {
    "name": "Nicolas Pinto",
    "id": "30017846",
    "h_index": 15,
    "papers": 31
   },
   {
    "name": "Joseph P. Turian",
    "id": "153160559",
    "h_index": 16,
    "papers": 26
   }
  ],
  "comment": "Empirical Methods in Natural Language Processing (EMNLP), 2020",
  "topics": [
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2004.10151v3",
  "pdf_url": "https://arxiv.org/pdf/2004.10151v3",
  "html_url": "https://arxiv.org/html/2004.10151v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.65
 },
 {
  "id": "2004.07219",
  "slug": "d4rl-datasets-for-deep-data-driven-reinforcement-learning",
  "title": "D4RL: Datasets for Deep Data-Driven Reinforcement Learning",
  "abstract": "The offline reinforcement learning (RL) setting (also known as full batch RL), where a policy is learned from a static dataset, is compelling as progress enables RL methods to take advantage of large, previously-collected datasets, much like how the rise of large datasets has fueled results in supervised learning. However, existing online RL benchmarks are not tailored towards the offline setting and existing offline RL benchmarks are restricted to data generated by partially-trained agents, making progress in offline RL difficult to measure. In this work, we introduce benchmarks specifically designed for the offline setting, guided by key properties of datasets relevant to real-world applications of offline RL. With a focus on dataset collection, examples of such properties include: datasets generated via hand-designed controllers and human demonstrators, multitask datasets where an agent performs different tasks in the same environment, and datasets collected with mixtures of policies. By moving beyond simple benchmark tasks and data collected by partially-trained RL agents, we reveal important and unappreciated deficiencies of existing algorithms. To facilitate research, we have released our benchmark tasks and datasets with a comprehensive evaluation of existing algorithms, an evaluation protocol, and open-source examples. This serves as a common starting point for the community to identify shortcomings in existing offline RL methods and a collaborative route for progress in this emerging area.",
  "published": "2020-04-15",
  "updated": "2021-02-06",
  "year": "2020",
  "authors": [
   "Justin Fu",
   "Aviral Kumar",
   "Ofir Nachum",
   "George Tucker",
   "Sergey Levine"
  ],
  "author_count": 5,
  "categories": [
   "cs.LG",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 1893,
  "influential_citations": 617,
  "tldr": "This work introduces benchmarks specifically designed for the offline setting, guided by key properties of datasets relevant to real-world applications of offline RL, and releases benchmark tasks and datasets with a comprehensive evaluation of existing algorithms and an evaluation protocol together with an open-source codebase.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Justin Fu",
    "id": "2550764",
    "h_index": 22,
    "papers": 31
   },
   {
    "name": "Aviral Kumar",
    "id": "1488785534",
    "h_index": 46,
    "papers": 83
   },
   {
    "name": "Ofir Nachum",
    "id": "7624658",
    "h_index": 47,
    "papers": 92
   },
   {
    "name": "G. Tucker",
    "id": "145499435",
    "h_index": 34,
    "papers": 55
   },
   {
    "name": "S. Levine",
    "id": "1736651",
    "h_index": 167,
    "papers": 551
   }
  ],
  "comment": "Website available at https://sites.google.com/view/d4rl/home",
  "topics": [
   "sim2real",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2004.07219v4",
  "pdf_url": "https://arxiv.org/pdf/2004.07219v4",
  "html_url": "https://arxiv.org/html/2004.07219v4",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.0
 },
 {
  "id": "2004.04136",
  "slug": "curl-contrastive-unsupervised-representations-for-reinforcement-learni",
  "title": "CURL: Contrastive Unsupervised Representations for Reinforcement Learning",
  "abstract": "We present CURL: Contrastive Unsupervised Representations for Reinforcement Learning. CURL extracts high-level features from raw pixels using contrastive learning and performs off-policy control on top of the extracted features. CURL outperforms prior pixel-based methods, both model-based and model-free, on complex tasks in the DeepMind Control Suite and Atari Games showing 1.9x and 1.2x performance gains at the 100K environment and interaction steps benchmarks respectively. On the DeepMind Control Suite, CURL is the first image-based algorithm to nearly match the sample-efficiency of methods that use state-based features. Our code is open-sourced and available at https://github.com/MishaLaskin/curl.",
  "published": "2020-04-08",
  "updated": "2020-09-21",
  "year": "2020",
  "authors": [
   "Aravind Srinivas",
   "Michael Laskin",
   "Pieter Abbeel"
  ],
  "author_count": 3,
  "categories": [
   "cs.LG",
   "cs.CV",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "ICML",
  "venue_source": "semantic-scholar",
  "citations": 1331,
  "influential_citations": 198,
  "tldr": "CURL extracts high-level features from raw pixels using contrastive learning and performs off-policy control on top of the extracted features and is the first image-based algorithm to nearly match the sample-efficiency of methods that use state-based features.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "A. Srinivas",
    "id": "41207614",
    "h_index": 20,
    "papers": 35
   },
   {
    "name": "M. Laskin",
    "id": "51093256",
    "h_index": 22,
    "papers": 39
   },
   {
    "name": "P. Abbeel",
    "id": "1689992",
    "h_index": 165,
    "papers": 596
   }
  ],
  "comment": "First two authors contributed equally, website: https://mishalaskin.github.io/curl code: https://github.com/MishaLaskin/curl",
  "topics": [
   "rl-control"
  ],
  "orgs": [
   "Google DeepMind"
  ],
  "abs_url": "https://arxiv.org/abs/2004.04136v4",
  "pdf_url": "https://arxiv.org/pdf/2004.04136v4",
  "html_url": "https://arxiv.org/html/2004.04136v4",
  "code_url": "https://mishalaskin.github.io/curl",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 6.0
 },
 {
  "id": "2004.03691",
  "slug": "soft-bubble-grippers-for-robust-and-perceptive-manipulation",
  "title": "Soft-Bubble grippers for robust and perceptive manipulation",
  "abstract": "Manipulation in cluttered environments like homes requires stable grasps, precise placement and robustness against external contact. We present the Soft-Bubble gripper system with a highly compliant gripping surface and dense-geometry visuotactile sensing, capable of multiple kinds of tactile perception. We first present various mechanical design advances and a fabrication technique to deposit custom patterns to the internal surface of the sensor that enable tracking of shear-induced displacement of the manipuland. The depth maps output by the internal imaging sensor are used in an in-hand proximity pose estimation framework -- the method better captures distances to corners or edges on the manipuland geometry. We also extend our previous work on tactile classification and integrate the system within a robust manipulation pipeline for cluttered home environments. The capabilities of the proposed system are demonstrated through robust execution multiple real-world manipulation tasks. A video of the system in action can be found here: [https://youtu.be/G_wBsbQyBfc].",
  "published": "2020-04-07",
  "updated": "2020-04-28",
  "year": "2020",
  "authors": [
   "Naveen Kuppuswamy",
   "Alex Alspach",
   "Avinash Uttamchandani",
   "Sam Creasey",
   "Takuya Ikeda",
   "Russ Tedrake"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS",
  "venue_source": "semantic-scholar",
  "citations": 118,
  "influential_citations": 10,
  "tldr": "This work presents the Soft-bubble gripper system, a system that combines highly compliant gripping surfaces with dense-geometry visuotactile sensing and facilitates multiple kinds of tactile perception and extends previous work on tactile classification to integrate the system within a robust manipulation pipeline for cluttered home environments.",
  "doi": "10.1109/IROS45743.2020.9341534",
  "oa_pdf": "https://arxiv.org/pdf/2004.03691",
  "s2_authors": [
   {
    "name": "N. Kuppuswamy",
    "id": "2529186",
    "h_index": 13,
    "papers": 42
   },
   {
    "name": "A. Alspach",
    "id": "2980876",
    "h_index": 15,
    "papers": 25
   },
   {
    "name": "Avinash Uttamchandani",
    "id": "101804298",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "S. Creasey",
    "id": "1620487062",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Takuya Ikeda",
    "id": "2114913896",
    "h_index": 4,
    "papers": 19
   },
   {
    "name": "Russ Tedrake",
    "id": "1726802",
    "h_index": 76,
    "papers": 270
   }
  ],
  "comment": "8 pages, conference",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "hardware-codesign",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2004.03691v2",
  "pdf_url": "https://arxiv.org/pdf/2004.03691v2",
  "html_url": "https://arxiv.org/html/2004.03691v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.58
 },
 {
  "id": "2004.02857",
  "slug": "beyond-the-nav-graph-vision-and-language-navigation-in-continuous-envi",
  "title": "Beyond the Nav-Graph: Vision-and-Language Navigation in Continuous Environments",
  "abstract": "We develop a language-guided navigation task set in a continuous 3D environment where agents must execute low-level actions to follow natural language navigation directions. By being situated in continuous environments, this setting lifts a number of assumptions implicit in prior work that represents environments as a sparse graph of panoramas with edges corresponding to navigability. Specifically, our setting drops the presumptions of known environment topologies, short-range oracle navigation, and perfect agent localization. To contextualize this new task, we develop models that mirror many of the advances made in prior settings as well as single-modality baselines. While some of these techniques transfer, we find significantly lower absolute performance in the continuous setting -- suggesting that performance in prior `navigation-graph' settings may be inflated by the strong implicit assumptions.",
  "published": "2020-04-06",
  "updated": "2020-05-01",
  "year": "2020",
  "authors": [
   "Jacob Krantz",
   "Erik Wijmans",
   "Arjun Majumdar",
   "Dhruv Batra",
   "Stefan Lee"
  ],
  "author_count": 5,
  "categories": [
   "cs.CV",
   "cs.CL",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "ECCV",
  "venue_source": "semantic-scholar",
  "citations": 618,
  "influential_citations": 134,
  "tldr": "A language-guided navigation task set in a continuous 3D environment where agents must execute low-level actions to follow natural language navigation directions is developed, suggesting that performance in prior `navigation-graph' settings may be inflated by the strong implicit assumptions.",
  "doi": "10.1007/978-3-030-58604-1_7",
  "oa_pdf": "https://arxiv.org/pdf/2004.02857",
  "s2_authors": [
   {
    "name": "Jacob Krantz",
    "id": "51050450",
    "h_index": 11,
    "papers": 14
   },
   {
    "name": "Erik Wijmans",
    "id": "8405939",
    "h_index": 13,
    "papers": 17
   },
   {
    "name": "Arjun Majumdar",
    "id": "2905057",
    "h_index": 15,
    "papers": 28
   },
   {
    "name": "Dhruv Batra",
    "id": "1746610",
    "h_index": 87,
    "papers": 327
   },
   {
    "name": "Stefan Lee",
    "id": "1607486000",
    "h_index": 18,
    "papers": 26
   }
  ],
  "comment": "",
  "topics": [
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2004.02857v2",
  "pdf_url": "https://arxiv.org/pdf/2004.02857v2",
  "html_url": "https://arxiv.org/html/2004.02857v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.29
 },
 {
  "id": "2004.00849",
  "slug": "pixel-bert-aligning-image-pixels-with-text-by-deep-multi-modal-transfo",
  "title": "Pixel-BERT: Aligning Image Pixels with Text by Deep Multi-Modal Transformers",
  "abstract": "We propose Pixel-BERT to align image pixels with text by deep multi-modal transformers that jointly learn visual and language embedding in a unified end-to-end framework. We aim to build a more accurate and thorough connection between image pixels and language semantics directly from image and sentence pairs instead of using region-based image features as the most recent vision and language tasks. Our Pixel-BERT which aligns semantic connection in pixel and text level solves the limitation of task-specific visual representation for vision and language tasks. It also relieves the cost of bounding box annotations and overcomes the unbalance between semantic labels in visual task and language semantic. To provide a better representation for down-stream tasks, we pre-train a universal end-to-end model with image and sentence pairs from Visual Genome dataset and MS-COCO dataset. We propose to use a random pixel sampling mechanism to enhance the robustness of visual representation and to apply the Masked Language Model and Image-Text Matching as pre-training tasks. Extensive experiments on downstream tasks with our pre-trained model show that our approach makes the most state-of-the-arts in downstream tasks, including Visual Question Answering (VQA), image-text retrieval, Natural Language for Visual Reasoning for Real (NLVR). Particularly, we boost the performance of a single model in VQA task by 2.17 points compared with SOTA under fair comparison.",
  "published": "2020-04-02",
  "updated": "2020-06-22",
  "year": "2020",
  "authors": [
   "Zhicheng Huang",
   "Zhaoyang Zeng",
   "Bei Liu",
   "Dongmei Fu",
   "Jianlong Fu"
  ],
  "author_count": 5,
  "categories": [
   "cs.CV",
   "cs.CL",
   "cs.LG",
   "cs.MM"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 476,
  "influential_citations": 40,
  "tldr": "The Pixel-BERT which aligns semantic connection in pixel and text level solves the limitation of task-specific visual representation for vision and language tasks and relieves the cost of bounding box annotations and overcomes the unbalance between semantic labels in visual task and language semantic.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Zhicheng Huang",
    "id": "47272083",
    "h_index": 10,
    "papers": 17
   },
   {
    "name": "Zhaoyang Zeng",
    "id": "2075413603",
    "h_index": 15,
    "papers": 28
   },
   {
    "name": "Bei Liu",
    "id": "2108662284",
    "h_index": 11,
    "papers": 20
   },
   {
    "name": "Dongmei Fu",
    "id": "2061347150",
    "h_index": 19,
    "papers": 66
   },
   {
    "name": "Jianlong Fu",
    "id": "3247966",
    "h_index": 55,
    "papers": 118
   }
  ],
  "comment": "",
  "topics": [
   "foundation-pretraining",
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2004.00849v2",
  "pdf_url": "https://arxiv.org/pdf/2004.00849v2",
  "html_url": "https://arxiv.org/html/2004.00849v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.68
 },
 {
  "id": "2003.13350",
  "slug": "agent57-outperforming-the-atari-human-benchmark",
  "title": "Agent57: Outperforming the Atari Human Benchmark",
  "abstract": "Atari games have been a long-standing benchmark in the reinforcement learning (RL) community for the past decade. This benchmark was proposed to test general competency of RL algorithms. Previous work has achieved good average performance by doing outstandingly well on many games of the set, but very poorly in several of the most challenging games. We propose Agent57, the first deep RL agent that outperforms the standard human benchmark on all 57 Atari games. To achieve this result, we train a neural network which parameterizes a family of policies ranging from very exploratory to purely exploitative. We propose an adaptive mechanism to choose which policy to prioritize throughout the training process. Additionally, we utilize a novel parameterization of the architecture that allows for more consistent and stable learning.",
  "published": "2020-03-30",
  "updated": "2020-03-30",
  "year": "2020",
  "authors": [
   "Adri\u00e0 Puigdom\u00e8nech Badia",
   "Bilal Piot",
   "Steven Kapturowski",
   "Pablo Sprechmann",
   "Alex Vitvitskyi",
   "Daniel Guo",
   "Charles Blundell"
  ],
  "author_count": 7,
  "categories": [
   "cs.LG",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "ICML",
  "venue_source": "semantic-scholar",
  "citations": 598,
  "influential_citations": 44,
  "tldr": "This work proposes Agent57, the first deep RL agent that outperforms the standard human benchmark on all 57 Atari games and trains a neural network which parameterizes a family of policies ranging from very exploratory to purely exploitative.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Adri\u00e0 Puigdom\u00e8nech Badia",
    "id": "36045539",
    "h_index": 13,
    "papers": 16
   },
   {
    "name": "Bilal Piot",
    "id": "1808897",
    "h_index": 44,
    "papers": 80
   },
   {
    "name": "Steven Kapturowski",
    "id": "67007190",
    "h_index": 12,
    "papers": 15
   },
   {
    "name": "P. Sprechmann",
    "id": "2905900",
    "h_index": 28,
    "papers": 54
   },
   {
    "name": "Alex Vitvitskyi",
    "id": "1492155662",
    "h_index": 10,
    "papers": 13
   },
   {
    "name": "Daniel Guo",
    "id": "1999078",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "C. Blundell",
    "id": "1723876",
    "h_index": 43,
    "papers": 70
   }
  ],
  "comment": "",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2003.13350v1",
  "pdf_url": "https://arxiv.org/pdf/2003.13350v1",
  "html_url": "https://arxiv.org/html/2003.13350v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.28
 },
 {
  "id": "2003.08876",
  "slug": "learning-to-fly-via-deep-model-based-reinforcement-learning",
  "title": "Learning to Fly via Deep Model-Based Reinforcement Learning",
  "abstract": "Learning to control robots without requiring engineered models has been a long-term goal, promising diverse and novel applications. Yet, reinforcement learning has only achieved limited impact on real-time robot control due to its high demand of real-world interactions. In this work, by leveraging a learnt probabilistic model of drone dynamics, we learn a thrust-attitude controller for a quadrotor through model-based reinforcement learning. No prior knowledge of the flight dynamics is assumed; instead, a sequential latent variable model, used generatively and as an online filter, is learnt from raw sensory input. The controller and value function are optimised entirely by propagating stochastic analytic gradients through generated latent trajectories. We show that \"learning to fly\" can be achieved with less than 30 minutes of experience with a single drone, and can be deployed solely using onboard computational resources and sensors, on a self-built drone.",
  "published": "2020-03-19",
  "updated": "2020-08-04",
  "year": "2020",
  "authors": [
   "Philip Becker-Ehmck",
   "Maximilian Karl",
   "Jan Peters",
   "Patrick van der Smagt"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG",
   "stat.ML"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 40,
  "influential_citations": 2,
  "tldr": "It is shown that \"learning to fly\" can be achieved with less than 30 minutes of experience with a single drone, and can be deployed solely using onboard computational resources and sensors, on a self-built drone.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Philip Becker-Ehmck",
    "id": "1402913772",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Maximilian Karl",
    "id": "36543095",
    "h_index": 8,
    "papers": 24
   },
   {
    "name": "J. Peters",
    "id": "144719340",
    "h_index": 25,
    "papers": 70
   },
   {
    "name": "Patrick van der Smagt",
    "id": "1715782",
    "h_index": 44,
    "papers": 211
   }
  ],
  "comment": "",
  "topics": [
   "rl-control",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2003.08876v3",
  "pdf_url": "https://arxiv.org/pdf/2003.08876v3",
  "html_url": "https://arxiv.org/html/2003.08876v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.61
 },
 {
  "id": "2003.06085",
  "slug": "learning-to-generalize-across-long-horizon-tasks-from-human-demonstrat",
  "title": "Learning to Generalize Across Long-Horizon Tasks from Human Demonstrations",
  "abstract": "Imitation learning is an effective and safe technique to train robot policies in the real world because it does not depend on an expensive random exploration process. However, due to the lack of exploration, learning policies that generalize beyond the demonstrated behaviors is still an open challenge. We present a novel imitation learning framework to enable robots to 1) learn complex real world manipulation tasks efficiently from a small number of human demonstrations, and 2) synthesize new behaviors not contained in the collected demonstrations. Our key insight is that multi-task domains often present a latent structure, where demonstrated trajectories for different tasks intersect at common regions of the state space. We present Generalization Through Imitation (GTI), a two-stage offline imitation learning algorithm that exploits this intersecting structure to train goal-directed policies that generalize to unseen start and goal state combinations. In the first stage of GTI, we train a stochastic policy that leverages trajectory intersections to have the capacity to compose behaviors from different demonstration trajectories together. In the second stage of GTI, we collect a small set of rollouts from the unconditioned stochastic policy of the first stage, and train a goal-directed agent to generalize to novel start and goal configurations. We validate GTI in both simulated domains and a challenging long-horizon robotic manipulation domain in the real world. Additional results and videos are available at https://sites.google.com/view/gti2020/ .",
  "published": "2020-03-13",
  "updated": "2021-06-23",
  "year": "2020",
  "authors": [
   "Ajay Mandlekar",
   "Danfei Xu",
   "Roberto Mart\u00edn-Mart\u00edn",
   "Silvio Savarese",
   "Li Fei-Fei"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "RSS",
  "venue_source": "semantic-scholar",
  "citations": 182,
  "influential_citations": 7,
  "tldr": "This work presents Generalization Through Imitation (GTI), a two-stage offline imitation learning algorithm that exploits this intersecting structure to train goal-directed policies that generalize to unseen start and goal state combinations.",
  "doi": "10.15607/rss.2020.xvi.061",
  "oa_pdf": "https://doi.org/10.15607/rss.2020.xvi.061",
  "s2_authors": [
   {
    "name": "A. Mandlekar",
    "id": "49686756",
    "h_index": 36,
    "papers": 67
   },
   {
    "name": "Danfei Xu",
    "id": "2068265",
    "h_index": 33,
    "papers": 45
   },
   {
    "name": "Roberto Mart\u00edn-Mart\u00edn",
    "id": "1382655067",
    "h_index": 27,
    "papers": 39
   },
   {
    "name": "S. Savarese",
    "id": "1702137",
    "h_index": 115,
    "papers": 346
   },
   {
    "name": "Li Fei-Fei",
    "id": "48004138",
    "h_index": 143,
    "papers": 606
   }
  ],
  "comment": "RSS 2020; First two authors contributed equally",
  "topics": [
   "egocentric-data",
   "imitation-diffusion",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2003.06085v2",
  "pdf_url": "https://arxiv.org/pdf/2003.06085v2",
  "html_url": "https://arxiv.org/html/2003.06085v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.76
 },
 {
  "id": "2002.09291",
  "slug": "transformer-hawkes-process",
  "title": "Transformer Hawkes Process",
  "abstract": "Modern data acquisition routinely produce massive amounts of event sequence data in various domains, such as social media, healthcare, and financial markets. These data often exhibit complicated short-term and long-term temporal dependencies. However, most of the existing recurrent neural network based point process models fail to capture such dependencies, and yield unreliable prediction performance. To address this issue, we propose a Transformer Hawkes Process (THP) model, which leverages the self-attention mechanism to capture long-term dependencies and meanwhile enjoys computational efficiency. Numerical experiments on various datasets show that THP outperforms existing models in terms of both likelihood and event prediction accuracy by a notable margin. Moreover, THP is quite general and can incorporate additional structural knowledge. We provide a concrete example, where THP achieves improved prediction performance for learning multiple point processes when incorporating their relational information.",
  "published": "2020-02-21",
  "updated": "2021-02-21",
  "year": "2020",
  "authors": [
   "Simiao Zuo",
   "Haoming Jiang",
   "Zichong Li",
   "Tuo Zhao",
   "Hongyuan Zha"
  ],
  "author_count": 5,
  "categories": [
   "cs.LG",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "ICML",
  "venue_source": "semantic-scholar",
  "citations": 409,
  "influential_citations": 118,
  "tldr": "A Transformer Hawkes Process (THP) model is proposed, which leverages the self-attention mechanism to capture long-term dependencies and meanwhile enjoys computational efficiency.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Simiao Zuo",
    "id": "52194893",
    "h_index": 15,
    "papers": 32
   },
   {
    "name": "Haoming Jiang",
    "id": "5795999",
    "h_index": 34,
    "papers": 115
   },
   {
    "name": "Zichong Li",
    "id": "2145275135",
    "h_index": 8,
    "papers": 15
   },
   {
    "name": "T. Zhao",
    "id": "36345161",
    "h_index": 41,
    "papers": 122
   },
   {
    "name": "H. Zha",
    "id": "145203884",
    "h_index": 83,
    "papers": 452
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2002.09291v5",
  "pdf_url": "https://arxiv.org/pdf/2002.09291v5",
  "html_url": "https://arxiv.org/html/2002.09291v5",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.11
 },
 {
  "id": "2002.08550",
  "slug": "learning-to-walk-in-the-real-world-with-minimal-human-effort",
  "title": "Learning to Walk in the Real World with Minimal Human Effort",
  "abstract": "Reliable and stable locomotion has been one of the most fundamental challenges for legged robots. Deep reinforcement learning (deep RL) has emerged as a promising method for developing such control policies autonomously. In this paper, we develop a system for learning legged locomotion policies with deep RL in the real world with minimal human effort. The key difficulties for on-robot learning systems are automatic data collection and safety. We overcome these two challenges by developing a multi-task learning procedure and a safety-constrained RL framework. We tested our system on the task of learning to walk on three different terrains: flat ground, a soft mattress, and a doormat with crevices. Our system can automatically and efficiently learn locomotion skills on a Minitaur robot with little human intervention. The supplemental video can be found at: \\url{https://youtu.be/cwyiq6dCgOc}.",
  "published": "2020-02-20",
  "updated": "2020-11-03",
  "year": "2020",
  "authors": [
   "Sehoon Ha",
   "Peng Xu",
   "Zhenyu Tan",
   "Sergey Levine",
   "Jie Tan"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 226,
  "influential_citations": 27,
  "tldr": "This paper develops a system for learning legged locomotion policies with deep RL in the real world with minimal human effort by developing a multi-task learning procedure, an automatic reset controller, and a safety-constrained RL framework.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Sehoon Ha",
    "id": "2248552",
    "h_index": 26,
    "papers": 69
   },
   {
    "name": "P. Xu",
    "id": "1796652",
    "h_index": 9,
    "papers": 33
   },
   {
    "name": "Zhenyu Tan",
    "id": "2093186792",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "S. Levine",
    "id": "1736651",
    "h_index": 167,
    "papers": 551
   },
   {
    "name": "Jie Tan",
    "id": "1739176520",
    "h_index": 37,
    "papers": 68
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "rl-control",
   "data-teleop",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2002.08550v3",
  "pdf_url": "https://arxiv.org/pdf/2002.08550v3",
  "html_url": "https://arxiv.org/html/2002.08550v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.86
 },
 {
  "id": "2002.07217",
  "slug": "decision-making-with-auto-encoding-variational-bayes",
  "title": "Decision-Making with Auto-Encoding Variational Bayes",
  "abstract": "To make decisions based on a model fit with auto-encoding variational Bayes (AEVB), practitioners often let the variational distribution serve as a surrogate for the posterior distribution. This approach yields biased estimates of the expected risk, and therefore leads to poor decisions for two reasons. First, the model fit with AEVB may not equal the underlying data distribution. Second, the variational distribution may not equal the posterior distribution under the fitted model. We explore how fitting the variational distribution based on several objective functions other than the ELBO, while continuing to fit the generative model based on the ELBO, affects the quality of downstream decisions. For the probabilistic principal component analysis model, we investigate how importance sampling error, as well as the bias of the model parameter estimates, varies across several approximate posteriors when used as proposal distributions. Our theoretical results suggest that a posterior approximation distinct from the variational distribution should be used for making decisions. Motivated by these theoretical results, we propose learning several approximate proposals for the best model and combining them using multiple importance sampling for decision-making. In addition to toy examples, we present a full-fledged case study of single-cell RNA sequencing. In this challenging instance of multiple hypothesis testing, our proposed approach surpasses the current state of the art.",
  "published": "2020-02-17",
  "updated": "2020-10-21",
  "year": "2020",
  "authors": [
   "Romain Lopez",
   "Pierre Boyeau",
   "Nir Yosef",
   "Michael I. Jordan",
   "Jeffrey Regier"
  ],
  "author_count": 5,
  "categories": [
   "stat.ML",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "stat.ML",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 24709,
  "influential_citations": 2747,
  "tldr": "This work describes the error of importance sampling as a function of posterior variance and shows that proposal distributions learned with evidence upper bounds are better than the current state of the art.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Romain Lopez",
    "id": "39848341",
    "h_index": 20,
    "papers": 31
   },
   {
    "name": "Pierre Boyeau",
    "id": "1441878314",
    "h_index": 13,
    "papers": 22
   },
   {
    "name": "N. Yosef",
    "id": "2163873",
    "h_index": 69,
    "papers": 240
   },
   {
    "name": "Michael I. Jordan",
    "id": "1694621",
    "h_index": 188,
    "papers": 900
   },
   {
    "name": "J. Regier",
    "id": "39967607",
    "h_index": 18,
    "papers": 48
   }
  ],
  "comment": "",
  "topics": [
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2002.07217v3",
  "pdf_url": "https://arxiv.org/pdf/2002.07217v3",
  "html_url": "https://arxiv.org/html/2002.07217v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "2002.05709",
  "slug": "a-simple-framework-for-contrastive-learning-of-visual-representations",
  "title": "A Simple Framework for Contrastive Learning of Visual Representations",
  "abstract": "This paper presents SimCLR: a simple framework for contrastive learning of visual representations. We simplify recently proposed contrastive self-supervised learning algorithms without requiring specialized architectures or a memory bank. In order to understand what enables the contrastive prediction tasks to learn useful representations, we systematically study the major components of our framework. We show that (1) composition of data augmentations plays a critical role in defining effective predictive tasks, (2) introducing a learnable nonlinear transformation between the representation and the contrastive loss substantially improves the quality of the learned representations, and (3) contrastive learning benefits from larger batch sizes and more training steps compared to supervised learning. By combining these findings, we are able to considerably outperform previous methods for self-supervised and semi-supervised learning on ImageNet. A linear classifier trained on self-supervised representations learned by SimCLR achieves 76.5% top-1 accuracy, which is a 7% relative improvement over previous state-of-the-art, matching the performance of a supervised ResNet-50. When fine-tuned on only 1% of the labels, we achieve 85.8% top-5 accuracy, outperforming AlexNet with 100X fewer labels.",
  "published": "2020-02-13",
  "updated": "2020-07-01",
  "year": "2020",
  "authors": [
   "Ting Chen",
   "Simon Kornblith",
   "Mohammad Norouzi",
   "Geoffrey Hinton"
  ],
  "author_count": 4,
  "categories": [
   "cs.LG",
   "cs.CV",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "ICML",
  "venue_source": "semantic-scholar",
  "citations": 25861,
  "influential_citations": 4164,
  "tldr": "It is shown that composition of data augmentations plays a critical role in defining effective predictive tasks, and introducing a learnable nonlinear transformation between the representation and the contrastive loss substantially improves the quality of the learned representations, and contrastive learning benefits from larger batch sizes and more training steps compared to supervised learning.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ting Chen",
    "id": "145358498",
    "h_index": 28,
    "papers": 44
   },
   {
    "name": "Simon Kornblith",
    "id": "40464924",
    "h_index": 36,
    "papers": 80
   },
   {
    "name": "Mohammad Norouzi",
    "id": "144739074",
    "h_index": 71,
    "papers": 179
   },
   {
    "name": "Geoffrey E. Hinton",
    "id": "1695689",
    "h_index": 160,
    "papers": 466
   }
  ],
  "comment": "ICML'2020. Code and pretrained models at https://github.com/google-research/simclr",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2002.05709v3",
  "pdf_url": "https://arxiv.org/pdf/2002.05709v3",
  "html_url": "https://arxiv.org/html/2002.05709v3",
  "code_url": "https://github.com/google-research/simclr",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.5
 },
 {
  "id": "2002.05202",
  "slug": "glu-variants-improve-transformer",
  "title": "GLU Variants Improve Transformer",
  "abstract": "Gated Linear Units (arXiv:1612.08083) consist of the component-wise product of two linear projections, one of which is first passed through a sigmoid function. Variations on GLU are possible, using different nonlinear (or even linear) functions in place of sigmoid. We test these variants in the feed-forward sublayers of the Transformer (arXiv:1706.03762) sequence-to-sequence model, and find that some of them yield quality improvements over the typically-used ReLU or GELU activations.",
  "published": "2020-02-12",
  "updated": "2020-02-12",
  "year": "2020",
  "authors": [
   "Noam Shazeer"
  ],
  "author_count": 1,
  "categories": [
   "cs.LG",
   "cs.NE",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 2170,
  "influential_citations": 132,
  "tldr": "Gated Linear Units (GLU) consist of the component-wise product of two linear projections, one of which is first passed through a sigmoid function, and it is found that some of them yield quality improvements over the typically-used ReLU or GELU activations.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Noam Shazeer",
    "id": "1846258",
    "h_index": 40,
    "papers": 146
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2002.05202v1",
  "pdf_url": "https://arxiv.org/pdf/2002.05202v1",
  "html_url": "https://arxiv.org/html/2002.05202v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.0
 },
 {
  "id": "2002.04839",
  "slug": "laprop-separating-momentum-and-adaptivity-in-adam",
  "title": "LaProp: Separating Momentum and Adaptivity in Adam",
  "abstract": "We identity a by-far-unrecognized problem of Adam-style optimizers which results from unnecessary coupling between momentum and adaptivity. The coupling leads to instability and divergence when the momentum and adaptivity parameters are mismatched. In this work, we propose a method, Laprop, which decouples momentum and adaptivity in the Adam-style methods. We show that the decoupling leads to greater flexibility in the hyperparameters and allows for a straightforward interpolation between the signed gradient methods and the adaptive gradient methods. We experimentally show that Laprop has consistently improved speed and stability over Adam on a variety of tasks. We also bound the regret of Laprop on a convex problem and show that our bound differs from that of Adam by a key factor, which demonstrates its advantage.",
  "published": "2020-02-12",
  "updated": "2021-06-13",
  "year": "2020",
  "authors": [
   "Liu Ziyin",
   "Zhikang T. Wang",
   "Masahito Ueda"
  ],
  "author_count": 3,
  "categories": [
   "cs.LG",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 26,
  "influential_citations": 1,
  "tldr": "This work proposes a method, Laprop, which decouples momentum and adaptivity in the Adam-style methods, and shows that the decoupling leads to greater flexibility in the hyperparameters and allows for a straightforward interpolation between the signed gradient methods and the adaptive gradient methods.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Liu Ziyin",
    "id": "12907562",
    "h_index": 19,
    "papers": 50
   },
   {
    "name": "Zhikang T.Wang",
    "id": "1844276615",
    "h_index": 1,
    "papers": 1
   },
   {
    "name": "Masahito Ueda",
    "id": "2815318",
    "h_index": 69,
    "papers": 527
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2002.04839v3",
  "pdf_url": "https://arxiv.org/pdf/2002.04839v3",
  "html_url": "https://arxiv.org/html/2002.04839v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.43
 },
 {
  "id": "2002.03236",
  "slug": "tactile-dexterity-manipulation-primitives-with-tactile-feedback",
  "title": "Tactile Dexterity: Manipulation Primitives with Tactile Feedback",
  "abstract": "This paper develops closed-loop tactile controllers for dexterous robotic manipulation with a dual-palm robotic system. Tactile dexterity is an approach to dexterous manipulation that plans for robot/object interactions that render interpretable tactile information for control. We divide the role of tactile control into two goals: 1) control the contact state between the end-effector and the object (contact/no-contact, stick/slip) by regulating the stability of planned contact configurations and monitoring undesired slip events; and 2) control the object state by tactile-based tracking and iterative replanning of the object and robot trajectories. Key to this formulation is the decomposition of manipulation plans into sequences of manipulation primitives with simple mechanics and efficient planners. We consider the scenario of manipulating an object from an initial pose to a target pose on a flat surface while correcting for external perturbations and uncertainty in the initial pose of the object. We experimentally validate the approach with an ABB YuMi dual-arm robot and demonstrate the ability of the tactile controller to react to external perturbations.",
  "published": "2020-02-08",
  "updated": "2020-04-30",
  "year": "2020",
  "authors": [
   "Francois R. Hogan",
   "Jose Ballester",
   "Siyuan Dong",
   "Alberto Rodriguez"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA",
  "venue_source": "semantic-scholar",
  "citations": 122,
  "influential_citations": 2,
  "tldr": "This paper develops closed-loop tactile controllers for dexterous robotic manipulation with a dual-palm robotic system and demonstrates the ability of the tactile controller to react to external perturbations.",
  "doi": "10.1109/ICRA40945.2020.9196976",
  "oa_pdf": "https://dspace.mit.edu/bitstream/1721.1/139614.2/1/2002.03236.pdf",
  "s2_authors": [
   {
    "name": "F. Hogan",
    "id": "15820310",
    "h_index": 14,
    "papers": 28
   },
   {
    "name": "Jos\u00e9 Ballester",
    "id": "2054255124",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Siyuan Dong",
    "id": "3308345",
    "h_index": 28,
    "papers": 42
   },
   {
    "name": "Alberto Rodriguez",
    "id": "152532021",
    "h_index": 49,
    "papers": 113
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/2002.03236v2",
  "pdf_url": "https://arxiv.org/pdf/2002.03236v2",
  "html_url": "https://arxiv.org/html/2002.03236v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.59
 },
 {
  "id": "1912.06680",
  "slug": "dota-2-with-large-scale-deep-reinforcement-learning",
  "title": "Dota 2 with Large Scale Deep Reinforcement Learning",
  "abstract": "On April 13th, 2019, OpenAI Five became the first AI system to defeat the world champions at an esports game. The game of Dota 2 presents novel challenges for AI systems such as long time horizons, imperfect information, and complex, continuous state-action spaces, all challenges which will become increasingly central to more capable AI systems. OpenAI Five leveraged existing reinforcement learning techniques, scaled to learn from batches of approximately 2 million frames every 2 seconds. We developed a distributed training system and tools for continual training which allowed us to train OpenAI Five for 10 months. By defeating the Dota 2 world champion (Team OG), OpenAI Five demonstrates that self-play reinforcement learning can achieve superhuman performance on a difficult task.",
  "published": "2019-12-13",
  "updated": "2019-12-13",
  "year": "2019",
  "authors": [
   " OpenAI",
   " :",
   "Christopher Berner",
   "Greg Brockman",
   "Brooke Chan",
   "Vicki Cheung",
   "Przemys\u0142aw D\u0119biak",
   "Christy Dennison",
   "David Farhi",
   "Quirin Fischer",
   "Shariq Hashme",
   "Chris Hesse",
   "Rafal J\u00f3zefowicz",
   "Scott Gray",
   "Catherine Olsson",
   "Jakub Pachocki",
   "Michael Petrov",
   "Henrique P. d. O. Pinto",
   "Jonathan Raiman",
   "Tim Salimans",
   "Jeremy Schlatter",
   "Jonas Schneider",
   "Szymon Sidor",
   "Ilya Sutskever",
   "Jie Tang",
   "Filip Wolski",
   "Susan Zhang"
  ],
  "author_count": 27,
  "categories": [
   "cs.LG",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 2192,
  "influential_citations": 99,
  "tldr": "By defeating the Dota 2 world champion (Team OG), OpenAI Five demonstrates that self-play reinforcement learning can achieve superhuman performance on a difficult task.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Christopher Berner",
    "id": "133740015",
    "h_index": 8,
    "papers": 53
   },
   {
    "name": "Greg Brockman",
    "id": "2065151121",
    "h_index": 11,
    "papers": 39
   },
   {
    "name": "Brooke Chan",
    "id": "1466431052",
    "h_index": 8,
    "papers": 39
   },
   {
    "name": "Vicki Cheung",
    "id": "34415167",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Przemyslaw Debiak",
    "id": "1474237245",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Christy Dennison",
    "id": "1468636850",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "David Farhi",
    "id": "2065430571",
    "h_index": 11,
    "papers": 32
   },
   {
    "name": "Quirin Fischer",
    "id": "2065826788",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Shariq Hashme",
    "id": "1468728082",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Christopher Hesse",
    "id": "144239765",
    "h_index": 9,
    "papers": 19
   },
   {
    "name": "R. J\u00f3zefowicz",
    "id": "1944541",
    "h_index": 14,
    "papers": 29
   },
   {
    "name": "Scott Gray",
    "id": "145565184",
    "h_index": 14,
    "papers": 57
   },
   {
    "name": "Catherine Olsson",
    "id": "2061321863",
    "h_index": 20,
    "papers": 27
   },
   {
    "name": "J. Pachocki",
    "id": "2713380",
    "h_index": 25,
    "papers": 54
   },
   {
    "name": "Michael Petrov",
    "id": "2136008481",
    "h_index": 7,
    "papers": 27
   },
   {
    "name": "Henrique Pond\u00e9 de Oliveira Pinto",
    "id": "1463773776",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Jonathan Raiman",
    "id": "34042420",
    "h_index": 12,
    "papers": 31
   },
   {
    "name": "Tim Salimans",
    "id": "2887364",
    "h_index": 36,
    "papers": 66
   },
   {
    "name": "Jeremy Schlatter",
    "id": "1468877169",
    "h_index": 2,
    "papers": 3
   },
   {
    "name": "Jonas Schneider",
    "id": "2113526509",
    "h_index": 11,
    "papers": 21
   },
   {
    "name": "Szymon Sidor",
    "id": "2700360",
    "h_index": 14,
    "papers": 54
   },
   {
    "name": "I. Sutskever",
    "id": "1701686",
    "h_index": 75,
    "papers": 166
   },
   {
    "name": "Jie Tang",
    "id": "2109541439",
    "h_index": 31,
    "papers": 51
   },
   {
    "name": "Filip Wolski",
    "id": "143909660",
    "h_index": 5,
    "papers": 8
   },
   {
    "name": "Susan Zhang",
    "id": "2108244542",
    "h_index": 7,
    "papers": 10
   }
  ],
  "comment": "",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1912.06680v1",
  "pdf_url": "https://arxiv.org/pdf/1912.06680v1",
  "html_url": "https://arxiv.org/html/1912.06680v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.0
 },
 {
  "id": "1912.04443",
  "slug": "avid-learning-multi-stage-tasks-via-pixel-level-translation-of-human-v",
  "title": "AVID: Learning Multi-Stage Tasks via Pixel-Level Translation of Human Videos",
  "abstract": "Robotic reinforcement learning (RL) holds the promise of enabling robots to learn complex behaviors through experience. However, realizing this promise for long-horizon tasks in the real world requires mechanisms to reduce human burden in terms of defining the task and scaffolding the learning process. In this paper, we study how these challenges can be alleviated with an automated robotic learning framework, in which multi-stage tasks are defined simply by providing videos of a human demonstrator and then learned autonomously by the robot from raw image observations. A central challenge in imitating human videos is the difference in appearance between the human and robot, which typically requires manual correspondence. We instead take an automated approach and perform pixel-level image translation via CycleGAN to convert the human demonstration into a video of a robot, which can then be used to construct a reward function for a model-based RL algorithm. The robot then learns the task one stage at a time, automatically learning how to reset each stage to retry it multiple times without human-provided resets. This makes the learning process largely automatic, from intuitive task specification via a video to automated training with minimal human intervention. We demonstrate that our approach is capable of learning complex tasks, such as operating a coffee machine, directly from raw image observations, requiring only 20 minutes to provide human demonstrations and about 180 minutes of robot interaction.",
  "published": "2019-12-10",
  "updated": "2020-06-21",
  "year": "2019",
  "authors": [
   "Laura Smith",
   "Nikita Dhawan",
   "Marvin Zhang",
   "Pieter Abbeel",
   "Sergey Levine"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "RSS",
  "venue_source": "semantic-scholar",
  "citations": 189,
  "influential_citations": 5,
  "tldr": "This paper takes an automated approach and performs pixel-level image translation via CycleGAN to convert the human demonstration into a video of a robot, which can then be used to construct a reward function for a model-based RL algorithm.",
  "doi": "10.15607/rss.2020.xvi.024",
  "oa_pdf": "https://doi.org/10.15607/rss.2020.xvi.024",
  "s2_authors": [
   {
    "name": "Laura M. Smith",
    "id": "152447364",
    "h_index": 15,
    "papers": 16
   },
   {
    "name": "Nikita Dhawan",
    "id": "2106046430",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Marvin Zhang",
    "id": "2634261",
    "h_index": 9,
    "papers": 12
   },
   {
    "name": "P. Abbeel",
    "id": "1689992",
    "h_index": 165,
    "papers": 596
   },
   {
    "name": "S. Levine",
    "id": "1736651",
    "h_index": 167,
    "papers": 551
   }
  ],
  "comment": "Robotics: Science and Systems (RSS) 2020 camera ready submission. Project website: https://sites.google.com/view/rss20avid",
  "topics": [
   "egocentric-data",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1912.04443v3",
  "pdf_url": "https://arxiv.org/pdf/1912.04443v3",
  "html_url": "https://arxiv.org/html/1912.04443v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.78
 },
 {
  "id": "1912.04344",
  "slug": "grasping-in-the-wild-learning-6dof-closed-loop-grasping-from-low-cost",
  "title": "Grasping in the Wild:Learning 6DoF Closed-Loop Grasping from Low-Cost Demonstrations",
  "abstract": "Intelligent manipulation benefits from the capacity to flexibly control an end-effector with high degrees of freedom (DoF) and dynamically react to the environment. However, due to the challenges of collecting effective training data and learning efficiently, most grasping algorithms today are limited to top-down movements and open-loop execution. In this work, we propose a new low-cost hardware interface for collecting grasping demonstrations by people in diverse environments. Leveraging this data, we show that it is possible to train a robust end-to-end 6DoF closed-loop grasping model with reinforcement learning that transfers to real robots. A key aspect of our grasping model is that it uses \"action-view\" based rendering to simulate future states with respect to different possible actions. By evaluating these states using a learned value function (Q-function), our method is able to better select corresponding actions that maximize total rewards (i.e., grasping success). Our final grasping system is able to achieve reliable 6DoF closed-loop grasping of novel objects across various scene configurations, as well as dynamic scenes with moving objects.",
  "published": "2019-12-09",
  "updated": "2020-06-17",
  "year": "2019",
  "authors": [
   "Shuran Song",
   "Andy Zeng",
   "Johnny Lee",
   "Thomas Funkhouser"
  ],
  "author_count": 4,
  "categories": [
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "RA-L",
  "venue_source": "semantic-scholar",
  "citations": 278,
  "influential_citations": 11,
  "tldr": "This work proposes a new low-cost hardware interface for collecting grasping demonstrations by people in diverse environments that makes it possible to train a robust end-to-end 6DoF closed-loop grasping model with reinforcement learning that transfers to real robots.",
  "doi": "10.1109/LRA.2020.3004787",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Shuran Song",
    "id": "3340170",
    "h_index": 59,
    "papers": 90
   },
   {
    "name": "Andy Zeng",
    "id": "38591293",
    "h_index": 34,
    "papers": 50
   },
   {
    "name": "Johnny Lee",
    "id": "2108488231",
    "h_index": 11,
    "papers": 14
   },
   {
    "name": "T. Funkhouser",
    "id": "1807080",
    "h_index": 89,
    "papers": 208
   }
  ],
  "comment": "Project Webpage https://graspinwild.cs.columbia.edu/",
  "topics": [
   "dexterous-manipulation",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1912.04344v2",
  "pdf_url": "https://arxiv.org/pdf/1912.04344v2",
  "html_url": "https://arxiv.org/html/1912.04344v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.95
 },
 {
  "id": "1912.01734",
  "slug": "alfred-a-benchmark-for-interpreting-grounded-instructions-for-everyday",
  "title": "ALFRED: A Benchmark for Interpreting Grounded Instructions for Everyday Tasks",
  "abstract": "We present ALFRED (Action Learning From Realistic Environments and Directives), a benchmark for learning a mapping from natural language instructions and egocentric vision to sequences of actions for household tasks. ALFRED includes long, compositional tasks with non-reversible state changes to shrink the gap between research benchmarks and real-world applications. ALFRED consists of expert demonstrations in interactive visual environments for 25k natural language directives. These directives contain both high-level goals like \"Rinse off a mug and place it in the coffee maker.\" and low-level language instructions like \"Walk to the coffee maker on the right.\" ALFRED tasks are more complex in terms of sequence length, action space, and language than existing vision-and-language task datasets. We show that a baseline model based on recent embodied vision-and-language tasks performs poorly on ALFRED, suggesting that there is significant room for developing innovative grounded visual language understanding models with this benchmark.",
  "published": "2019-12-03",
  "updated": "2020-03-31",
  "year": "2019",
  "authors": [
   "Mohit Shridhar",
   "Jesse Thomason",
   "Daniel Gordon",
   "Yonatan Bisk",
   "Winson Han",
   "Roozbeh Mottaghi",
   "Luke Zettlemoyer",
   "Dieter Fox"
  ],
  "author_count": 8,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.CL",
   "cs.LG",
   "cs.RO"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 1179,
  "influential_citations": 179,
  "tldr": "It is shown that a baseline model based on recent embodied vision-and-language tasks performs poorly on ALFRED, suggesting that there is significant room for developing innovative grounded visual language understanding models with this benchmark.",
  "doi": "10.1109/cvpr42600.2020.01075",
  "oa_pdf": "https://arxiv.org/pdf/1912.01734",
  "s2_authors": [
   {
    "name": "Mohit Shridhar",
    "id": "33516562",
    "h_index": 15,
    "papers": 23
   },
   {
    "name": "Jesse Thomason",
    "id": "2665873",
    "h_index": 27,
    "papers": 63
   },
   {
    "name": "Daniel Gordon",
    "id": "152462964",
    "h_index": 11,
    "papers": 16
   },
   {
    "name": "Yonatan Bisk",
    "id": "3312309",
    "h_index": 46,
    "papers": 144
   },
   {
    "name": "Winson Han",
    "id": "1443358534",
    "h_index": 12,
    "papers": 17
   },
   {
    "name": "Roozbeh Mottaghi",
    "id": "3012475",
    "h_index": 46,
    "papers": 102
   },
   {
    "name": "Luke Zettlemoyer",
    "id": "1982950",
    "h_index": 118,
    "papers": 278
   },
   {
    "name": "D. Fox",
    "id": "145197953",
    "h_index": 133,
    "papers": 428
   }
  ],
  "comment": "Computer Vision and Pattern Recognition (CVPR) 2020 ; https://askforalfred.com/",
  "topics": [
   "egocentric-data"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1912.01734v2",
  "pdf_url": "https://arxiv.org/pdf/1912.01734v2",
  "html_url": "https://arxiv.org/html/1912.01734v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "1912.01603",
  "slug": "dream-to-control-learning-behaviors-by-latent-imagination",
  "title": "Dream to Control: Learning Behaviors by Latent Imagination",
  "abstract": "Learned world models summarize an agent's experience to facilitate learning complex behaviors. While learning world models from high-dimensional sensory inputs is becoming feasible through deep learning, there are many potential ways for deriving behaviors from them. We present Dreamer, a reinforcement learning agent that solves long-horizon tasks from images purely by latent imagination. We efficiently learn behaviors by propagating analytic gradients of learned state values back through trajectories imagined in the compact state space of a learned world model. On 20 challenging visual control tasks, Dreamer exceeds existing approaches in data-efficiency, computation time, and final performance.",
  "published": "2019-12-03",
  "updated": "2020-03-17",
  "year": "2019",
  "authors": [
   "Danijar Hafner",
   "Timothy Lillicrap",
   "Jimmy Ba",
   "Mohammad Norouzi"
  ],
  "author_count": 4,
  "categories": [
   "cs.LG",
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.LG",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 2186,
  "influential_citations": 289,
  "tldr": "Dreamer is presented, a reinforcement learning agent that solves long-horizon tasks purely by latent imagination and efficiently learn behaviors by backpropagating analytic gradients of learned state values through trajectories imagined in the compact state space of a learned world model.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Danijar Hafner",
    "id": "35006479",
    "h_index": 25,
    "papers": 47
   },
   {
    "name": "T. Lillicrap",
    "id": "2542999",
    "h_index": 69,
    "papers": 154
   },
   {
    "name": "Jimmy Ba",
    "id": "2503659",
    "h_index": 46,
    "papers": 84
   },
   {
    "name": "Mohammad Norouzi",
    "id": "144739074",
    "h_index": 71,
    "papers": 179
   }
  ],
  "comment": "9 pages, 12 figures",
  "topics": [
   "world-models",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1912.01603v3",
  "pdf_url": "https://arxiv.org/pdf/1912.01603v3",
  "html_url": "https://arxiv.org/html/1912.01603v3",
  "code_url": null,
  "presented": [
   {
    "club": "saturday-robotics",
    "club_name": "Saturday Robotics",
    "session_number": 7,
    "session_title": "Robotics & World Models Reading Club 07: Learning to Dream: World Models, Imagination, Path to Foundation Models for Control \u2014 Los Altos",
    "date_text": "Saturday, May 9, 2026 | 2:00 PM \u2013 5:00 PM",
    "city": "",
    "url": "https://lu.ma/srhe0vuo",
    "listed_as": "Dream to Control (2019)"
   }
  ],
  "club_note": "Replaces search/planning with actor-critic trained entirely in imagination",
  "featured": true,
  "signal": 7.5
 },
 {
  "id": "1912.01588",
  "slug": "leveraging-procedural-generation-to-benchmark-reinforcement-learning",
  "title": "Leveraging Procedural Generation to Benchmark Reinforcement Learning",
  "abstract": "We introduce Procgen Benchmark, a suite of 16 procedurally generated game-like environments designed to benchmark both sample efficiency and generalization in reinforcement learning. We believe that the community will benefit from increased access to high quality training environments, and we provide detailed experimental protocols for using this benchmark. We empirically demonstrate that diverse environment distributions are essential to adequately train and evaluate RL agents, thereby motivating the extensive use of procedural content generation. We then use this benchmark to investigate the effects of scaling model size, finding that larger models significantly improve both sample efficiency and generalization.",
  "published": "2019-12-03",
  "updated": "2020-07-26",
  "year": "2019",
  "authors": [
   "Karl Cobbe",
   "Christopher Hesse",
   "Jacob Hilton",
   "John Schulman"
  ],
  "author_count": 4,
  "categories": [
   "cs.LG",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "ICML",
  "venue_source": "semantic-scholar",
  "citations": 736,
  "influential_citations": 158,
  "tldr": "This work empirically demonstrate that diverse environment distributions are essential to adequately train and evaluate RL agents, thereby motivating the extensive use of procedural content generation and uses this benchmark to investigate the effects of scaling model size.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "K. Cobbe",
    "id": "6062736",
    "h_index": 11,
    "papers": 53
   },
   {
    "name": "Christopher Hesse",
    "id": "144239765",
    "h_index": 9,
    "papers": 19
   },
   {
    "name": "Jacob Hilton",
    "id": "144890163",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "John Schulman",
    "id": "47971768",
    "h_index": 45,
    "papers": 69
   }
  ],
  "comment": "",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1912.01588v2",
  "pdf_url": "https://arxiv.org/pdf/1912.01588v2",
  "html_url": "https://arxiv.org/html/1912.01588v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.37
 },
 {
  "id": "1911.08265",
  "slug": "mastering-atari-go-chess-and-shogi-by-planning-with-a-learned-model",
  "title": "Mastering Atari, Go, Chess and Shogi by Planning with a Learned Model",
  "abstract": "Constructing agents with planning capabilities has long been one of the main challenges in the pursuit of artificial intelligence. Tree-based planning methods have enjoyed huge success in challenging domains, such as chess and Go, where a perfect simulator is available. However, in real-world problems the dynamics governing the environment are often complex and unknown. In this work we present the MuZero algorithm which, by combining a tree-based search with a learned model, achieves superhuman performance in a range of challenging and visually complex domains, without any knowledge of their underlying dynamics. MuZero learns a model that, when applied iteratively, predicts the quantities most directly relevant to planning: the reward, the action-selection policy, and the value function. When evaluated on 57 different Atari games - the canonical video game environment for testing AI techniques, in which model-based planning approaches have historically struggled - our new algorithm achieved a new state of the art. When evaluated on Go, chess and shogi, without any knowledge of the game rules, MuZero matched the superhuman performance of the AlphaZero algorithm that was supplied with the game rules.",
  "published": "2019-11-19",
  "updated": "2020-02-21",
  "year": "2019",
  "authors": [
   "Julian Schrittwieser",
   "Ioannis Antonoglou",
   "Thomas Hubert",
   "Karen Simonyan",
   "Laurent Sifre",
   "Simon Schmitt",
   "Arthur Guez",
   "Edward Lockhart",
   "Demis Hassabis",
   "Thore Graepel",
   "Timothy Lillicrap",
   "David Silver"
  ],
  "author_count": 12,
  "categories": [
   "cs.LG",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "Nature",
  "venue_source": "semantic-scholar",
  "citations": 2685,
  "influential_citations": 196,
  "tldr": "The MuZero algorithm is presented, which, by combining a tree-based search with a learned model, achieves superhuman performance in a range of challenging and visually complex domains, without any knowledge of their underlying dynamics.",
  "doi": "10.1038/s41586-020-03051-4",
  "oa_pdf": "https://arxiv.org/pdf/1911.08265",
  "s2_authors": [
   {
    "name": "Julian Schrittwieser",
    "id": "4337102",
    "h_index": 20,
    "papers": 32
   },
   {
    "name": "Ioannis Antonoglou",
    "id": "2460849",
    "h_index": 20,
    "papers": 35
   },
   {
    "name": "T. Hubert",
    "id": "2067208983",
    "h_index": 12,
    "papers": 18
   },
   {
    "name": "K. Simonyan",
    "id": "34838386",
    "h_index": 65,
    "papers": 108
   },
   {
    "name": "L. Sifre",
    "id": "2175946",
    "h_index": 28,
    "papers": 42
   },
   {
    "name": "Simon Schmitt",
    "id": "152380508",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "A. Guez",
    "id": "35099444",
    "h_index": 27,
    "papers": 50
   },
   {
    "name": "Edward Lockhart",
    "id": "49860549",
    "h_index": 14,
    "papers": 19
   },
   {
    "name": "D. Hassabis",
    "id": "48987704",
    "h_index": 92,
    "papers": 160
   },
   {
    "name": "T. Graepel",
    "id": "1686971",
    "h_index": 60,
    "papers": 185
   },
   {
    "name": "T. Lillicrap",
    "id": "2542999",
    "h_index": 69,
    "papers": 154
   },
   {
    "name": "David Silver",
    "id": "145824029",
    "h_index": 80,
    "papers": 120
   }
  ],
  "comment": "",
  "topics": [
   "sim2real"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1911.08265v2",
  "pdf_url": "https://arxiv.org/pdf/1911.08265v2",
  "html_url": "https://arxiv.org/html/1911.08265v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "1911.05722",
  "slug": "momentum-contrast-for-unsupervised-visual-representation-learning",
  "title": "Momentum Contrast for Unsupervised Visual Representation Learning",
  "abstract": "We present Momentum Contrast (MoCo) for unsupervised visual representation learning. From a perspective on contrastive learning as dictionary look-up, we build a dynamic dictionary with a queue and a moving-averaged encoder. This enables building a large and consistent dictionary on-the-fly that facilitates contrastive unsupervised learning. MoCo provides competitive results under the common linear protocol on ImageNet classification. More importantly, the representations learned by MoCo transfer well to downstream tasks. MoCo can outperform its supervised pre-training counterpart in 7 detection/segmentation tasks on PASCAL VOC, COCO, and other datasets, sometimes surpassing it by large margins. This suggests that the gap between unsupervised and supervised representation learning has been largely closed in many vision tasks.",
  "published": "2019-11-13",
  "updated": "2020-03-23",
  "year": "2019",
  "authors": [
   "Kaiming He",
   "Haoqi Fan",
   "Yuxin Wu",
   "Saining Xie",
   "Ross Girshick"
  ],
  "author_count": 5,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 15597,
  "influential_citations": 2136,
  "tldr": "",
  "doi": "10.1109/cvpr42600.2020.00975",
  "oa_pdf": "https://arxiv.org/pdf/1911.05722",
  "s2_authors": [
   {
    "name": "Kaiming He",
    "id": "39353098",
    "h_index": 67,
    "papers": 85
   },
   {
    "name": "Haoqi Fan",
    "id": "146884473",
    "h_index": 27,
    "papers": 36
   },
   {
    "name": "Yuxin Wu",
    "id": "98264506",
    "h_index": 26,
    "papers": 80
   },
   {
    "name": "Saining Xie",
    "id": "1817030",
    "h_index": 32,
    "papers": 46
   },
   {
    "name": "Ross B. Girshick",
    "id": "2983898",
    "h_index": 80,
    "papers": 113
   }
  ],
  "comment": "CVPR 2020 camera-ready. Code: https://github.com/facebookresearch/moco",
  "topics": [
   "foundation-pretraining"
  ],
  "orgs": [
   "Meta FAIR"
  ],
  "abs_url": "https://arxiv.org/abs/1911.05722v3",
  "pdf_url": "https://arxiv.org/pdf/1911.05722v3",
  "html_url": "https://arxiv.org/html/1911.05722v3",
  "code_url": "https://github.com/facebookresearch/moco",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 6.0
 },
 {
  "id": "1910.11215",
  "slug": "robonet-large-scale-multi-robot-learning",
  "title": "RoboNet: Large-Scale Multi-Robot Learning",
  "abstract": "Robot learning has emerged as a promising tool for taming the complexity and diversity of the real world. Methods based on high-capacity models, such as deep networks, hold the promise of providing effective generalization to a wide range of open-world environments. However, these same methods typically require large amounts of diverse training data to generalize effectively. In contrast, most robotic learning experiments are small-scale, single-domain, and single-robot. This leads to a frequent tension in robotic learning: how can we learn generalizable robotic controllers without having to collect impractically large amounts of data for each separate experiment? In this paper, we propose RoboNet, an open database for sharing robotic experience, which provides an initial pool of 15 million video frames, from 7 different robot platforms, and study how it can be used to learn generalizable models for vision-based robotic manipulation. We combine the dataset with two different learning algorithms: visual foresight, which uses forward video prediction models, and supervised inverse models. Our experiments test the learned algorithms' ability to work across new objects, new tasks, new scenes, new camera viewpoints, new grippers, or even entirely new robots. In our final experiment, we find that by pre-training on RoboNet and fine-tuning on data from a held-out Franka or Kuka robot, we can exceed the performance of a robot-specific training approach that uses 4x-20x more data. For videos and data, see the project webpage: https://www.robonet.wiki/",
  "published": "2019-10-24",
  "updated": "2020-01-02",
  "year": "2019",
  "authors": [
   "Sudeep Dasari",
   "Frederik Ebert",
   "Stephen Tian",
   "Suraj Nair",
   "Bernadette Bucher",
   "Karl Schmeckpeper",
   "Siddharth Singh",
   "Sergey Levine",
   "Chelsea Finn"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 452,
  "influential_citations": 23,
  "tldr": "This paper proposes RoboNet, an open database for sharing robotic experience, which provides an initial pool of 15 million video frames, from 7 different robot platforms, and studies how it can be used to learn generalizable models for vision-based robotic manipulation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "S. Dasari",
    "id": "36076404",
    "h_index": 23,
    "papers": 39
   },
   {
    "name": "F. Ebert",
    "id": "27535721",
    "h_index": 15,
    "papers": 22
   },
   {
    "name": "Stephen Tian",
    "id": "71692259",
    "h_index": 13,
    "papers": 14
   },
   {
    "name": "Suraj Nair",
    "id": "4734949",
    "h_index": 20,
    "papers": 36
   },
   {
    "name": "Bernadette Bucher",
    "id": "47015098",
    "h_index": 8,
    "papers": 27
   },
   {
    "name": "Karl Schmeckpeper",
    "id": "88726258",
    "h_index": 14,
    "papers": 35
   },
   {
    "name": "Siddharth Singh",
    "id": "2130661865",
    "h_index": 5,
    "papers": 13
   },
   {
    "name": "S. Levine",
    "id": "1736651",
    "h_index": 167,
    "papers": 551
   },
   {
    "name": "Chelsea Finn",
    "id": "46881670",
    "h_index": 94,
    "papers": 240
   }
  ],
  "comment": "accepted at the Conference on Robot Learning (CoRL) 2019",
  "topics": [
   "world-models",
   "foundation-pretraining",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1910.11215v2",
  "pdf_url": "https://arxiv.org/pdf/1910.11215v2",
  "html_url": "https://arxiv.org/html/1910.11215v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.16
 },
 {
  "id": "1910.10897",
  "slug": "meta-world-a-benchmark-and-evaluation-for-multi-task-and-meta-reinforc",
  "title": "Meta-World: A Benchmark and Evaluation for Multi-Task and Meta Reinforcement Learning",
  "abstract": "Meta-reinforcement learning algorithms can enable robots to acquire new skills much more quickly, by leveraging prior experience to learn how to learn. However, much of the current research on meta-reinforcement learning focuses on task distributions that are very narrow. For example, a commonly used meta-reinforcement learning benchmark uses different running velocities for a simulated robot as different tasks. When policies are meta-trained on such narrow task distributions, they cannot possibly generalize to more quickly acquire entirely new tasks. Therefore, if the aim of these methods is to enable faster acquisition of entirely new behaviors, we must evaluate them on task distributions that are sufficiently broad to enable generalization to new behaviors. In this paper, we propose an open-source simulated benchmark for meta-reinforcement learning and multi-task learning consisting of 50 distinct robotic manipulation tasks. Our aim is to make it possible to develop algorithms that generalize to accelerate the acquisition of entirely new, held-out tasks. We evaluate 7 state-of-the-art meta-reinforcement learning and multi-task learning algorithms on these tasks. Surprisingly, while each task and its variations (e.g., with different object positions) can be learned with reasonable success, these algorithms struggle to learn with multiple tasks at the same time, even with as few as ten distinct training tasks. Our analysis and open-source environments pave the way for future research in multi-task learning and meta-learning that can enable meaningful generalization, thereby unlocking the full potential of these methods.",
  "published": "2019-10-24",
  "updated": "2021-06-14",
  "year": "2019",
  "authors": [
   "Tianhe Yu",
   "Deirdre Quillen",
   "Zhanpeng He",
   "Ryan Julian",
   "Avnish Narayan",
   "Hayden Shively",
   "Adithya Bellathur",
   "Karol Hausman",
   "Chelsea Finn",
   "Sergey Levine"
  ],
  "author_count": 10,
  "categories": [
   "cs.LG",
   "cs.AI",
   "cs.RO",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 1750,
  "influential_citations": 318,
  "tldr": "An open-source simulated benchmark for meta-reinforcement learning and multi-task learning consisting of 50 distinct robotic manipulation tasks is proposed to make it possible to develop algorithms that generalize to accelerate the acquisition of entirely new, held-out tasks.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Tianhe Yu",
    "id": "10909315",
    "h_index": 31,
    "papers": 48
   },
   {
    "name": "Deirdre Quillen",
    "id": "47202040",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Zhanpeng He",
    "id": "3402875",
    "h_index": 12,
    "papers": 21
   },
   {
    "name": "Ryan C. Julian",
    "id": "144885996",
    "h_index": 19,
    "papers": 34
   },
   {
    "name": "Karol Hausman",
    "id": "1944801",
    "h_index": 47,
    "papers": 122
   },
   {
    "name": "Chelsea Finn",
    "id": "46881670",
    "h_index": 94,
    "papers": 240
   },
   {
    "name": "S. Levine",
    "id": "1736651",
    "h_index": 167,
    "papers": 551
   }
  ],
  "comment": "This is an update version of a manuscript that originally appeared at CoRL 2019. Videos are here: meta-world.github.io, open-sourced code are available at: https://github.com/rlworkgroup/metaworld, and the baselines can be found at https://github.com/rlworkgroup/garage",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1910.10897v2",
  "pdf_url": "https://arxiv.org/pdf/1910.10897v2",
  "html_url": "https://arxiv.org/html/1910.10897v2",
  "code_url": "https://github.com/rlworkgroup/metaworld",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.5
 },
 {
  "id": "1910.10683",
  "slug": "exploring-the-limits-of-transfer-learning-with-a-unified-text-to-text",
  "title": "Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer",
  "abstract": "Transfer learning, where a model is first pre-trained on a data-rich task before being fine-tuned on a downstream task, has emerged as a powerful technique in natural language processing (NLP). The effectiveness of transfer learning has given rise to a diversity of approaches, methodology, and practice. In this paper, we explore the landscape of transfer learning techniques for NLP by introducing a unified framework that converts all text-based language problems into a text-to-text format. Our systematic study compares pre-training objectives, architectures, unlabeled data sets, transfer approaches, and other factors on dozens of language understanding tasks. By combining the insights from our exploration with scale and our new ``Colossal Clean Crawled Corpus'', we achieve state-of-the-art results on many benchmarks covering summarization, question answering, text classification, and more. To facilitate future work on transfer learning for NLP, we release our data set, pre-trained models, and code.",
  "published": "2019-10-23",
  "updated": "2023-09-19",
  "year": "2019",
  "authors": [
   "Colin Raffel",
   "Noam Shazeer",
   "Adam Roberts",
   "Katherine Lee",
   "Sharan Narang",
   "Michael Matena",
   "Yanqi Zhou",
   "Wei Li",
   "Peter J. Liu"
  ],
  "author_count": 9,
  "categories": [
   "cs.LG",
   "cs.CL",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 27033,
  "influential_citations": 2606,
  "tldr": "This systematic study compares pre-training objectives, architectures, unlabeled datasets, transfer approaches, and other factors on dozens of language understanding tasks and achieves state-of-the-art results on many benchmarks covering summarization, question answering, text classification, and more.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Colin Raffel",
    "id": "2402716",
    "h_index": 59,
    "papers": 105
   },
   {
    "name": "Noam Shazeer",
    "id": "1846258",
    "h_index": 40,
    "papers": 146
   },
   {
    "name": "Adam Roberts",
    "id": "145625142",
    "h_index": 40,
    "papers": 60
   },
   {
    "name": "Katherine Lee",
    "id": "3844009",
    "h_index": 20,
    "papers": 25
   },
   {
    "name": "Sharan Narang",
    "id": "46617804",
    "h_index": 31,
    "papers": 105
   },
   {
    "name": "Michael Matena",
    "id": "1380243217",
    "h_index": 5,
    "papers": 21
   },
   {
    "name": "Yanqi Zhou",
    "id": "2389316",
    "h_index": 30,
    "papers": 73
   },
   {
    "name": "Wei Li",
    "id": "2157338362",
    "h_index": 3,
    "papers": 5
   },
   {
    "name": "Peter J. Liu",
    "id": "35025299",
    "h_index": 20,
    "papers": 35
   }
  ],
  "comment": "",
  "topics": [
   "navigation",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1910.10683v4",
  "pdf_url": "https://arxiv.org/pdf/1910.10683v4",
  "html_url": "https://arxiv.org/html/1910.10683v4",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.0
 },
 {
  "id": "1910.07113",
  "slug": "solving-rubik-s-cube-with-a-robot-hand",
  "title": "Solving Rubik's Cube with a Robot Hand",
  "abstract": "We demonstrate that models trained only in simulation can be used to solve a manipulation problem of unprecedented complexity on a real robot. This is made possible by two key components: a novel algorithm, which we call automatic domain randomization (ADR) and a robot platform built for machine learning. ADR automatically generates a distribution over randomized environments of ever-increasing difficulty. Control policies and vision state estimators trained with ADR exhibit vastly improved sim2real transfer. For control policies, memory-augmented models trained on an ADR-generated distribution of environments show clear signs of emergent meta-learning at test time. The combination of ADR with our custom robot platform allows us to solve a Rubik's cube with a humanoid robot hand, which involves both control and state estimation problems. Videos summarizing our results are available: https://openai.com/blog/solving-rubiks-cube/",
  "published": "2019-10-16",
  "updated": "2019-10-16",
  "year": "2019",
  "authors": [
   " OpenAI",
   "Ilge Akkaya",
   "Marcin Andrychowicz",
   "Maciek Chociej",
   "Mateusz Litwin",
   "Bob McGrew",
   "Arthur Petron",
   "Alex Paino",
   "Matthias Plappert",
   "Glenn Powell",
   "Raphael Ribas",
   "Jonas Schneider",
   "Nikolas Tezak",
   "Jerry Tworek",
   "Peter Welinder",
   "Lilian Weng",
   "Qiming Yuan",
   "Wojciech Zaremba",
   "Lei Zhang"
  ],
  "author_count": 19,
  "categories": [
   "cs.LG",
   "cs.AI",
   "cs.CV",
   "cs.RO",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 1518,
  "influential_citations": 66,
  "tldr": "It is demonstrated that models trained only in simulation can be used to solve a manipulation problem of unprecedented complexity on a real robot, made possible by a novel algorithm, which is called automatic domain randomization (ADR), and a robot platform built for machine learning.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "OpenAI",
    "id": "51139888",
    "h_index": 6,
    "papers": 16
   },
   {
    "name": "Ilge Akkaya",
    "id": "2258629",
    "h_index": 13,
    "papers": 48
   },
   {
    "name": "Marcin Andrychowicz",
    "id": "2206490",
    "h_index": 24,
    "papers": 35
   },
   {
    "name": "Maciek Chociej",
    "id": "36045639",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Ma-teusz Litwin",
    "id": "1380985420",
    "h_index": 8,
    "papers": 49
   },
   {
    "name": "Bob McGrew",
    "id": "39593364",
    "h_index": 14,
    "papers": 30
   },
   {
    "name": "Arthur Petron",
    "id": "6817951",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "A. Paino",
    "id": "34800652",
    "h_index": 10,
    "papers": 27
   },
   {
    "name": "Matthias Plappert",
    "id": "3407285",
    "h_index": 15,
    "papers": 44
   },
   {
    "name": "Glenn Powell",
    "id": "2059171221",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "R. Ribas",
    "id": "1380603785",
    "h_index": 1,
    "papers": 2
   },
   {
    "name": "Jonas Schneider",
    "id": "2113526509",
    "h_index": 11,
    "papers": 21
   },
   {
    "name": "N. Tezak",
    "id": "145950540",
    "h_index": 18,
    "papers": 50
   },
   {
    "name": "Jerry Tworek",
    "id": "2065005836",
    "h_index": 13,
    "papers": 61
   },
   {
    "name": "Peter Welinder",
    "id": "2930640",
    "h_index": 17,
    "papers": 37
   },
   {
    "name": "Lilian Weng",
    "id": "2065741038",
    "h_index": 9,
    "papers": 13
   },
   {
    "name": "Qiming Yuan",
    "id": "153930486",
    "h_index": 10,
    "papers": 22
   },
   {
    "name": "Wojciech Zaremba",
    "id": "2563432",
    "h_index": 31,
    "papers": 40
   },
   {
    "name": "Lei M. Zhang",
    "id": "2152836492",
    "h_index": 4,
    "papers": 8
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "sim2real"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1910.07113v1",
  "pdf_url": "https://arxiv.org/pdf/1910.07113v1",
  "html_url": "https://arxiv.org/html/1910.07113v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.0
 },
 {
  "id": "1910.04142",
  "slug": "imagined-value-gradients-model-based-policy-optimization-with-transfer",
  "title": "Imagined Value Gradients: Model-Based Policy Optimization with Transferable Latent Dynamics Models",
  "abstract": "Humans are masters at quickly learning many complex tasks, relying on an approximate understanding of the dynamics of their environments. In much the same way, we would like our learning agents to quickly adapt to new tasks. In this paper, we explore how model-based Reinforcement Learning (RL) can facilitate transfer to new tasks. We develop an algorithm that learns an action-conditional, predictive model of expected future observations, rewards and values from which a policy can be derived by following the gradient of the estimated value along imagined trajectories. We show how robust policy optimization can be achieved in robot manipulation tasks even with approximate models that are learned directly from vision and proprioception. We evaluate the efficacy of our approach in a transfer learning scenario, re-using previously learned models on tasks with different reward structures and visual distractors, and show a significant improvement in learning speed compared to strong off-policy baselines. Videos with results can be found at https://sites.google.com/view/ivg-corl19",
  "published": "2019-10-09",
  "updated": "2019-10-09",
  "year": "2019",
  "authors": [
   "Arunkumar Byravan",
   "Jost Tobias Springenberg",
   "Abbas Abdolmaleki",
   "Roland Hafner",
   "Michael Neunert",
   "Thomas Lampe",
   "Noah Siegel",
   "Nicolas Heess",
   "Martin Riedmiller"
  ],
  "author_count": 9,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG",
   "cs.NE"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 45,
  "influential_citations": 3,
  "tldr": "An algorithm is developed that learns an action-conditional, predictive model of expected future observations, rewards and values from which a policy can be derived by following the gradient of the estimated value along imagined trajectories.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Arunkumar Byravan",
    "id": "2631257",
    "h_index": 20,
    "papers": 41
   },
   {
    "name": "Jost Tobias Springenberg",
    "id": "2060551",
    "h_index": 44,
    "papers": 93
   },
   {
    "name": "A. Abdolmaleki",
    "id": "2799799",
    "h_index": 29,
    "papers": 95
   },
   {
    "name": "Roland Hafner",
    "id": "49512734",
    "h_index": 21,
    "papers": 36
   },
   {
    "name": "M. Neunert",
    "id": "2366050",
    "h_index": 29,
    "papers": 50
   },
   {
    "name": "Thomas Lampe",
    "id": "2066153554",
    "h_index": 22,
    "papers": 36
   },
   {
    "name": "Noah Siegel",
    "id": "1500370330",
    "h_index": 15,
    "papers": 19
   },
   {
    "name": "N. Heess",
    "id": "2801204",
    "h_index": 73,
    "papers": 192
   },
   {
    "name": "Martin A. Riedmiller",
    "id": "3137672",
    "h_index": 55,
    "papers": 219
   }
  ],
  "comment": "To appear at the 3rd annual Conference on Robot Learning, Osaka, Japan (CoRL 2019). 24 pages including appendix (main paper - 8 pages)",
  "topics": [
   "world-models",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1910.04142v1",
  "pdf_url": "https://arxiv.org/pdf/1910.04142v1",
  "html_url": "https://arxiv.org/html/1910.04142v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.16
 },
 {
  "id": "1910.03973",
  "slug": "towards-learning-to-detect-and-predict-contact-events-on-vision-based",
  "title": "Towards Learning to Detect and Predict Contact Events on Vision-based Tactile Sensors",
  "abstract": "In essence, successful grasp boils down to correct responses to multiple contact events between fingertips and objects. In most scenarios, tactile sensing is adequate to distinguish contact events. Due to the nature of high dimensionality of tactile information, classifying spatiotemporal tactile signals using conventional model-based methods is difficult. In this work, we propose to predict and classify tactile signal using deep learning methods, seeking to enhance the adaptability of the robotic grasp system to external event changes that may lead to grasping failure. We develop a deep learning framework and collect 6650 tactile image sequences with a vision-based tactile sensor, and the neural network is integrated into a contact-event-based robotic grasping system. In grasping experiments, we achieved 52% increase in terms of object lifting success rate with contact detection, significantly higher robustness under unexpected loads with slip prediction compared with open-loop grasps, demonstrating that integration of the proposed framework into robotic grasping system substantially improves picking success rate and capability to withstand external disturbances.",
  "published": "2019-10-09",
  "updated": "2019-10-09",
  "year": "2019",
  "authors": [
   "Yazhan Zhang",
   "Weihao Yuan",
   "Zicheng Kan",
   "Michael Yu Wang"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 48,
  "influential_citations": 1,
  "tldr": "This work develops a deep learning framework and collects 6650 tactile image sequences with a vision-based tactile sensor, and the neural network is integrated into a contact-event-based robotic grasping system, seeking to enhance the adaptability of the robotic grasp system to external event changes that may lead to grasping failure.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yazhan Zhang",
    "id": "2108008698",
    "h_index": 13,
    "papers": 19
   },
   {
    "name": "Weihao Yuan",
    "id": "11349534",
    "h_index": 15,
    "papers": 24
   },
   {
    "name": "Zicheng Kan",
    "id": "81238216",
    "h_index": 8,
    "papers": 10
   },
   {
    "name": "M. Wang",
    "id": "2108608469",
    "h_index": 63,
    "papers": 379
   }
  ],
  "comment": "10 pages, 7 figures, Accepted to conference on Robot Learning (CoRL 2019)",
  "topics": [
   "dexterous-manipulation",
   "tactile",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1910.03973v1",
  "pdf_url": "https://arxiv.org/pdf/1910.03973v1",
  "html_url": "https://arxiv.org/html/1910.03973v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.19
 },
 {
  "id": "1910.02054",
  "slug": "zero-memory-optimizations-toward-training-trillion-parameter-models",
  "title": "ZeRO: Memory Optimizations Toward Training Trillion Parameter Models",
  "abstract": "Large deep learning models offer significant accuracy gains, but training billions to trillions of parameters is challenging. Existing solutions such as data and model parallelisms exhibit fundamental limitations to fit these models into limited device memory, while obtaining computation, communication and development efficiency. We develop a novel solution, Zero Redundancy Optimizer (ZeRO), to optimize memory, vastly improving training speed while increasing the model size that can be efficiently trained. ZeRO eliminates memory redundancies in data- and model-parallel training while retaining low communication volume and high computational granularity, allowing us to scale the model size proportional to the number of devices with sustained high efficiency. Our analysis on memory requirements and communication volume demonstrates: ZeRO has the potential to scale beyond 1 Trillion parameters using today's hardware. We implement and evaluate ZeRO: it trains large models of over 100B parameter with super-linear speedup on 400 GPUs, achieving throughput of 15 Petaflops. This represents an 8x increase in model size and 10x increase in achievable performance over state-of-the-art. In terms of usability, ZeRO can train large models of up to 13B parameters (e.g., larger than Megatron GPT 8.3B and T5 11B) without requiring model parallelism which is harder for scientists to apply. Last but not the least, researchers have used the system breakthroughs of ZeRO to create the world's largest language model (Turing-NLG, 17B parameters) with record breaking accuracy.",
  "published": "2019-10-04",
  "updated": "2020-05-13",
  "year": "2019",
  "authors": [
   "Samyam Rajbhandari",
   "Jeff Rasley",
   "Olatunji Ruwase",
   "Yuxiong He"
  ],
  "author_count": 4,
  "categories": [
   "cs.LG",
   "cs.DC",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 1997,
  "influential_citations": 163,
  "tldr": "A novel solution, Zero Redundancy Optimizer (ZeRO), to optimize memory, vastly improving training speed while increasing the model size that can be efficiently trained, allowing to scale the model size proportional to the number of devices with sustained high efficiency.",
  "doi": "10.1109/SC41405.2020.00024",
  "oa_pdf": "https://arxiv.org/pdf/1910.02054",
  "s2_authors": [
   {
    "name": "Samyam Rajbhandari",
    "id": "32817044",
    "h_index": 23,
    "papers": 55
   },
   {
    "name": "Jeff Rasley",
    "id": "3299496",
    "h_index": 16,
    "papers": 55
   },
   {
    "name": "Olatunji Ruwase",
    "id": "2537545",
    "h_index": 25,
    "papers": 64
   },
   {
    "name": "Yuxiong He",
    "id": "2300208615",
    "h_index": 23,
    "papers": 41
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1910.02054v3",
  "pdf_url": "https://arxiv.org/pdf/1910.02054v3",
  "html_url": "https://arxiv.org/html/1910.02054v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.0
 },
 {
  "id": "1910.02860",
  "slug": "cable-manipulation-with-a-tactile-reactive-gripper",
  "title": "Cable Manipulation with a Tactile-Reactive Gripper",
  "abstract": "Cables are complex, high dimensional, and dynamic objects. Standard approaches to manipulate them often rely on conservative strategies that involve long series of very slow and incremental deformations, or various mechanical fixtures such as clamps, pins or rings. We are interested in manipulating freely moving cables, in real time, with a pair of robotic grippers, and with no added mechanical constraints. The main contribution of this paper is a perception and control framework that moves in that direction, and uses real-time tactile feedback to accomplish the task of following a dangling cable. The approach relies on a vision-based tactile sensor, GelSight, that estimates the pose of the cable in the grip, and the friction forces during cable sliding. We achieve the behavior by combining two tactile-based controllers: 1) Cable grip controller, where a PD controller combined with a leaky integrator regulates the gripping force to maintain the frictional sliding forces close to a suitable value; and 2) Cable pose controller, where an LQR controller based on a learned linear model of the cable sliding dynamics keeps the cable centered and aligned on the fingertips to prevent the cable from falling from the grip. This behavior is possible by a reactive gripper fitted with GelSight-based high-resolution tactile sensors. The robot can follow one meter of cable in random configurations within 2-3 hand regrasps, adapting to cables of different materials and thicknesses. We demonstrate a robot grasping a headphone cable, sliding the fingers to the jack connector, and inserting it. To the best of our knowledge, this is the first implementation of real-time cable following without the aid of mechanical fixtures.",
  "published": "2019-10-03",
  "updated": "2020-06-23",
  "year": "2019",
  "authors": [
   "Yu She",
   "Shaoxiong Wang",
   "Siyuan Dong",
   "Neha Sunil",
   "Alberto Rodriguez",
   "Edward Adelson"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "eess.IV",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "RSS",
  "venue_source": "semantic-scholar",
  "citations": 326,
  "influential_citations": 17,
  "tldr": "A perception and control framework that moves in that direction, and uses real-time tactile feedback to accomplish the task of following a dangling cable, is presented, believed to be the first implementation of real- time cable following without the aid of mechanical fixtures.",
  "doi": "10.1177/02783649211027233",
  "oa_pdf": "https://journals.sagepub.com/doi/pdf/10.1177/02783649211027233",
  "s2_authors": [
   {
    "name": "Y. She",
    "id": "2392034",
    "h_index": 17,
    "papers": 43
   },
   {
    "name": "Shaoxiong Wang",
    "id": "7488549",
    "h_index": 14,
    "papers": 16
   },
   {
    "name": "Siyuan Dong",
    "id": "3308345",
    "h_index": 28,
    "papers": 42
   },
   {
    "name": "N. Sunil",
    "id": "1389062449",
    "h_index": 3,
    "papers": 6
   },
   {
    "name": "Alberto Rodriguez",
    "id": "152532021",
    "h_index": 49,
    "papers": 113
   },
   {
    "name": "E. Adelson",
    "id": "145358192",
    "h_index": 87,
    "papers": 272
   }
  ],
  "comment": "Accepted to RSS 2020",
  "topics": [
   "dexterous-manipulation",
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1910.02860v3",
  "pdf_url": "https://arxiv.org/pdf/1910.02860v3",
  "html_url": "https://arxiv.org/html/1910.02860v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.01
 },
 {
  "id": "1910.01741",
  "slug": "improving-sample-efficiency-in-model-free-reinforcement-learning-from",
  "title": "Improving Sample Efficiency in Model-Free Reinforcement Learning from Images",
  "abstract": "Training an agent to solve control tasks directly from high-dimensional images with model-free reinforcement learning (RL) has proven difficult. A promising approach is to learn a latent representation together with the control policy. However, fitting a high-capacity encoder using a scarce reward signal is sample inefficient and leads to poor performance. Prior work has shown that auxiliary losses, such as image reconstruction, can aid efficient representation learning. However, incorporating reconstruction loss into an off-policy learning algorithm often leads to training instability. We explore the underlying reasons and identify variational autoencoders, used by previous investigations, as the cause of the divergence. Following these findings, we propose effective techniques to improve training stability. This results in a simple approach capable of matching state-of-the-art model-free and model-based algorithms on MuJoCo control tasks. Furthermore, our approach demonstrates robustness to observational noise, surpassing existing approaches in this setting. Code, results, and videos are anonymously available at https://sites.google.com/view/sac-ae/home.",
  "published": "2019-10-02",
  "updated": "2020-07-09",
  "year": "2019",
  "authors": [
   "Denis Yarats",
   "Amy Zhang",
   "Ilya Kostrikov",
   "Brandon Amos",
   "Joelle Pineau",
   "Rob Fergus"
  ],
  "author_count": 6,
  "categories": [
   "cs.LG",
   "cs.AI",
   "cs.RO",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 526,
  "influential_citations": 67,
  "tldr": "A simple approach capable of matching state-of-the-art model-free and model-based algorithms on MuJoCo control tasks and demonstrating robustness to observational noise, surpassing existing approaches in this setting.",
  "doi": "10.1609/aaai.v35i12.17276",
  "oa_pdf": "https://doi.org/10.1609/aaai.v35i12.17276",
  "s2_authors": [
   {
    "name": "Denis Yarats",
    "id": "13759615",
    "h_index": 21,
    "papers": 31
   },
   {
    "name": "Amy Zhang",
    "id": "2111672235",
    "h_index": 28,
    "papers": 54
   },
   {
    "name": "Ilya Kostrikov",
    "id": "2000906",
    "h_index": 29,
    "papers": 43
   },
   {
    "name": "Brandon Amos",
    "id": "1773498",
    "h_index": 38,
    "papers": 70
   },
   {
    "name": "Joelle Pineau",
    "id": "145134886",
    "h_index": 71,
    "papers": 289
   },
   {
    "name": "R. Fergus",
    "id": "2276554",
    "h_index": 78,
    "papers": 125
   }
  ],
  "comment": "",
  "topics": [
   "rl-control",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1910.01741v3",
  "pdf_url": "https://arxiv.org/pdf/1910.01741v3",
  "html_url": "https://arxiv.org/html/1910.01741v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.72
 },
 {
  "id": "1909.11652",
  "slug": "deep-dynamics-models-for-learning-dexterous-manipulation",
  "title": "Deep Dynamics Models for Learning Dexterous Manipulation",
  "abstract": "Dexterous multi-fingered hands can provide robots with the ability to flexibly perform a wide range of manipulation skills. However, many of the more complex behaviors are also notoriously difficult to control: Performing in-hand object manipulation, executing finger gaits to move objects, and exhibiting precise fine motor skills such as writing, all require finely balancing contact forces, breaking and reestablishing contacts repeatedly, and maintaining control of unactuated objects. Learning-based techniques provide the appealing possibility of acquiring these skills directly from data, but current learning approaches either require large amounts of data and produce task-specific policies, or they have not yet been shown to scale up to more complex and realistic tasks requiring fine motor skills. In this work, we demonstrate that our method of online planning with deep dynamics models (PDDM) addresses both of these limitations; we show that improvements in learned dynamics models, together with improvements in online model-predictive control, can indeed enable efficient and effective learning of flexible contact-rich dexterous manipulation skills -- and that too, on a 24-DoF anthropomorphic hand in the real world, using just 4 hours of purely real-world data to learn to simultaneously coordinate multiple free-floating objects. Videos can be found at https://sites.google.com/view/pddm/",
  "published": "2019-09-25",
  "updated": "2019-09-25",
  "year": "2019",
  "authors": [
   "Anusha Nagabandi",
   "Kurt Konoglie",
   "Sergey Levine",
   "Vikash Kumar"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 481,
  "influential_citations": 29,
  "tldr": "It is shown that improvements in learned dynamics models, together with improvements in online model-predictive control, can indeed enable efficient and effective learning of flexible contact-rich dexterous manipulation skills -- and that too, on a 24-DoF anthropomorphic hand in the real world, using just 4 hours of purely real-world data to learn to simultaneously coordinate multiple free-floating objects.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Anusha Nagabandi",
    "id": "3195183",
    "h_index": 13,
    "papers": 23
   },
   {
    "name": "K. Konolige",
    "id": "70162540",
    "h_index": 49,
    "papers": 86
   },
   {
    "name": "S. Levine",
    "id": "1736651",
    "h_index": 167,
    "papers": 551
   },
   {
    "name": "Vikash Kumar",
    "id": "2109446216",
    "h_index": 38,
    "papers": 76
   }
  ],
  "comment": "project website https://sites.google.com/view/pddm/",
  "topics": [
   "dexterous-manipulation",
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1909.11652v1",
  "pdf_url": "https://arxiv.org/pdf/1909.11652v1",
  "html_url": "https://arxiv.org/html/1909.11652v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.18
 },
 {
  "id": "1909.11583",
  "slug": "off-policy-actor-critic-with-shared-experience-replay",
  "title": "Off-Policy Actor-Critic with Shared Experience Replay",
  "abstract": "We investigate the combination of actor-critic reinforcement learning algorithms with uniform large-scale experience replay and propose solutions for two challenges: (a) efficient actor-critic learning with experience replay (b) stability of off-policy learning where agents learn from other agents behaviour. We employ those insights to accelerate hyper-parameter sweeps in which all participating agents run concurrently and share their experience via a common replay module. To this end we analyze the bias-variance tradeoffs in V-trace, a form of importance sampling for actor-critic methods. Based on our analysis, we then argue for mixing experience sampled from replay with on-policy experience, and propose a new trust region scheme that scales effectively to data distributions where V-trace becomes unstable. We provide extensive empirical validation of the proposed solution. We further show the benefits of this setup by demonstrating state-of-the-art data efficiency on Atari among agents trained up until 200M environment frames.",
  "published": "2019-09-25",
  "updated": "2019-11-18",
  "year": "2019",
  "authors": [
   "Simon Schmitt",
   "Matteo Hessel",
   "Karen Simonyan"
  ],
  "author_count": 3,
  "categories": [
   "cs.LG",
   "cs.AI",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "ICML",
  "venue_source": "semantic-scholar",
  "citations": 71,
  "influential_citations": 13,
  "tldr": "This work analyzes the bias-variance tradeoffs in V- Trace, a form of importance sampling for actor-critic methods, and argues for mixing experience sampled from replay with on-policy experience, and proposes a new trust region scheme that scales effectively to data distributions where V-trace becomes unstable.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Simon Schmitt",
    "id": "152380508",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "Matteo Hessel",
    "id": "39357484",
    "h_index": 25,
    "papers": 39
   },
   {
    "name": "K. Simonyan",
    "id": "34838386",
    "h_index": 65,
    "papers": 108
   }
  ],
  "comment": "",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1909.11583v2",
  "pdf_url": "https://arxiv.org/pdf/1909.11583v2",
  "html_url": "https://arxiv.org/html/1909.11583v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.36
 },
 {
  "id": "1909.10080",
  "slug": "whole-body-geometric-retargeting-for-humanoid-robots",
  "title": "Whole-Body Geometric Retargeting for Humanoid Robots",
  "abstract": "Humanoid robot teleoperation allows humans to integrate their cognitive capabilities with the apparatus to perform tasks that need high strength, manoeuvrability and dexterity. This paper presents a framework for teleoperation of humanoid robots using a novel approach for motion retargeting through inverse kinematics over the robot model. The proposed method enhances scalability for retargeting, i.e., it allows teleoperating different robots by different human users with minimal changes to the proposed system. Our framework enables an intuitive and natural interaction between the human operator and the humanoid robot at the configuration space level. We validate our approach by demonstrating whole-body retargeting with multiple robot models. Furthermore, we present experimental validation through teleoperation experiments using two state-of-the-art whole-body controllers for humanoid robots.",
  "published": "2019-09-22",
  "updated": "2019-09-22",
  "year": "2019",
  "authors": [
   "Kourosh Darvish",
   "Yeshasvi Tirupachuri",
   "Giulio Romualdi",
   "Lorenzo Rapetti",
   "Diego Ferigo",
   "Francisco Javier Andrade Chavez",
   "Daniele Pucci"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "Humanoids",
  "venue_source": "semantic-scholar",
  "citations": 83,
  "influential_citations": 3,
  "tldr": "This paper presents a framework for teleoperation of humanoid robots using a novel approach for motion retargeting through inverse kinematics over the robot model, which enhances scalability for Retargeting and enables an intuitive and natural interaction between the human operator and the humanoid robot at the configuration space level.",
  "doi": "10.1109/Humanoids43949.2019.9035059",
  "oa_pdf": "https://arxiv.org/pdf/1909.10080",
  "s2_authors": [
   {
    "name": "K. Darvish",
    "id": "34308317",
    "h_index": 16,
    "papers": 47
   },
   {
    "name": "Yeshasvi Tirupachuri",
    "id": "7862218",
    "h_index": 9,
    "papers": 26
   },
   {
    "name": "Giulio Romualdi",
    "id": "51451149",
    "h_index": 12,
    "papers": 32
   },
   {
    "name": "Lorenzo Rapetti",
    "id": "27558790",
    "h_index": 13,
    "papers": 32
   },
   {
    "name": "Diego Ferigo",
    "id": "10029059",
    "h_index": 9,
    "papers": 18
   },
   {
    "name": "Francisco Javier Andrade Chavez",
    "id": "7513152",
    "h_index": 10,
    "papers": 16
   },
   {
    "name": "D. Pucci",
    "id": "2202742",
    "h_index": 26,
    "papers": 139
   }
  ],
  "comment": "Equal author contribution from Kourosh Darvish and Yeshasvi Tirupachuri",
  "topics": [
   "dexterous-manipulation",
   "humanoids",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1909.10080v1",
  "pdf_url": "https://arxiv.org/pdf/1909.10080v1",
  "html_url": "https://arxiv.org/html/1909.10080v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.42
 },
 {
  "id": "1909.08593",
  "slug": "fine-tuning-language-models-from-human-preferences",
  "title": "Fine-Tuning Language Models from Human Preferences",
  "abstract": "Reward learning enables the application of reinforcement learning (RL) to tasks where reward is defined by human judgment, building a model of reward by asking humans questions. Most work on reward learning has used simulated environments, but complex information about values is often expressed in natural language, and we believe reward learning for language is a key to making RL practical and safe for real-world tasks. In this paper, we build on advances in generative pretraining of language models to apply reward learning to four natural language tasks: continuing text with positive sentiment or physically descriptive language, and summarization tasks on the TL;DR and CNN/Daily Mail datasets. For stylistic continuation we achieve good results with only 5,000 comparisons evaluated by humans. For summarization, models trained with 60,000 comparisons copy whole sentences from the input but skip irrelevant preamble; this leads to reasonable ROUGE scores and very good performance according to our human labelers, but may be exploiting the fact that labelers rely on simple heuristics.",
  "published": "2019-09-18",
  "updated": "2020-01-08",
  "year": "2019",
  "authors": [
   "Daniel M. Ziegler",
   "Nisan Stiennon",
   "Jeffrey Wu",
   "Tom B. Brown",
   "Alec Radford",
   "Dario Amodei",
   "Paul Christiano",
   "Geoffrey Irving"
  ],
  "author_count": 8,
  "categories": [
   "cs.CL",
   "cs.LG",
   "stat.ML"
  ],
  "primary_category": "cs.CL",
  "venue": "",
  "venue_source": "",
  "citations": 2695,
  "influential_citations": 210,
  "tldr": "This paper builds on advances in generative pretraining of language models to apply reward learning to four natural language tasks: continuing text with positive sentiment or physically descriptive language, and summarization tasks on the TL;DR and CNN/Daily Mail datasets.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Daniel M. Ziegler",
    "id": "2052152920",
    "h_index": 12,
    "papers": 67
   },
   {
    "name": "Nisan Stiennon",
    "id": "1387983862",
    "h_index": 6,
    "papers": 10
   },
   {
    "name": "Jeff Wu",
    "id": "49387725",
    "h_index": 11,
    "papers": 12
   },
   {
    "name": "Tom B. Brown",
    "id": "31035595",
    "h_index": 25,
    "papers": 30
   },
   {
    "name": "Alec Radford",
    "id": "38909097",
    "h_index": 33,
    "papers": 153
   },
   {
    "name": "Dario Amodei",
    "id": "2698777",
    "h_index": 30,
    "papers": 61
   },
   {
    "name": "Paul Christiano",
    "id": "145370786",
    "h_index": 15,
    "papers": 45
   },
   {
    "name": "G. Irving",
    "id": "2060655766",
    "h_index": 22,
    "papers": 27
   }
  ],
  "comment": "",
  "topics": [
   "rl-control",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1909.08593v2",
  "pdf_url": "https://arxiv.org/pdf/1909.08593v2",
  "html_url": "https://arxiv.org/html/1909.08593v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.0
 },
 {
  "id": "1908.09186",
  "slug": "efficient-learning-on-point-clouds-with-basis-point-sets",
  "title": "Efficient Learning on Point Clouds with Basis Point Sets",
  "abstract": "With the increased availability of 3D scanning technology, point clouds are moving into the focus of computer vision as a rich representation of everyday scenes. However, they are hard to handle for machine learning algorithms due to their unordered structure. One common approach is to apply occupancy grid mapping, which dramatically increases the amount of data stored and at the same time loses details through discretization. Recently, deep learning models were proposed to handle point clouds directly and achieve input permutation invariance. However, these architectures often use an increased number of parameters and are computationally inefficient. In this work, we propose basis point sets (BPS) as a highly efficient and fully general way to process point clouds with machine learning algorithms. The basis point set representation is a residual representation that can be computed efficiently and can be used with standard neural network architectures and other machine learning algorithms. Using the proposed representation as the input to a simple fully connected network allows us to match the performance of PointNet on a shape classification task while using three orders of magnitude less floating-point operations. In a second experiment, we show how the proposed representation can be used for registering high-resolution meshes to noisy 3D scans. Here, we present the first method for single-pass high-resolution mesh registration, avoiding time-consuming per-scan optimization and allowing real-time execution.",
  "published": "2019-08-24",
  "updated": "2019-08-24",
  "year": "2019",
  "authors": [
   "Sergey Prokudin",
   "Christoph Lassner",
   "Javier Romero"
  ],
  "author_count": 3,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "ICCV",
  "venue_source": "semantic-scholar",
  "citations": 169,
  "influential_citations": 21,
  "tldr": "This work proposes basis point sets as a highly efficient and fully general way to process point clouds with machine learning algorithms and achieves performance comparable to the state-of-the-art computationally intense multi-step frameworks in one network pass that can be done in less than 1ms.",
  "doi": "10.1109/ICCV.2019.00443",
  "oa_pdf": "https://arxiv.org/pdf/1908.09186",
  "s2_authors": [
   {
    "name": "S. Prokudin",
    "id": "15968671",
    "h_index": 12,
    "papers": 31
   },
   {
    "name": "Christoph Lassner",
    "id": "3266545",
    "h_index": 21,
    "papers": 39
   },
   {
    "name": "J. Romero",
    "id": "143881914",
    "h_index": 27,
    "papers": 42
   }
  ],
  "comment": "ICCV 2019",
  "topics": [
   "spatial-3d",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1908.09186v1",
  "pdf_url": "https://arxiv.org/pdf/1908.09186v1",
  "html_url": "https://arxiv.org/html/1908.09186v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.73
 },
 {
  "id": "1908.04683",
  "slug": "is-deep-reinforcement-learning-really-superhuman-on-atari-leveling-the",
  "title": "Is Deep Reinforcement Learning Really Superhuman on Atari? Leveling the playing field",
  "abstract": "Consistent and reproducible evaluation of Deep Reinforcement Learning (DRL) is not straightforward. In the Arcade Learning Environment (ALE), small changes in environment parameters such as stochasticity or the maximum allowed play time can lead to very different performance. In this work, we discuss the difficulties of comparing different agents trained on ALE. In order to take a step further towards reproducible and comparable DRL, we introduce SABER, a Standardized Atari BEnchmark for general Reinforcement learning algorithms. Our methodology extends previous recommendations and contains a complete set of environment parameters as well as train and test procedures. We then use SABER to evaluate the current state of the art, Rainbow. Furthermore, we introduce a human world records baseline, and argue that previous claims of expert or superhuman performance of DRL might not be accurate. Finally, we propose Rainbow-IQN by extending Rainbow with Implicit Quantile Networks (IQN) leading to new state-of-the-art performance. Source code is available for reproducibility.",
  "published": "2019-08-13",
  "updated": "2019-11-08",
  "year": "2019",
  "authors": [
   "Marin Toromanoff",
   "Emilie Wirbel",
   "Fabien Moutarde"
  ],
  "author_count": 3,
  "categories": [
   "cs.AI"
  ],
  "primary_category": "cs.AI",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 24,
  "influential_citations": 3,
  "tldr": "This work introduces SABER, a Standardized Atari BEnchmark for general Reinforcement learning algorithms and uses it to evaluate the current state of the art, Rainbow, and introduces a human world records baseline, and argues that previous claims of expert or superhuman performance of DRL might not be accurate.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Marin Toromanoff",
    "id": "24306829",
    "h_index": 7,
    "papers": 11
   },
   {
    "name": "\u00c9. Wirbel",
    "id": "3422914",
    "h_index": 12,
    "papers": 16
   },
   {
    "name": "F. Moutarde",
    "id": "1748488",
    "h_index": 27,
    "papers": 117
   }
  ],
  "comment": "Paper currently in review",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1908.04683v5",
  "pdf_url": "https://arxiv.org/pdf/1908.04683v5",
  "html_url": "https://arxiv.org/html/1908.04683v5",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.9
 },
 {
  "id": "1908.03568",
  "slug": "behaviour-suite-for-reinforcement-learning",
  "title": "Behaviour Suite for Reinforcement Learning",
  "abstract": "This paper introduces the Behaviour Suite for Reinforcement Learning, or bsuite for short. bsuite is a collection of carefully-designed experiments that investigate core capabilities of reinforcement learning (RL) agents with two objectives. First, to collect clear, informative and scalable problems that capture key issues in the design of general and efficient learning algorithms. Second, to study agent behaviour through their performance on these shared benchmarks. To complement this effort, we open source github.com/deepmind/bsuite, which automates evaluation and analysis of any agent on bsuite. This library facilitates reproducible and accessible research on the core issues in RL, and ultimately the design of superior learning algorithms. Our code is Python, and easy to use within existing projects. We include examples with OpenAI Baselines, Dopamine as well as new reference implementations. Going forward, we hope to incorporate more excellent experiments from the research community, and commit to a periodic review of bsuite from a committee of prominent researchers.",
  "published": "2019-08-09",
  "updated": "2020-02-14",
  "year": "2019",
  "authors": [
   "Ian Osband",
   "Yotam Doron",
   "Matteo Hessel",
   "John Aslanides",
   "Eren Sezener",
   "Andre Saraiva",
   "Katrina McKinney",
   "Tor Lattimore",
   "Csaba Szepesvari",
   "Satinder Singh",
   "Benjamin Van Roy",
   "Richard Sutton",
   "David Silver",
   "Hado Van Hasselt"
  ],
  "author_count": 14,
  "categories": [
   "cs.LG",
   "cs.AI",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 213,
  "influential_citations": 26,
  "tldr": "This paper introduces the Behaviour Suite for Reinforcement Learning, or bsuite for short, a collection of carefully-designed experiments that investigate core capabilities of reinforcement learning agents with two objectives: to collect clear, informative and scalable problems that capture key issues in the design of general and efficient learning algorithms.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ian Osband",
    "id": "2561924",
    "h_index": 28,
    "papers": 57
   },
   {
    "name": "Yotam Doron",
    "id": "2895238",
    "h_index": 7,
    "papers": 9
   },
   {
    "name": "Matteo Hessel",
    "id": "39357484",
    "h_index": 25,
    "papers": 39
   },
   {
    "name": "John Aslanides",
    "id": "9958912",
    "h_index": 14,
    "papers": 19
   },
   {
    "name": "Eren Sezener",
    "id": "1413718981",
    "h_index": 9,
    "papers": 10
   },
   {
    "name": "Andre Saraiva",
    "id": "2064274251",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Katrina McKinney",
    "id": "152182296",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Tor Lattimore",
    "id": "2989692",
    "h_index": 39,
    "papers": 117
   },
   {
    "name": "Csaba Szepesvari",
    "id": "40868287",
    "h_index": 79,
    "papers": 375
   },
   {
    "name": "Satinder Singh",
    "id": "2108384183",
    "h_index": 28,
    "papers": 67
   },
   {
    "name": "Benjamin Van Roy",
    "id": "1731282",
    "h_index": 56,
    "papers": 204
   },
   {
    "name": "R. Sutton",
    "id": "1699645",
    "h_index": 75,
    "papers": 277
   },
   {
    "name": "David Silver",
    "id": "145824029",
    "h_index": 80,
    "papers": 120
   },
   {
    "name": "H. V. Hasselt",
    "id": "7634925",
    "h_index": 40,
    "papers": 67
   }
  ],
  "comment": "",
  "topics": [
   "rl-control"
  ],
  "orgs": [
   "Google DeepMind"
  ],
  "abs_url": "https://arxiv.org/abs/1908.03568v3",
  "pdf_url": "https://arxiv.org/pdf/1908.03568v3",
  "html_url": "https://arxiv.org/html/1908.03568v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.33
 },
 {
  "id": "1907.08225",
  "slug": "dynamical-distance-learning-for-semi-supervised-and-unsupervised-skill",
  "title": "Dynamical Distance Learning for Semi-Supervised and Unsupervised Skill Discovery",
  "abstract": "Reinforcement learning requires manual specification of a reward function to learn a task. While in principle this reward function only needs to specify the task goal, in practice reinforcement learning can be very time-consuming or even infeasible unless the reward function is shaped so as to provide a smooth gradient towards a successful outcome. This shaping is difficult to specify by hand, particularly when the task is learned from raw observations, such as images. In this paper, we study how we can automatically learn dynamical distances: a measure of the expected number of time steps to reach a given goal state from any other state. These dynamical distances can be used to provide well-shaped reward functions for reaching new goals, making it possible to learn complex tasks efficiently. We show that dynamical distances can be used in a semi-supervised regime, where unsupervised interaction with the environment is used to learn the dynamical distances, while a small amount of preference supervision is used to determine the task goal, without any manually engineered reward function or goal examples. We evaluate our method both on a real-world robot and in simulation. We show that our method can learn to turn a valve with a real-world 9-DoF hand, using raw image observations and just ten preference labels, without any other supervision. Videos of the learned skills can be found on the project website: https://sites.google.com/view/dynamical-distance-learning.",
  "published": "2019-07-18",
  "updated": "2020-02-14",
  "year": "2019",
  "authors": [
   "Kristian Hartikainen",
   "Xinyang Geng",
   "Tuomas Haarnoja",
   "Sergey Levine"
  ],
  "author_count": 4,
  "categories": [
   "cs.LG",
   "cs.AI",
   "cs.CV",
   "cs.RO",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 97,
  "influential_citations": 10,
  "tldr": "This paper studies how to automatically learn dynamical distances: a measure of the expected number of time steps to reach a given goal state from any other state, which can be used to provide well-shaped reward functions for reaching new goals, making it possible to learn complex tasks efficiently.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Kristian Hartikainen",
    "id": "41016704",
    "h_index": 10,
    "papers": 13
   },
   {
    "name": "Xinyang Geng",
    "id": "3468192",
    "h_index": 19,
    "papers": 33
   },
   {
    "name": "Tuomas Haarnoja",
    "id": "2587648",
    "h_index": 15,
    "papers": 31
   },
   {
    "name": "S. Levine",
    "id": "1736651",
    "h_index": 167,
    "papers": 551
   }
  ],
  "comment": "11+6 pages, 6+2 figures, last two authors (Tuomas Haarnoja, Sergey Levine) advised equally",
  "topics": [
   "rl-control",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1907.08225v4",
  "pdf_url": "https://arxiv.org/pdf/1907.08225v4",
  "html_url": "https://arxiv.org/html/1907.08225v4",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.49
 },
 {
  "id": "1907.04957",
  "slug": "vision-and-dialog-navigation",
  "title": "Vision-and-Dialog Navigation",
  "abstract": "Robots navigating in human environments should use language to ask for assistance and be able to understand human responses. To study this challenge, we introduce Cooperative Vision-and-Dialog Navigation, a dataset of over 2k embodied, human-human dialogs situated in simulated, photorealistic home environments. The Navigator asks questions to their partner, the Oracle, who has privileged access to the best next steps the Navigator should take according to a shortest path planner. To train agents that search an environment for a goal location, we define the Navigation from Dialog History task. An agent, given a target object and a dialog history between humans cooperating to find that object, must infer navigation actions towards the goal in unexplored environments. We establish an initial, multi-modal sequence-to-sequence model and demonstrate that looking farther back in the dialog history improves performance. Sourcecode and a live interface demo can be found at https://cvdn.dev/",
  "published": "2019-07-10",
  "updated": "2019-10-13",
  "year": "2019",
  "authors": [
   "Jesse Thomason",
   "Michael Murray",
   "Maya Cakmak",
   "Luke Zettlemoyer"
  ],
  "author_count": 4,
  "categories": [
   "cs.CL",
   "cs.AI",
   "cs.CV",
   "cs.RO"
  ],
  "primary_category": "cs.CL",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 418,
  "influential_citations": 52,
  "tldr": "This work introduces Cooperative Vision-and-Dialog Navigation, a dataset of over 2k embodied, human-human dialogs situated in simulated, photorealistic home environments and establishes an initial, multi-modal sequence-to-sequence model.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jesse Thomason",
    "id": "2665873",
    "h_index": 27,
    "papers": 63
   },
   {
    "name": "Michael Murray",
    "id": "2114300655",
    "h_index": 4,
    "papers": 13
   },
   {
    "name": "M. Cakmak",
    "id": "35096370",
    "h_index": 43,
    "papers": 149
   },
   {
    "name": "Luke Zettlemoyer",
    "id": "1982950",
    "h_index": 118,
    "papers": 278
   }
  ],
  "comment": "Conference on Robot Learning (CoRL) 2019",
  "topics": [
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1907.04957v3",
  "pdf_url": "https://arxiv.org/pdf/1907.04957v3",
  "html_url": "https://arxiv.org/html/1907.04957v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.12
 },
 {
  "id": "1907.03687",
  "slug": "general-non-linear-bellman-equations",
  "title": "General non-linear Bellman equations",
  "abstract": "We consider a general class of non-linear Bellman equations. These open up a design space of algorithms that have interesting properties, which has two potential advantages. First, we can perhaps better model natural phenomena. For instance, hyperbolic discounting has been proposed as a mathematical model that matches human and animal data well, and can therefore be used to explain preference orderings. We present a different mathematical model that matches the same data, but that makes very different predictions under other circumstances. Second, the larger design space can perhaps lead to algorithms that perform better, similar to how discount factors are often used in practice even when the true objective is undiscounted. We show that many of the resulting Bellman operators still converge to a fixed point, and therefore that the resulting algorithms are reasonable and inherit many beneficial properties of their linear counterparts.",
  "published": "2019-07-08",
  "updated": "2019-07-08",
  "year": "2019",
  "authors": [
   "Hado van Hasselt",
   "John Quan",
   "Matteo Hessel",
   "Zhongwen Xu",
   "Diana Borsa",
   "Andre Barreto"
  ],
  "author_count": 6,
  "categories": [
   "cs.LG",
   "cs.AI",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 14,
  "influential_citations": 0,
  "tldr": "A general class of non-linear Bellman equations is considered, which opens up a design space of algorithms that have interesting properties and may lead to algorithms that perform better, similar to how discount factors are often used in practice even when the true objective is undiscounted.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "H. V. Hasselt",
    "id": "7634925",
    "h_index": 40,
    "papers": 67
   },
   {
    "name": "John Quan",
    "id": "34660073",
    "h_index": 16,
    "papers": 22
   },
   {
    "name": "Matteo Hessel",
    "id": "39357484",
    "h_index": 25,
    "papers": 39
   },
   {
    "name": "Zhongwen Xu",
    "id": "2351434",
    "h_index": 27,
    "papers": 62
   },
   {
    "name": "Diana Borsa",
    "id": "2311858",
    "h_index": 17,
    "papers": 30
   },
   {
    "name": "Andr\u00e9 Barreto",
    "id": "1689289",
    "h_index": 17,
    "papers": 38
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1907.03687v1",
  "pdf_url": "https://arxiv.org/pdf/1907.03687v1",
  "html_url": "https://arxiv.org/html/1907.03687v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.18
 },
 {
  "id": "1907.03613",
  "slug": "data-efficient-reinforcement-learning-for-legged-robots",
  "title": "Data Efficient Reinforcement Learning for Legged Robots",
  "abstract": "We present a model-based framework for robot locomotion that achieves walking based on only 4.5 minutes (45,000 control steps) of data collected on a quadruped robot. To accurately model the robot's dynamics over a long horizon, we introduce a loss function that tracks the model's prediction over multiple timesteps. We adapt model predictive control to account for planning latency, which allows the learned model to be used for real time control. Additionally, to ensure safe exploration during model learning, we embed prior knowledge of leg trajectories into the action space. The resulting system achieves fast and robust locomotion. Unlike model-free methods, which optimize for a particular task, our planner can use the same learned dynamics for various tasks, simply by changing the reward function. To the best of our knowledge, our approach is more than an order of magnitude more sample efficient than current model-free methods.",
  "published": "2019-07-08",
  "updated": "2019-10-06",
  "year": "2019",
  "authors": [
   "Yuxiang Yang",
   "Ken Caluwaerts",
   "Atil Iscen",
   "Tingnan Zhang",
   "Jie Tan",
   "Vikas Sindhwani"
  ],
  "author_count": 6,
  "categories": [
   "cs.LG",
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.LG",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 172,
  "influential_citations": 9,
  "tldr": "A model-based framework for robot locomotion that achieves walking based on only 4.5 minutes of data collected on a quadruped robot is presented, which is more than an order of magnitude more sample efficient than current model-free methods.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yuxiang Yang",
    "id": "2108795581",
    "h_index": 18,
    "papers": 50
   },
   {
    "name": "Ken Caluwaerts",
    "id": "2758571",
    "h_index": 22,
    "papers": 46
   },
   {
    "name": "Atil Iscen",
    "id": "2106754",
    "h_index": 21,
    "papers": 41
   },
   {
    "name": "Tingnan Zhang",
    "id": "28292148",
    "h_index": 26,
    "papers": 53
   },
   {
    "name": "Jie Tan",
    "id": "1739176520",
    "h_index": 37,
    "papers": 68
   },
   {
    "name": "Vikas Sindhwani",
    "id": "1808676",
    "h_index": 52,
    "papers": 174
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "rl-control",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1907.03613v2",
  "pdf_url": "https://arxiv.org/pdf/1907.03613v2",
  "html_url": "https://arxiv.org/html/1907.03613v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.74
 },
 {
  "id": "1907.02057",
  "slug": "benchmarking-model-based-reinforcement-learning",
  "title": "Benchmarking Model-Based Reinforcement Learning",
  "abstract": "Model-based reinforcement learning (MBRL) is widely seen as having the potential to be significantly more sample efficient than model-free RL. However, research in model-based RL has not been very standardized. It is fairly common for authors to experiment with self-designed environments, and there are several separate lines of research, which are sometimes closed-sourced or not reproducible. Accordingly, it is an open question how these various existing MBRL algorithms perform relative to each other. To facilitate research in MBRL, in this paper we gather a wide collection of MBRL algorithms and propose over 18 benchmarking environments specially designed for MBRL. We benchmark these algorithms with unified problem settings, including noisy environments. Beyond cataloguing performance, we explore and unify the underlying algorithmic differences across MBRL algorithms. We characterize three key research challenges for future MBRL research: the dynamics bottleneck, the planning horizon dilemma, and the early-termination dilemma. Finally, to maximally facilitate future research on MBRL, we open-source our benchmark in http://www.cs.toronto.edu/~tingwuwang/mbrl.html.",
  "published": "2019-07-03",
  "updated": "2019-07-03",
  "year": "2019",
  "authors": [
   "Tingwu Wang",
   "Xuchan Bao",
   "Ignasi Clavera",
   "Jerrick Hoang",
   "Yeming Wen",
   "Eric Langlois",
   "Shunshi Zhang",
   "Guodong Zhang",
   "Pieter Abbeel",
   "Jimmy Ba"
  ],
  "author_count": 10,
  "categories": [
   "cs.LG",
   "cs.AI",
   "cs.RO",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 395,
  "influential_citations": 30,
  "tldr": "This paper gathers a wide collection of MBRL algorithms and proposes over 18 benchmarking environments specially designed for MBRL, and describes three key research challenges for future MBRL research: the dynamics bottleneck, the planning horizon dilemma, and the early-termination dilemma.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Tingwu Wang",
    "id": "3428549",
    "h_index": 10,
    "papers": 14
   },
   {
    "name": "Xuchan Bao",
    "id": "8538131",
    "h_index": 5,
    "papers": 7
   },
   {
    "name": "Ignasi Clavera",
    "id": "15593386",
    "h_index": 14,
    "papers": 15
   },
   {
    "name": "Jerrick Hoang",
    "id": "150148498",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Yeming Wen",
    "id": "38356166",
    "h_index": 12,
    "papers": 15
   },
   {
    "name": "Eric D. Langlois",
    "id": "144006195",
    "h_index": 3,
    "papers": 9
   },
   {
    "name": "Matthew Shunshi Zhang",
    "id": "6303990",
    "h_index": 8,
    "papers": 9
   },
   {
    "name": "Guodong Zhang",
    "id": "46266081",
    "h_index": 19,
    "papers": 30
   },
   {
    "name": "P. Abbeel",
    "id": "1689992",
    "h_index": 165,
    "papers": 596
   },
   {
    "name": "Jimmy Ba",
    "id": "2503659",
    "h_index": 46,
    "papers": 84
   }
  ],
  "comment": "8 main pages, 8 figures; 14 appendix pages, 25 figures",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1907.02057v1",
  "pdf_url": "https://arxiv.org/pdf/1907.02057v1",
  "html_url": "https://arxiv.org/html/1907.02057v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.6
 },
 {
  "id": "1907.01657",
  "slug": "dynamics-aware-unsupervised-discovery-of-skills",
  "title": "Dynamics-Aware Unsupervised Discovery of Skills",
  "abstract": "Conventionally, model-based reinforcement learning (MBRL) aims to learn a global model for the dynamics of the environment. A good model can potentially enable planning algorithms to generate a large variety of behaviors and solve diverse tasks. However, learning an accurate model for complex dynamical systems is difficult, and even then, the model might not generalize well outside the distribution of states on which it was trained. In this work, we combine model-based learning with model-free learning of primitives that make model-based planning easy. To that end, we aim to answer the question: how can we discover skills whose outcomes are easy to predict? We propose an unsupervised learning algorithm, Dynamics-Aware Discovery of Skills (DADS), which simultaneously discovers predictable behaviors and learns their dynamics. Our method can leverage continuous skill spaces, theoretically, allowing us to learn infinitely many behaviors even for high-dimensional state-spaces. We demonstrate that zero-shot planning in the learned latent space significantly outperforms standard MBRL and model-free goal-conditioned RL, can handle sparse-reward tasks, and substantially improves over prior hierarchical RL methods for unsupervised skill discovery.",
  "published": "2019-07-02",
  "updated": "2020-02-14",
  "year": "2019",
  "authors": [
   "Archit Sharma",
   "Shixiang Gu",
   "Sergey Levine",
   "Vikash Kumar",
   "Karol Hausman"
  ],
  "author_count": 5,
  "categories": [
   "cs.LG",
   "cs.RO",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 499,
  "influential_citations": 61,
  "tldr": "This work proposes an unsupervised learning algorithm, Dynamics-Aware Discovery of Skills (DADS), which simultaneously discovers predictable behaviors and learns their dynamics, and demonstrates that zero-shot planning in the learned latent space significantly outperforms standard MBRL and model-free goal-conditioned RL, and substantially improves over prior hierarchical RL methods for unsuper supervised skill discovery.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Archit Sharma",
    "id": "50465276",
    "h_index": 22,
    "papers": 36
   },
   {
    "name": "S. Gu",
    "id": "2046135",
    "h_index": 43,
    "papers": 71
   },
   {
    "name": "S. Levine",
    "id": "1736651",
    "h_index": 167,
    "papers": 551
   },
   {
    "name": "Vikash Kumar",
    "id": "2109446216",
    "h_index": 38,
    "papers": 76
   },
   {
    "name": "Karol Hausman",
    "id": "1944801",
    "h_index": 47,
    "papers": 122
   }
  ],
  "comment": "",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1907.01657v2",
  "pdf_url": "https://arxiv.org/pdf/1907.01657v2",
  "html_url": "https://arxiv.org/html/1907.01657v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.2
 },
 {
  "id": "1907.00953",
  "slug": "stochastic-latent-actor-critic-deep-reinforcement-learning-with-a-late",
  "title": "Stochastic Latent Actor-Critic: Deep Reinforcement Learning with a Latent Variable Model",
  "abstract": "Deep reinforcement learning (RL) algorithms can use high-capacity deep networks to learn directly from image observations. However, these high-dimensional observation spaces present a number of challenges in practice, since the policy must now solve two problems: representation learning and task learning. In this work, we tackle these two problems separately, by explicitly learning latent representations that can accelerate reinforcement learning from images. We propose the stochastic latent actor-critic (SLAC) algorithm: a sample-efficient and high-performing RL algorithm for learning policies for complex continuous control tasks directly from high-dimensional image inputs. SLAC provides a novel and principled approach for unifying stochastic sequential models and RL into a single method, by learning a compact latent representation and then performing RL in the model's learned latent space. Our experimental evaluation demonstrates that our method outperforms both model-free and model-based alternatives in terms of final performance and sample efficiency, on a range of difficult image-based control tasks. Our code and videos of our results are available at our website.",
  "published": "2019-07-01",
  "updated": "2020-10-26",
  "year": "2019",
  "authors": [
   "Alex X. Lee",
   "Anusha Nagabandi",
   "Pieter Abbeel",
   "Sergey Levine"
  ],
  "author_count": 4,
  "categories": [
   "cs.LG",
   "cs.AI",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 435,
  "influential_citations": 49,
  "tldr": "The stochastic latent actor-critic (SLAC) algorithm is proposed: a sample-efficient and high-performing RL algorithm for learning policies for complex continuous control tasks directly from high-dimensional image inputs.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Alex X. Lee",
    "id": "49250083",
    "h_index": 17,
    "papers": 24
   },
   {
    "name": "Anusha Nagabandi",
    "id": "3195183",
    "h_index": 13,
    "papers": 23
   },
   {
    "name": "P. Abbeel",
    "id": "1689992",
    "h_index": 165,
    "papers": 596
   },
   {
    "name": "S. Levine",
    "id": "1736651",
    "h_index": 167,
    "papers": 551
   }
  ],
  "comment": "Project website: https://alexlee-gk.github.io/slac/",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1907.00953v4",
  "pdf_url": "https://arxiv.org/pdf/1907.00953v4",
  "html_url": "https://arxiv.org/html/1907.00953v4",
  "code_url": "https://alexlee-gk.github.io/slac/",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.14
 },
 {
  "id": "1906.09237",
  "slug": "shaping-belief-states-with-generative-environment-models-for-rl",
  "title": "Shaping Belief States with Generative Environment Models for RL",
  "abstract": "When agents interact with a complex environment, they must form and maintain beliefs about the relevant aspects of that environment. We propose a way to efficiently train expressive generative models in complex environments. We show that a predictive algorithm with an expressive generative model can form stable belief-states in visually rich and dynamic 3D environments. More precisely, we show that the learned representation captures the layout of the environment as well as the position and orientation of the agent. Our experiments show that the model substantially improves data-efficiency on a number of reinforcement learning (RL) tasks compared with strong model-free baseline agents. We find that predicting multiple steps into the future (overshooting), in combination with an expressive generative model, is critical for stable representations to emerge. In practice, using expressive generative models in RL is computationally expensive and we propose a scheme to reduce this computational burden, allowing us to build agents that are competitive with model-free baselines.",
  "published": "2019-06-21",
  "updated": "2019-06-24",
  "year": "2019",
  "authors": [
   "Karol Gregor",
   "Danilo Jimenez Rezende",
   "Frederic Besse",
   "Yan Wu",
   "Hamza Merzic",
   "Aaron van den Oord"
  ],
  "author_count": 6,
  "categories": [
   "cs.LG",
   "cs.AI",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 125,
  "influential_citations": 7,
  "tldr": "It is found that predicting multiple steps into the future (overshooting) is critical for stable representations to emerge and a scheme to reduce this computational burden is proposed, allowing us to build agents that are competitive with model-free baselines.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Karol Gregor",
    "id": "144717963",
    "h_index": 25,
    "papers": 40
   },
   {
    "name": "Danilo Jimenez Rezende",
    "id": "1748523",
    "h_index": 47,
    "papers": 82
   },
   {
    "name": "F. Besse",
    "id": "143923544",
    "h_index": 12,
    "papers": 19
   },
   {
    "name": "Yan Wu",
    "id": "1834814485",
    "h_index": 9,
    "papers": 15
   },
   {
    "name": "Hamza Merzic",
    "id": "20896818",
    "h_index": 9,
    "papers": 11
   },
   {
    "name": "A\u00e4ron van den Oord",
    "id": "3422336",
    "h_index": 42,
    "papers": 58
   }
  ],
  "comment": "pre-print",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1906.09237v2",
  "pdf_url": "https://arxiv.org/pdf/1906.09237v2",
  "html_url": "https://arxiv.org/html/1906.09237v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.6
 },
 {
  "id": "1906.08649",
  "slug": "exploring-model-based-planning-with-policy-networks",
  "title": "Exploring Model-based Planning with Policy Networks",
  "abstract": "Model-based reinforcement learning (MBRL) with model-predictive control or online planning has shown great potential for locomotion control tasks in terms of both sample efficiency and asymptotic performance. Despite their initial successes, the existing planning methods search from candidate sequences randomly generated in the action space, which is inefficient in complex high-dimensional environments. In this paper, we propose a novel MBRL algorithm, model-based policy planning (POPLIN), that combines policy networks with online planning. More specifically, we formulate action planning at each time-step as an optimization problem using neural networks. We experiment with both optimization w.r.t. the action sequences initialized from the policy network, and also online optimization directly w.r.t. the parameters of the policy network. We show that POPLIN obtains state-of-the-art performance in the MuJoCo benchmarking environments, being about 3x more sample efficient than the state-of-the-art algorithms, such as PETS, TD3 and SAC. To explain the effectiveness of our algorithm, we show that the optimization surface in parameter space is smoother than in action space. Further more, we found the distilled policy network can be effectively applied without the expansive model predictive control during test time for some environments such as Cheetah. Code is released in https://github.com/WilsonWangTHU/POPLIN.",
  "published": "2019-06-20",
  "updated": "2019-06-20",
  "year": "2019",
  "authors": [
   "Tingwu Wang",
   "Jimmy Ba"
  ],
  "author_count": 2,
  "categories": [
   "cs.LG",
   "cs.AI",
   "cs.RO",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 165,
  "influential_citations": 15,
  "tldr": "This paper proposes a novel MBRL algorithm, model-based policy planning (POPLIN), that combines policy networks with online planning and shows that POPLIN obtains state-of-the-art performance in the MuJoCo benchmarking environments, being about 3x more sample efficient than the state of theart algorithms, such as PETS, TD3 and SAC.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Tingwu Wang",
    "id": "3428549",
    "h_index": 10,
    "papers": 14
   },
   {
    "name": "Jimmy Ba",
    "id": "2503659",
    "h_index": 46,
    "papers": 84
   }
  ],
  "comment": "8 pages, 7 figures",
  "topics": [
   "humanoids",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1906.08649v1",
  "pdf_url": "https://arxiv.org/pdf/1906.08649v1",
  "html_url": "https://arxiv.org/html/1906.08649v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.72
 },
 {
  "id": "1906.08226",
  "slug": "unsupervised-state-representation-learning-in-atari",
  "title": "Unsupervised State Representation Learning in Atari",
  "abstract": "State representation learning, or the ability to capture latent generative factors of an environment, is crucial for building intelligent agents that can perform a wide variety of tasks. Learning such representations without supervision from rewards is a challenging open problem. We introduce a method that learns state representations by maximizing mutual information across spatially and temporally distinct features of a neural encoder of the observations. We also introduce a new benchmark based on Atari 2600 games where we evaluate representations based on how well they capture the ground truth state variables. We believe this new framework for evaluating representation learning models will be crucial for future representation learning research. Finally, we compare our technique with other state-of-the-art generative and contrastive representation learning methods. The code associated with this work is available at https://github.com/mila-iqia/atari-representation-learning",
  "published": "2019-06-19",
  "updated": "2020-11-05",
  "year": "2019",
  "authors": [
   "Ankesh Anand",
   "Evan Racah",
   "Sherjil Ozair",
   "Yoshua Bengio",
   "Marc-Alexandre C\u00f4t\u00e9",
   "R Devon Hjelm"
  ],
  "author_count": 6,
  "categories": [
   "cs.LG",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 284,
  "influential_citations": 27,
  "tldr": "A method that learns state representations by maximizing mutual information across spatially and temporally distinct features of a neural encoder of the observations is introduced, and a new benchmark based on Atari 2600 games is introduced.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ankesh Anand",
    "id": "12679121",
    "h_index": 14,
    "papers": 28
   },
   {
    "name": "Evan Racah",
    "id": "3159503",
    "h_index": 14,
    "papers": 27
   },
   {
    "name": "Sherjil Ozair",
    "id": "1955694",
    "h_index": 16,
    "papers": 32
   },
   {
    "name": "Yoshua Bengio",
    "id": "1751762",
    "h_index": 212,
    "papers": 813
   },
   {
    "name": "Marc-Alexandre C\u00f4t\u00e9",
    "id": "40638665",
    "h_index": 26,
    "papers": 68
   },
   {
    "name": "R. Devon Hjelm",
    "id": "40482726",
    "h_index": 28,
    "papers": 59
   }
  ],
  "comment": "NeurIPS 2019; v6 fixes a broken figure reference",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1906.08226v6",
  "pdf_url": "https://arxiv.org/pdf/1906.08226v6",
  "html_url": "https://arxiv.org/html/1906.08226v6",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.95
 },
 {
  "id": "1906.07343",
  "slug": "language-as-an-abstraction-for-hierarchical-deep-reinforcement-learnin",
  "title": "Language as an Abstraction for Hierarchical Deep Reinforcement Learning",
  "abstract": "Solving complex, temporally-extended tasks is a long-standing problem in reinforcement learning (RL). We hypothesize that one critical element of solving such problems is the notion of compositionality. With the ability to learn concepts and sub-skills that can be composed to solve longer tasks, i.e. hierarchical RL, we can acquire temporally-extended behaviors. However, acquiring effective yet general abstractions for hierarchical RL is remarkably challenging. In this paper, we propose to use language as the abstraction, as it provides unique compositional structure, enabling fast learning and combinatorial generalization, while retaining tremendous flexibility, making it suitable for a variety of problems. Our approach learns an instruction-following low-level policy and a high-level policy that can reuse abstractions across tasks, in essence, permitting agents to reason using structured language. To study compositional task learning, we introduce an open-source object interaction environment built using the MuJoCo physics engine and the CLEVR engine. We find that, using our approach, agents can learn to solve to diverse, temporally-extended tasks such as object sorting and multi-object rearrangement, including from raw pixel observations. Our analysis reveals that the compositional nature of language is critical for learning diverse sub-skills and systematically generalizing to new sub-skills in comparison to non-compositional abstractions that use the same supervision.",
  "published": "2019-06-18",
  "updated": "2019-11-18",
  "year": "2019",
  "authors": [
   "Yiding Jiang",
   "Shixiang Gu",
   "Kevin Murphy",
   "Chelsea Finn"
  ],
  "author_count": 4,
  "categories": [
   "cs.LG",
   "cs.AI",
   "cs.CL",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 250,
  "influential_citations": 19,
  "tldr": "This paper introduces an open-source object interaction environment built using the MuJoCo physics engine and the CLEVR engine and finds that, using the approach, agents can learn to solve to diverse, temporally-extended tasks such as object sorting and multi-object rearrangement, including from raw pixel observations.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yiding Jiang",
    "id": "4614137",
    "h_index": 14,
    "papers": 17
   },
   {
    "name": "S. Gu",
    "id": "2046135",
    "h_index": 43,
    "papers": 71
   },
   {
    "name": "K. Murphy",
    "id": "1702318",
    "h_index": 52,
    "papers": 68
   },
   {
    "name": "Chelsea Finn",
    "id": "46881670",
    "h_index": 94,
    "papers": 240
   }
  ],
  "comment": "Published in Neural Information Processing Systems (NeurIPS) 2019; Supplementary materials: https://sites.google.com/view/hal-demo",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1906.07343v2",
  "pdf_url": "https://arxiv.org/pdf/1906.07343v2",
  "html_url": "https://arxiv.org/html/1906.07343v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.9
 },
 {
  "id": "1906.05841",
  "slug": "deep-reinforcement-learning-for-industrial-insertion-tasks-with-visual",
  "title": "Deep Reinforcement Learning for Industrial Insertion Tasks with Visual Inputs and Natural Rewards",
  "abstract": "Connector insertion and many other tasks commonly found in modern manufacturing settings involve complex contact dynamics and friction. Since it is difficult to capture related physical effects with first-order modeling, traditional control methods often result in brittle and inaccurate controllers, which have to be manually tuned. Reinforcement learning (RL) methods have been demonstrated to be capable of learning controllers in such environments from autonomous interaction with the environment, but running RL algorithms in the real world poses sample efficiency and safety challenges. Moreover, in practical real-world settings we cannot assume access to perfect state information or dense reward signals. In this paper, we consider a variety of difficult industrial insertion tasks with visual inputs and different natural reward specifications, namely sparse rewards and goal images. We show that methods that combine RL with prior information, such as classical controllers or demonstrations, can solve these tasks from a reasonable amount of real-world interaction.",
  "published": "2019-06-13",
  "updated": "2019-08-02",
  "year": "2019",
  "authors": [
   "Gerrit Schoettler",
   "Ashvin Nair",
   "Jianlan Luo",
   "Shikhar Bahl",
   "Juan Aparicio Ojea",
   "Eugen Solowjow",
   "Sergey Levine"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS",
  "venue_source": "semantic-scholar",
  "citations": 220,
  "influential_citations": 14,
  "tldr": "This paper considers a variety of difficult industrial insertion tasks with visual inputs and different natural reward specifications, namely sparse rewards and goal images and shows that methods that combine RL with prior information can solve these tasks from a reasonable amount of real-world interaction.",
  "doi": "10.1109/IROS45743.2020.9341714",
  "oa_pdf": "https://arxiv.org/pdf/1906.05841",
  "s2_authors": [
   {
    "name": "Gerrit Schoettler",
    "id": "146843417",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "Ashvin Nair",
    "id": "3422774",
    "h_index": 18,
    "papers": 22
   },
   {
    "name": "Jianlan Luo",
    "id": "2591708",
    "h_index": 11,
    "papers": 17
   },
   {
    "name": "Shikhar Bahl",
    "id": "8527563",
    "h_index": 20,
    "papers": 38
   },
   {
    "name": "J. A. Ojea",
    "id": "32027425",
    "h_index": 13,
    "papers": 17
   },
   {
    "name": "Eugen Solowjow",
    "id": "2419277",
    "h_index": 16,
    "papers": 48
   },
   {
    "name": "S. Levine",
    "id": "1736651",
    "h_index": 167,
    "papers": 551
   }
  ],
  "comment": "",
  "topics": [
   "rl-control",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1906.05841v2",
  "pdf_url": "https://arxiv.org/pdf/1906.05841v2",
  "html_url": "https://arxiv.org/html/1906.05841v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.84
 },
 {
  "id": "1906.03853",
  "slug": "densephysnet-learning-dense-physical-object-representations-via-multi",
  "title": "DensePhysNet: Learning Dense Physical Object Representations via Multi-step Dynamic Interactions",
  "abstract": "We study the problem of learning physical object representations for robot manipulation. Understanding object physics is critical for successful object manipulation, but also challenging because physical object properties can rarely be inferred from the object's static appearance. In this paper, we propose DensePhysNet, a system that actively executes a sequence of dynamic interactions (e.g., sliding and colliding), and uses a deep predictive model over its visual observations to learn dense, pixel-wise representations that reflect the physical properties of observed objects. Our experiments in both simulation and real settings demonstrate that the learned representations carry rich physical information, and can directly be used to decode physical object properties such as friction and mass. The use of dense representation enables DensePhysNet to generalize well to novel scenes with more objects than in training. With knowledge of object physics, the learned representation also leads to more accurate and efficient manipulation in downstream tasks than the state-of-the-art.",
  "published": "2019-06-10",
  "updated": "2019-06-11",
  "year": "2019",
  "authors": [
   "Zhenjia Xu",
   "Jiajun Wu",
   "Andy Zeng",
   "Joshua B. Tenenbaum",
   "Shuran Song"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "RSS",
  "venue_source": "semantic-scholar",
  "citations": 150,
  "influential_citations": 13,
  "tldr": "DensePhysNet is proposed, a system that actively executes a sequence of dynamic interactions, and uses a deep predictive model over its visual observations to learn dense, pixel-wise representations that reflect the physical properties of observed objects.",
  "doi": "10.15607/RSS.2019.XV.046",
  "oa_pdf": "https://doi.org/10.15607/rss.2019.xv.046",
  "s2_authors": [
   {
    "name": "Zhenjia Xu",
    "id": "74498275",
    "h_index": 15,
    "papers": 22
   },
   {
    "name": "Jiajun Wu",
    "id": "3045089",
    "h_index": 80,
    "papers": 228
   },
   {
    "name": "Andy Zeng",
    "id": "38591293",
    "h_index": 34,
    "papers": 50
   },
   {
    "name": "J. Tenenbaum",
    "id": "1763295",
    "h_index": 140,
    "papers": 785
   },
   {
    "name": "Shuran Song",
    "id": "3340170",
    "h_index": 59,
    "papers": 90
   }
  ],
  "comment": "RSS 2019. Project page: http://zhenjiaxu.com/DensePhysNet",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1906.03853v2",
  "pdf_url": "https://arxiv.org/pdf/1906.03853v2",
  "html_url": "https://arxiv.org/html/1906.03853v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.68
 },
 {
  "id": "1906.03327",
  "slug": "howto100m-learning-a-text-video-embedding-by-watching-hundred-million",
  "title": "HowTo100M: Learning a Text-Video Embedding by Watching Hundred Million Narrated Video Clips",
  "abstract": "Learning text-video embeddings usually requires a dataset of video clips with manually provided captions. However, such datasets are expensive and time consuming to create and therefore difficult to obtain on a large scale. In this work, we propose instead to learn such embeddings from video data with readily available natural language annotations in the form of automatically transcribed narrations. The contributions of this work are three-fold. First, we introduce HowTo100M: a large-scale dataset of 136 million video clips sourced from 1.22M narrated instructional web videos depicting humans performing and describing over 23k different visual tasks. Our data collection procedure is fast, scalable and does not require any additional manual annotation. Second, we demonstrate that a text-video embedding trained on this data leads to state-of-the-art results for text-to-video retrieval and action localization on instructional video datasets such as YouCook2 or CrossTask. Finally, we show that this embedding transfers well to other domains: fine-tuning on generic Youtube videos (MSR-VTT dataset) and movies (LSMDC dataset) outperforms models trained on these datasets alone. Our dataset, code and models will be publicly available at: www.di.ens.fr/willow/research/howto100m/.",
  "published": "2019-06-07",
  "updated": "2019-07-31",
  "year": "2019",
  "authors": [
   "Antoine Miech",
   "Dimitri Zhukov",
   "Jean-Baptiste Alayrac",
   "Makarand Tapaswi",
   "Ivan Laptev",
   "Josef Sivic"
  ],
  "author_count": 6,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "ICCV",
  "venue_source": "semantic-scholar",
  "citations": 1510,
  "influential_citations": 245,
  "tldr": "It is demonstrated that a text-video embedding trained on this data leads to state-of-the-art results for text-to-video retrieval and action localization on instructional video datasets such as YouCook2 or CrossTask.",
  "doi": "10.1109/ICCV.2019.00272",
  "oa_pdf": "https://arxiv.org/pdf/1906.03327",
  "s2_authors": [
   {
    "name": "Antoine Miech",
    "id": "19200186",
    "h_index": 21,
    "papers": 42
   },
   {
    "name": "D. Zhukov",
    "id": "35838466",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "Jean-Baptiste Alayrac",
    "id": "2285263",
    "h_index": 33,
    "papers": 67
   },
   {
    "name": "Makarand Tapaswi",
    "id": "2103464",
    "h_index": 25,
    "papers": 90
   },
   {
    "name": "I. Laptev",
    "id": "143991676",
    "h_index": 76,
    "papers": 193
   },
   {
    "name": "Josef Sivic",
    "id": "1782755",
    "h_index": 79,
    "papers": 193
   }
  ],
  "comment": "Accepted at ICCV 2019",
  "topics": [
   "data-teleop",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1906.03327v2",
  "pdf_url": "https://arxiv.org/pdf/1906.03327v2",
  "html_url": "https://arxiv.org/html/1906.03327v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "1906.02736",
  "slug": "deepmdp-learning-continuous-latent-space-models-for-representation-lea",
  "title": "DeepMDP: Learning Continuous Latent Space Models for Representation Learning",
  "abstract": "Many reinforcement learning (RL) tasks provide the agent with high-dimensional observations that can be simplified into low-dimensional continuous states. To formalize this process, we introduce the concept of a DeepMDP, a parameterized latent space model that is trained via the minimization of two tractable losses: prediction of rewards and prediction of the distribution over next latent states. We show that the optimization of these objectives guarantees (1) the quality of the latent space as a representation of the state space and (2) the quality of the DeepMDP as a model of the environment. We connect these results to prior work in the bisimulation literature, and explore the use of a variety of metrics. Our theoretical findings are substantiated by the experimental result that a trained DeepMDP recovers the latent structure underlying high-dimensional observations on a synthetic environment. Finally, we show that learning a DeepMDP as an auxiliary task in the Atari 2600 domain leads to large performance improvements over model-free RL.",
  "published": "2019-06-06",
  "updated": "2019-06-06",
  "year": "2019",
  "authors": [
   "Carles Gelada",
   "Saurabh Kumar",
   "Jacob Buckman",
   "Ofir Nachum",
   "Marc G. Bellemare"
  ],
  "author_count": 5,
  "categories": [
   "cs.LG",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "ICML",
  "venue_source": "semantic-scholar",
  "citations": 362,
  "influential_citations": 34,
  "tldr": "This work introduces the concept of a DeepMDP, a parameterized latent space model that is trained via the minimization of two tractable losses: prediction of rewards and prediction of the distribution over next latent states, and shows that the optimization of these objectives guarantees the quality of the latent space as a representation of the state space.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Carles Gelada",
    "id": "52382152",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Saurabh Kumar",
    "id": "2121434953",
    "h_index": 11,
    "papers": 14
   },
   {
    "name": "Jacob Buckman",
    "id": "47619311",
    "h_index": 8,
    "papers": 11
   },
   {
    "name": "Ofir Nachum",
    "id": "7624658",
    "h_index": 47,
    "papers": 92
   },
   {
    "name": "Marc G. Bellemare",
    "id": "1792298",
    "h_index": 44,
    "papers": 92
   }
  ],
  "comment": "13 pages main text, 16 pages appendix. ICML 2019",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1906.02736v1",
  "pdf_url": "https://arxiv.org/pdf/1906.02736v1",
  "html_url": "https://arxiv.org/html/1906.02736v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.06
 },
 {
  "id": "1906.00446",
  "slug": "generating-diverse-high-fidelity-images-with-vq-vae-2",
  "title": "Generating Diverse High-Fidelity Images with VQ-VAE-2",
  "abstract": "We explore the use of Vector Quantized Variational AutoEncoder (VQ-VAE) models for large scale image generation. To this end, we scale and enhance the autoregressive priors used in VQ-VAE to generate synthetic samples of much higher coherence and fidelity than possible before. We use simple feed-forward encoder and decoder networks, making our model an attractive candidate for applications where the encoding and/or decoding speed is critical. Additionally, VQ-VAE requires sampling an autoregressive model only in the compressed latent space, which is an order of magnitude faster than sampling in the pixel space, especially for large images. We demonstrate that a multi-scale hierarchical organization of VQ-VAE, augmented with powerful priors over the latent codes, is able to generate samples with quality that rivals that of state of the art Generative Adversarial Networks on multifaceted datasets such as ImageNet, while not suffering from GAN's known shortcomings such as mode collapse and lack of diversity.",
  "published": "2019-06-02",
  "updated": "2019-06-02",
  "year": "2019",
  "authors": [
   "Ali Razavi",
   "Aaron van den Oord",
   "Oriol Vinyals"
  ],
  "author_count": 3,
  "categories": [
   "cs.LG",
   "cs.CV",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 2447,
  "influential_citations": 186,
  "tldr": "It is demonstrated that a multi-scale hierarchical organization of VQ-VAE, augmented with powerful priors over the latent codes, is able to generate samples with quality that rivals that of state of the art Generative Adversarial Networks on multifaceted datasets such as ImageNet, while not suffering from GAN's known shortcomings such as mode collapse and lack of diversity.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ali Razavi",
    "id": "143653164",
    "h_index": 14,
    "papers": 17
   },
   {
    "name": "A\u00e4ron van den Oord",
    "id": "3422336",
    "h_index": 42,
    "papers": 58
   },
   {
    "name": "O. Vinyals",
    "id": "1689108",
    "h_index": 103,
    "papers": 204
   }
  ],
  "comment": "",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1906.00446v1",
  "pdf_url": "https://arxiv.org/pdf/1906.00446v1",
  "html_url": "https://arxiv.org/html/1906.00446v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "1905.12681",
  "slug": "what-makes-training-multi-modal-classification-networks-hard",
  "title": "What Makes Training Multi-Modal Classification Networks Hard?",
  "abstract": "Consider end-to-end training of a multi-modal vs. a single-modal network on a task with multiple input modalities: the multi-modal network receives more information, so it should match or outperform its single-modal counterpart. In our experiments, however, we observe the opposite: the best single-modal network always outperforms the multi-modal network. This observation is consistent across different combinations of modalities and on different tasks and benchmarks. This paper identifies two main causes for this performance drop: first, multi-modal networks are often prone to overfitting due to increased capacity. Second, different modalities overfit and generalize at different rates, so training them jointly with a single optimization strategy is sub-optimal. We address these two problems with a technique we call Gradient Blending, which computes an optimal blend of modalities based on their overfitting behavior. We demonstrate that Gradient Blending outperforms widely-used baselines for avoiding overfitting and achieves state-of-the-art accuracy on various tasks including human action recognition, ego-centric action recognition, and acoustic event detection.",
  "published": "2019-05-29",
  "updated": "2020-04-03",
  "year": "2019",
  "authors": [
   "Weiyao Wang",
   "Du Tran",
   "Matt Feiszli"
  ],
  "author_count": 3,
  "categories": [
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 695,
  "influential_citations": 95,
  "tldr": "This paper identifies two main causes for this performance drop: first, multi-modal networks are often prone to overfitting due to increased capacity and second, different modalities overfit and generalize at different rates, so training them jointly with a single optimization strategy is sub-optimal.",
  "doi": "10.1109/CVPR42600.2020.01271",
  "oa_pdf": "https://arxiv.org/pdf/1905.12681",
  "s2_authors": [
   {
    "name": "Weiyao Wang",
    "id": "7634810",
    "h_index": 13,
    "papers": 26
   },
   {
    "name": "Du Tran",
    "id": "1687325",
    "h_index": 29,
    "papers": 53
   },
   {
    "name": "Matt Feiszli",
    "id": "3429328",
    "h_index": 20,
    "papers": 46
   }
  ],
  "comment": "CVPR 2020",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1905.12681v5",
  "pdf_url": "https://arxiv.org/pdf/1905.12681v5",
  "html_url": "https://arxiv.org/html/1905.12681v5",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.34
 },
 {
  "id": "1905.12340",
  "slug": "rethinking-full-connectivity-in-recurrent-neural-networks",
  "title": "Rethinking Full Connectivity in Recurrent Neural Networks",
  "abstract": "Recurrent neural networks (RNNs) are omnipresent in sequence modeling tasks. Practical models usually consist of several layers of hundreds or thousands of neurons which are fully connected. This places a heavy computational and memory burden on hardware, restricting adoption in practical low-cost and low-power devices. Compared to fully convolutional models, the costly sequential operation of RNNs severely hinders performance on parallel hardware. This paper challenges the convention of full connectivity in RNNs. We study structurally sparse RNNs, showing that they are well suited for acceleration on parallel hardware, with a greatly reduced cost of the recurrent operations as well as orders of magnitude less recurrent weights. Extensive experiments on challenging tasks ranging from language modeling and speech recognition to video action recognition reveal that structurally sparse RNNs achieve competitive performance as compared to fully-connected networks. This allows for using large sparse RNNs for a wide range of real-world tasks that previously were too costly with fully connected networks.",
  "published": "2019-05-29",
  "updated": "2019-05-29",
  "year": "2019",
  "authors": [
   "Matthijs Van Keirsbilck",
   "Alexander Keller",
   "Xiaodong Yang"
  ],
  "author_count": 3,
  "categories": [
   "cs.LG",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 18,
  "influential_citations": 0,
  "tldr": "Structurally sparse RNNs are studied, showing that they are well suited for acceleration on parallel hardware, with a greatly reduced cost of the recurrent operations as well as orders of magnitude less recurrent weights.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Matthijs Van Keirsbilck",
    "id": "40897085",
    "h_index": 7,
    "papers": 10
   },
   {
    "name": "A. Keller",
    "id": "145661463",
    "h_index": 35,
    "papers": 115
   },
   {
    "name": "Xiaodong Yang",
    "id": "38101706",
    "h_index": 40,
    "papers": 93
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1905.12340v1",
  "pdf_url": "https://arxiv.org/pdf/1905.12340v1",
  "html_url": "https://arxiv.org/html/1905.12340v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.28
 },
 {
  "id": "1905.06922",
  "slug": "on-variational-bounds-of-mutual-information",
  "title": "On Variational Bounds of Mutual Information",
  "abstract": "Estimating and optimizing Mutual Information (MI) is core to many problems in machine learning; however, bounding MI in high dimensions is challenging. To establish tractable and scalable objectives, recent work has turned to variational bounds parameterized by neural networks, but the relationships and tradeoffs between these bounds remains unclear. In this work, we unify these recent developments in a single framework. We find that the existing variational lower bounds degrade when the MI is large, exhibiting either high bias or high variance. To address this problem, we introduce a continuum of lower bounds that encompasses previous bounds and flexibly trades off bias and variance. On high-dimensional, controlled problems, we empirically characterize the bias and variance of the bounds and their gradients and demonstrate the effectiveness of our new bounds for estimation and representation learning.",
  "published": "2019-05-16",
  "updated": "2019-05-16",
  "year": "2019",
  "authors": [
   "Ben Poole",
   "Sherjil Ozair",
   "Aaron van den Oord",
   "Alexander A. Alemi",
   "George Tucker"
  ],
  "author_count": 5,
  "categories": [
   "cs.LG",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "ICML",
  "venue_source": "semantic-scholar",
  "citations": 1059,
  "influential_citations": 116,
  "tldr": "This work introduces a continuum of lower bounds that encompasses previous bounds and flexibly trades off bias and variance and demonstrates the effectiveness of these new bounds for estimation and representation learning.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ben Poole",
    "id": "16443937",
    "h_index": 40,
    "papers": 64
   },
   {
    "name": "Sherjil Ozair",
    "id": "1955694",
    "h_index": 16,
    "papers": 32
   },
   {
    "name": "A\u00e4ron van den Oord",
    "id": "3422336",
    "h_index": 42,
    "papers": 58
   },
   {
    "name": "Alexander A. Alemi",
    "id": "122113652",
    "h_index": 25,
    "papers": 68
   },
   {
    "name": "G. Tucker",
    "id": "145499435",
    "h_index": 34,
    "papers": 55
   }
  ],
  "comment": "ICML 2019",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1905.06922v1",
  "pdf_url": "https://arxiv.org/pdf/1905.06922v1",
  "html_url": "https://arxiv.org/html/1905.06922v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "1906.08857",
  "slug": "deep-neuroevolution-of-recurrent-and-discrete-world-models",
  "title": "Deep Neuroevolution of Recurrent and Discrete World Models",
  "abstract": "Neural architectures inspired by our own human cognitive system, such as the recently introduced world models, have been shown to outperform traditional deep reinforcement learning (RL) methods in a variety of different domains. Instead of the relatively simple architectures employed in most RL experiments, world models rely on multiple different neural components that are responsible for visual information processing, memory, and decision-making. However, so far the components of these models have to be trained separately and through a variety of specialized training methods. This paper demonstrates the surprising finding that models with the same precise parts can be instead efficiently trained end-to-end through a genetic algorithm (GA), reaching a comparable performance to the original world model by solving a challenging car racing task. An analysis of the evolved visual and memory system indicates that they include a similar effective representation to the system trained through gradient descent. Additionally, in contrast to gradient descent methods that struggle with discrete variables, GAs also work directly with such representations, opening up opportunities for classical planning in latent space. This paper adds additional evidence on the effectiveness of deep neuroevolution for tasks that require the intricate orchestration of multiple components in complex heterogeneous architectures.",
  "published": "2019-04-28",
  "updated": "2019-04-28",
  "year": "2019",
  "authors": [
   "Sebastian Risi",
   "Kenneth O. Stanley"
  ],
  "author_count": 2,
  "categories": [
   "cs.NE",
   "cs.AI"
  ],
  "primary_category": "cs.NE",
  "venue": "",
  "venue_source": "",
  "citations": 58,
  "influential_citations": 2,
  "tldr": "This paper demonstrates the surprising finding that models with the same precise parts can be instead efficiently trained end-to-end through a genetic algorithm (GA), reaching a comparable performance to the original world model by solving a challenging car racing task.",
  "doi": "10.1145/3321707.3321817",
  "oa_pdf": "https://pure.itu.dk/portal/files/84783035/1906.08857.pdf",
  "s2_authors": [
   {
    "name": "S. Risi",
    "id": "1745664",
    "h_index": 38,
    "papers": 171
   },
   {
    "name": "Kenneth O. Stanley",
    "id": "1846883",
    "h_index": 68,
    "papers": 212
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1906.08857v1",
  "pdf_url": "https://arxiv.org/pdf/1906.08857v1",
  "html_url": "https://arxiv.org/html/1906.08857v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.77
 },
 {
  "id": "1904.10666",
  "slug": "segmenting-the-future",
  "title": "Segmenting the Future",
  "abstract": "Predicting the future is an important aspect for decision-making in robotics or autonomous driving systems, which heavily rely upon visual scene understanding. While prior work attempts to predict future video pixels, anticipate activities or forecast future scene semantic segments from segmentation of the preceding frames, methods that predict future semantic segmentation solely from the previous frame RGB data in a single end-to-end trainable model do not exist. In this paper, we propose a temporal encoder-decoder network architecture that encodes RGB frames from the past and decodes the future semantic segmentation. The network is coupled with a new knowledge distillation training framework specific for the forecasting task. Our method, only seeing preceding video frames, implicitly models the scene segments while simultaneously accounting for the object dynamics to infer the future scene semantic segments. Our results on Cityscapes and Apolloscape outperform the baseline and current state-of-the-art methods. Code is available at https://github.com/eddyhkchiu/segmenting_the_future/.",
  "published": "2019-04-24",
  "updated": "2019-12-12",
  "year": "2019",
  "authors": [
   "Hsu-kuang Chiu",
   "Ehsan Adeli",
   "Juan Carlos Niebles"
  ],
  "author_count": 3,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "RA-L",
  "venue_source": "semantic-scholar",
  "citations": 54,
  "influential_citations": 2,
  "tldr": "A temporal encoder-decoder network architecture that encodes RGB frames from the past and decodes the future semantic segmentation of the scene segments while simultaneously accounting for the object dynamics to infer the future scene semantic segments is proposed.",
  "doi": "10.1109/LRA.2020.2992184",
  "oa_pdf": "https://arxiv.org/pdf/1904.10666",
  "s2_authors": [
   {
    "name": "Hsu-kuang Chiu",
    "id": "3306760",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "Ehsan Adeli",
    "id": "2397747865",
    "h_index": 15,
    "papers": 27
   },
   {
    "name": "Juan Carlos Niebles",
    "id": "9200530",
    "h_index": 66,
    "papers": 174
   }
  ],
  "comment": "",
  "topics": [
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1904.10666v2",
  "pdf_url": "https://arxiv.org/pdf/1904.10666v2",
  "html_url": "https://arxiv.org/html/1904.10666v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.24
 },
 {
  "id": "1904.10079",
  "slug": "the-minerl-2019-competition-on-sample-efficient-reinforcement-learning",
  "title": "The MineRL 2019 Competition on Sample Efficient Reinforcement Learning using Human Priors",
  "abstract": "Though deep reinforcement learning has led to breakthroughs in many difficult domains, these successes have required an ever-increasing number of samples. As state-of-the-art reinforcement learning (RL) systems require an exponentially increasing number of samples, their development is restricted to a continually shrinking segment of the AI community. Likewise, many of these systems cannot be applied to real-world problems, where environment samples are expensive. Resolution of these limitations requires new, sample-efficient methods. To facilitate research in this direction, we introduce the MineRL Competition on Sample Efficient Reinforcement Learning using Human Priors. The primary goal of the competition is to foster the development of algorithms which can efficiently leverage human demonstrations to drastically reduce the number of samples needed to solve complex, hierarchical, and sparse environments. To that end, we introduce: (1) the Minecraft ObtainDiamond task, a sequential decision making environment requiring long-term planning, hierarchical control, and efficient exploration methods; and (2) the MineRL-v0 dataset, a large-scale collection of over 60 million state-action pairs of human demonstrations that can be resimulated into embodied trajectories with arbitrary modifications to game state and visuals. Participants will compete to develop systems which solve the ObtainDiamond task with a limited number of samples from the environment simulator, Malmo. The competition is structured into two rounds in which competitors are provided several paired versions of the dataset and environment with different game textures. At the end of each round, competitors will submit containerized versions of their learning algorithms and they will then be trained/evaluated from scratch on a hold-out dataset-environment pair for a total of 4-days on a prespecified hardware platform.",
  "published": "2019-04-22",
  "updated": "2021-01-19",
  "year": "2019",
  "authors": [
   "William H. Guss",
   "Cayden Codel",
   "Katja Hofmann",
   "Brandon Houghton",
   "Noboru Kuno",
   "Stephanie Milani",
   "Sharada Mohanty",
   "Diego Perez Liebana",
   "Ruslan Salakhutdinov",
   "Nicholay Topin",
   "Manuela Veloso",
   "Phillip Wang"
  ],
  "author_count": 12,
  "categories": [
   "cs.LG",
   "cs.AI",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "NeurIPS 2019",
  "venue_source": "arxiv-comment",
  "citations": 72,
  "influential_citations": 4,
  "tldr": "The MineRL Competition on Sample Efficient Reinforcement Learning using Human Priors is introduced, to foster the development of algorithms which can efficiently leverage human demonstrations to drastically reduce the number of samples needed to solve complex, hierarchical, and sparse environments.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "William H. Guss",
    "id": "39121861",
    "h_index": 12,
    "papers": 20
   },
   {
    "name": "Cayden R. Codel",
    "id": "104300088",
    "h_index": 4,
    "papers": 17
   },
   {
    "name": "Katja Hofmann",
    "id": "2256702356",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "Brandon Houghton",
    "id": "103681415",
    "h_index": 9,
    "papers": 16
   },
   {
    "name": "Noburu Kuno",
    "id": "67230338",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "Stephanie Milani",
    "id": "144177520",
    "h_index": 14,
    "papers": 29
   },
   {
    "name": "S. Mohanty",
    "id": "24178944",
    "h_index": 22,
    "papers": 42
   },
   {
    "name": "Diego Perez Liebana",
    "id": "2717017",
    "h_index": 29,
    "papers": 89
   },
   {
    "name": "R. Salakhutdinov",
    "id": "145124475",
    "h_index": 120,
    "papers": 369
   },
   {
    "name": "Nicholay Topin",
    "id": "34887814",
    "h_index": 18,
    "papers": 28
   },
   {
    "name": "Manuela M. Veloso",
    "id": "2285981900",
    "h_index": 6,
    "papers": 15
   },
   {
    "name": "Phillip Wang",
    "id": "2108692541",
    "h_index": 4,
    "papers": 4
   }
  ],
  "comment": "accepted at NeurIPS 2019, 28 pages",
  "topics": [
   "egocentric-data",
   "sim2real",
   "rl-control",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1904.10079v3",
  "pdf_url": "https://arxiv.org/pdf/1904.10079v3",
  "html_url": "https://arxiv.org/html/1904.10079v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.36
 },
 {
  "id": "1904.05538",
  "slug": "improvisation-through-physical-understanding-using-novel-objects-as-to",
  "title": "Improvisation through Physical Understanding: Using Novel Objects as Tools with Visual Foresight",
  "abstract": "Machine learning techniques have enabled robots to learn narrow, yet complex tasks and also perform broad, yet simple skills with a wide variety of objects. However, learning a model that can both perform complex tasks and generalize to previously unseen objects and goals remains a significant challenge. We study this challenge in the context of \"improvisational\" tool use: a robot is presented with novel objects and a user-specified goal (e.g., sweep some clutter into the dustpan), and must figure out, using only raw image observations, how to accomplish the goal using the available objects as tools. We approach this problem by training a model with both a visual and physical understanding of multi-object interactions, and develop a sampling-based optimizer that can leverage these interactions to accomplish tasks. We do so by combining diverse demonstration data with self-supervised interaction data, aiming to leverage the interaction data to build generalizable models and the demonstration data to guide the model-based RL planner to solve complex tasks. Our experiments show that our approach can solve a variety of complex tool use tasks from raw pixel inputs, outperforming both imitation learning and self-supervised learning individually. Furthermore, we show that the robot can perceive and use novel objects as tools, including objects that are not conventional tools, while also choosing dynamically to use or not use tools depending on whether or not they are required.",
  "published": "2019-04-11",
  "updated": "2019-04-11",
  "year": "2019",
  "authors": [
   "Annie Xie",
   "Frederik Ebert",
   "Sergey Levine",
   "Chelsea Finn"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "RSS",
  "venue_source": "semantic-scholar",
  "citations": 96,
  "influential_citations": 3,
  "tldr": "This work training a model with both a visual and physical understanding of multi-object interactions, and develops a sampling-based optimizer that can leverage these interactions to accomplish tasks, shows that the robot can perceive and use novel objects as tools, including objects that are not conventional tools, while also choosing dynamically to use or not use tools depending on whether or not they are required.",
  "doi": "10.15607/RSS.2019.XV.001",
  "oa_pdf": "https://doi.org/10.15607/rss.2019.xv.001",
  "s2_authors": [
   {
    "name": "Annie Xie",
    "id": "14484808",
    "h_index": 21,
    "papers": 23
   },
   {
    "name": "F. Ebert",
    "id": "27535721",
    "h_index": 15,
    "papers": 22
   },
   {
    "name": "S. Levine",
    "id": "1736651",
    "h_index": 167,
    "papers": 551
   },
   {
    "name": "Chelsea Finn",
    "id": "46881670",
    "h_index": 94,
    "papers": 240
   }
  ],
  "comment": "Videos available at https://sites.google.com/view/gvf-tool",
  "topics": [
   "imitation-diffusion"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1904.05538v1",
  "pdf_url": "https://arxiv.org/pdf/1904.05538v1",
  "html_url": "https://arxiv.org/html/1904.05538v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.49
 },
 {
  "id": "1904.04196",
  "slug": "pushing-the-envelope-for-rgb-based-dense-3d-hand-pose-estimation-via-n",
  "title": "Pushing the Envelope for RGB-based Dense 3D Hand Pose Estimation via Neural Rendering",
  "abstract": "Estimating 3D hand meshes from single RGB images is challenging, due to intrinsic 2D-3D mapping ambiguities and limited training data. We adopt a compact parametric 3D hand model that represents deformable and articulated hand meshes. To achieve the model fitting to RGB images, we investigate and contribute in three ways: 1) Neural rendering: inspired by recent work on human body, our hand mesh estimator (HME) is implemented by a neural network and a differentiable renderer, supervised by 2D segmentation masks and 3D skeletons. HME demonstrates good performance for estimating diverse hand shapes and improves pose estimation accuracies. 2) Iterative testing refinement: Our fitting function is differentiable. We iteratively refine the initial estimate using the gradients, in the spirit of iterative model fitting methods like ICP. The idea is supported by the latest research on human body. 3) Self-data augmentation: collecting sized RGB-mesh (or segmentation mask)-skeleton triplets for training is a big hurdle. Once the model is successfully fitted to input RGB images, its meshes i.e. shapes and articulations, are realistic, and we augment view-points on top of estimated dense hand poses. Experiments using three RGB-based benchmarks show that our framework offers beyond state-of-the-art accuracy in 3D pose estimation, as well as recovers dense 3D hand shapes. Each technical component above meaningfully improves the accuracy in the ablation study.",
  "published": "2019-04-08",
  "updated": "2019-04-09",
  "year": "2019",
  "authors": [
   "Seungryul Baek",
   "Kwang In Kim",
   "Tae-Kyun Kim"
  ],
  "author_count": 3,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 243,
  "influential_citations": 14,
  "tldr": "Experiments using three RGB-based benchmarks show that the framework offers beyond state-of-the-art accuracy in 3D pose estimation, as well as recovers dense 3D hand shapes.",
  "doi": "10.1109/CVPR.2019.00116",
  "oa_pdf": "https://arxiv.org/pdf/1904.04196",
  "s2_authors": [
   {
    "name": "Seungryul Baek",
    "id": "2637535",
    "h_index": 14,
    "papers": 34
   },
   {
    "name": "K. Kim",
    "id": "1808255",
    "h_index": 27,
    "papers": 71
   },
   {
    "name": "Tae-Kyun Kim",
    "id": "143617697",
    "h_index": 47,
    "papers": 131
   }
  ],
  "comment": "Accepted to CVPR 2019",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1904.04196v2",
  "pdf_url": "https://arxiv.org/pdf/1904.04196v2",
  "html_url": "https://arxiv.org/html/1904.04196v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.89
 },
 {
  "id": "1904.03754",
  "slug": "contactgrasp-functional-multi-finger-grasp-synthesis-from-contact",
  "title": "ContactGrasp: Functional Multi-finger Grasp Synthesis from Contact",
  "abstract": "Grasping and manipulating objects is an important human skill. Since most objects are designed to be manipulated by human hands, anthropomorphic hands can enable richer human-robot interaction. Desirable grasps are not only stable, but also functional: they enable post-grasp actions with the object. However, functional grasp synthesis for high degree-of-freedom anthropomorphic hands from object shape alone is challenging because of the large optimization space. We present ContactGrasp, a framework for functional grasp synthesis from object shape and contact on the object surface. Contact can be manually specified or obtained through demonstrations. Our contact representation is object-centric and allows functional grasp synthesis even for hand models different than the one used for demonstration. Using a dataset of contact demonstrations from humans grasping diverse household objects, we synthesize functional grasps for three hand models and two functional intents. The project webpage is https://contactdb.cc.gatech.edu/contactgrasp.html.",
  "published": "2019-04-07",
  "updated": "2019-07-25",
  "year": "2019",
  "authors": [
   "Samarth Brahmbhatt",
   "Ankur Handa",
   "James Hays",
   "Dieter Fox"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO",
   "cs.CV"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS",
  "venue_source": "semantic-scholar",
  "citations": 168,
  "influential_citations": 16,
  "tldr": "This work synthesizes functional grasps for three hand models and two functional intents for functional grasp synthesis from object shape and contact on the object surface using a dataset of contact demonstrations from humans grasping diverse household objects.",
  "doi": "10.1109/IROS40897.2019.8967960",
  "oa_pdf": "http://arxiv.org/pdf/1904.03754",
  "s2_authors": [
   {
    "name": "Samarth Brahmbhatt",
    "id": "2928511",
    "h_index": 12,
    "papers": 37
   },
   {
    "name": "Ankur Handa",
    "id": "34653454",
    "h_index": 33,
    "papers": 55
   },
   {
    "name": "James Hays",
    "id": "48966748",
    "h_index": 52,
    "papers": 123
   },
   {
    "name": "D. Fox",
    "id": "145197953",
    "h_index": 133,
    "papers": 428
   }
  ],
  "comment": "IROS 2019 camera ready version",
  "topics": [
   "dexterous-manipulation",
   "hri"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1904.03754v3",
  "pdf_url": "https://arxiv.org/pdf/1904.03754v3",
  "html_url": "https://arxiv.org/html/1904.03754v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.73
 },
 {
  "id": "1903.04128",
  "slug": "manipulation-by-feel-touch-based-control-with-deep-predictive-models",
  "title": "Manipulation by Feel: Touch-Based Control with Deep Predictive Models",
  "abstract": "Touch sensing is widely acknowledged to be important for dexterous robotic manipulation, but exploiting tactile sensing for continuous, non-prehensile manipulation is challenging. General purpose control techniques that are able to effectively leverage tactile sensing as well as accurate physics models of contacts and forces remain largely elusive, and it is unclear how to even specify a desired behavior in terms of tactile percepts. In this paper, we take a step towards addressing these issues by combining high-resolution tactile sensing with data-driven modeling using deep neural network dynamics models. We propose deep tactile MPC, a framework for learning to perform tactile servoing from raw tactile sensor inputs, without manual supervision. We show that this method enables a robot equipped with a GelSight-style tactile sensor to manipulate a ball, analog stick, and 20-sided die, learning from unsupervised autonomous interaction and then using the learned tactile predictive model to reposition each object to user-specified configurations, indicated by a goal tactile reading. Videos, visualizations and the code are available here: https://sites.google.com/view/deeptactilempc",
  "published": "2019-03-11",
  "updated": "2019-03-11",
  "year": "2019",
  "authors": [
   "Stephen Tian",
   "Frederik Ebert",
   "Dinesh Jayaraman",
   "Mayur Mudigonda",
   "Chelsea Finn",
   "Roberto Calandra",
   "Sergey Levine"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA",
  "venue_source": "semantic-scholar",
  "citations": 157,
  "influential_citations": 3,
  "tldr": "This paper proposes deep tactile MPC, a framework for learning to perform tactile servoing from raw tactile sensor inputs, without manual supervision, and shows that this method enables a robot equipped with a GelSight-style tactile sensor to manipulate a ball, analog stick, and 20-sided die.",
  "doi": "10.1109/ICRA.2019.8794219",
  "oa_pdf": "http://arxiv.org/pdf/1903.04128",
  "s2_authors": [
   {
    "name": "Stephen Tian",
    "id": "71692259",
    "h_index": 13,
    "papers": 14
   },
   {
    "name": "F. Ebert",
    "id": "27535721",
    "h_index": 15,
    "papers": 22
   },
   {
    "name": "Dinesh Jayaraman",
    "id": "144348441",
    "h_index": 33,
    "papers": 72
   },
   {
    "name": "M. Mudigonda",
    "id": "2045638",
    "h_index": 10,
    "papers": 30
   },
   {
    "name": "Chelsea Finn",
    "id": "46881670",
    "h_index": 94,
    "papers": 240
   },
   {
    "name": "R. Calandra",
    "id": "35159852",
    "h_index": 40,
    "papers": 75
   },
   {
    "name": "S. Levine",
    "id": "1736651",
    "h_index": 167,
    "papers": 551
   }
  ],
  "comment": "Accepted to ICRA 2019",
  "topics": [
   "dexterous-manipulation",
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1903.04128v1",
  "pdf_url": "https://arxiv.org/pdf/1903.04128v1",
  "html_url": "https://arxiv.org/html/1903.04128v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.7
 },
 {
  "id": "1903.03698",
  "slug": "skew-fit-state-covering-self-supervised-reinforcement-learning",
  "title": "Skew-Fit: State-Covering Self-Supervised Reinforcement Learning",
  "abstract": "Autonomous agents that must exhibit flexible and broad capabilities will need to be equipped with large repertoires of skills. Defining each skill with a manually-designed reward function limits this repertoire and imposes a manual engineering burden. Self-supervised agents that set their own goals can automate this process, but designing appropriate goal setting objectives can be difficult, and often involves heuristic design decisions. In this paper, we propose a formal exploration objective for goal-reaching policies that maximizes state coverage. We show that this objective is equivalent to maximizing goal reaching performance together with the entropy of the goal distribution, where goals correspond to full state observations. To instantiate this principle, we present an algorithm called Skew-Fit for learning a maximum-entropy goal distributions. We prove that, under regularity conditions, Skew-Fit converges to a uniform distribution over the set of valid states, even when we do not know this set beforehand. Our experiments show that combining Skew-Fit for learning goal distributions with existing goal-reaching methods outperforms a variety of prior methods on open-sourced visual goal-reaching tasks. Moreover, we demonstrate that Skew-Fit enables a real-world robot to learn to open a door, entirely from scratch, from pixels, and without any manually-designed reward function.",
  "published": "2019-03-08",
  "updated": "2020-08-04",
  "year": "2019",
  "authors": [
   "Vitchyr H. Pong",
   "Murtaza Dalal",
   "Steven Lin",
   "Ashvin Nair",
   "Shikhar Bahl",
   "Sergey Levine"
  ],
  "author_count": 6,
  "categories": [
   "cs.LG",
   "cs.AI",
   "cs.RO",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "ICML",
  "venue_source": "semantic-scholar",
  "citations": 316,
  "influential_citations": 38,
  "tldr": "This paper proposes a formal exploration objective for goal-reaching policies that maximizes state coverage and presents an algorithm called Skew-Fit, which enables a real-world robot to learn to open a door, entirely from scratch, from pixels, and without any manually-designed reward function.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Vitchyr H. Pong",
    "id": "144401061",
    "h_index": 15,
    "papers": 36
   },
   {
    "name": "Murtaza Dalal",
    "id": "35904540",
    "h_index": 13,
    "papers": 16
   },
   {
    "name": "Steven Lin",
    "id": "2110473490",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Ashvin Nair",
    "id": "3422774",
    "h_index": 18,
    "papers": 22
   },
   {
    "name": "Shikhar Bahl",
    "id": "8527563",
    "h_index": 20,
    "papers": 38
   },
   {
    "name": "S. Levine",
    "id": "1736651",
    "h_index": 167,
    "papers": 551
   }
  ],
  "comment": "ICML 2020. 8 pages, 8 figures; 9 pages appendix (6 additional figures)",
  "topics": [
   "rl-control",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1903.03698v4",
  "pdf_url": "https://arxiv.org/pdf/1903.03698v4",
  "html_url": "https://arxiv.org/html/1903.03698v4",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.0
 },
 {
  "id": "1903.03591",
  "slug": "learning-to-identify-object-instances-by-touch-tactile-recognition-via",
  "title": "Learning to Identify Object Instances by Touch: Tactile Recognition via Multimodal Matching",
  "abstract": "Much of the literature on robotic perception focuses on the visual modality. Vision provides a global observation of a scene, making it broadly useful. However, in the domain of robotic manipulation, vision alone can sometimes prove inadequate: in the presence of occlusions or poor lighting, visual object identification might be difficult. The sense of touch can provide robots with an alternative mechanism for recognizing objects. In this paper, we study the problem of touch-based instance recognition. We propose a novel framing of the problem as multi-modal recognition: the goal of our system is to recognize, given a visual and tactile observation, whether or not these observations correspond to the same object. To our knowledge, our work is the first to address this type of multi-modal instance recognition problem on such a large-scale with our analysis spanning 98 different objects. We employ a robot equipped with two GelSight touch sensors, one on each finger, and a self-supervised, autonomous data collection procedure to collect a dataset of tactile observations and images. Our experimental results show that it is possible to accurately recognize object instances by touch alone, including instances of novel objects that were never seen during training. Our learned model outperforms other methods on this complex task, including that of human volunteers.",
  "published": "2019-03-08",
  "updated": "2019-03-08",
  "year": "2019",
  "authors": [
   "Justin Lin",
   "Roberto Calandra",
   "Sergey Levine"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA",
  "venue_source": "semantic-scholar",
  "citations": 68,
  "influential_citations": 1,
  "tldr": "This work is the first to address this type of multi-modal instance recognition problem on such a large-scale with its analysis spanning 98 different objects, and its learned model outperforms other methods on this complex task, including that of human volunteers.",
  "doi": "10.1109/ICRA.2019.8793885",
  "oa_pdf": "https://arxiv.org/pdf/1903.03591",
  "s2_authors": [
   {
    "name": "Justin Lin",
    "id": "2144488483",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "R. Calandra",
    "id": "35159852",
    "h_index": 40,
    "papers": 75
   },
   {
    "name": "S. Levine",
    "id": "1736651",
    "h_index": 167,
    "papers": 551
   }
  ],
  "comment": "6 pages; accepted to IEEE International Conference on Robotics and Automation 2019 (ICRA 2019)",
  "topics": [
   "tactile",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1903.03591v1",
  "pdf_url": "https://arxiv.org/pdf/1903.03591v1",
  "html_url": "https://arxiv.org/html/1903.03591v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.34
 },
 {
  "id": "1903.01973",
  "slug": "learning-latent-plans-from-play",
  "title": "Learning Latent Plans from Play",
  "abstract": "Acquiring a diverse repertoire of general-purpose skills remains an open challenge for robotics. In this work, we propose self-supervising control on top of human teleoperated play data as a way to scale up skill learning. Play has two properties that make it attractive compared to conventional task demonstrations. Play is cheap, as it can be collected in large quantities quickly without task segmenting, labeling, or resetting to an initial state. Play is naturally rich, covering ~4x more interaction space than task demonstrations for the same amount of collection time. To learn control from play, we introduce Play-LMP, a self-supervised method that learns to organize play behaviors in a latent space, then reuse them at test time to achieve specific goals. Combining self-supervised control with a diverse play dataset shifts the focus of skill learning from a narrow and discrete set of tasks to the full continuum of behaviors available in an environment. We find that this combination generalizes well empirically---after self-supervising on unlabeled play, our method substantially outperforms individual expert-trained policies on 18 difficult user-specified visual manipulation tasks in a simulated robotic tabletop environment. We additionally find that play-supervised models, unlike their expert-trained counterparts, are more robust to perturbations and exhibit retrying-till-success behaviors. Finally, we find that our agent organizes its latent plan space around functional tasks, despite never being trained with task labels. Videos, code and data are available at learning-from-play.github.io",
  "published": "2019-03-05",
  "updated": "2019-12-20",
  "year": "2019",
  "authors": [
   "Corey Lynch",
   "Mohi Khansari",
   "Ted Xiao",
   "Vikash Kumar",
   "Jonathan Tompson",
   "Sergey Levine",
   "Pierre Sermanet"
  ],
  "author_count": 7,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 522,
  "influential_citations": 55,
  "tldr": "Play-LMP is introduced, a method designed to handle variability in the LfP setting by organizing it in an embedding space and finding that play-supervised models, unlike their expert-trained counterparts, are more robust to perturbations and exhibit retrying-till-success.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Corey Lynch",
    "id": "32245472",
    "h_index": 20,
    "papers": 27
   },
   {
    "name": "Mohi Khansari",
    "id": "30559411",
    "h_index": 14,
    "papers": 22
   },
   {
    "name": "Ted Xiao",
    "id": "9961095",
    "h_index": 33,
    "papers": 45
   },
   {
    "name": "Vikash Kumar",
    "id": "2109446216",
    "h_index": 38,
    "papers": 76
   },
   {
    "name": "Jonathan Tompson",
    "id": "2704494",
    "h_index": 43,
    "papers": 71
   },
   {
    "name": "S. Levine",
    "id": "1736651",
    "h_index": 167,
    "papers": 551
   },
   {
    "name": "P. Sermanet",
    "id": "3142556",
    "h_index": 39,
    "papers": 77
   }
  ],
  "comment": "Published at CoRL 2019 (3rd Conference on Robot Learning, Osaka, Japan)",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1903.01973v2",
  "pdf_url": "https://arxiv.org/pdf/1903.01973v2",
  "html_url": "https://arxiv.org/html/1903.01973v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.22
 },
 {
  "id": "1903.00374",
  "slug": "model-based-reinforcement-learning-for-atari",
  "title": "Model-Based Reinforcement Learning for Atari",
  "abstract": "Model-free reinforcement learning (RL) can be used to learn effective policies for complex tasks, such as Atari games, even from image observations. However, this typically requires very large amounts of interaction -- substantially more, in fact, than a human would need to learn the same games. How can people learn so quickly? Part of the answer may be that people can learn how the game works and predict which actions will lead to desirable outcomes. In this paper, we explore how video prediction models can similarly enable agents to solve Atari games with fewer interactions than model-free methods. We describe Simulated Policy Learning (SimPLe), a complete model-based deep RL algorithm based on video prediction models and present a comparison of several model architectures, including a novel architecture that yields the best results in our setting. Our experiments evaluate SimPLe on a range of Atari games in low data regime of 100k interactions between the agent and the environment, which corresponds to two hours of real-time play. In most games SimPLe outperforms state-of-the-art model-free algorithms, in some games by over an order of magnitude.",
  "published": "2019-03-01",
  "updated": "2024-04-03",
  "year": "2019",
  "authors": [
   "Lukasz Kaiser",
   "Mohammad Babaeizadeh",
   "Piotr Milos",
   "Blazej Osinski",
   "Roy H Campbell",
   "Konrad Czechowski",
   "Dumitru Erhan",
   "Chelsea Finn",
   "Piotr Kozakowski",
   "Sergey Levine",
   "Afroz Mohiuddin",
   "Ryan Sepassi",
   "George Tucker",
   "Henryk Michalewski"
  ],
  "author_count": 14,
  "categories": [
   "cs.LG",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 1008,
  "influential_citations": 87,
  "tldr": "Simulated Policy Learning (SimPLe), a complete model-based deep RL algorithm based on video prediction models, is described and a comparison of several model architectures is presented, including a novel architecture that yields the best results in the authors' setting.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Lukasz Kaiser",
    "id": "40527594",
    "h_index": 31,
    "papers": 69
   },
   {
    "name": "M. Babaeizadeh",
    "id": "3365707",
    "h_index": 15,
    "papers": 24
   },
   {
    "name": "Piotr Milos",
    "id": "29900205",
    "h_index": 11,
    "papers": 33
   },
   {
    "name": "B. Osinski",
    "id": "41022287",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "R. Campbell",
    "id": "143775101",
    "h_index": 67,
    "papers": 496
   },
   {
    "name": "K. Czechowski",
    "id": "30770224",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "D. Erhan",
    "id": "1761978",
    "h_index": 37,
    "papers": 60
   },
   {
    "name": "Chelsea Finn",
    "id": "46881670",
    "h_index": 94,
    "papers": 240
   },
   {
    "name": "Piotr Kozakowski",
    "id": "49265654",
    "h_index": 4,
    "papers": 10
   },
   {
    "name": "S. Levine",
    "id": "1736651",
    "h_index": 167,
    "papers": 551
   },
   {
    "name": "Afroz Mohiuddin",
    "id": "1579862074",
    "h_index": 7,
    "papers": 16
   },
   {
    "name": "Ryan Sepassi",
    "id": "35474601",
    "h_index": 8,
    "papers": 15
   },
   {
    "name": "G. Tucker",
    "id": "145499435",
    "h_index": 34,
    "papers": 55
   },
   {
    "name": "H. Michalewski",
    "id": "47407464",
    "h_index": 25,
    "papers": 108
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "rl-control",
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1903.00374v5",
  "pdf_url": "https://arxiv.org/pdf/1903.00374v5",
  "html_url": "https://arxiv.org/html/1903.00374v5",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "1902.09305",
  "slug": "end-to-end-hand-mesh-recovery-from-a-monocular-rgb-image",
  "title": "End-to-end Hand Mesh Recovery from a Monocular RGB Image",
  "abstract": "In this paper, we present a HAnd Mesh Recovery (HAMR) framework to tackle the problem of reconstructing the full 3D mesh of a human hand from a single RGB image. In contrast to existing research on 2D or 3D hand pose estimation from RGB or/and depth image data, HAMR can provide a more expressive and useful mesh representation for monocular hand image understanding. In particular, the mesh representation is achieved by parameterizing a generic 3D hand model with shape and relative 3D joint angles. By utilizing this mesh representation, we can easily compute the 3D joint locations via linear interpolations between the vertexes of the mesh, while obtain the 2D joint locations with a projection of the 3D joints.To this end, a differentiable re-projection loss can be defined in terms of the derived representations and the ground-truth labels, thus making our framework end-to-end trainable.Qualitative experiments show that our framework is capable of recovering appealing 3D hand mesh even in the presence of severe occlusions.Quantitatively, our approach also outperforms the state-of-the-art methods for both 2D and 3D hand pose estimation from a monocular RGB image on several benchmark datasets.",
  "published": "2019-02-25",
  "updated": "2019-09-07",
  "year": "2019",
  "authors": [
   "Xiong Zhang",
   "Qiang Li",
   "Hong Mo",
   "Wenbo Zhang",
   "Wen Zheng"
  ],
  "author_count": 5,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "ICCV",
  "venue_source": "semantic-scholar",
  "citations": 259,
  "influential_citations": 31,
  "tldr": "Qualitative experiments show that the HAMR framework is capable of recovering appealing 3D hand mesh even in the presence of severe occlusions, and outperforms the state-of-the-art methods for both 2D and3D hand pose estimation from a monocular RGB image on several benchmark datasets.",
  "doi": "10.1109/ICCV.2019.00244",
  "oa_pdf": "https://arxiv.org/pdf/1902.09305",
  "s2_authors": [
   {
    "name": "Xiong Zhang",
    "id": "2108289655",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Qiang Li",
    "id": "48934185",
    "h_index": 29,
    "papers": 217
   },
   {
    "name": "Wenbo Zhang",
    "id": "2155265026",
    "h_index": 3,
    "papers": 7
   },
   {
    "name": "Wen Zheng",
    "id": "2152934280",
    "h_index": 12,
    "papers": 19
   }
  ],
  "comment": "11 pages;",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1902.09305v3",
  "pdf_url": "https://arxiv.org/pdf/1902.09305v3",
  "html_url": "https://arxiv.org/html/1902.09305v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.91
 },
 {
  "id": "1902.03451",
  "slug": "3d-hand-shape-and-pose-from-images-in-the-wild",
  "title": "3D Hand Shape and Pose from Images in the Wild",
  "abstract": "We present in this work the first end-to-end deep learning based method that predicts both 3D hand shape and pose from RGB images in the wild. Our network consists of the concatenation of a deep convolutional encoder, and a fixed model-based decoder. Given an input image, and optionally 2D joint detections obtained from an independent CNN, the encoder predicts a set of hand and view parameters. The decoder has two components: A pre-computed articulated mesh deformation hand model that generates a 3D mesh from the hand parameters, and a re-projection module controlled by the view parameters that projects the generated hand into the image domain. We show that using the shape and pose prior knowledge encoded in the hand model within a deep learning framework yields state-of-the-art performance in 3D pose prediction from images on standard benchmarks, and produces geometrically valid and plausible 3D reconstructions. Additionally, we show that training with weak supervision in the form of 2D joint annotations on datasets of images in the wild, in conjunction with full supervision in the form of 3D joint annotations on limited available datasets allows for good generalization to 3D shape and pose predictions on images in the wild.",
  "published": "2019-02-09",
  "updated": "2019-02-09",
  "year": "2019",
  "authors": [
   "Adnane Boukhayma",
   "Rodrigo de Bem",
   "Philip H. S. Torr"
  ],
  "author_count": 3,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 417,
  "influential_citations": 60,
  "tldr": "This work presents the first end-to-end deep learning based method that predicts both 3D hand shape and pose from RGB images in the wild, consisting of the concatenation of a deep convolutional encoder, and a fixed model-based decoder.",
  "doi": "10.1109/CVPR.2019.01110",
  "oa_pdf": "https://arxiv.org/pdf/1902.03451",
  "s2_authors": [
   {
    "name": "A. Boukhayma",
    "id": "2984583",
    "h_index": 13,
    "papers": 47
   },
   {
    "name": "Rodrigo de Bem",
    "id": "145582147",
    "h_index": 7,
    "papers": 22
   },
   {
    "name": "Philip H. S. Torr",
    "id": "143635540",
    "h_index": 111,
    "papers": 657
   }
  ],
  "comment": "",
  "topics": [
   "spatial-3d",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1902.03451v1",
  "pdf_url": "https://arxiv.org/pdf/1902.03451v1",
  "html_url": "https://arxiv.org/html/1902.03451v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.12
 },
 {
  "id": "1902.01240",
  "slug": "pipps-flexible-model-based-policy-search-robust-to-the-curse-of-chaos",
  "title": "PIPPS: Flexible Model-Based Policy Search Robust to the Curse of Chaos",
  "abstract": "Previously, the exploding gradient problem has been explained to be central in deep learning and model-based reinforcement learning, because it causes numerical issues and instability in optimization. Our experiments in model-based reinforcement learning imply that the problem is not just a numerical issue, but it may be caused by a fundamental chaos-like nature of long chains of nonlinear computations. Not only do the magnitudes of the gradients become large, the direction of the gradients becomes essentially random. We show that reparameterization gradients suffer from the problem, while likelihood ratio gradients are robust. Using our insights, we develop a model-based policy search framework, Probabilistic Inference for Particle-Based Policy Search (PIPPS), which is easily extensible, and allows for almost arbitrary models and policies, while simultaneously matching the performance of previous data-efficient learning algorithms. Finally, we invent the total propagation algorithm, which efficiently computes a union over all pathwise derivative depths during a single backwards pass, automatically giving greater weight to estimators with lower variance, sometimes improving over reparameterization gradients by $10^6$ times.",
  "published": "2019-02-04",
  "updated": "2019-02-04",
  "year": "2019",
  "authors": [
   "Paavo Parmas",
   "Carl Edward Rasmussen",
   "Jan Peters",
   "Kenji Doya"
  ],
  "author_count": 4,
  "categories": [
   "cs.LG",
   "cs.AI",
   "cs.RO",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "ICML",
  "venue_source": "semantic-scholar",
  "citations": 102,
  "influential_citations": 8,
  "tldr": "A model-based policy search framework, Probabilistic Inference for Particle-Based Policy Search (PIPPS), which is easily extensible, and allows for almost arbitrary models and policies, while simultaneously matching the performance of previous data-efficient learning algorithms.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Paavo Parmas",
    "id": "51135177",
    "h_index": 6,
    "papers": 19
   },
   {
    "name": "C. Rasmussen",
    "id": "3472959",
    "h_index": 50,
    "papers": 134
   },
   {
    "name": "Jan Peters",
    "id": "145197867",
    "h_index": 86,
    "papers": 611
   },
   {
    "name": "K. Doya",
    "id": "1714997",
    "h_index": 60,
    "papers": 368
   }
  ],
  "comment": "ICML 2018",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1902.01240v1",
  "pdf_url": "https://arxiv.org/pdf/1902.01240v1",
  "html_url": "https://arxiv.org/html/1902.01240v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.51
 },
 {
  "id": "1901.02705",
  "slug": "model-predictive-policy-learning-with-uncertainty-regularization-for-d",
  "title": "Model-Predictive Policy Learning with Uncertainty Regularization for Driving in Dense Traffic",
  "abstract": "Learning a policy using only observational data is challenging because the distribution of states it induces at execution time may differ from the distribution observed during training. We propose to train a policy by unrolling a learned model of the environment dynamics over multiple time steps while explicitly penalizing two costs: the original cost the policy seeks to optimize, and an uncertainty cost which represents its divergence from the states it is trained on. We measure this second cost by using the uncertainty of the dynamics model about its own predictions, using recent ideas from uncertainty estimation for deep networks. We evaluate our approach using a large-scale observational dataset of driving behavior recorded from traffic cameras, and show that we are able to learn effective driving policies from purely observational data, with no environment interaction.",
  "published": "2019-01-08",
  "updated": "2019-01-08",
  "year": "2019",
  "authors": [
   "Mikael Henaff",
   "Alfredo Canziani",
   "Yann LeCun"
  ],
  "author_count": 3,
  "categories": [
   "cs.LG",
   "cs.AI",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 129,
  "influential_citations": 10,
  "tldr": "This work proposes to train a policy by unrolling a learned model of the environment dynamics over multiple time steps while explicitly penalizing two costs: the original cost the policy seeks to optimize, and an uncertainty cost which represents its divergence from the states it is trained on.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Mikael Henaff",
    "id": "39713408",
    "h_index": 22,
    "papers": 27
   },
   {
    "name": "A. Canziani",
    "id": "2067767",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Yann LeCun",
    "id": "1688882",
    "h_index": 139,
    "papers": 406
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1901.02705v1",
  "pdf_url": "https://arxiv.org/pdf/1901.02705v1",
  "html_url": "https://arxiv.org/html/1901.02705v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.61
 },
 {
  "id": "1812.07035",
  "slug": "on-the-continuity-of-rotation-representations-in-neural-networks",
  "title": "On the Continuity of Rotation Representations in Neural Networks",
  "abstract": "In neural networks, it is often desirable to work with various representations of the same space. For example, 3D rotations can be represented with quaternions or Euler angles. In this paper, we advance a definition of a continuous representation, which can be helpful for training deep neural networks. We relate this to topological concepts such as homeomorphism and embedding. We then investigate what are continuous and discontinuous representations for 2D, 3D, and n-dimensional rotations. We demonstrate that for 3D rotations, all representations are discontinuous in the real Euclidean spaces of four or fewer dimensions. Thus, widely used representations such as quaternions and Euler angles are discontinuous and difficult for neural networks to learn. We show that the 3D rotations have continuous representations in 5D and 6D, which are more suitable for learning. We also present continuous representations for the general case of the n-dimensional rotation group SO(n). While our main focus is on rotations, we also show that our constructions apply to other groups such as the orthogonal group and similarity transforms. We finally present empirical results, which show that our continuous rotation representations outperform discontinuous ones for several practical problems in graphics and vision, including a simple autoencoder sanity test, a rotation estimator for 3D point clouds, and an inverse kinematics solver for 3D human poses.",
  "published": "2018-12-17",
  "updated": "2020-06-08",
  "year": "2018",
  "authors": [
   "Yi Zhou",
   "Connelly Barnes",
   "Jingwan Lu",
   "Jimei Yang",
   "Hao Li"
  ],
  "author_count": 5,
  "categories": [
   "cs.LG",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 1904,
  "influential_citations": 127,
  "tldr": "A definition of a continuous representation is advanced, which can be helpful for training deep neural networks and related to topological concepts such as homeomorphism and embedding, and results show that continuous rotation representations outperform discontinuous ones for several practical problems in graphics and vision.",
  "doi": "10.1109/CVPR.2019.00589",
  "oa_pdf": "https://arxiv.org/pdf/1812.07035",
  "s2_authors": [
   {
    "name": "Yi Zhou",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Connelly Barnes",
    "id": "2496412",
    "h_index": 29,
    "papers": 67
   },
   {
    "name": "Jingwan Lu",
    "id": "2054975",
    "h_index": 34,
    "papers": 65
   },
   {
    "name": "Jimei Yang",
    "id": "1768964",
    "h_index": 47,
    "papers": 75
   },
   {
    "name": "Hao Li",
    "id": "79482877",
    "h_index": 22,
    "papers": 25
   }
  ],
  "comment": "",
  "topics": [
   "spatial-3d"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1812.07035v4",
  "pdf_url": "https://arxiv.org/pdf/1812.07035v4",
  "html_url": "https://arxiv.org/html/1812.07035v4",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "1812.06110",
  "slug": "dopamine-a-research-framework-for-deep-reinforcement-learning",
  "title": "Dopamine: A Research Framework for Deep Reinforcement Learning",
  "abstract": "Deep reinforcement learning (deep RL) research has grown significantly in recent years. A number of software offerings now exist that provide stable, comprehensive implementations for benchmarking. At the same time, recent deep RL research has become more diverse in its goals. In this paper we introduce Dopamine, a new research framework for deep RL that aims to support some of that diversity. Dopamine is open-source, TensorFlow-based, and provides compact and reliable implementations of some state-of-the-art deep RL agents. We complement this offering with a taxonomy of the different research objectives in deep RL research. While by no means exhaustive, our analysis highlights the heterogeneity of research in the field, and the value of frameworks such as ours.",
  "published": "2018-12-14",
  "updated": "2018-12-14",
  "year": "2018",
  "authors": [
   "Pablo Samuel Castro",
   "Subhodeep Moitra",
   "Carles Gelada",
   "Saurabh Kumar",
   "Marc G. Bellemare"
  ],
  "author_count": 5,
  "categories": [
   "cs.LG",
   "cs.AI"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 296,
  "influential_citations": 32,
  "tldr": "Dopamine is an open-source, TensorFlow-based, and compact and reliable implementations of some state-of-the-art deep RL agents that complement this offering with a taxonomy of the different research objectives in deep RL research.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "P. S. Castro",
    "id": "39163115",
    "h_index": 29,
    "papers": 55
   },
   {
    "name": "Subhodeep Moitra",
    "id": "3316330",
    "h_index": 9,
    "papers": 15
   },
   {
    "name": "Carles Gelada",
    "id": "52382152",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "Saurabh Kumar",
    "id": "2121434953",
    "h_index": 11,
    "papers": 14
   },
   {
    "name": "Marc G. Bellemare",
    "id": "1792298",
    "h_index": 44,
    "papers": 92
   }
  ],
  "comment": "",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1812.06110v1",
  "pdf_url": "https://arxiv.org/pdf/1812.06110v1",
  "html_url": "https://arxiv.org/html/1812.06110v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.47
 },
 {
  "id": "1812.05905",
  "slug": "soft-actor-critic-algorithms-and-applications",
  "title": "Soft Actor-Critic Algorithms and Applications",
  "abstract": "Model-free deep reinforcement learning (RL) algorithms have been successfully applied to a range of challenging sequential decision making and control tasks. However, these methods typically suffer from two major challenges: high sample complexity and brittleness to hyperparameters. Both of these challenges limit the applicability of such methods to real-world domains. In this paper, we describe Soft Actor-Critic (SAC), our recently introduced off-policy actor-critic algorithm based on the maximum entropy RL framework. In this framework, the actor aims to simultaneously maximize expected return and entropy. That is, to succeed at the task while acting as randomly as possible. We extend SAC to incorporate a number of modifications that accelerate training and improve stability with respect to the hyperparameters, including a constrained formulation that automatically tunes the temperature hyperparameter. We systematically evaluate SAC on a range of benchmark tasks, as well as real-world challenging tasks such as locomotion for a quadrupedal robot and robotic manipulation with a dexterous hand. With these improvements, SAC achieves state-of-the-art performance, outperforming prior on-policy and off-policy methods in sample-efficiency and asymptotic performance. Furthermore, we demonstrate that, in contrast to other off-policy algorithms, our approach is very stable, achieving similar performance across different random seeds. These results suggest that SAC is a promising candidate for learning in real-world robotics tasks.",
  "published": "2018-12-13",
  "updated": "2019-01-29",
  "year": "2018",
  "authors": [
   "Tuomas Haarnoja",
   "Aurick Zhou",
   "Kristian Hartikainen",
   "George Tucker",
   "Sehoon Ha",
   "Jie Tan",
   "Vikash Kumar",
   "Henry Zhu",
   "Abhishek Gupta",
   "Pieter Abbeel",
   "Sergey Levine"
  ],
  "author_count": 11,
  "categories": [
   "cs.LG",
   "cs.AI",
   "cs.RO",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 3264,
  "influential_citations": 665,
  "tldr": "Soft Actor-Critic (SAC), the recently introduced off-policy actor-critic algorithm based on the maximum entropy RL framework, achieves state-of-the-art performance, outperforming prior on-policy and off- policy methods in sample-efficiency and asymptotic performance.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Tuomas Haarnoja",
    "id": "2587648",
    "h_index": 15,
    "papers": 31
   },
   {
    "name": "Aurick Zhou",
    "id": "35499972",
    "h_index": 12,
    "papers": 13
   },
   {
    "name": "Kristian Hartikainen",
    "id": "41016704",
    "h_index": 10,
    "papers": 13
   },
   {
    "name": "G. Tucker",
    "id": "145499435",
    "h_index": 34,
    "papers": 55
   },
   {
    "name": "Sehoon Ha",
    "id": "2248552",
    "h_index": 26,
    "papers": 69
   },
   {
    "name": "Jie Tan",
    "id": "1739176520",
    "h_index": 37,
    "papers": 68
   },
   {
    "name": "Vikash Kumar",
    "id": "2109446216",
    "h_index": 38,
    "papers": 76
   },
   {
    "name": "Henry Zhu",
    "id": "2117693732",
    "h_index": 5,
    "papers": 9
   },
   {
    "name": "Abhishek Gupta",
    "id": "2129458064",
    "h_index": 57,
    "papers": 156
   },
   {
    "name": "P. Abbeel",
    "id": "1689992",
    "h_index": 165,
    "papers": 596
   },
   {
    "name": "S. Levine",
    "id": "1736651",
    "h_index": 167,
    "papers": 551
   }
  ],
  "comment": "arXiv admin note: substantial text overlap with arXiv:1801.01290",
  "topics": [
   "dexterous-manipulation",
   "humanoids",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1812.05905v2",
  "pdf_url": "https://arxiv.org/pdf/1812.05905v2",
  "html_url": "https://arxiv.org/html/1812.05905v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.0
 },
 {
  "id": "1812.00568",
  "slug": "visual-foresight-model-based-deep-reinforcement-learning-for-vision-ba",
  "title": "Visual Foresight: Model-Based Deep Reinforcement Learning for Vision-Based Robotic Control",
  "abstract": "Deep reinforcement learning (RL) algorithms can learn complex robotic skills from raw sensory inputs, but have yet to achieve the kind of broad generalization and applicability demonstrated by deep learning methods in supervised domains. We present a deep RL method that is practical for real-world robotics tasks, such as robotic manipulation, and generalizes effectively to never-before-seen tasks and objects. In these settings, ground truth reward signals are typically unavailable, and we therefore propose a self-supervised model-based approach, where a predictive model learns to directly predict the future from raw sensory readings, such as camera images. At test time, we explore three distinct goal specification methods: designated pixels, where a user specifies desired object manipulation tasks by selecting particular pixels in an image and corresponding goal positions, goal images, where the desired goal state is specified with an image, and image classifiers, which define spaces of goal states. Our deep predictive models are trained using data collected autonomously and continuously by a robot interacting with hundreds of objects, without human supervision. We demonstrate that visual MPC can generalize to never-before-seen objects---both rigid and deformable---and solve a range of user-defined object manipulation tasks using the same model.",
  "published": "2018-12-03",
  "updated": "2018-12-03",
  "year": "2018",
  "authors": [
   "Frederik Ebert",
   "Chelsea Finn",
   "Sudeep Dasari",
   "Annie Xie",
   "Alex Lee",
   "Sergey Levine"
  ],
  "author_count": 6,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.CV",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "",
  "venue_source": "",
  "citations": 482,
  "influential_citations": 26,
  "tldr": "It is demonstrated that visual MPC can generalize to never-before-seen objects---both rigid and deformable---and solve a range of user-defined object manipulation tasks using the same model.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "F. Ebert",
    "id": "27535721",
    "h_index": 15,
    "papers": 22
   },
   {
    "name": "Chelsea Finn",
    "id": "46881670",
    "h_index": 94,
    "papers": 240
   },
   {
    "name": "S. Dasari",
    "id": "36076404",
    "h_index": 23,
    "papers": 39
   },
   {
    "name": "Annie Xie",
    "id": "14484808",
    "h_index": 21,
    "papers": 23
   },
   {
    "name": "Alex X. Lee",
    "id": "49250083",
    "h_index": 17,
    "papers": 24
   },
   {
    "name": "S. Levine",
    "id": "1736651",
    "h_index": 167,
    "papers": 551
   }
  ],
  "comment": "",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1812.00568v1",
  "pdf_url": "https://arxiv.org/pdf/1812.00568v1",
  "html_url": "https://arxiv.org/html/1812.00568v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.68
 },
 {
  "id": "1811.11359",
  "slug": "unsupervised-control-through-non-parametric-discriminative-rewards",
  "title": "Unsupervised Control Through Non-Parametric Discriminative Rewards",
  "abstract": "Learning to control an environment without hand-crafted rewards or expert data remains challenging and is at the frontier of reinforcement learning research. We present an unsupervised learning algorithm to train agents to achieve perceptually-specified goals using only a stream of observations and actions. Our agent simultaneously learns a goal-conditioned policy and a goal achievement reward function that measures how similar a state is to the goal state. This dual optimization leads to a co-operative game, giving rise to a learned reward function that reflects similarity in controllable aspects of the environment instead of distance in the space of observations. We demonstrate the efficacy of our agent to learn, in an unsupervised manner, to reach a diverse set of goals on three domains -- Atari, the DeepMind Control Suite and DeepMind Lab.",
  "published": "2018-11-28",
  "updated": "2018-11-28",
  "year": "2018",
  "authors": [
   "David Warde-Farley",
   "Tom Van de Wiele",
   "Tejas Kulkarni",
   "Catalin Ionescu",
   "Steven Hansen",
   "Volodymyr Mnih"
  ],
  "author_count": 6,
  "categories": [
   "cs.LG",
   "cs.AI",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 193,
  "influential_citations": 23,
  "tldr": "An unsupervised learning algorithm to train agents to achieve perceptually-specified goals using only a stream of observations and actions, which leads to a co-operative game and a learned reward function that reflects similarity in controllable aspects of the environment instead of distance in the space of observations.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "David Warde-Farley",
    "id": "1393680089",
    "h_index": 25,
    "papers": 36
   },
   {
    "name": "T. Wiele",
    "id": "8023592",
    "h_index": 35,
    "papers": 186
   },
   {
    "name": "Tejas D. Kulkarni",
    "id": "1954876",
    "h_index": 21,
    "papers": 45
   },
   {
    "name": "Catalin Ionescu",
    "id": "2273228",
    "h_index": 14,
    "papers": 29
   },
   {
    "name": "S. Hansen",
    "id": "35231584",
    "h_index": 10,
    "papers": 23
   },
   {
    "name": "Volodymyr Mnih",
    "id": "3255983",
    "h_index": 34,
    "papers": 47
   }
  ],
  "comment": "10 pages + references & 5 page appendix",
  "topics": [
   "rl-control"
  ],
  "orgs": [
   "Google DeepMind"
  ],
  "abs_url": "https://arxiv.org/abs/1811.11359v1",
  "pdf_url": "https://arxiv.org/pdf/1811.11359v1",
  "html_url": "https://arxiv.org/html/1811.11359v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.29
 },
 {
  "id": "1811.09656",
  "slug": "hierarchical-visuomotor-control-of-humanoids",
  "title": "Hierarchical visuomotor control of humanoids",
  "abstract": "We aim to build complex humanoid agents that integrate perception, motor control, and memory. In this work, we partly factor this problem into low-level motor control from proprioception and high-level coordination of the low-level skills informed by vision. We develop an architecture capable of surprisingly flexible, task-directed motor control of a relatively high-DoF humanoid body by combining pre-training of low-level motor controllers with a high-level, task-focused controller that switches among low-level sub-policies. The resulting system is able to control a physically-simulated humanoid body to solve tasks that require coupling visual perception from an unstabilized egocentric RGB camera during locomotion in the environment. For a supplementary video link, see https://youtu.be/7GISvfbykLE .",
  "published": "2018-11-23",
  "updated": "2019-01-15",
  "year": "2018",
  "authors": [
   "Josh Merel",
   "Arun Ahuja",
   "Vu Pham",
   "Saran Tunyasuvunakool",
   "Siqi Liu",
   "Dhruva Tirumala",
   "Nicolas Heess",
   "Greg Wayne"
  ],
  "author_count": 8,
  "categories": [
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.AI",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 103,
  "influential_citations": 2,
  "tldr": "An architecture capable of surprisingly flexible, task-directed motor control of a relatively high-DoF humanoid body is developed by combining pre-training of low-level motor controllers with a high-level,task-focused controller that switches among low- level sub-policies.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "J. Merel",
    "id": "1879232",
    "h_index": 28,
    "papers": 51
   },
   {
    "name": "Arun Ahuja",
    "id": "37968006",
    "h_index": 22,
    "papers": 49
   },
   {
    "name": "Vu Pham",
    "id": "40094154",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "S. Tunyasuvunakool",
    "id": "47985172",
    "h_index": 18,
    "papers": 26
   },
   {
    "name": "Siqi Liu",
    "id": "2108637290",
    "h_index": 13,
    "papers": 20
   },
   {
    "name": "Dhruva Tirumala",
    "id": "7794353",
    "h_index": 13,
    "papers": 21
   },
   {
    "name": "N. Heess",
    "id": "2801204",
    "h_index": 73,
    "papers": 192
   },
   {
    "name": "Greg Wayne",
    "id": "89504302",
    "h_index": 33,
    "papers": 47
   }
  ],
  "comment": "Accepted as a conference paper at ICLR 2019",
  "topics": [
   "humanoids",
   "egocentric-data",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1811.09656v2",
  "pdf_url": "https://arxiv.org/pdf/1811.09656v2",
  "html_url": "https://arxiv.org/html/1811.09656v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.52
 },
 {
  "id": "1811.06407",
  "slug": "neural-predictive-belief-representations",
  "title": "Neural Predictive Belief Representations",
  "abstract": "Unsupervised representation learning has succeeded with excellent results in many applications. It is an especially powerful tool to learn a good representation of environments with partial or noisy observations. In partially observable domains it is important for the representation to encode a belief state, a sufficient statistic of the observations seen so far. In this paper, we investigate whether it is possible to learn such a belief representation using modern neural architectures. Specifically, we focus on one-step frame prediction and two variants of contrastive predictive coding (CPC) as the objective functions to learn the representations. To evaluate these learned representations, we test how well they can predict various pieces of information about the underlying state of the environment, e.g., position of the agent in a 3D maze. We show that all three methods are able to learn belief representations of the environment, they encode not only the state information, but also its uncertainty, a crucial aspect of belief states. We also find that for CPC multi-step predictions and action-conditioning are critical for accurate belief representations in visually complex environments. The ability of neural representations to capture the belief information has the potential to spur new advances for learning and planning in partially observable domains, where leveraging uncertainty is essential for optimal decision making.",
  "published": "2018-11-15",
  "updated": "2019-08-19",
  "year": "2018",
  "authors": [
   "Zhaohan Daniel Guo",
   "Mohammad Gheshlaghi Azar",
   "Bilal Piot",
   "Bernardo A. Pires",
   "R\u00e9mi Munos"
  ],
  "author_count": 5,
  "categories": [
   "cs.LG",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 83,
  "influential_citations": 7,
  "tldr": "It is shown that for CPC multi-step predictions and action-conditioning are critical for accurate belief representations in visually complex environments, and all three methods are able to learn belief representations of the environment.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Z. Guo",
    "id": "3407143",
    "h_index": 13,
    "papers": 20
   },
   {
    "name": "M. G. Azar",
    "id": "37666967",
    "h_index": 32,
    "papers": 52
   },
   {
    "name": "Bilal Piot",
    "id": "1808897",
    "h_index": 44,
    "papers": 80
   },
   {
    "name": "B. '. Pires",
    "id": "3429927",
    "h_index": 18,
    "papers": 32
   },
   {
    "name": "Tobias Pohlen",
    "id": "3408089",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "R. Munos",
    "id": "1708654",
    "h_index": 90,
    "papers": 246
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1811.06407v2",
  "pdf_url": "https://arxiv.org/pdf/1811.06407v2",
  "html_url": "https://arxiv.org/html/1811.06407v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.92
 },
 {
  "id": "1811.04551",
  "slug": "learning-latent-dynamics-for-planning-from-pixels",
  "title": "Learning Latent Dynamics for Planning from Pixels",
  "abstract": "Planning has been very successful for control tasks with known environment dynamics. To leverage planning in unknown environments, the agent needs to learn the dynamics from interactions with the world. However, learning dynamics models that are accurate enough for planning has been a long-standing challenge, especially in image-based domains. We propose the Deep Planning Network (PlaNet), a purely model-based agent that learns the environment dynamics from images and chooses actions through fast online planning in latent space. To achieve high performance, the dynamics model must accurately predict the rewards ahead for multiple time steps. We approach this using a latent dynamics model with both deterministic and stochastic transition components. Moreover, we propose a multi-step variational inference objective that we name latent overshooting. Using only pixel observations, our agent solves continuous control tasks with contact dynamics, partial observability, and sparse rewards, which exceed the difficulty of tasks that were previously solved by planning with learned models. PlaNet uses substantially fewer episodes and reaches final performance close to and sometimes higher than strong model-free algorithms.",
  "published": "2018-11-12",
  "updated": "2019-06-04",
  "year": "2018",
  "authors": [
   "Danijar Hafner",
   "Timothy Lillicrap",
   "Ian Fischer",
   "Ruben Villegas",
   "David Ha",
   "Honglak Lee",
   "James Davidson"
  ],
  "author_count": 7,
  "categories": [
   "cs.LG",
   "cs.AI",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "ICML",
  "venue_source": "semantic-scholar",
  "citations": 2029,
  "influential_citations": 253,
  "tldr": "The Deep Planning Network (PlaNet) is proposed, a purely model-based agent that learns the environment dynamics from images and chooses actions through fast online planning in latent space using a latent dynamics model with both deterministic and stochastic transition components.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Danijar Hafner",
    "id": "35006479",
    "h_index": 25,
    "papers": 47
   },
   {
    "name": "T. Lillicrap",
    "id": "2542999",
    "h_index": 69,
    "papers": 154
   },
   {
    "name": "Ian S. Fischer",
    "id": "33091759",
    "h_index": 18,
    "papers": 25
   },
   {
    "name": "Ruben Villegas",
    "id": "144543406",
    "h_index": 18,
    "papers": 25
   },
   {
    "name": "David R Ha",
    "id": "39810222",
    "h_index": 20,
    "papers": 92
   },
   {
    "name": "Honglak Lee",
    "id": "1697141",
    "h_index": 88,
    "papers": 188
   },
   {
    "name": "James Davidson",
    "id": "2068894907",
    "h_index": 13,
    "papers": 20
   }
  ],
  "comment": "20 pages, 12 figures, 1 table",
  "topics": [
   "world-models"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1811.04551v5",
  "pdf_url": "https://arxiv.org/pdf/1811.04551v5",
  "html_url": "https://arxiv.org/html/1811.04551v5",
  "code_url": null,
  "presented": [
   {
    "club": "saturday-robotics",
    "club_name": "Saturday Robotics",
    "session_number": 7,
    "session_title": "Robotics & World Models Reading Club 07: Learning to Dream: World Models, Imagination, Path to Foundation Models for Control \u2014 Los Altos",
    "date_text": "Saturday, May 9, 2026 | 2:00 PM \u2013 5:00 PM",
    "city": "",
    "url": "https://lu.ma/srhe0vuo",
    "listed_as": "PlaNet (2018)"
   }
  ],
  "club_note": "First strong demonstration of planning directly in latent space (RSSM)",
  "featured": true,
  "signal": 7.5
 },
 {
  "id": "1811.04251",
  "slug": "formal-limitations-on-the-measurement-of-mutual-information",
  "title": "Formal Limitations on the Measurement of Mutual Information",
  "abstract": "Measuring mutual information from finite data is difficult. Recent work has considered variational methods maximizing a lower bound. In this paper, we prove that serious statistical limitations are inherent to any method of measuring mutual information. More specifically, we show that any distribution-free high-confidence lower bound on mutual information estimated from N samples cannot be larger than O(ln N ).",
  "published": "2018-11-10",
  "updated": "2020-05-20",
  "year": "2018",
  "authors": [
   "David McAllester",
   "Karl Stratos"
  ],
  "author_count": 2,
  "categories": [
   "cs.IT",
   "cs.LG",
   "stat.ML"
  ],
  "primary_category": "cs.IT",
  "venue": "",
  "venue_source": "",
  "citations": 343,
  "influential_citations": 35,
  "tldr": "It is proved that any distribution-free high-confidence lower bound on mutual information estimated from N samples cannot be larger than O(ln N ).",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "D. McAllester",
    "id": "46948352",
    "h_index": 12,
    "papers": 97
   },
   {
    "name": "K. Stratos",
    "id": "1714215",
    "h_index": 25,
    "papers": 89
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1811.04251v4",
  "pdf_url": "https://arxiv.org/pdf/1811.04251v4",
  "html_url": "https://arxiv.org/html/1811.04251v4",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.54
 },
 {
  "id": "1811.02790",
  "slug": "roboturk-a-crowdsourcing-platform-for-robotic-skill-learning-through-i",
  "title": "RoboTurk: A Crowdsourcing Platform for Robotic Skill Learning through Imitation",
  "abstract": "Imitation Learning has empowered recent advances in learning robotic manipulation tasks by addressing shortcomings of Reinforcement Learning such as exploration and reward specification. However, research in this area has been limited to modest-sized datasets due to the difficulty of collecting large quantities of task demonstrations through existing mechanisms. This work introduces RoboTurk to address this challenge. RoboTurk is a crowdsourcing platform for high quality 6-DoF trajectory based teleoperation through the use of widely available mobile devices (e.g. iPhone). We evaluate RoboTurk on three manipulation tasks of varying timescales (15-120s) and observe that our user interface is statistically similar to special purpose hardware such as virtual reality controllers in terms of task completion times. Furthermore, we observe that poor network conditions, such as low bandwidth and high delay links, do not substantially affect the remote users' ability to perform task demonstrations successfully on RoboTurk. Lastly, we demonstrate the efficacy of RoboTurk through the collection of a pilot dataset; using RoboTurk, we collected 137.5 hours of manipulation data from remote workers, amounting to over 2200 successful task demonstrations in 22 hours of total system usage. We show that the data obtained through RoboTurk enables policy learning on multi-step manipulation tasks with sparse rewards and that using larger quantities of demonstrations during policy learning provides benefits in terms of both learning consistency and final performance. For additional results, videos, and to download our pilot dataset, visit $\\href{http://roboturk.stanford.edu/}{\\texttt{roboturk.stanford.edu}}$",
  "published": "2018-11-07",
  "updated": "2018-11-07",
  "year": "2018",
  "authors": [
   "Ajay Mandlekar",
   "Yuke Zhu",
   "Animesh Garg",
   "Jonathan Booher",
   "Max Spero",
   "Albert Tung",
   "Julian Gao",
   "John Emmons",
   "Anchit Gupta",
   "Emre Orbay",
   "Silvio Savarese",
   "Li Fei-Fei"
  ],
  "author_count": 12,
  "categories": [
   "cs.RO",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "cs.RO",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 391,
  "influential_citations": 17,
  "tldr": "It is shown that the data obtained through RoboTurk enables policy learning on multi-step manipulation tasks with sparse rewards and that using larger quantities of demonstrations during policy learning provides benefits in terms of both learning consistency and final performance.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "A. Mandlekar",
    "id": "49686756",
    "h_index": 36,
    "papers": 67
   },
   {
    "name": "Yuke Zhu",
    "id": "2117748",
    "h_index": 57,
    "papers": 130
   },
   {
    "name": "Animesh Garg",
    "id": "1873736",
    "h_index": 60,
    "papers": 163
   },
   {
    "name": "Jonathan Booher",
    "id": "51455502",
    "h_index": 4,
    "papers": 11
   },
   {
    "name": "Max Spero",
    "id": "81096394",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "Albert Tung",
    "id": "2064365382",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "Julian Gao",
    "id": "27545851",
    "h_index": 2,
    "papers": 4
   },
   {
    "name": "John Emmons",
    "id": "39465576",
    "h_index": 9,
    "papers": 17
   },
   {
    "name": "Anchit Gupta",
    "id": "3377939",
    "h_index": 11,
    "papers": 22
   },
   {
    "name": "Emre Orbay",
    "id": "40912158",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "S. Savarese",
    "id": "1702137",
    "h_index": 115,
    "papers": 346
   },
   {
    "name": "Li Fei-Fei",
    "id": "48004138",
    "h_index": 143,
    "papers": 606
   }
  ],
  "comment": "Published at the Conference on Robot Learning (CoRL) 2018",
  "topics": [
   "imitation-diffusion",
   "rl-control",
   "navigation",
   "data-teleop"
  ],
  "orgs": [
   "Stanford"
  ],
  "abs_url": "https://arxiv.org/abs/1811.02790v1",
  "pdf_url": "https://arxiv.org/pdf/1811.02790v1",
  "html_url": "https://arxiv.org/html/1811.02790v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.59
 },
 {
  "id": "1811.01848",
  "slug": "plan-online-learn-offline-efficient-learning-and-exploration-via-model",
  "title": "Plan Online, Learn Offline: Efficient Learning and Exploration via Model-Based Control",
  "abstract": "We propose a plan online and learn offline (POLO) framework for the setting where an agent, with an internal model, needs to continually act and learn in the world. Our work builds on the synergistic relationship between local model-based control, global value function learning, and exploration. We study how local trajectory optimization can cope with approximation errors in the value function, and can stabilize and accelerate value function learning. Conversely, we also study how approximate value functions can help reduce the planning horizon and allow for better policies beyond local solutions. Finally, we also demonstrate how trajectory optimization can be used to perform temporally coordinated exploration in conjunction with estimating uncertainty in value function approximation. This exploration is critical for fast and stable learning of the value function. Combining these components enable solutions to complex simulated control tasks, like humanoid locomotion and dexterous in-hand manipulation, in the equivalent of a few minutes of experience in the real world.",
  "published": "2018-11-05",
  "updated": "2019-01-28",
  "year": "2018",
  "authors": [
   "Kendall Lowrey",
   "Aravind Rajeswaran",
   "Sham Kakade",
   "Emanuel Todorov",
   "Igor Mordatch"
  ],
  "author_count": 5,
  "categories": [
   "cs.LG",
   "cs.AI",
   "cs.RO",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 271,
  "influential_citations": 18,
  "tldr": "A plan online and learn offline (POLO) framework for the setting where an agent, with an internal model, needs to continually act and learn in the world and how trajectory optimization can be used to perform temporally coordinated exploration in conjunction with estimating uncertainty in value function approximation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Kendall Lowrey",
    "id": "33557393",
    "h_index": 11,
    "papers": 15
   },
   {
    "name": "A. Rajeswaran",
    "id": "19275599",
    "h_index": 34,
    "papers": 58
   },
   {
    "name": "S. Kakade",
    "id": "144695232",
    "h_index": 97,
    "papers": 314
   },
   {
    "name": "E. Todorov",
    "id": "144832491",
    "h_index": 57,
    "papers": 153
   },
   {
    "name": "Igor Mordatch",
    "id": "2316241382",
    "h_index": 7,
    "papers": 7
   }
  ],
  "comment": "The first two authors contributed equally. Accepted at ICLR 2019. Supplementary videos available at: https://sites.google.com/view/polo-mpc",
  "topics": [
   "dexterous-manipulation",
   "humanoids",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1811.01848v3",
  "pdf_url": "https://arxiv.org/pdf/1811.01848v3",
  "html_url": "https://arxiv.org/html/1811.01848v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.93
 },
 {
  "id": "1810.13381",
  "slug": "maintaining-grasps-within-slipping-bound-by-monitoring-incipient-slip",
  "title": "Maintaining Grasps within Slipping Bound by Monitoring Incipient Slip",
  "abstract": "In this paper, we propose an approach to detect incipient slip, i.e. predict slip, by using a high-resolution vision-based tactile sensor, GelSlim. The sensor dynamically captures the tactile imprints of the contact object and their changes with a soft gel pad. The method assumes the object is mostly rigid and treats the motion of object's imprint on sensor surface as a 2D rigid-body motion. We use the deviation of the true motion field from that of a 2D planar rigid transformation as a measure of slip. The output is a dense slip field which we use to detect when small areas of the contact patch start to slip (incipient slip). The method can detect both translational and rotational incipient slip without any prior knowledge of the object at 24 Hz. We test the method on 10 objects 240 times and achieve 86.25% detection accuracy. We further show how the slip feedback can be used to monitor the gripping force to avoid slip with a closed-loop bottle-cap screwing and unscrewing experiment with incipient slip detection feedback. The method was demonstrated to be useful for the robot to apply proper gripping force and stop screwing at the right point before breaking objects. The method can be applied to many manipulation tasks in both structured and unstructured environments.",
  "published": "2018-10-31",
  "updated": "2018-10-31",
  "year": "2018",
  "authors": [
   "Siyuan Dong",
   "Daolin Ma",
   "Elliott Donlon",
   "Alberto Rodriguez"
  ],
  "author_count": 4,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA",
  "venue_source": "semantic-scholar",
  "citations": 113,
  "influential_citations": 12,
  "tldr": "An approach to detect incipient slip, i.e. predict slip, by using a high-resolution vision-based tactile sensor, GelSlim, which dynamically captures the tactile imprints of the grasped object and their changes with a soft gel pad is proposed.",
  "doi": "10.1109/ICRA.2019.8793538",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Siyuan Dong",
    "id": "3308345",
    "h_index": 28,
    "papers": 42
   },
   {
    "name": "Daolin Ma",
    "id": "145572541",
    "h_index": 14,
    "papers": 25
   },
   {
    "name": "E. Donlon",
    "id": "39425333",
    "h_index": 7,
    "papers": 9
   },
   {
    "name": "Alberto Rodriguez",
    "id": "152532021",
    "h_index": 49,
    "papers": 113
   }
  ],
  "comment": "",
  "topics": [
   "dexterous-manipulation",
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1810.13381v1",
  "pdf_url": "https://arxiv.org/pdf/1810.13381v1",
  "html_url": "https://arxiv.org/html/1810.13381v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.56
 },
 {
  "id": "1810.12894",
  "slug": "exploration-by-random-network-distillation",
  "title": "Exploration by Random Network Distillation",
  "abstract": "We introduce an exploration bonus for deep reinforcement learning methods that is easy to implement and adds minimal overhead to the computation performed. The bonus is the error of a neural network predicting features of the observations given by a fixed randomly initialized neural network. We also introduce a method to flexibly combine intrinsic and extrinsic rewards. We find that the random network distillation (RND) bonus combined with this increased flexibility enables significant progress on several hard exploration Atari games. In particular we establish state of the art performance on Montezuma's Revenge, a game famously difficult for deep reinforcement learning methods. To the best of our knowledge, this is the first method that achieves better than average human performance on this game without using demonstrations or having access to the underlying state of the game, and occasionally completes the first level.",
  "published": "2018-10-30",
  "updated": "2018-10-30",
  "year": "2018",
  "authors": [
   "Yuri Burda",
   "Harrison Edwards",
   "Amos Storkey",
   "Oleg Klimov"
  ],
  "author_count": 4,
  "categories": [
   "cs.LG",
   "cs.AI",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 1738,
  "influential_citations": 399,
  "tldr": "An exploration bonus for deep reinforcement learning methods that is easy to implement and adds minimal overhead to the computation performed and a method to flexibly combine intrinsic and extrinsic rewards that enables significant progress on several hard exploration Atari games is introduced.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yuri Burda",
    "id": "3080409",
    "h_index": 10,
    "papers": 25
   },
   {
    "name": "Harrison Edwards",
    "id": "144632352",
    "h_index": 13,
    "papers": 17
   },
   {
    "name": "A. Storkey",
    "id": "1728216",
    "h_index": 48,
    "papers": 290
   },
   {
    "name": "Oleg Klimov",
    "id": "2067138712",
    "h_index": 6,
    "papers": 6
   }
  ],
  "comment": "",
  "topics": [
   "rl-control",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1810.12894v1",
  "pdf_url": "https://arxiv.org/pdf/1810.12894v1",
  "html_url": "https://arxiv.org/html/1810.12894v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "1810.06187",
  "slug": "robust-learning-of-tactile-force-estimation-through-robot-interaction",
  "title": "Robust Learning of Tactile Force Estimation through Robot Interaction",
  "abstract": "Current methods for estimating force from tactile sensor signals are either inaccurate analytic models or task-specific learned models. In this paper, we explore learning a robust model that maps tactile sensor signals to force. We specifically explore learning a mapping for the SynTouch BioTac sensor via neural networks. We propose a voxelized input feature layer for spatial signals and leverage information about the sensor surface to regularize the loss function. To learn a robust tactile force model that transfers across tasks, we generate ground truth data from three different sources: (1) the BioTac rigidly mounted to a force torque~(FT) sensor, (2) a robot interacting with a ball rigidly attached to the same FT sensor, and (3) through force inference on a planar pushing task by formalizing the mechanics as a system of particles and optimizing over the object motion. A total of 140k samples were collected from the three sources. We achieve a median angular accuracy of 3.5 degrees in predicting force direction (66% improvement over the current state of the art) and a median magnitude accuracy of 0.06 N (93% improvement) on a test dataset. Additionally, we evaluate the learned force model in a force feedback grasp controller performing object lifting and gentle placement. Our results can be found on https://sites.google.com/view/tactile-force.",
  "published": "2018-10-15",
  "updated": "2019-03-05",
  "year": "2018",
  "authors": [
   "Balakumar Sundaralingam",
   "Alexander Lambert",
   "Ankur Handa",
   "Byron Boots",
   "Tucker Hermans",
   "Stan Birchfield",
   "Nathan Ratliff",
   "Dieter Fox"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA",
  "venue_source": "semantic-scholar",
  "citations": 67,
  "influential_citations": 2,
  "tldr": "This paper explores learning a robust model that maps tactile sensor signals to force via neural networks for the SynTouch BioTac sensor and proposes a voxelized input feature layer for spatial signals and leverage information about the sensor surface to regularize the loss function.",
  "doi": "10.1109/ICRA.2019.8793502",
  "oa_pdf": "https://arxiv.org/pdf/1810.06187",
  "s2_authors": [
   {
    "name": "Balakumar Sundaralingam",
    "id": "32469503",
    "h_index": 26,
    "papers": 47
   },
   {
    "name": "Alexander Lambert",
    "id": "2052569995",
    "h_index": 10,
    "papers": 14
   },
   {
    "name": "Ankur Handa",
    "id": "34653454",
    "h_index": 33,
    "papers": 55
   },
   {
    "name": "Byron Boots",
    "id": "3288815",
    "h_index": 49,
    "papers": 182
   },
   {
    "name": "Tucker Hermans",
    "id": "145767346",
    "h_index": 34,
    "papers": 107
   },
   {
    "name": "Stan Birchfield",
    "id": "2238841",
    "h_index": 52,
    "papers": 152
   },
   {
    "name": "Nathan D. Ratliff",
    "id": "13693897",
    "h_index": 34,
    "papers": 79
   },
   {
    "name": "D. Fox",
    "id": "145197953",
    "h_index": 133,
    "papers": 428
   }
  ],
  "comment": "accepted to ICRA 2019 (camera ready version)",
  "topics": [
   "dexterous-manipulation",
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1810.06187v4",
  "pdf_url": "https://arxiv.org/pdf/1810.06187v4",
  "html_url": "https://arxiv.org/html/1810.06187v4",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.33
 },
 {
  "id": "1810.01257",
  "slug": "near-optimal-representation-learning-for-hierarchical-reinforcement-le",
  "title": "Near-Optimal Representation Learning for Hierarchical Reinforcement Learning",
  "abstract": "We study the problem of representation learning in goal-conditioned hierarchical reinforcement learning. In such hierarchical structures, a higher-level controller solves tasks by iteratively communicating goals which a lower-level policy is trained to reach. Accordingly, the choice of representation -- the mapping of observation space to goal space -- is crucial. To study this problem, we develop a notion of sub-optimality of a representation, defined in terms of expected reward of the optimal hierarchical policy using this representation. We derive expressions which bound the sub-optimality and show how these expressions can be translated to representation learning objectives which may be optimized in practice. Results on a number of difficult continuous-control tasks show that our approach to representation learning yields qualitatively better representations as well as quantitatively better hierarchical policies, compared to existing methods (see videos at https://sites.google.com/view/representation-hrl).",
  "published": "2018-10-02",
  "updated": "2019-01-09",
  "year": "2018",
  "authors": [
   "Ofir Nachum",
   "Shixiang Gu",
   "Honglak Lee",
   "Sergey Levine"
  ],
  "author_count": 4,
  "categories": [
   "cs.AI"
  ],
  "primary_category": "cs.AI",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 235,
  "influential_citations": 30,
  "tldr": "Results on a number of difficult continuous-control tasks show that the developed notion of sub-optimality of a representation, defined in terms of expected reward of the optimal hierarchical policy using this representation, yields qualitatively better representations as well as quantitatively better hierarchical policies compared to existing methods.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ofir Nachum",
    "id": "7624658",
    "h_index": 47,
    "papers": 92
   },
   {
    "name": "S. Gu",
    "id": "2046135",
    "h_index": 43,
    "papers": 71
   },
   {
    "name": "Honglak Lee",
    "id": "1697141",
    "h_index": 88,
    "papers": 188
   },
   {
    "name": "S. Levine",
    "id": "1736651",
    "h_index": 167,
    "papers": 551
   }
  ],
  "comment": "ICLR 2019 Conference Paper",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1810.01257v2",
  "pdf_url": "https://arxiv.org/pdf/1810.01257v2",
  "html_url": "https://arxiv.org/html/1810.01257v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.87
 },
 {
  "id": "1810.00597",
  "slug": "taming-vaes",
  "title": "Taming VAEs",
  "abstract": "In spite of remarkable progress in deep latent variable generative modeling, training still remains a challenge due to a combination of optimization and generalization issues. In practice, a combination of heuristic algorithms (such as hand-crafted annealing of KL-terms) is often used in order to achieve the desired results, but such solutions are not robust to changes in model architecture or dataset. The best settings can often vary dramatically from one problem to another, which requires doing expensive parameter sweeps for each new case. Here we develop on the idea of training VAEs with additional constraints as a way to control their behaviour. We first present a detailed theoretical analysis of constrained VAEs, expanding our understanding of how these models work. We then introduce and analyze a practical algorithm termed Generalized ELBO with Constrained Optimization, GECO. The main advantage of GECO for the machine learning practitioner is a more intuitive, yet principled, process of tuning the loss. This involves defining of a set of constraints, which typically have an explicit relation to the desired model performance, in contrast to tweaking abstract hyper-parameters which implicitly affect the model behavior. Encouraging experimental results in several standard datasets indicate that GECO is a very robust and effective tool to balance reconstruction and compression constraints.",
  "published": "2018-10-01",
  "updated": "2018-10-01",
  "year": "2018",
  "authors": [
   "Danilo Jimenez Rezende",
   "Fabio Viola"
  ],
  "author_count": 2,
  "categories": [
   "stat.ML",
   "cs.LG"
  ],
  "primary_category": "stat.ML",
  "venue": "",
  "venue_source": "",
  "citations": 201,
  "influential_citations": 27,
  "tldr": "A detailed theoretical analysis of constrained VAEs is presented, expanding the understanding of how these models work, and a practical algorithm termed Generalized ELBO with Constrained Optimization, GECO is introduced, which is a very robust and effective tool to balance reconstruction and compression constraints.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Danilo Jimenez Rezende",
    "id": "1748523",
    "h_index": 47,
    "papers": 82
   },
   {
    "name": "Fabio Viola",
    "id": "47963165",
    "h_index": 19,
    "papers": 32
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1810.00597v1",
  "pdf_url": "https://arxiv.org/pdf/1810.00597v1",
  "html_url": "https://arxiv.org/html/1810.00597v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.31
 },
 {
  "id": "1809.04474",
  "slug": "multi-task-deep-reinforcement-learning-with-popart",
  "title": "Multi-task Deep Reinforcement Learning with PopArt",
  "abstract": "The reinforcement learning community has made great strides in designing algorithms capable of exceeding human performance on specific tasks. These algorithms are mostly trained one task at the time, each new task requiring to train a brand new agent instance. This means the learning algorithm is general, but each solution is not; each agent can only solve the one task it was trained on. In this work, we study the problem of learning to master not one but multiple sequential-decision tasks at once. A general issue in multi-task learning is that a balance must be found between the needs of multiple tasks competing for the limited resources of a single learning system. Many learning algorithms can get distracted by certain tasks in the set of tasks to solve. Such tasks appear more salient to the learning process, for instance because of the density or magnitude of the in-task rewards. This causes the algorithm to focus on those salient tasks at the expense of generality. We propose to automatically adapt the contribution of each task to the agent's updates, so that all tasks have a similar impact on the learning dynamics. This resulted in state of the art performance on learning to play all games in a set of 57 diverse Atari games. Excitingly, our method learned a single trained policy - with a single set of weights - that exceeds median human performance. To our knowledge, this was the first time a single agent surpassed human-level performance on this multi-task domain. The same approach also demonstrated state of the art performance on a set of 30 tasks in the 3D reinforcement learning platform DeepMind Lab.",
  "published": "2018-09-12",
  "updated": "2018-09-12",
  "year": "2018",
  "authors": [
   "Matteo Hessel",
   "Hubert Soyer",
   "Lasse Espeholt",
   "Wojciech Czarnecki",
   "Simon Schmitt",
   "Hado van Hasselt"
  ],
  "author_count": 6,
  "categories": [
   "cs.LG",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 365,
  "influential_citations": 29,
  "tldr": "This work proposes to automatically adapt the contribution of each task to the agent\u2019s updates, so that all tasks have a similar impact on the learning dynamics, and learns a single trained policy that exceeds median human performance on this multi-task domain.",
  "doi": "10.1609/AAAI.V33I01.33013796",
  "oa_pdf": "https://doi.org/10.1609/aaai.v33i01.33013796",
  "s2_authors": [
   {
    "name": "Matteo Hessel",
    "id": "39357484",
    "h_index": 25,
    "papers": 39
   },
   {
    "name": "Hubert Soyer",
    "id": "2794457",
    "h_index": 15,
    "papers": 27
   },
   {
    "name": "L. Espeholt",
    "id": "2311318",
    "h_index": 12,
    "papers": 21
   },
   {
    "name": "Wojciech M. Czarnecki",
    "id": "144792148",
    "h_index": 35,
    "papers": 78
   },
   {
    "name": "Simon Schmitt",
    "id": "152380508",
    "h_index": 8,
    "papers": 12
   },
   {
    "name": "H. V. Hasselt",
    "id": "7634925",
    "h_index": 40,
    "papers": 67
   }
  ],
  "comment": "",
  "topics": [
   "rl-control"
  ],
  "orgs": [
   "Google DeepMind"
  ],
  "abs_url": "https://arxiv.org/abs/1809.04474v1",
  "pdf_url": "https://arxiv.org/pdf/1809.04474v1",
  "html_url": "https://arxiv.org/html/1809.04474v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.06
 },
 {
  "id": "1809.01999",
  "slug": "recurrent-world-models-facilitate-policy-evolution",
  "title": "Recurrent World Models Facilitate Policy Evolution",
  "abstract": "A generative recurrent neural network is quickly trained in an unsupervised manner to model popular reinforcement learning environments through compressed spatio-temporal representations. The world model's extracted features are fed into compact and simple policies trained by evolution, achieving state of the art results in various environments. We also train our agent entirely inside of an environment generated by its own internal world model, and transfer this policy back into the actual environment. Interactive version of paper at https://worldmodels.github.io",
  "published": "2018-09-04",
  "updated": "2018-09-04",
  "year": "2018",
  "authors": [
   "David Ha",
   "J\u00fcrgen Schmidhuber"
  ],
  "author_count": 2,
  "categories": [
   "cs.LG",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 1395,
  "influential_citations": 103,
  "tldr": "A generative recurrent neural network is quickly trained in an unsupervised manner to model popular reinforcement learning environments through compressed spatio-temporal representations.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "David R Ha",
    "id": "39810222",
    "h_index": 20,
    "papers": 92
   },
   {
    "name": "J. Schmidhuber",
    "id": "145341374",
    "h_index": 101,
    "papers": 467
   }
  ],
  "comment": "To appear at NIPS 2018, selected for an oral presentation. arXiv admin note: substantial text overlap with arXiv:1803.10122",
  "topics": [
   "world-models",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1809.01999v1",
  "pdf_url": "https://arxiv.org/pdf/1809.01999v1",
  "html_url": "https://arxiv.org/html/1809.01999v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "1808.00177",
  "slug": "learning-dexterous-in-hand-manipulation",
  "title": "Learning Dexterous In-Hand Manipulation",
  "abstract": "We use reinforcement learning (RL) to learn dexterous in-hand manipulation policies which can perform vision-based object reorientation on a physical Shadow Dexterous Hand. The training is performed in a simulated environment in which we randomize many of the physical properties of the system like friction coefficients and an object's appearance. Our policies transfer to the physical robot despite being trained entirely in simulation. Our method does not rely on any human demonstrations, but many behaviors found in human manipulation emerge naturally, including finger gaiting, multi-finger coordination, and the controlled use of gravity. Our results were obtained using the same distributed RL system that was used to train OpenAI Five. We also include a video of our results: https://youtu.be/jwSbzNHGflM",
  "published": "2018-08-01",
  "updated": "2019-01-18",
  "year": "2018",
  "authors": [
   " OpenAI",
   "Marcin Andrychowicz",
   "Bowen Baker",
   "Maciek Chociej",
   "Rafal Jozefowicz",
   "Bob McGrew",
   "Jakub Pachocki",
   "Arthur Petron",
   "Matthias Plappert",
   "Glenn Powell",
   "Alex Ray",
   "Jonas Schneider",
   "Szymon Sidor",
   "Josh Tobin",
   "Peter Welinder",
   "Lilian Weng",
   "Wojciech Zaremba"
  ],
  "author_count": 17,
  "categories": [
   "cs.LG",
   "cs.AI",
   "cs.RO",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "Int. J. Robotics Res.",
  "venue_source": "semantic-scholar",
  "citations": 2263,
  "influential_citations": 104,
  "tldr": "This work uses reinforcement learning (RL) to learn dexterous in-hand manipulation policies that can perform vision-based object reorientation on a physical Shadow Dexterous Hand, and these policies transfer to the physical robot despite being trained entirely in simulation.",
  "doi": "10.1177/0278364919887447",
  "oa_pdf": "https://journals.sagepub.com/doi/pdf/10.1177/0278364919887447",
  "s2_authors": [
   {
    "name": "Marcin Andrychowicz",
    "id": "2206490",
    "h_index": 24,
    "papers": 35
   },
   {
    "name": "Bowen Baker",
    "id": "40566201",
    "h_index": 12,
    "papers": 18
   },
   {
    "name": "Maciek Chociej",
    "id": "36045639",
    "h_index": 5,
    "papers": 5
   },
   {
    "name": "R. J\u00f3zefowicz",
    "id": "1944541",
    "h_index": 14,
    "papers": 29
   },
   {
    "name": "Bob McGrew",
    "id": "39593364",
    "h_index": 14,
    "papers": 30
   },
   {
    "name": "J. Pachocki",
    "id": "2713380",
    "h_index": 25,
    "papers": 54
   },
   {
    "name": "Arthur Petron",
    "id": "6817951",
    "h_index": 5,
    "papers": 11
   },
   {
    "name": "Matthias Plappert",
    "id": "3407285",
    "h_index": 15,
    "papers": 44
   },
   {
    "name": "Glenn Powell",
    "id": "2059171221",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Alex Ray",
    "id": "2064770039",
    "h_index": 7,
    "papers": 8
   },
   {
    "name": "Jonas Schneider",
    "id": "2113526509",
    "h_index": 11,
    "papers": 21
   },
   {
    "name": "Szymon Sidor",
    "id": "2700360",
    "h_index": 14,
    "papers": 54
   },
   {
    "name": "Joshua Tobin",
    "id": "2052880384",
    "h_index": 8,
    "papers": 11
   },
   {
    "name": "Peter Welinder",
    "id": "2930640",
    "h_index": 17,
    "papers": 37
   },
   {
    "name": "Lilian Weng",
    "id": "2065741038",
    "h_index": 9,
    "papers": 13
   },
   {
    "name": "Wojciech Zaremba",
    "id": "2563432",
    "h_index": 31,
    "papers": 40
   }
  ],
  "comment": "Making OpenAI the first author. We wish this paper to be cited as \"Learning Dexterous In-Hand Manipulation\" by OpenAI et al. We are replicating the approach from the physics community: arXiv:1812.06489",
  "topics": [
   "dexterous-manipulation",
   "egocentric-data",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1808.00177v5",
  "pdf_url": "https://arxiv.org/pdf/1808.00177v5",
  "html_url": "https://arxiv.org/html/1808.00177v5",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "1807.10299",
  "slug": "variational-option-discovery-algorithms",
  "title": "Variational Option Discovery Algorithms",
  "abstract": "We explore methods for option discovery based on variational inference and make two algorithmic contributions. First: we highlight a tight connection between variational option discovery methods and variational autoencoders, and introduce Variational Autoencoding Learning of Options by Reinforcement (VALOR), a new method derived from the connection. In VALOR, the policy encodes contexts from a noise distribution into trajectories, and the decoder recovers the contexts from the complete trajectories. Second: we propose a curriculum learning approach where the number of contexts seen by the agent increases whenever the agent's performance is strong enough (as measured by the decoder) on the current set of contexts. We show that this simple trick stabilizes training for VALOR and prior variational option discovery methods, allowing a single agent to learn many more modes of behavior than it could with a fixed context distribution. Finally, we investigate other topics related to variational option discovery, including fundamental limitations of the general approach and the applicability of learned options to downstream tasks.",
  "published": "2018-07-26",
  "updated": "2018-07-26",
  "year": "2018",
  "authors": [
   "Joshua Achiam",
   "Harrison Edwards",
   "Dario Amodei",
   "Pieter Abbeel"
  ],
  "author_count": 4,
  "categories": [
   "cs.AI"
  ],
  "primary_category": "cs.AI",
  "venue": "",
  "venue_source": "",
  "citations": 210,
  "influential_citations": 32,
  "tldr": "A tight connection between variational option discovery methods and variational autoencoders is highlighted, and Variational Autoencoding Learning of Options by Reinforcement (VALOR), a new method derived from the connection is introduced, and a curriculum learning approach is proposed.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Joshua Achiam",
    "id": "3381809",
    "h_index": 9,
    "papers": 14
   },
   {
    "name": "Harrison Edwards",
    "id": "144632352",
    "h_index": 13,
    "papers": 17
   },
   {
    "name": "Dario Amodei",
    "id": "2330246600",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "P. Abbeel",
    "id": "1689992",
    "h_index": 165,
    "papers": 596
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1807.10299v1",
  "pdf_url": "https://arxiv.org/pdf/1807.10299v1",
  "html_url": "https://arxiv.org/html/1807.10299v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.32
 },
 {
  "id": "1807.04742",
  "slug": "visual-reinforcement-learning-with-imagined-goals",
  "title": "Visual Reinforcement Learning with Imagined Goals",
  "abstract": "For an autonomous agent to fulfill a wide range of user-specified goals at test time, it must be able to learn broadly applicable and general-purpose skill repertoires. Furthermore, to provide the requisite level of generality, these skills must handle raw sensory input such as images. In this paper, we propose an algorithm that acquires such general-purpose skills by combining unsupervised representation learning and reinforcement learning of goal-conditioned policies. Since the particular goals that might be required at test-time are not known in advance, the agent performs a self-supervised \"practice\" phase where it imagines goals and attempts to achieve them. We learn a visual representation with three distinct purposes: sampling goals for self-supervised practice, providing a structured transformation of raw sensory inputs, and computing a reward signal for goal reaching. We also propose a retroactive goal relabeling scheme to further improve the sample-efficiency of our method. Our off-policy algorithm is efficient enough to learn policies that operate on raw image observations and goals for a real-world robotic system, and substantially outperforms prior techniques.",
  "published": "2018-07-12",
  "updated": "2018-12-04",
  "year": "2018",
  "authors": [
   "Ashvin Nair",
   "Vitchyr Pong",
   "Murtaza Dalal",
   "Shikhar Bahl",
   "Steven Lin",
   "Sergey Levine"
  ],
  "author_count": 6,
  "categories": [
   "cs.LG",
   "cs.CV",
   "cs.RO",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 604,
  "influential_citations": 57,
  "tldr": "An algorithm is proposed that acquires general-purpose skills by combining unsupervised representation learning and reinforcement learning of goal-conditioned policies, efficient enough to learn policies that operate on raw image observations and goals for a real-world robotic system, and substantially outperforms prior techniques.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ashvin Nair",
    "id": "3422774",
    "h_index": 18,
    "papers": 22
   },
   {
    "name": "Vitchyr H. Pong",
    "id": "144401061",
    "h_index": 15,
    "papers": 36
   },
   {
    "name": "Murtaza Dalal",
    "id": "35904540",
    "h_index": 13,
    "papers": 16
   },
   {
    "name": "Shikhar Bahl",
    "id": "8527563",
    "h_index": 20,
    "papers": 38
   },
   {
    "name": "Steven Lin",
    "id": "2110473490",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "S. Levine",
    "id": "1736651",
    "h_index": 167,
    "papers": 551
   }
  ],
  "comment": "15 pages, NeurIPS 2018",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1807.04742v2",
  "pdf_url": "https://arxiv.org/pdf/1807.04742v2",
  "html_url": "https://arxiv.org/html/1807.04742v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.28
 },
 {
  "id": "1807.03748",
  "slug": "representation-learning-with-contrastive-predictive-coding",
  "title": "Representation Learning with Contrastive Predictive Coding",
  "abstract": "While supervised learning has enabled great progress in many applications, unsupervised learning has not seen such widespread adoption, and remains an important and challenging endeavor for artificial intelligence. In this work, we propose a universal unsupervised learning approach to extract useful representations from high-dimensional data, which we call Contrastive Predictive Coding. The key insight of our model is to learn such representations by predicting the future in latent space by using powerful autoregressive models. We use a probabilistic contrastive loss which induces the latent space to capture information that is maximally useful to predict future samples. It also makes the model tractable by using negative sampling. While most prior work has focused on evaluating representations for a particular modality, we demonstrate that our approach is able to learn useful representations achieving strong performance on four distinct domains: speech, images, text and reinforcement learning in 3D environments.",
  "published": "2018-07-10",
  "updated": "2019-01-22",
  "year": "2018",
  "authors": [
   "Aaron van den Oord",
   "Yazhe Li",
   "Oriol Vinyals"
  ],
  "author_count": 3,
  "categories": [
   "cs.LG",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 14281,
  "influential_citations": 1577,
  "tldr": "This work proposes a universal unsupervised learning approach to extract useful representations from high-dimensional data, which it calls Contrastive Predictive Coding, and demonstrates that the approach is able to learn useful representations achieving strong performance on four distinct domains: speech, images, text and reinforcement learning in 3D environments.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "A\u00e4ron van den Oord",
    "id": "3422336",
    "h_index": 42,
    "papers": 58
   },
   {
    "name": "Yazhe Li",
    "id": "2144417088",
    "h_index": 12,
    "papers": 24
   },
   {
    "name": "O. Vinyals",
    "id": "1689108",
    "h_index": 103,
    "papers": 204
   }
  ],
  "comment": "",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1807.03748v2",
  "pdf_url": "https://arxiv.org/pdf/1807.03748v2",
  "html_url": "https://arxiv.org/html/1807.03748v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.0
 },
 {
  "id": "1807.03039",
  "slug": "glow-generative-flow-with-invertible-1x1-convolutions",
  "title": "Glow: Generative Flow with Invertible 1x1 Convolutions",
  "abstract": "Flow-based generative models (Dinh et al., 2014) are conceptually attractive due to tractability of the exact log-likelihood, tractability of exact latent-variable inference, and parallelizability of both training and synthesis. In this paper we propose Glow, a simple type of generative flow using an invertible 1x1 convolution. Using our method we demonstrate a significant improvement in log-likelihood on standard benchmarks. Perhaps most strikingly, we demonstrate that a generative model optimized towards the plain log-likelihood objective is capable of efficient realistic-looking synthesis and manipulation of large images. The code for our model is available at https://github.com/openai/glow",
  "published": "2018-07-09",
  "updated": "2018-07-10",
  "year": "2018",
  "authors": [
   "Diederik P. Kingma",
   "Prafulla Dhariwal"
  ],
  "author_count": 2,
  "categories": [
   "stat.ML",
   "cs.AI",
   "cs.LG"
  ],
  "primary_category": "stat.ML",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 3679,
  "influential_citations": 538,
  "tldr": "Glow, a simple type of generative flow using an invertible 1x1 convolution, is proposed, demonstrating that a generative model optimized towards the plain log-likelihood objective is capable of efficient realistic-looking synthesis and manipulation of large images.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Diederik P. Kingma",
    "id": "1726807",
    "h_index": 35,
    "papers": 46
   },
   {
    "name": "Prafulla Dhariwal",
    "id": "6515819",
    "h_index": 20,
    "papers": 43
   }
  ],
  "comment": "15 pages; fixed typo in abstract",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1807.03039v2",
  "pdf_url": "https://arxiv.org/pdf/1807.03039v2",
  "html_url": "https://arxiv.org/html/1807.03039v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "1807.01675",
  "slug": "sample-efficient-reinforcement-learning-with-stochastic-ensemble-value",
  "title": "Sample-Efficient Reinforcement Learning with Stochastic Ensemble Value Expansion",
  "abstract": "Integrating model-free and model-based approaches in reinforcement learning has the potential to achieve the high performance of model-free algorithms with low sample complexity. However, this is difficult because an imperfect dynamics model can degrade the performance of the learning algorithm, and in sufficiently complex environments, the dynamics model will almost always be imperfect. As a result, a key challenge is to combine model-based approaches with model-free learning in such a way that errors in the model do not degrade performance. We propose stochastic ensemble value expansion (STEVE), a novel model-based technique that addresses this issue. By dynamically interpolating between model rollouts of various horizon lengths for each individual example, STEVE ensures that the model is only utilized when doing so does not introduce significant errors. Our approach outperforms model-free baselines on challenging continuous control benchmarks with an order-of-magnitude increase in sample efficiency, and in contrast to previous model-based approaches, performance does not degrade in complex environments.",
  "published": "2018-07-04",
  "updated": "2019-06-07",
  "year": "2018",
  "authors": [
   "Jacob Buckman",
   "Danijar Hafner",
   "George Tucker",
   "Eugene Brevdo",
   "Honglak Lee"
  ],
  "author_count": 5,
  "categories": [
   "cs.LG",
   "cs.AI",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 374,
  "influential_citations": 37,
  "tldr": "Stochastic ensemble value expansion (STEVE), a novel model-based technique that addresses this issue by dynamically interpolating between model rollouts of various horizon lengths for each individual example, outperforms model-free baselines on challenging continuous control benchmarks with an order-of-magnitude increase in sample efficiency.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Jacob Buckman",
    "id": "47619311",
    "h_index": 8,
    "papers": 11
   },
   {
    "name": "Danijar Hafner",
    "id": "35006479",
    "h_index": 25,
    "papers": 47
   },
   {
    "name": "G. Tucker",
    "id": "145499435",
    "h_index": 34,
    "papers": 55
   },
   {
    "name": "E. Brevdo",
    "id": "2445241",
    "h_index": 13,
    "papers": 24
   },
   {
    "name": "Honglak Lee",
    "id": "1697141",
    "h_index": 88,
    "papers": 188
   }
  ],
  "comment": "",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1807.01675v2",
  "pdf_url": "https://arxiv.org/pdf/1807.01675v2",
  "html_url": "https://arxiv.org/html/1807.01675v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.07
 },
 {
  "id": "1806.10293",
  "slug": "qt-opt-scalable-deep-reinforcement-learning-for-vision-based-robotic-m",
  "title": "QT-Opt: Scalable Deep Reinforcement Learning for Vision-Based Robotic Manipulation",
  "abstract": "In this paper, we study the problem of learning vision-based dynamic manipulation skills using a scalable reinforcement learning approach. We study this problem in the context of grasping, a longstanding challenge in robotic manipulation. In contrast to static learning behaviors that choose a grasp point and then execute the desired grasp, our method enables closed-loop vision-based control, whereby the robot continuously updates its grasp strategy based on the most recent observations to optimize long-horizon grasp success. To that end, we introduce QT-Opt, a scalable self-supervised vision-based reinforcement learning framework that can leverage over 580k real-world grasp attempts to train a deep neural network Q-function with over 1.2M parameters to perform closed-loop, real-world grasping that generalizes to 96% grasp success on unseen objects. Aside from attaining a very high success rate, our method exhibits behaviors that are quite distinct from more standard grasping systems: using only RGB vision-based perception from an over-the-shoulder camera, our method automatically learns regrasping strategies, probes objects to find the most effective grasps, learns to reposition objects and perform other non-prehensile pre-grasp manipulations, and responds dynamically to disturbances and perturbations.",
  "published": "2018-06-27",
  "updated": "2018-11-28",
  "year": "2018",
  "authors": [
   "Dmitry Kalashnikov",
   "Alex Irpan",
   "Peter Pastor",
   "Julian Ibarz",
   "Alexander Herzog",
   "Eric Jang",
   "Deirdre Quillen",
   "Ethan Holly",
   "Mrinal Kalakrishnan",
   "Vincent Vanhoucke",
   "Sergey Levine"
  ],
  "author_count": 11,
  "categories": [
   "cs.LG",
   "cs.AI",
   "cs.CV",
   "cs.RO",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "CoRL",
  "venue_source": "semantic-scholar",
  "citations": 1755,
  "influential_citations": 105,
  "tldr": "QT-Opt is introduced, a scalable self-supervised vision-based reinforcement learning framework that can leverage over 580k real-world grasp attempts to train a deep neural network Q-function with over 1.2M parameters to perform closed-loop, real- world grasping that generalizes to 96% grasp success on unseen objects.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Dmitry Kalashnikov",
    "id": "48313860",
    "h_index": 21,
    "papers": 34
   },
   {
    "name": "A. Irpan",
    "id": "17818078",
    "h_index": 22,
    "papers": 32
   },
   {
    "name": "P. Pastor",
    "id": "143970835",
    "h_index": 32,
    "papers": 47
   },
   {
    "name": "Julian Ibarz",
    "id": "46920727",
    "h_index": 22,
    "papers": 35
   },
   {
    "name": "Alexander Herzog",
    "id": "1505793452",
    "h_index": 22,
    "papers": 35
   },
   {
    "name": "Eric Jang",
    "id": "145116380",
    "h_index": 20,
    "papers": 30
   },
   {
    "name": "Deirdre Quillen",
    "id": "47202040",
    "h_index": 6,
    "papers": 8
   },
   {
    "name": "E. Holly",
    "id": "29891985",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Mrinal Kalakrishnan",
    "id": "1729262",
    "h_index": 35,
    "papers": 58
   },
   {
    "name": "Vincent Vanhoucke",
    "id": "2657155",
    "h_index": 34,
    "papers": 60
   },
   {
    "name": "S. Levine",
    "id": "1736651",
    "h_index": 167,
    "papers": 551
   }
  ],
  "comment": "CoRL 2018 camera ready. 23 pages, 14 figures",
  "topics": [
   "dexterous-manipulation",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1806.10293v3",
  "pdf_url": "https://arxiv.org/pdf/1806.10293v3",
  "html_url": "https://arxiv.org/html/1806.10293v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "1806.06923",
  "slug": "implicit-quantile-networks-for-distributional-reinforcement-learning",
  "title": "Implicit Quantile Networks for Distributional Reinforcement Learning",
  "abstract": "In this work, we build on recent advances in distributional reinforcement learning to give a generally applicable, flexible, and state-of-the-art distributional variant of DQN. We achieve this by using quantile regression to approximate the full quantile function for the state-action return distribution. By reparameterizing a distribution over the sample space, this yields an implicitly defined return distribution and gives rise to a large class of risk-sensitive policies. We demonstrate improved performance on the 57 Atari 2600 games in the ALE, and use our algorithm's implicitly defined distributions to study the effects of risk-sensitive policies in Atari games.",
  "published": "2018-06-14",
  "updated": "2018-06-14",
  "year": "2018",
  "authors": [
   "Will Dabney",
   "Georg Ostrovski",
   "David Silver",
   "R\u00e9mi Munos"
  ],
  "author_count": 4,
  "categories": [
   "cs.LG",
   "cs.AI",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "ICML",
  "venue_source": "semantic-scholar",
  "citations": 734,
  "influential_citations": 173,
  "tldr": "This work builds on recent advances in distributional reinforcement learning to give a generally applicable, flexible, and state-of-the-art distributional variant of DQN by using quantile regression to approximate the full quantile function for the state-action return distribution.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Will Dabney",
    "id": "2605877",
    "h_index": 34,
    "papers": 64
   },
   {
    "name": "Georg Ostrovski",
    "id": "2273072",
    "h_index": 22,
    "papers": 30
   },
   {
    "name": "David Silver",
    "id": "145824029",
    "h_index": 80,
    "papers": 120
   },
   {
    "name": "R. Munos",
    "id": "1708654",
    "h_index": 90,
    "papers": 246
   }
  ],
  "comment": "ICML 2018",
  "topics": [
   "rl-control",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1806.06923v1",
  "pdf_url": "https://arxiv.org/pdf/1806.06923v1",
  "html_url": "https://arxiv.org/html/1806.06923v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.37
 },
 {
  "id": "1806.06920",
  "slug": "maximum-a-posteriori-policy-optimisation",
  "title": "Maximum a Posteriori Policy Optimisation",
  "abstract": "We introduce a new algorithm for reinforcement learning called Maximum aposteriori Policy Optimisation (MPO) based on coordinate ascent on a relative entropy objective. We show that several existing methods can directly be related to our derivation. We develop two off-policy algorithms and demonstrate that they are competitive with the state-of-the-art in deep reinforcement learning. In particular, for continuous control, our method outperforms existing methods with respect to sample efficiency, premature convergence and robustness to hyperparameter settings while achieving similar or better final performance.",
  "published": "2018-06-14",
  "updated": "2018-06-14",
  "year": "2018",
  "authors": [
   "Abbas Abdolmaleki",
   "Jost Tobias Springenberg",
   "Yuval Tassa",
   "Remi Munos",
   "Nicolas Heess",
   "Martin Riedmiller"
  ],
  "author_count": 6,
  "categories": [
   "cs.LG",
   "cs.AI",
   "cs.IT",
   "cs.RO",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 601,
  "influential_citations": 88,
  "tldr": "This work introduces a new algorithm for reinforcement learning called Maximum aposteriori Policy Optimisation (MPO) based on coordinate ascent on a relative entropy objective and develops two off-policy algorithms that are competitive with the state-of-the-art in deep reinforcement learning.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "A. Abdolmaleki",
    "id": "2799799",
    "h_index": 29,
    "papers": 95
   },
   {
    "name": "Jost Tobias Springenberg",
    "id": "2060551",
    "h_index": 44,
    "papers": 93
   },
   {
    "name": "Yuval Tassa",
    "id": "2109481",
    "h_index": 39,
    "papers": 67
   },
   {
    "name": "R. Munos",
    "id": "1708654",
    "h_index": 90,
    "papers": 246
   },
   {
    "name": "N. Heess",
    "id": "2801204",
    "h_index": 73,
    "papers": 192
   },
   {
    "name": "Martin A. Riedmiller",
    "id": "3137672",
    "h_index": 55,
    "papers": 219
   }
  ],
  "comment": "",
  "topics": [
   "rl-control",
   "safety-eval"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1806.06920v1",
  "pdf_url": "https://arxiv.org/pdf/1806.06920v1",
  "html_url": "https://arxiv.org/html/1806.06920v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.28
 },
 {
  "id": "1806.04613",
  "slug": "improving-regression-performance-with-distributional-losses",
  "title": "Improving Regression Performance with Distributional Losses",
  "abstract": "There is growing evidence that converting targets to soft targets in supervised learning can provide considerable gains in performance. Much of this work has considered classification, converting hard zero-one values to soft labels---such as by adding label noise, incorporating label ambiguity or using distillation. In parallel, there is some evidence from a regression setting in reinforcement learning that learning distributions can improve performance. In this work, we investigate the reasons for this improvement, in a regression setting. We introduce a novel distributional regression loss, and similarly find it significantly improves prediction accuracy. We investigate several common hypotheses, around reducing overfitting and improved representations. We instead find evidence for an alternative hypothesis: this loss is easier to optimize, with better behaved gradients, resulting in improved generalization. We provide theoretical support for this alternative hypothesis, by characterizing the norm of the gradients of this loss.",
  "published": "2018-06-12",
  "updated": "2018-06-12",
  "year": "2018",
  "authors": [
   "Ehsan Imani",
   "Martha White"
  ],
  "author_count": 2,
  "categories": [
   "stat.ML",
   "cs.LG"
  ],
  "primary_category": "stat.ML",
  "venue": "ICML",
  "venue_source": "semantic-scholar",
  "citations": 102,
  "influential_citations": 22,
  "tldr": "This work introduces a novel distributional regression loss, and finds it significantly improves prediction accuracy, and provides theoretical support for an alternative hypothesis: this loss is easier to optimize, with better behaved gradients, resulting in improved generalization.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ehsan Imani",
    "id": "29905816",
    "h_index": 7,
    "papers": 13
   },
   {
    "name": "Martha White",
    "id": "144542337",
    "h_index": 32,
    "papers": 106
   }
  ],
  "comment": "12 pages, 4 figures. To appear in Proceedings of the 35th International Conference on Machine Learning, 2018",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1806.04613v1",
  "pdf_url": "https://arxiv.org/pdf/1806.04613v1",
  "html_url": "https://arxiv.org/html/1806.04613v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.51
 },
 {
  "id": "1806.03107",
  "slug": "temporal-difference-variational-auto-encoder",
  "title": "Temporal Difference Variational Auto-Encoder",
  "abstract": "To act and plan in complex environments, we posit that agents should have a mental simulator of the world with three characteristics: (a) it should build an abstract state representing the condition of the world; (b) it should form a belief which represents uncertainty on the world; (c) it should go beyond simple step-by-step simulation, and exhibit temporal abstraction. Motivated by the absence of a model satisfying all these requirements, we propose TD-VAE, a generative sequence model that learns representations containing explicit beliefs about states several steps into the future, and that can be rolled out directly without single-step transitions. TD-VAE is trained on pairs of temporally separated time points, using an analogue of temporal difference learning used in reinforcement learning.",
  "published": "2018-06-08",
  "updated": "2019-01-02",
  "year": "2018",
  "authors": [
   "Karol Gregor",
   "George Papamakarios",
   "Frederic Besse",
   "Lars Buesing",
   "Theophane Weber"
  ],
  "author_count": 5,
  "categories": [
   "cs.LG",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 132,
  "influential_citations": 17,
  "tldr": "TD-VAE is proposed, a generative sequence model that learns representations containing explicit beliefs about states several steps into the future, and that can be rolled out directly without single-step transitions.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Karol Gregor",
    "id": "144717963",
    "h_index": 25,
    "papers": 40
   },
   {
    "name": "F. Besse",
    "id": "143923544",
    "h_index": 12,
    "papers": 19
   }
  ],
  "comment": "",
  "topics": [
   "sim2real",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1806.03107v3",
  "pdf_url": "https://arxiv.org/pdf/1806.03107v3",
  "html_url": "https://arxiv.org/html/1806.03107v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.62
 },
 {
  "id": "1806.02813",
  "slug": "self-consistent-trajectory-autoencoder-hierarchical-reinforcement-lear",
  "title": "Self-Consistent Trajectory Autoencoder: Hierarchical Reinforcement Learning with Trajectory Embeddings",
  "abstract": "In this work, we take a representation learning perspective on hierarchical reinforcement learning, where the problem of learning lower layers in a hierarchy is transformed into the problem of learning trajectory-level generative models. We show that we can learn continuous latent representations of trajectories, which are effective in solving temporally extended and multi-stage problems. Our proposed model, SeCTAR, draws inspiration from variational autoencoders, and learns latent representations of trajectories. A key component of this method is to learn both a latent-conditioned policy and a latent-conditioned model which are consistent with each other. Given the same latent, the policy generates a trajectory which should match the trajectory predicted by the model. This model provides a built-in prediction mechanism, by predicting the outcome of closed loop policy behavior. We propose a novel algorithm for performing hierarchical RL with this model, combining model-based planning in the learned latent space with an unsupervised exploration objective. We show that our model is effective at reasoning over long horizons with sparse rewards for several simulated tasks, outperforming standard reinforcement learning methods and prior methods for hierarchical reasoning, model-based planning, and exploration.",
  "published": "2018-06-07",
  "updated": "2018-06-07",
  "year": "2018",
  "authors": [
   "John D. Co-Reyes",
   "YuXuan Liu",
   "Abhishek Gupta",
   "Benjamin Eysenbach",
   "Pieter Abbeel",
   "Sergey Levine"
  ],
  "author_count": 6,
  "categories": [
   "cs.LG",
   "cs.AI",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "ICML",
  "venue_source": "semantic-scholar",
  "citations": 160,
  "influential_citations": 16,
  "tldr": "This work shows that it can learn continuous latent representations of trajectories, which are effective in solving temporally extended and multi-stage problems and provides a built-in prediction mechanism, by predicting the outcome of closed loop policy behavior.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "John D. Co-Reyes",
    "id": "1388383230",
    "h_index": 10,
    "papers": 18
   },
   {
    "name": "Yuxuan Liu",
    "id": null,
    "h_index": 0,
    "papers": 0
   },
   {
    "name": "Abhishek Gupta",
    "id": "2129458064",
    "h_index": 57,
    "papers": 156
   },
   {
    "name": "Benjamin Eysenbach",
    "id": "8140754",
    "h_index": 34,
    "papers": 75
   },
   {
    "name": "P. Abbeel",
    "id": "1689992",
    "h_index": 165,
    "papers": 596
   },
   {
    "name": "S. Levine",
    "id": "1736651",
    "h_index": 167,
    "papers": 551
   }
  ],
  "comment": "Accepted at ICML 2018",
  "topics": [
   "rl-control",
   "navigation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1806.02813v1",
  "pdf_url": "https://arxiv.org/pdf/1806.02813v1",
  "html_url": "https://arxiv.org/html/1806.02813v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.71
 },
 {
  "id": "1806.02426",
  "slug": "deep-variational-reinforcement-learning-for-pomdps",
  "title": "Deep Variational Reinforcement Learning for POMDPs",
  "abstract": "Many real-world sequential decision making problems are partially observable by nature, and the environment model is typically unknown. Consequently, there is great need for reinforcement learning methods that can tackle such problems given only a stream of incomplete and noisy observations. In this paper, we propose deep variational reinforcement learning (DVRL), which introduces an inductive bias that allows an agent to learn a generative model of the environment and perform inference in that model to effectively aggregate the available information. We develop an n-step approximation to the evidence lower bound (ELBO), allowing the model to be trained jointly with the policy. This ensures that the latent state representation is suitable for the control task. In experiments on Mountain Hike and flickering Atari we show that our method outperforms previous approaches relying on recurrent neural networks to encode the past.",
  "published": "2018-06-06",
  "updated": "2018-06-06",
  "year": "2018",
  "authors": [
   "Maximilian Igl",
   "Luisa Zintgraf",
   "Tuan Anh Le",
   "Frank Wood",
   "Shimon Whiteson"
  ],
  "author_count": 5,
  "categories": [
   "cs.LG",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "ICML",
  "venue_source": "semantic-scholar",
  "citations": 312,
  "influential_citations": 33,
  "tldr": "Deep variational reinforcement learning (DVRL) is proposed, which introduces an inductive bias that allows an agent to learn a generative model of the environment and perform inference in that model to effectively aggregate the available information.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "M. Igl",
    "id": "27550002",
    "h_index": 17,
    "papers": 34
   },
   {
    "name": "Luisa M. Zintgraf",
    "id": "3378188",
    "h_index": 19,
    "papers": 35
   },
   {
    "name": "T. Le",
    "id": "2067951916",
    "h_index": 13,
    "papers": 29
   },
   {
    "name": "Frank Wood",
    "id": "144109143",
    "h_index": 19,
    "papers": 65
   },
   {
    "name": "Shimon Whiteson",
    "id": "1766767",
    "h_index": 68,
    "papers": 277
   }
  ],
  "comment": "",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1806.02426v1",
  "pdf_url": "https://arxiv.org/pdf/1806.02426v1",
  "html_url": "https://arxiv.org/html/1806.02426v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.0
 },
 {
  "id": "1805.12114",
  "slug": "deep-reinforcement-learning-in-a-handful-of-trials-using-probabilistic",
  "title": "Deep Reinforcement Learning in a Handful of Trials using Probabilistic Dynamics Models",
  "abstract": "Model-based reinforcement learning (RL) algorithms can attain excellent sample efficiency, but often lag behind the best model-free algorithms in terms of asymptotic performance. This is especially true with high-capacity parametric function approximators, such as deep networks. In this paper, we study how to bridge this gap, by employing uncertainty-aware dynamics models. We propose a new algorithm called probabilistic ensembles with trajectory sampling (PETS) that combines uncertainty-aware deep network dynamics models with sampling-based uncertainty propagation. Our comparison to state-of-the-art model-based and model-free deep RL algorithms shows that our approach matches the asymptotic performance of model-free algorithms on several challenging benchmark tasks, while requiring significantly fewer samples (e.g., 8 and 125 times fewer samples than Soft Actor Critic and Proximal Policy Optimization respectively on the half-cheetah task).",
  "published": "2018-05-30",
  "updated": "2018-11-02",
  "year": "2018",
  "authors": [
   "Kurtland Chua",
   "Roberto Calandra",
   "Rowan McAllister",
   "Sergey Levine"
  ],
  "author_count": 4,
  "categories": [
   "cs.LG",
   "cs.AI",
   "cs.RO",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 1589,
  "influential_citations": 244,
  "tldr": "This paper proposes a new algorithm called probabilistic ensembles with trajectory sampling (PETS) that combines uncertainty-aware deep network dynamics models with sampling-based uncertainty propagation, which matches the asymptotic performance of model-free algorithms on several challenging benchmark tasks, while requiring significantly fewer samples.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Kurtland Chua",
    "id": "46224156",
    "h_index": 4,
    "papers": 6
   },
   {
    "name": "R. Calandra",
    "id": "35159852",
    "h_index": 40,
    "papers": 75
   },
   {
    "name": "R. McAllister",
    "id": "49686609",
    "h_index": 21,
    "papers": 44
   },
   {
    "name": "S. Levine",
    "id": "1736651",
    "h_index": 167,
    "papers": 551
   }
  ],
  "comment": "NIPS 2018, video and code available at https://sites.google.com/view/drl-in-a-handful-of-trials/",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1805.12114v2",
  "pdf_url": "https://arxiv.org/pdf/1805.12114v2",
  "html_url": "https://arxiv.org/html/1805.12114v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "1805.11085",
  "slug": "more-than-a-feeling-learning-to-grasp-and-regrasp-using-vision-and-tou",
  "title": "More Than a Feeling: Learning to Grasp and Regrasp using Vision and Touch",
  "abstract": "For humans, the process of grasping an object relies heavily on rich tactile feedback. Most recent robotic grasping work, however, has been based only on visual input, and thus cannot easily benefit from feedback after initiating contact. In this paper, we investigate how a robot can learn to use tactile information to iteratively and efficiently adjust its grasp. To this end, we propose an end-to-end action-conditional model that learns regrasping policies from raw visuo-tactile data. This model -- a deep, multimodal convolutional network -- predicts the outcome of a candidate grasp adjustment, and then executes a grasp by iteratively selecting the most promising actions. Our approach requires neither calibration of the tactile sensors, nor any analytical modeling of contact forces, thus reducing the engineering effort required to obtain efficient grasping policies. We train our model with data from about 6,450 grasping trials on a two-finger gripper equipped with GelSight high-resolution tactile sensors on each finger. Across extensive experiments, our approach outperforms a variety of baselines at (i) estimating grasp adjustment outcomes, (ii) selecting efficient grasp adjustments for quick grasping, and (iii) reducing the amount of force applied at the fingers, while maintaining competitive performance. Finally, we study the choices made by our model and show that it has successfully acquired useful and interpretable grasping behaviors.",
  "published": "2018-05-28",
  "updated": "2018-07-26",
  "year": "2018",
  "authors": [
   "Roberto Calandra",
   "Andrew Owens",
   "Dinesh Jayaraman",
   "Justin Lin",
   "Wenzhen Yuan",
   "Jitendra Malik",
   "Edward H. Adelson",
   "Sergey Levine"
  ],
  "author_count": 8,
  "categories": [
   "cs.RO",
   "cs.LG",
   "stat.ML"
  ],
  "primary_category": "cs.RO",
  "venue": "RA-L",
  "venue_source": "semantic-scholar",
  "citations": 441,
  "influential_citations": 18,
  "tldr": "An end-to-end action-conditional model that learns regrasping policies from raw visuo-tactile data and outperforms a variety of baselines at estimating grasp adjustment outcomes, selecting efficient grasp adjustments for quick grasping, and reducing the amount of force applied at the fingers, while maintaining competitive performance.",
  "doi": "10.1109/LRA.2018.2852779",
  "oa_pdf": "https://arxiv.org/pdf/1805.11085",
  "s2_authors": [
   {
    "name": "R. Calandra",
    "id": "35159852",
    "h_index": 40,
    "papers": 75
   },
   {
    "name": "Andrew Owens",
    "id": "144956994",
    "h_index": 27,
    "papers": 40
   },
   {
    "name": "Dinesh Jayaraman",
    "id": "144348441",
    "h_index": 33,
    "papers": 72
   },
   {
    "name": "Justin Lin",
    "id": "2144488483",
    "h_index": 4,
    "papers": 5
   },
   {
    "name": "Wenzhen Yuan",
    "id": "3333169",
    "h_index": 27,
    "papers": 41
   },
   {
    "name": "Jitendra Malik",
    "id": "143751119",
    "h_index": 139,
    "papers": 397
   },
   {
    "name": "E. Adelson",
    "id": "145358192",
    "h_index": 87,
    "papers": 272
   },
   {
    "name": "S. Levine",
    "id": "1736651",
    "h_index": 167,
    "papers": 551
   }
  ],
  "comment": "8 pages. Published on IEEE Robotics and Automation Letters (RAL). Website: https://sites.google.com/view/more-than-a-feeling",
  "topics": [
   "dexterous-manipulation",
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1805.11085v2",
  "pdf_url": "https://arxiv.org/pdf/1805.11085v2",
  "html_url": "https://arxiv.org/html/1805.11085v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.15
 },
 {
  "id": "1805.08296",
  "slug": "data-efficient-hierarchical-reinforcement-learning",
  "title": "Data-Efficient Hierarchical Reinforcement Learning",
  "abstract": "Hierarchical reinforcement learning (HRL) is a promising approach to extend traditional reinforcement learning (RL) methods to solve more complex tasks. Yet, the majority of current HRL methods require careful task-specific design and on-policy training, making them difficult to apply in real-world scenarios. In this paper, we study how we can develop HRL algorithms that are general, in that they do not make onerous additional assumptions beyond standard RL algorithms, and efficient, in the sense that they can be used with modest numbers of interaction samples, making them suitable for real-world problems such as robotic control. For generality, we develop a scheme where lower-level controllers are supervised with goals that are learned and proposed automatically by the higher-level controllers. To address efficiency, we propose to use off-policy experience for both higher and lower-level training. This poses a considerable challenge, since changes to the lower-level behaviors change the action space for the higher-level policy, and we introduce an off-policy correction to remedy this challenge. This allows us to take advantage of recent advances in off-policy model-free RL to learn both higher- and lower-level policies using substantially fewer environment interactions than on-policy algorithms. We term the resulting HRL agent HIRO and find that it is generally applicable and highly sample-efficient. Our experiments show that HIRO can be used to learn highly complex behaviors for simulated robots, such as pushing objects and utilizing them to reach target locations, learning from only a few million samples, equivalent to a few days of real-time interaction. In comparisons with a number of prior HRL methods, we find that our approach substantially outperforms previous state-of-the-art techniques.",
  "published": "2018-05-21",
  "updated": "2018-10-05",
  "year": "2018",
  "authors": [
   "Ofir Nachum",
   "Shixiang Gu",
   "Honglak Lee",
   "Sergey Levine"
  ],
  "author_count": 4,
  "categories": [
   "cs.LG",
   "cs.AI",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "NeurIPS",
  "venue_source": "semantic-scholar",
  "citations": 1047,
  "influential_citations": 121,
  "tldr": "This paper studies how to develop HRL algorithms that are general, in that they do not make onerous additional assumptions beyond standard RL algorithms, and efficient, in the sense that they can be used with modest numbers of interaction samples, making them suitable for real-world problems such as robotic control.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Ofir Nachum",
    "id": "7624658",
    "h_index": 47,
    "papers": 92
   },
   {
    "name": "S. Gu",
    "id": "2046135",
    "h_index": 43,
    "papers": 71
   },
   {
    "name": "Honglak Lee",
    "id": "1697141",
    "h_index": 88,
    "papers": 188
   },
   {
    "name": "S. Levine",
    "id": "1736651",
    "h_index": 167,
    "papers": 551
   }
  ],
  "comment": "NIPS 2018",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1805.08296v4",
  "pdf_url": "https://arxiv.org/pdf/1805.08296v4",
  "html_url": "https://arxiv.org/html/1805.08296v4",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "1805.07813",
  "slug": "learning-real-world-robot-policies-by-dreaming",
  "title": "Learning Real-World Robot Policies by Dreaming",
  "abstract": "Learning to control robots directly based on images is a primary challenge in robotics. However, many existing reinforcement learning approaches require iteratively obtaining millions of robot samples to learn a policy, which can take significant time. In this paper, we focus on learning a realistic world model capturing the dynamics of scene changes conditioned on robot actions. Our dreaming model can emulate samples equivalent to a sequence of images from the actual environment, technically by learning an action-conditioned future representation/scene regressor. This allows the agent to learn action policies (i.e., visuomotor policies) by interacting with the dreaming model rather than the real-world. We experimentally confirm that our dreaming model enables robot learning of policies that transfer to the real-world.",
  "published": "2018-05-20",
  "updated": "2019-08-01",
  "year": "2018",
  "authors": [
   "AJ Piergiovanni",
   "Alan Wu",
   "Michael S. Ryoo"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.CV",
   "stat.ML"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS",
  "venue_source": "semantic-scholar",
  "citations": 33,
  "influential_citations": 0,
  "tldr": "The dreaming model can emulate samples equivalent to a sequence of images from the actual environment, technically by learning an action-conditioned future representation/scene regressor, and enables robot learning of policies that transfer to the real-world.",
  "doi": "10.1109/IROS40897.2019.8967559",
  "oa_pdf": "https://arxiv.org/pdf/1805.07813",
  "s2_authors": [
   {
    "name": "A. Piergiovanni",
    "id": "8797855",
    "h_index": 27,
    "papers": 63
   },
   {
    "name": "Alan Wu",
    "id": "145825349",
    "h_index": 3,
    "papers": 8
   },
   {
    "name": "M. Ryoo",
    "id": "1766489",
    "h_index": 50,
    "papers": 164
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1805.07813v4",
  "pdf_url": "https://arxiv.org/pdf/1805.07813v4",
  "html_url": "https://arxiv.org/html/1805.07813v4",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.03
 },
 {
  "id": "1804.09627",
  "slug": "actor-and-observer-joint-modeling-of-first-and-third-person-videos",
  "title": "Actor and Observer: Joint Modeling of First and Third-Person Videos",
  "abstract": "Several theories in cognitive neuroscience suggest that when people interact with the world, or simulate interactions, they do so from a first-person egocentric perspective, and seamlessly transfer knowledge between third-person (observer) and first-person (actor). Despite this, learning such models for human action recognition has not been achievable due to the lack of data. This paper takes a step in this direction, with the introduction of Charades-Ego, a large-scale dataset of paired first-person and third-person videos, involving 112 people, with 4000 paired videos. This enables learning the link between the two, actor and observer perspectives. Thereby, we address one of the biggest bottlenecks facing egocentric vision research, providing a link from first-person to the abundant third-person data on the web. We use this data to learn a joint representation of first and third-person videos, with only weak supervision, and show its effectiveness for transferring knowledge from the third-person to the first-person domain.",
  "published": "2018-04-25",
  "updated": "2018-04-25",
  "year": "2018",
  "authors": [
   "Gunnar A. Sigurdsson",
   "Abhinav Gupta",
   "Cordelia Schmid",
   "Ali Farhadi",
   "Karteek Alahari"
  ],
  "author_count": 5,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 191,
  "influential_citations": 32,
  "tldr": "Charades-Ego is introduced, a large-scale dataset of paired first-person and third-person videos, involving 112 people, with 4000 paired videos, which enables learning the link between the two, actor and observer perspectives, and addresses one of the biggest bottlenecks facing egocentric vision research.",
  "doi": "10.1145/3265987.3265995",
  "oa_pdf": "https://hal.inria.fr/hal-01755547/file/Sigurdsson18.pdf",
  "s2_authors": [
   {
    "name": "Gunnar A. Sigurdsson",
    "id": "34280810",
    "h_index": 13,
    "papers": 34
   },
   {
    "name": "A. Gupta",
    "id": "1726095131",
    "h_index": 96,
    "papers": 211
   },
   {
    "name": "Cordelia Schmid",
    "id": "2462253",
    "h_index": 153,
    "papers": 466
   },
   {
    "name": "Ali Farhadi",
    "id": "143787583",
    "h_index": 78,
    "papers": 208
   },
   {
    "name": "Alahari Karteek",
    "id": "72492981",
    "h_index": 36,
    "papers": 122
   }
  ],
  "comment": "CVPR 2018 spotlight presentation",
  "topics": [
   "egocentric-data"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1804.09627v1",
  "pdf_url": "https://arxiv.org/pdf/1804.09627v1",
  "html_url": "https://arxiv.org/html/1804.09627v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.78
 },
 {
  "id": "1804.09235",
  "slug": "on-the-effectiveness-of-task-granularity-for-transfer-learning",
  "title": "On the effectiveness of task granularity for transfer learning",
  "abstract": "We describe a DNN for video classification and captioning, trained end-to-end, with shared features, to solve tasks at different levels of granularity, exploring the link between granularity in a source task and the quality of learned features for transfer learning. For solving the new task domain in transfer learning, we freeze the trained encoder and fine-tune a neural net on the target domain. We train on the Something-Something dataset with over 220, 000 videos, and multiple levels of target granularity, including 50 action groups, 174 fine-grained action categories and captions. Classification and captioning with Something-Something are challenging because of the subtle differences between actions, applied to thousands of different object classes, and the diversity of captions penned by crowd actors. Our model performs better than existing classification baselines for SomethingSomething, with impressive fine-grained results. And it yields a strong baseline on the new Something-Something captioning task. Experiments reveal that training with more fine-grained tasks tends to produce better features for transfer learning.",
  "published": "2018-04-24",
  "updated": "2018-11-29",
  "year": "2018",
  "authors": [
   "Farzaneh Mahdisoltani",
   "Guillaume Berger",
   "Waseem Gharbieh",
   "David Fleet",
   "Roland Memisevic"
  ],
  "author_count": 5,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "",
  "venue_source": "",
  "citations": 71,
  "influential_citations": 4,
  "tldr": "A DNN for video classification and captioning, trained end-to-end, with shared features, to solve tasks at different levels of granularity, exploring the link between granularity in a source task and the quality of learned features for transfer learning.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "F. Mahdisoltani",
    "id": "2454800",
    "h_index": 6,
    "papers": 14
   },
   {
    "name": "Guillaume Berger",
    "id": "40586522",
    "h_index": 7,
    "papers": 16
   },
   {
    "name": "W. Gharbieh",
    "id": "3462264",
    "h_index": 9,
    "papers": 18
   },
   {
    "name": "David J. Fleet",
    "id": "1793739",
    "h_index": 79,
    "papers": 221
   },
   {
    "name": "R. Memisevic",
    "id": "1710604",
    "h_index": 38,
    "papers": 71
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1804.09235v2",
  "pdf_url": "https://arxiv.org/pdf/1804.09235v2",
  "html_url": "https://arxiv.org/html/1804.09235v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.86
 },
 {
  "id": "1804.08617",
  "slug": "distributed-distributional-deterministic-policy-gradients",
  "title": "Distributed Distributional Deterministic Policy Gradients",
  "abstract": "This work adopts the very successful distributional perspective on reinforcement learning and adapts it to the continuous control setting. We combine this within a distributed framework for off-policy learning in order to develop what we call the Distributed Distributional Deep Deterministic Policy Gradient algorithm, D4PG. We also combine this technique with a number of additional, simple improvements such as the use of $N$-step returns and prioritized experience replay. Experimentally we examine the contribution of each of these individual components, and show how they interact, as well as their combined contributions. Our results show that across a wide variety of simple control tasks, difficult manipulation tasks, and a set of hard obstacle-based locomotion tasks the D4PG algorithm achieves state of the art performance.",
  "published": "2018-04-23",
  "updated": "2018-04-23",
  "year": "2018",
  "authors": [
   "Gabriel Barth-Maron",
   "Matthew W. Hoffman",
   "David Budden",
   "Will Dabney",
   "Dan Horgan",
   "Dhruva TB",
   "Alistair Muldal",
   "Nicolas Heess",
   "Timothy Lillicrap"
  ],
  "author_count": 9,
  "categories": [
   "cs.LG",
   "cs.AI",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 547,
  "influential_citations": 60,
  "tldr": "The results show that across a wide variety of simple control tasks, difficult manipulation tasks, and a set of hard obstacle-based locomotion tasks the D4PG algorithm achieves state of the art performance.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Gabriel Barth-Maron",
    "id": "1403998955",
    "h_index": 15,
    "papers": 29
   },
   {
    "name": "Matthew W. Hoffman",
    "id": "3243579",
    "h_index": 23,
    "papers": 34
   },
   {
    "name": "D. Budden",
    "id": "2508525",
    "h_index": 27,
    "papers": 73
   },
   {
    "name": "Will Dabney",
    "id": "2605877",
    "h_index": 34,
    "papers": 64
   },
   {
    "name": "Dan Horgan",
    "id": "48257711",
    "h_index": 11,
    "papers": 18
   },
   {
    "name": "TB Dhruva",
    "id": "22216833",
    "h_index": 6,
    "papers": 6
   },
   {
    "name": "Alistair Muldal",
    "id": "50654556",
    "h_index": 12,
    "papers": 15
   },
   {
    "name": "N. Heess",
    "id": "2801204",
    "h_index": 73,
    "papers": 192
   },
   {
    "name": "T. Lillicrap",
    "id": "2542999",
    "h_index": 69,
    "papers": 154
   }
  ],
  "comment": "",
  "topics": [
   "humanoids",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1804.08617v1",
  "pdf_url": "https://arxiv.org/pdf/1804.08617v1",
  "html_url": "https://arxiv.org/html/1804.08617v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.24
 },
 {
  "id": "1804.06318",
  "slug": "learning-awareness-models",
  "title": "Learning Awareness Models",
  "abstract": "We consider the setting of an agent with a fixed body interacting with an unknown and uncertain external world. We show that models trained to predict proprioceptive information about the agent's body come to represent objects in the external world. In spite of being trained with only internally available signals, these dynamic body models come to represent external objects through the necessity of predicting their effects on the agent's own body. That is, the model learns holistic persistent representations of objects in the world, even though the only training signals are body signals. Our dynamics model is able to successfully predict distributions over 132 sensor readings over 100 steps into the future and we demonstrate that even when the body is no longer in contact with an object, the latent variables of the dynamics model continue to represent its shape. We show that active data collection by maximizing the entropy of predictions about the body---touch sensors, proprioception and vestibular information---leads to learning of dynamic models that show superior performance when used for control. We also collect data from a real robotic hand and show that the same models can be used to answer questions about properties of objects in the real world. Videos with qualitative results of our models are available at https://goo.gl/mZuqAV.",
  "published": "2018-04-17",
  "updated": "2018-04-17",
  "year": "2018",
  "authors": [
   "Brandon Amos",
   "Laurent Dinh",
   "Serkan Cabi",
   "Thomas Roth\u00f6rl",
   "Sergio G\u00f3mez Colmenarejo",
   "Alistair Muldal",
   "Tom Erez",
   "Yuval Tassa",
   "Nando de Freitas",
   "Misha Denil"
  ],
  "author_count": 10,
  "categories": [
   "cs.AI",
   "cs.NE",
   "cs.RO"
  ],
  "primary_category": "cs.AI",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 49,
  "influential_citations": 5,
  "tldr": "The setting of an agent with a fixed body interacting with an unknown and uncertain external world is considered and it is demonstrated that even when the body is no longer in contact with an object, the latent variables of the dynamics model continue to represent its shape.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Brandon Amos",
    "id": "1773498",
    "h_index": 38,
    "papers": 70
   },
   {
    "name": "Laurent Dinh",
    "id": "46573521",
    "h_index": 21,
    "papers": 33
   },
   {
    "name": "Serkan Cabi",
    "id": "12159303",
    "h_index": 19,
    "papers": 29
   },
   {
    "name": "Thomas Roth\u00f6rl",
    "id": "8282805",
    "h_index": 7,
    "papers": 9
   },
   {
    "name": "Sergio Gomez Colmenarejo",
    "id": "2016840",
    "h_index": 21,
    "papers": 24
   },
   {
    "name": "Alistair Muldal",
    "id": "50654556",
    "h_index": 12,
    "papers": 15
   },
   {
    "name": "Tom Erez",
    "id": "1968210",
    "h_index": 26,
    "papers": 52
   },
   {
    "name": "Yuval Tassa",
    "id": "2109481",
    "h_index": 39,
    "papers": 67
   },
   {
    "name": "Nando de Freitas",
    "id": "1737568",
    "h_index": 79,
    "papers": 193
   },
   {
    "name": "Misha Denil",
    "id": "1715051",
    "h_index": 31,
    "papers": 51
   }
  ],
  "comment": "Accepted to ICLR 2018",
  "topics": [
   "dexterous-manipulation",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1804.06318v1",
  "pdf_url": "https://arxiv.org/pdf/1804.06318v1",
  "html_url": "https://arxiv.org/html/1804.06318v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.2
 },
 {
  "id": "1804.02748",
  "slug": "scaling-egocentric-vision-the-epic-kitchens-dataset",
  "title": "Scaling Egocentric Vision: The EPIC-KITCHENS Dataset",
  "abstract": "First-person vision is gaining interest as it offers a unique viewpoint on people's interaction with objects, their attention, and even intention. However, progress in this challenging domain has been relatively slow due to the lack of sufficiently large datasets. In this paper, we introduce EPIC-KITCHENS, a large-scale egocentric video benchmark recorded by 32 participants in their native kitchen environments. Our videos depict nonscripted daily activities: we simply asked each participant to start recording every time they entered their kitchen. Recording took place in 4 cities (in North America and Europe) by participants belonging to 10 different nationalities, resulting in highly diverse cooking styles. Our dataset features 55 hours of video consisting of 11.5M frames, which we densely labeled for a total of 39.6K action segments and 454.3K object bounding boxes. Our annotation is unique in that we had the participants narrate their own videos (after recording), thus reflecting true intention, and we crowd-sourced ground-truths based on these. We describe our object, action and anticipation challenges, and evaluate several baselines over two test splits, seen and unseen kitchens. Dataset and Project page: http://epic-kitchens.github.io",
  "published": "2018-04-08",
  "updated": "2018-07-31",
  "year": "2018",
  "authors": [
   "Dima Damen",
   "Hazel Doughty",
   "Giovanni Maria Farinella",
   "Sanja Fidler",
   "Antonino Furnari",
   "Evangelos Kazakos",
   "Davide Moltisanti",
   "Jonathan Munro",
   "Toby Perrett",
   "Will Price",
   "Michael Wray"
  ],
  "author_count": 11,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "ECCV 2018",
  "venue_source": "arxiv-comment",
  "citations": 1400,
  "influential_citations": 204,
  "tldr": "This paper introduces EPIC-KITCHENS, a large-scale egocentric video benchmark recorded by 32 participants in their native kitchen environments, and had the participants narrate their own videos (after recording), thus reflecting true intention, and crowd-sourced ground-truths based on these.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "D. Damen",
    "id": "145089978",
    "h_index": 45,
    "papers": 201
   },
   {
    "name": "Hazel Doughty",
    "id": "28798386",
    "h_index": 17,
    "papers": 52
   },
   {
    "name": "G. Farinella",
    "id": "1729739",
    "h_index": 37,
    "papers": 320
   },
   {
    "name": "S. Fidler",
    "id": "37895334",
    "h_index": 95,
    "papers": 236
   },
   {
    "name": "Antonino Furnari",
    "id": "1792681",
    "h_index": 29,
    "papers": 152
   },
   {
    "name": "E. Kazakos",
    "id": "12387007",
    "h_index": 19,
    "papers": 80
   },
   {
    "name": "D. Moltisanti",
    "id": "3420479",
    "h_index": 10,
    "papers": 24
   },
   {
    "name": "Jonathan Munro",
    "id": "47077615",
    "h_index": 8,
    "papers": 19
   },
   {
    "name": "Toby Perrett",
    "id": "2682004",
    "h_index": 12,
    "papers": 33
   },
   {
    "name": "Will Price",
    "id": "50065546",
    "h_index": 8,
    "papers": 17
   },
   {
    "name": "Michael Wray",
    "id": "145032628",
    "h_index": 15,
    "papers": 51
   }
  ],
  "comment": "European Conference on Computer Vision (ECCV) 2018 Dataset and Project page: http://epic-kitchens.github.io",
  "topics": [
   "egocentric-data",
   "data-teleop"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1804.02748v2",
  "pdf_url": "https://arxiv.org/pdf/1804.02748v2",
  "html_url": "https://arxiv.org/html/1804.02748v2",
  "code_url": "https://epic-kitchens.github.io",
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.5
 },
 {
  "id": "1804.00645",
  "slug": "universal-planning-networks",
  "title": "Universal Planning Networks",
  "abstract": "A key challenge in complex visuomotor control is learning abstract representations that are effective for specifying goals, planning, and generalization. To this end, we introduce universal planning networks (UPN). UPNs embed differentiable planning within a goal-directed policy. This planning computation unrolls a forward model in a latent space and infers an optimal action plan through gradient descent trajectory optimization. The plan-by-gradient-descent process and its underlying representations are learned end-to-end to directly optimize a supervised imitation learning objective. We find that the representations learned are not only effective for goal-directed visual imitation via gradient-based trajectory optimization, but can also provide a metric for specifying goals using images. The learned representations can be leveraged to specify distance-based rewards to reach new target states for model-free reinforcement learning, resulting in substantially more effective learning when solving new tasks described via image-based goals. We were able to achieve successful transfer of visuomotor planning strategies across robots with significantly different morphologies and actuation capabilities.",
  "published": "2018-04-02",
  "updated": "2018-04-04",
  "year": "2018",
  "authors": [
   "Aravind Srinivas",
   "Allan Jabri",
   "Pieter Abbeel",
   "Sergey Levine",
   "Chelsea Finn"
  ],
  "author_count": 5,
  "categories": [
   "cs.LG",
   "cs.AI",
   "cs.CV",
   "cs.RO",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "ICML",
  "venue_source": "semantic-scholar",
  "citations": 147,
  "influential_citations": 9,
  "tldr": "This work finds that the representations learned are not only effective for goal-directed visual imitation via gradient-based trajectory optimization, but can also provide a metric for specifying goals using images.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "A. Srinivas",
    "id": "41207614",
    "h_index": 20,
    "papers": 35
   },
   {
    "name": "A. Jabri",
    "id": "14258597",
    "h_index": 16,
    "papers": 28
   },
   {
    "name": "P. Abbeel",
    "id": "1689992",
    "h_index": 165,
    "papers": 596
   },
   {
    "name": "S. Levine",
    "id": "1736651",
    "h_index": 167,
    "papers": 551
   },
   {
    "name": "Chelsea Finn",
    "id": "46881670",
    "h_index": 94,
    "papers": 240
   }
  ],
  "comment": "Videos available at https://sites.google.com/view/upn-public/home",
  "topics": [
   "imitation-diffusion",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1804.00645v2",
  "pdf_url": "https://arxiv.org/pdf/1804.00645v2",
  "html_url": "https://arxiv.org/html/1804.00645v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.67
 },
 {
  "id": "1803.11496",
  "slug": "predicting-future-instance-segmentation-by-forecasting-convolutional-f",
  "title": "Predicting Future Instance Segmentation by Forecasting Convolutional Features",
  "abstract": "Anticipating future events is an important prerequisite towards intelligent behavior. Video forecasting has been studied as a proxy task towards this goal. Recent work has shown that to predict semantic segmentation of future frames, forecasting at the semantic level is more effective than forecasting RGB frames and then segmenting these. In this paper we consider the more challenging problem of future instance segmentation, which additionally segments out individual objects. To deal with a varying number of output labels per image, we develop a predictive model in the space of fixed-sized convolutional features of the Mask R-CNN instance segmentation model. We apply the \"detection head'\" of Mask R-CNN on the predicted features to produce the instance segmentation of future frames. Experiments show that this approach significantly improves over strong baselines based on optical flow and repurposed instance segmentation architectures.",
  "published": "2018-03-30",
  "updated": "2018-10-03",
  "year": "2018",
  "authors": [
   "Pauline Luc",
   "Camille Couprie",
   "Yann LeCun",
   "Jakob Verbeek"
  ],
  "author_count": 4,
  "categories": [
   "cs.CV"
  ],
  "primary_category": "cs.CV",
  "venue": "ECCV",
  "venue_source": "semantic-scholar",
  "citations": 100,
  "influential_citations": 21,
  "tldr": "A predictive model is developed in the space of fixed-sized convolutional features of the Mask R-CNN instance segmentation model that significantly improves over baselines based on optical flow.",
  "doi": "10.1007/978-3-030-01240-3_36",
  "oa_pdf": "https://hal.inria.fr/hal-01757669/file/luc18eccv_arxiv.pdf",
  "s2_authors": [
   {
    "name": "Pauline Luc",
    "id": "152831141",
    "h_index": 11,
    "papers": 17
   },
   {
    "name": "C. Couprie",
    "id": "2341378",
    "h_index": 23,
    "papers": 50
   },
   {
    "name": "Yann LeCun",
    "id": "1688882",
    "h_index": 139,
    "papers": 406
   },
   {
    "name": "Jakob Verbeek",
    "id": "34602236",
    "h_index": 26,
    "papers": 61
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1803.11496v2",
  "pdf_url": "https://arxiv.org/pdf/1803.11496v2",
  "html_url": "https://arxiv.org/html/1803.11496v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.5
 },
 {
  "id": "1803.10760",
  "slug": "unsupervised-predictive-memory-in-a-goal-directed-agent",
  "title": "Unsupervised Predictive Memory in a Goal-Directed Agent",
  "abstract": "Animals execute goal-directed behaviours despite the limited range and scope of their sensors. To cope, they explore environments and store memories maintaining estimates of important information that is not presently available. Recently, progress has been made with artificial intelligence (AI) agents that learn to perform tasks from sensory input, even at a human level, by merging reinforcement learning (RL) algorithms with deep neural networks, and the excitement surrounding these results has led to the pursuit of related ideas as explanations of non-human animal learning. However, we demonstrate that contemporary RL algorithms struggle to solve simple tasks when enough information is concealed from the sensors of the agent, a property called \"partial observability\". An obvious requirement for handling partially observed tasks is access to extensive memory, but we show memory is not enough; it is critical that the right information be stored in the right format. We develop a model, the Memory, RL, and Inference Network (MERLIN), in which memory formation is guided by a process of predictive modeling. MERLIN facilitates the solution of tasks in 3D virtual reality environments for which partial observability is severe and memories must be maintained over long durations. Our model demonstrates a single learning agent architecture that can solve canonical behavioural tasks in psychology and neurobiology without strong simplifying assumptions about the dimensionality of sensory input or the duration of experiences.",
  "published": "2018-03-28",
  "updated": "2018-03-28",
  "year": "2018",
  "authors": [
   "Greg Wayne",
   "Chia-Chun Hung",
   "David Amos",
   "Mehdi Mirza",
   "Arun Ahuja",
   "Agnieszka Grabska-Barwinska",
   "Jack Rae",
   "Piotr Mirowski",
   "Joel Z. Leibo",
   "Adam Santoro",
   "Mevlana Gemici",
   "Malcolm Reynolds",
   "Tim Harley",
   "Josh Abramson",
   "Shakir Mohamed",
   "Danilo Rezende",
   "David Saxton",
   "Adam Cain",
   "Chloe Hillier",
   "David Silver",
   "Koray Kavukcuoglu",
   "Matt Botvinick",
   "Demis Hassabis",
   "Timothy Lillicrap"
  ],
  "author_count": 24,
  "categories": [
   "cs.LG",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 206,
  "influential_citations": 14,
  "tldr": "A model, the Memory, RL, and Inference Network (MERLIN), in which memory formation is guided by a process of predictive modeling, demonstrates a single learning agent architecture that can solve canonical behavioural tasks in psychology and neurobiology without strong simplifying assumptions about the dimensionality of sensory input or the duration of experiences.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Greg Wayne",
    "id": "89504302",
    "h_index": 33,
    "papers": 47
   },
   {
    "name": "Chia-Chun Hung",
    "id": "2143897443",
    "h_index": 5,
    "papers": 6
   },
   {
    "name": "David Amos",
    "id": "2064400086",
    "h_index": 7,
    "papers": 7
   },
   {
    "name": "Mehdi Mirza",
    "id": "153583218",
    "h_index": 18,
    "papers": 27
   },
   {
    "name": "Arun Ahuja",
    "id": "37968006",
    "h_index": 22,
    "papers": 49
   },
   {
    "name": "A. Grabska-Barwinska",
    "id": "1398898827",
    "h_index": 15,
    "papers": 24
   },
   {
    "name": "Jack W. Rae",
    "id": "34269227",
    "h_index": 24,
    "papers": 32
   },
   {
    "name": "Piotr Wojciech Mirowski",
    "id": "144705062",
    "h_index": 28,
    "papers": 58
   },
   {
    "name": "Joel Z. Leibo",
    "id": "1700356",
    "h_index": 46,
    "papers": 145
   },
   {
    "name": "Adam Santoro",
    "id": "35030998",
    "h_index": 38,
    "papers": 53
   },
   {
    "name": "Mevlana Gemici",
    "id": "2730646",
    "h_index": 6,
    "papers": 7
   },
   {
    "name": "Malcolm Reynolds",
    "id": "47447264",
    "h_index": 11,
    "papers": 12
   },
   {
    "name": "Tim Harley",
    "id": "3367786",
    "h_index": 13,
    "papers": 17
   },
   {
    "name": "Josh Abramson",
    "id": "3041463",
    "h_index": 7,
    "papers": 8
   },
   {
    "name": "S. Mohamed",
    "id": "14594344",
    "h_index": 38,
    "papers": 61
   },
   {
    "name": "Danilo Jimenez Rezende",
    "id": "1748523",
    "h_index": 47,
    "papers": 82
   },
   {
    "name": "D. Saxton",
    "id": "143810408",
    "h_index": 13,
    "papers": 24
   },
   {
    "name": "Adam Cain",
    "id": "2055913310",
    "h_index": 3,
    "papers": 4
   },
   {
    "name": "Chloe Hillier",
    "id": "38961760",
    "h_index": 6,
    "papers": 6
   },
   {
    "name": "David Silver",
    "id": "145824029",
    "h_index": 80,
    "papers": 120
   },
   {
    "name": "K. Kavukcuoglu",
    "id": "2645384",
    "h_index": 76,
    "papers": 124
   },
   {
    "name": "M. Botvinick",
    "id": "46378362",
    "h_index": 94,
    "papers": 226
   },
   {
    "name": "D. Hassabis",
    "id": "48987704",
    "h_index": 92,
    "papers": 160
   },
   {
    "name": "T. Lillicrap",
    "id": "2542999",
    "h_index": 69,
    "papers": 154
   }
  ],
  "comment": "",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1803.10760v1",
  "pdf_url": "https://arxiv.org/pdf/1803.10760v1",
  "html_url": "https://arxiv.org/html/1803.10760v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.32
 },
 {
  "id": "1803.10122",
  "slug": "world-models",
  "title": "World Models",
  "abstract": "We explore building generative neural network models of popular reinforcement learning environments. Our world model can be trained quickly in an unsupervised manner to learn a compressed spatial and temporal representation of the environment. By using features extracted from the world model as inputs to an agent, we can train a very compact and simple policy that can solve the required task. We can even train our agent entirely inside of its own hallucinated dream generated by its world model, and transfer this policy back into the actual environment. An interactive version of this paper is available at https://worldmodels.github.io/",
  "published": "2018-03-27",
  "updated": "2018-05-09",
  "year": "2018",
  "authors": [
   "David Ha",
   "J\u00fcrgen Schmidhuber"
  ],
  "author_count": 2,
  "categories": [
   "cs.LG",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 1987,
  "influential_citations": 99,
  "tldr": "This work explores building generative neural network models of popular reinforcement learning environments by using features extracted from the world model as inputs to an agent, and can train a very compact and simple policy that can solve the required task.",
  "doi": "10.5281/zenodo.1207631",
  "oa_pdf": "http://arxiv.org/pdf/1803.10122",
  "s2_authors": [
   {
    "name": "David R Ha",
    "id": "39810222",
    "h_index": 20,
    "papers": 92
   },
   {
    "name": "J. Schmidhuber",
    "id": "145341374",
    "h_index": 101,
    "papers": 467
   }
  ],
  "comment": "",
  "topics": [
   "world-models",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1803.10122v4",
  "pdf_url": "https://arxiv.org/pdf/1803.10122v4",
  "html_url": "https://arxiv.org/html/1803.10122v4",
  "code_url": null,
  "presented": [
   {
    "club": "saturday-robotics",
    "club_name": "Saturday Robotics",
    "session_number": 7,
    "session_title": "Robotics & World Models Reading Club 07: Learning to Dream: World Models, Imagination, Path to Foundation Models for Control \u2014 Los Altos",
    "date_text": "Saturday, May 9, 2026 | 2:00 PM \u2013 5:00 PM",
    "city": "",
    "url": "https://lu.ma/srhe0vuo",
    "listed_as": "Ha & Schmidhuber \u2014 World Models (2018)"
   }
  ],
  "club_note": "Introduced the core idea of learning a compressed latent simulator for control",
  "featured": true,
  "signal": 6.0
 },
 {
  "id": "1803.10081",
  "slug": "deepjdot-deep-joint-distribution-optimal-transport-for-unsupervised-do",
  "title": "DeepJDOT: Deep Joint Distribution Optimal Transport for Unsupervised Domain Adaptation",
  "abstract": "In computer vision, one is often confronted with problems of domain shifts, which occur when one applies a classifier trained on a source dataset to target data sharing similar characteristics (e.g. same classes), but also different latent data structures (e.g. different acquisition conditions). In such a situation, the model will perform poorly on the new data, since the classifier is specialized to recognize visual cues specific to the source domain. In this work we explore a solution, named DeepJDOT, to tackle this problem: through a measure of discrepancy on joint deep representations/labels based on optimal transport, we not only learn new data representations aligned between the source and target domain, but also simultaneously preserve the discriminative information used by the classifier. We applied DeepJDOT to a series of visual recognition tasks, where it compares favorably against state-of-the-art deep domain adaptation methods.",
  "published": "2018-03-27",
  "updated": "2018-09-05",
  "year": "2018",
  "authors": [
   "Bharath Bhushan Damodaran",
   "Benjamin Kellenberger",
   "R\u00e9mi Flamary",
   "Devis Tuia",
   "Nicolas Courty"
  ],
  "author_count": 5,
  "categories": [
   "cs.CV",
   "cs.AI"
  ],
  "primary_category": "cs.CV",
  "venue": "ECCV",
  "venue_source": "semantic-scholar",
  "citations": 552,
  "influential_citations": 72,
  "tldr": "Through a measure of discrepancy on joint deep representations/labels based on optimal transport, this work not only learns new data representations aligned between the source and target domain, but also simultaneously preserve the discriminative information used by the classifier.",
  "doi": "10.1007/978-3-030-01225-0_28",
  "oa_pdf": "https://research.wur.nl/en/publications/deepjdot-deep-joint-distribution-optimal-transport-for-unsupervis",
  "s2_authors": [
   {
    "name": "B. Damodaran",
    "id": "8162004",
    "h_index": 12,
    "papers": 33
   },
   {
    "name": "B. Kellenberger",
    "id": "31394603",
    "h_index": 18,
    "papers": 46
   },
   {
    "name": "R\u00e9mi Flamary",
    "id": "145258331",
    "h_index": 33,
    "papers": 130
   },
   {
    "name": "D. Tuia",
    "id": "2977931",
    "h_index": 68,
    "papers": 400
   },
   {
    "name": "N. Courty",
    "id": "145374281",
    "h_index": 36,
    "papers": 139
   }
  ],
  "comment": "European Conference on Computer Vision 2018 (ECCV-2018)",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1803.10081v3",
  "pdf_url": "https://arxiv.org/pdf/1803.10081v3",
  "html_url": "https://arxiv.org/html/1803.10081v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.24
 },
 {
  "id": "1803.07616",
  "slug": "intphys-a-framework-and-benchmark-for-visual-intuitive-physics-reasoni",
  "title": "IntPhys: A Framework and Benchmark for Visual Intuitive Physics Reasoning",
  "abstract": "In order to reach human performance on complexvisual tasks, artificial systems need to incorporate a sig-nificant amount of understanding of the world in termsof macroscopic objects, movements, forces, etc. Inspiredby work on intuitive physics in infants, we propose anevaluation benchmark which diagnoses how much a givensystem understands about physics by testing whether itcan tell apart well matched videos of possible versusimpossible events constructed with a game engine. Thetest requires systems to compute a physical plausibilityscore over an entire video. It is free of bias and cantest a range of basic physical reasoning concepts. Wethen describe two Deep Neural Networks systems aimedat learning intuitive physics in an unsupervised way,using only physically possible videos. The systems aretrained with a future semantic mask prediction objectiveand tested on the possible versus impossible discrimi-nation task. The analysis of their results compared tohuman data gives novel insights in the potentials andlimitations of next frame prediction architectures.",
  "published": "2018-03-20",
  "updated": "2020-02-11",
  "year": "2018",
  "authors": [
   "Ronan Riochet",
   "Mario Ynocente Castro",
   "Mathieu Bernard",
   "Adam Lerer",
   "Rob Fergus",
   "V\u00e9ronique Izard",
   "Emmanuel Dupoux"
  ],
  "author_count": 7,
  "categories": [
   "cs.AI",
   "cs.CV"
  ],
  "primary_category": "cs.AI",
  "venue": "",
  "venue_source": "",
  "citations": 91,
  "influential_citations": 14,
  "tldr": "An evaluation benchmark which diagnoses how much a given system understands about physics by testing whether it can tell apart well matched videos of possible versus impossible events constructed with a game engine is proposed.",
  "doi": "10.1109/TPAMI.2021.3083839",
  "oa_pdf": "https://ieeexplore.ieee.org/ielx7/34/9893110/09442261.pdf",
  "s2_authors": [
   {
    "name": "Ronan Riochet",
    "id": "40895941",
    "h_index": 4,
    "papers": 8
   },
   {
    "name": "M. Castro",
    "id": "40901792",
    "h_index": 6,
    "papers": 9
   },
   {
    "name": "Mathieu Bernard",
    "id": "2065529954",
    "h_index": 9,
    "papers": 14
   },
   {
    "name": "Adam Lerer",
    "id": "1977806",
    "h_index": 30,
    "papers": 69
   },
   {
    "name": "R. Fergus",
    "id": "2276554",
    "h_index": 78,
    "papers": 125
   },
   {
    "name": "V\u00e9ronique Izard",
    "id": "144186731",
    "h_index": 29,
    "papers": 65
   },
   {
    "name": "Emmanuel Dupoux",
    "id": "2202008",
    "h_index": 68,
    "papers": 286
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1803.07616v3",
  "pdf_url": "https://arxiv.org/pdf/1803.07616v3",
  "html_url": "https://arxiv.org/html/1803.07616v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.96
 },
 {
  "id": "1803.02291",
  "slug": "synthesizing-neural-network-controllers-with-probabilistic-model-based",
  "title": "Synthesizing Neural Network Controllers with Probabilistic Model based Reinforcement Learning",
  "abstract": "We present an algorithm for rapidly learning controllers for robotics systems. The algorithm follows the model-based reinforcement learning paradigm, and improves upon existing algorithms; namely Probabilistic learning in Control (PILCO) and a sample-based version of PILCO with neural network dynamics (Deep-PILCO). We propose training a neural network dynamics model using variational dropout with truncated Log-Normal noise. This allows us to obtain a dynamics model with calibrated uncertainty, which can be used to simulate controller executions via rollouts. We also describe set of techniques, inspired by viewing PILCO as a recurrent neural network model, that are crucial to improve the convergence of the method. We test our method on a variety of benchmark tasks, demonstrating data-efficiency that is competitive with PILCO, while being able to optimize complex neural network controllers. Finally, we assess the performance of the algorithm for learning motor controllers for a six legged autonomous underwater vehicle. This demonstrates the potential of the algorithm for scaling up the dimensionality and dataset sizes, in more complex control tasks.",
  "published": "2018-03-06",
  "updated": "2018-08-01",
  "year": "2018",
  "authors": [
   "Juan Camilo Gamboa Higuera",
   "David Meger",
   "Gregory Dudek"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO",
   "cs.AI"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS",
  "venue_source": "semantic-scholar",
  "citations": 41,
  "influential_citations": 2,
  "tldr": "An algorithm for rapidly learning neural network policies for robotics systems that follows the model-based reinforcement learning paradigm and improves upon existing algorithms: PILeO and a sample-based version of PILeo with neural network dynamics (Deep-PILeO).",
  "doi": "10.1109/IROS.2018.8594018",
  "oa_pdf": "https://arxiv.org/pdf/1803.02291",
  "s2_authors": [
   {
    "name": "J. A. G. Higuera",
    "id": "2550467",
    "h_index": 14,
    "papers": 24
   },
   {
    "name": "D. Meger",
    "id": "2462512",
    "h_index": 29,
    "papers": 116
   },
   {
    "name": "G. Dudek",
    "id": "1798721",
    "h_index": 48,
    "papers": 315
   }
  ],
  "comment": "8 pages, 7 figures",
  "topics": [
   "humanoids",
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1803.02291v3",
  "pdf_url": "https://arxiv.org/pdf/1803.02291v3",
  "html_url": "https://arxiv.org/html/1803.02291v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.12
 },
 {
  "id": "1803.01940",
  "slug": "tactile-regrasp-grasp-adjustments-via-simulated-tactile-transformation",
  "title": "Tactile Regrasp: Grasp Adjustments via Simulated Tactile Transformations",
  "abstract": "This paper presents a novel regrasp control policy that makes use of tactile sensing to plan local grasp adjustments. Our approach determines regrasp actions by virtually searching for local transformations of tactile measurements that improve the quality of the grasp. First, we construct a tactile-based grasp quality metric using a deep convolutional neural network trained on over 2800 grasps. The quality of each grasp, a continuous value between 0 and 1, is determined experimentally by measuring its resistance to external perturbations. Second, we simulate the tactile imprints associated with robot motions relative to the initial grasp by performing rigid-body transformations of the given tactile measurements. The newly generated tactile imprints are evaluated with the learned grasp quality network and the regrasp action is chosen to maximize the grasp quality. Results show that the grasp quality network can predict the outcome of grasps with an average accuracy of 85% on known objects and 75% on a cross validation set of 12 objects. The regrasp control policy improves the success rate of grasp actions by an average relative increase of 70% on a test set of 8 objects.",
  "published": "2018-03-05",
  "updated": "2018-10-09",
  "year": "2018",
  "authors": [
   "Francois R. Hogan",
   "Maria Bauza",
   "Oleguer Canal",
   "Elliott Donlon",
   "Alberto Rodriguez"
  ],
  "author_count": 5,
  "categories": [
   "cs.RO",
   "eess.SY"
  ],
  "primary_category": "cs.RO",
  "venue": "IROS",
  "venue_source": "semantic-scholar",
  "citations": 115,
  "influential_citations": 2,
  "tldr": "A novel regrasp control policy that makes use of tactile sensing to plan local grasp adjustments by virtually searching for local transformations of tactile measurements that improve the quality of the grasp is presented.",
  "doi": "10.1109/IROS.2018.8593528",
  "oa_pdf": "https://arxiv.org/pdf/1803.01940",
  "s2_authors": [
   {
    "name": "F. Hogan",
    "id": "15820310",
    "h_index": 14,
    "papers": 28
   },
   {
    "name": "Maria Bauz\u00e1",
    "id": "93223375",
    "h_index": 19,
    "papers": 30
   },
   {
    "name": "Oleguer Canal",
    "id": "46957461",
    "h_index": 2,
    "papers": 2
   },
   {
    "name": "E. Donlon",
    "id": "39425333",
    "h_index": 7,
    "papers": 9
   },
   {
    "name": "Alberto Rodriguez",
    "id": "152532021",
    "h_index": 49,
    "papers": 113
   }
  ],
  "comment": "Francois R. Hogan and Maria Bauza contributed equally to this work. 8 pages, 7 figures",
  "topics": [
   "dexterous-manipulation",
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1803.01940v2",
  "pdf_url": "https://arxiv.org/pdf/1803.01940v2",
  "html_url": "https://arxiv.org/html/1803.01940v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.56
 },
 {
  "id": "1803.00933",
  "slug": "distributed-prioritized-experience-replay",
  "title": "Distributed Prioritized Experience Replay",
  "abstract": "We propose a distributed architecture for deep reinforcement learning at scale, that enables agents to learn effectively from orders of magnitude more data than previously possible. The algorithm decouples acting from learning: the actors interact with their own instances of the environment by selecting actions according to a shared neural network, and accumulate the resulting experience in a shared experience replay memory; the learner replays samples of experience and updates the neural network. The architecture relies on prioritized experience replay to focus only on the most significant data generated by the actors. Our architecture substantially improves the state of the art on the Arcade Learning Environment, achieving better final performance in a fraction of the wall-clock training time.",
  "published": "2018-03-02",
  "updated": "2018-03-02",
  "year": "2018",
  "authors": [
   "Dan Horgan",
   "John Quan",
   "David Budden",
   "Gabriel Barth-Maron",
   "Matteo Hessel",
   "Hado van Hasselt",
   "David Silver"
  ],
  "author_count": 7,
  "categories": [
   "cs.LG"
  ],
  "primary_category": "cs.LG",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 838,
  "influential_citations": 95,
  "tldr": "This work proposes a distributed architecture for deep reinforcement learning at scale, that enables agents to learn effectively from orders of magnitude more data than previously possible, and substantially improves the state of the art on the Arcade Learning Environment.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Dan Horgan",
    "id": "48257711",
    "h_index": 11,
    "papers": 18
   },
   {
    "name": "John Quan",
    "id": "34660073",
    "h_index": 16,
    "papers": 22
   },
   {
    "name": "D. Budden",
    "id": "2508525",
    "h_index": 27,
    "papers": 73
   },
   {
    "name": "Gabriel Barth-Maron",
    "id": "1403998955",
    "h_index": 15,
    "papers": 29
   },
   {
    "name": "Matteo Hessel",
    "id": "39357484",
    "h_index": 25,
    "papers": 39
   },
   {
    "name": "H. V. Hasselt",
    "id": "7634925",
    "h_index": 40,
    "papers": 67
   },
   {
    "name": "David Silver",
    "id": "145824029",
    "h_index": 80,
    "papers": 120
   }
  ],
  "comment": "Accepted to International Conference on Learning Representations 2018",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1803.00933v1",
  "pdf_url": "https://arxiv.org/pdf/1803.00933v1",
  "html_url": "https://arxiv.org/html/1803.00933v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.42
 },
 {
  "id": "1803.00101",
  "slug": "model-based-value-estimation-for-efficient-model-free-reinforcement-le",
  "title": "Model-Based Value Estimation for Efficient Model-Free Reinforcement Learning",
  "abstract": "Recent model-free reinforcement learning algorithms have proposed incorporating learned dynamics models as a source of additional data with the intention of reducing sample complexity. Such methods hold the promise of incorporating imagined data coupled with a notion of model uncertainty to accelerate the learning of continuous control tasks. Unfortunately, they rely on heuristics that limit usage of the dynamics model. We present model-based value expansion, which controls for uncertainty in the model by only allowing imagination to fixed depth. By enabling wider use of learned dynamics models within a model-free reinforcement learning algorithm, we improve value estimation, which, in turn, reduces the sample complexity of learning.",
  "published": "2018-02-28",
  "updated": "2018-02-28",
  "year": "2018",
  "authors": [
   "Vladimir Feinberg",
   "Alvin Wan",
   "Ion Stoica",
   "Michael I. Jordan",
   "Joseph E. Gonzalez",
   "Sergey Levine"
  ],
  "author_count": 6,
  "categories": [
   "cs.LG",
   "cs.AI",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 355,
  "influential_citations": 37,
  "tldr": "By enabling wider use of learned dynamics models within a model-free reinforcement learning algorithm, this work improves value estimation, which, in turn, reduces the sample complexity of learning.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Vladimir Feinberg",
    "id": "2275181199",
    "h_index": 11,
    "papers": 38
   },
   {
    "name": "Alvin Wan",
    "id": "144546548",
    "h_index": 12,
    "papers": 18
   },
   {
    "name": "Ion Stoica",
    "id": "2055174324",
    "h_index": 41,
    "papers": 94
   },
   {
    "name": "Michael I. Jordan",
    "id": "1694621",
    "h_index": 188,
    "papers": 900
   },
   {
    "name": "Joseph E. Gonzalez",
    "id": "49988044",
    "h_index": 68,
    "papers": 221
   },
   {
    "name": "S. Levine",
    "id": "1736651",
    "h_index": 167,
    "papers": 551
   }
  ],
  "comment": "",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1803.00101v1",
  "pdf_url": "https://arxiv.org/pdf/1803.00101v1",
  "html_url": "https://arxiv.org/html/1803.00101v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.55
 },
 {
  "id": "1802.10592",
  "slug": "model-ensemble-trust-region-policy-optimization",
  "title": "Model-Ensemble Trust-Region Policy Optimization",
  "abstract": "Model-free reinforcement learning (RL) methods are succeeding in a growing number of tasks, aided by recent advances in deep learning. However, they tend to suffer from high sample complexity, which hinders their use in real-world domains. Alternatively, model-based reinforcement learning promises to reduce sample complexity, but tends to require careful tuning and to date have succeeded mainly in restrictive domains where simple models are sufficient for learning. In this paper, we analyze the behavior of vanilla model-based reinforcement learning methods when deep neural networks are used to learn both the model and the policy, and show that the learned policy tends to exploit regions where insufficient data is available for the model to be learned, causing instability in training. To overcome this issue, we propose to use an ensemble of models to maintain the model uncertainty and regularize the learning process. We further show that the use of likelihood ratio derivatives yields much more stable learning than backpropagation through time. Altogether, our approach Model-Ensemble Trust-Region Policy Optimization (ME-TRPO) significantly reduces the sample complexity compared to model-free deep RL methods on challenging continuous control benchmark tasks.",
  "published": "2018-02-28",
  "updated": "2018-10-05",
  "year": "2018",
  "authors": [
   "Thanard Kurutach",
   "Ignasi Clavera",
   "Yan Duan",
   "Aviv Tamar",
   "Pieter Abbeel"
  ],
  "author_count": 5,
  "categories": [
   "cs.LG",
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.LG",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 490,
  "influential_citations": 46,
  "tldr": "This paper analyzes the behavior of vanilla model-based reinforcement learning methods when deep neural networks are used to learn both the model and the policy, and shows that the learned policy tends to exploit regions where insufficient data is available for the model to be learned, causing instability in training.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Thanard Kurutach",
    "id": "2765564",
    "h_index": 9,
    "papers": 11
   },
   {
    "name": "Ignasi Clavera",
    "id": "15593386",
    "h_index": 14,
    "papers": 15
   },
   {
    "name": "Yan Duan",
    "id": "144581158",
    "h_index": 24,
    "papers": 34
   },
   {
    "name": "Aviv Tamar",
    "id": "3025260",
    "h_index": 40,
    "papers": 102
   },
   {
    "name": "P. Abbeel",
    "id": "1689992",
    "h_index": 165,
    "papers": 596
   }
  ],
  "comment": "",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1802.10592v2",
  "pdf_url": "https://arxiv.org/pdf/1802.10592v2",
  "html_url": "https://arxiv.org/html/1802.10592v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.19
 },
 {
  "id": "1802.10567",
  "slug": "learning-by-playing-solving-sparse-reward-tasks-from-scratch",
  "title": "Learning by Playing - Solving Sparse Reward Tasks from Scratch",
  "abstract": "We propose Scheduled Auxiliary Control (SAC-X), a new learning paradigm in the context of Reinforcement Learning (RL). SAC-X enables learning of complex behaviors - from scratch - in the presence of multiple sparse reward signals. To this end, the agent is equipped with a set of general auxiliary tasks, that it attempts to learn simultaneously via off-policy RL. The key idea behind our method is that active (learned) scheduling and execution of auxiliary policies allows the agent to efficiently explore its environment - enabling it to excel at sparse reward RL. Our experiments in several challenging robotic manipulation settings demonstrate the power of our approach.",
  "published": "2018-02-28",
  "updated": "2018-02-28",
  "year": "2018",
  "authors": [
   "Martin Riedmiller",
   "Roland Hafner",
   "Thomas Lampe",
   "Michael Neunert",
   "Jonas Degrave",
   "Tom Van de Wiele",
   "Volodymyr Mnih",
   "Nicolas Heess",
   "Jost Tobias Springenberg"
  ],
  "author_count": 9,
  "categories": [
   "cs.LG",
   "cs.RO",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "ICML",
  "venue_source": "semantic-scholar",
  "citations": 516,
  "influential_citations": 24,
  "tldr": "The key idea behind the method is that active (learned) scheduling and execution of auxiliary policies allows the agent to efficiently explore its environment - enabling it to excel at sparse reward RL.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Martin A. Riedmiller",
    "id": "3137672",
    "h_index": 55,
    "papers": 219
   },
   {
    "name": "Roland Hafner",
    "id": "49512734",
    "h_index": 21,
    "papers": 36
   },
   {
    "name": "Thomas Lampe",
    "id": "2066153554",
    "h_index": 22,
    "papers": 36
   },
   {
    "name": "M. Neunert",
    "id": "2366050",
    "h_index": 29,
    "papers": 50
   },
   {
    "name": "Jonas Degrave",
    "id": "3110620",
    "h_index": 15,
    "papers": 33
   },
   {
    "name": "T. Wiele",
    "id": "8023592",
    "h_index": 35,
    "papers": 186
   },
   {
    "name": "Volodymyr Mnih",
    "id": "3255983",
    "h_index": 34,
    "papers": 47
   },
   {
    "name": "N. Heess",
    "id": "2801204",
    "h_index": 73,
    "papers": 192
   },
   {
    "name": "Jost Tobias Springenberg",
    "id": "2060551",
    "h_index": 44,
    "papers": 93
   }
  ],
  "comment": "A video of the rich set of learned behaviours can be found at https://youtu.be/mPKyvocNe_M",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1802.10567v1",
  "pdf_url": "https://arxiv.org/pdf/1802.10567v1",
  "html_url": "https://arxiv.org/html/1802.10567v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.21
 },
 {
  "id": "1802.10153",
  "slug": "slip-detection-with-combined-tactile-and-visual-information",
  "title": "Slip Detection with Combined Tactile and Visual Information",
  "abstract": "Slip detection plays a vital role in robotic manipulation and it has long been a challenging problem in the robotic community. In this paper, we propose a new method based on deep neural network (DNN) to detect slip. The training data is acquired by a GelSight tactile sensor and a camera mounted on a gripper when we use a robot arm to grasp and lift 94 daily objects with different grasping forces and grasping positions. The DNN is trained to classify whether a slip occurred or not. To evaluate the performance of the DNN, we test 10 unseen objects in 152 grasps. A detection accuracy as high as 88.03% is achieved. It is anticipated that the accuracy can be further improved with a larger dataset. This method is beneficial for robots to make stable grasps, which can be widely applied to automatic force control, grasping strategy selection and fine manipulation.",
  "published": "2018-02-27",
  "updated": "2018-02-27",
  "year": "2018",
  "authors": [
   "Jianhua Li",
   "Siyuan Dong",
   "Edward Adelson"
  ],
  "author_count": 3,
  "categories": [
   "cs.RO"
  ],
  "primary_category": "cs.RO",
  "venue": "ICRA",
  "venue_source": "semantic-scholar",
  "citations": 179,
  "influential_citations": 18,
  "tldr": "A new method based on deep neural network (DNN) to detect slip and make stable grasps, which can be widely applied to automatic force control, grasping strategy selection and fine manipulation.",
  "doi": "10.1109/ICRA.2018.8460495",
  "oa_pdf": "https://arxiv.org/pdf/1802.10153",
  "s2_authors": [
   {
    "name": "Jianhua Li",
    "id": "2116451956",
    "h_index": 9,
    "papers": 27
   },
   {
    "name": "Siyuan Dong",
    "id": "3308345",
    "h_index": 28,
    "papers": 42
   },
   {
    "name": "E. Adelson",
    "id": "145358192",
    "h_index": 87,
    "papers": 272
   }
  ],
  "comment": "International Conference on Robotics and Automation (ICRA) 2018",
  "topics": [
   "dexterous-manipulation",
   "tactile"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1802.10153v1",
  "pdf_url": "https://arxiv.org/pdf/1802.10153v1",
  "html_url": "https://arxiv.org/html/1802.10153v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.76
 },
 {
  "id": "1802.08864",
  "slug": "one-big-net-for-everything",
  "title": "One Big Net For Everything",
  "abstract": "I apply recent work on \"learning to think\" (2015) and on PowerPlay (2011) to the incremental training of an increasingly general problem solver, continually learning to solve new tasks without forgetting previous skills. The problem solver is a single recurrent neural network (or similar general purpose computer) called ONE. ONE is unusual in the sense that it is trained in various ways, e.g., by black box optimization / reinforcement learning / artificial evolution as well as supervised / unsupervised learning. For example, ONE may learn through neuroevolution to control a robot through environment-changing actions, and learn through unsupervised gradient descent to predict future inputs and vector-valued reward signals as suggested in 1990. User-given tasks can be defined through extra goal-defining input patterns, also proposed in 1990. Suppose ONE has already learned many skills. Now a copy of ONE can be re-trained to learn a new skill, e.g., through neuroevolution without a teacher. Here it may profit from re-using previously learned subroutines, but it may also forget previous skills. Then ONE is retrained in PowerPlay style (2011) on stored input/output traces of (a) ONE's copy executing the new skill and (b) previous instances of ONE whose skills are still considered worth memorizing. Simultaneously, ONE is retrained on old traces (even those of unsuccessful trials) to become a better predictor, without additional expensive interaction with the enviroment. More and more control and prediction skills are thus collapsed into ONE, like in the chunker-automatizer system of the neural history compressor (1991). This forces ONE to relate partially analogous skills (with shared algorithmic information) to each other, creating common subroutines in form of shared subnetworks of ONE, to greatly speed up subsequent learning of additional, novel but algorithmically related skills.",
  "published": "2018-02-24",
  "updated": "2018-02-24",
  "year": "2018",
  "authors": [
   "Juergen Schmidhuber"
  ],
  "author_count": 1,
  "categories": [
   "cs.AI"
  ],
  "primary_category": "cs.AI",
  "venue": "",
  "venue_source": "",
  "citations": 35,
  "influential_citations": 3,
  "tldr": "The incremental training of an increasingly general problem solver, continually learning to solve new tasks without forgetting previous skills is applied, to greatly speed up subsequent learning of additional, novel but algorithmically related skills.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "J. Schmidhuber",
    "id": "145341374",
    "h_index": 101,
    "papers": 467
   }
  ],
  "comment": "17 pages, 107 references",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1802.08864v1",
  "pdf_url": "https://arxiv.org/pdf/1802.08864v1",
  "html_url": "https://arxiv.org/html/1802.08864v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 1.56
 },
 {
  "id": "1802.07687",
  "slug": "stochastic-video-generation-with-a-learned-prior",
  "title": "Stochastic Video Generation with a Learned Prior",
  "abstract": "Generating video frames that accurately predict future world states is challenging. Existing approaches either fail to capture the full distribution of outcomes, or yield blurry generations, or both. In this paper we introduce an unsupervised video generation model that learns a prior model of uncertainty in a given environment. Video frames are generated by drawing samples from this prior and combining them with a deterministic estimate of the future frame. The approach is simple and easily trained end-to-end on a variety of datasets. Sample generations are both varied and sharp, even many frames into the future, and compare favorably to those from existing approaches.",
  "published": "2018-02-21",
  "updated": "2018-03-02",
  "year": "2018",
  "authors": [
   "Remi Denton",
   "Rob Fergus"
  ],
  "author_count": 2,
  "categories": [
   "cs.CV",
   "cs.AI",
   "cs.LG",
   "stat.ML"
  ],
  "primary_category": "cs.CV",
  "venue": "ICML",
  "venue_source": "semantic-scholar",
  "citations": 590,
  "influential_citations": 107,
  "tldr": "An unsupervised video generation model that learns a prior model of uncertainty in a given environment and generates video frames by drawing samples from this prior and combining them with a deterministic estimate of the future frame.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Emily L. Denton",
    "id": "40081727",
    "h_index": 25,
    "papers": 36
   },
   {
    "name": "R. Fergus",
    "id": "2276554",
    "h_index": 78,
    "papers": 125
   }
  ],
  "comment": "",
  "topics": [
   "video-generation"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1802.07687v2",
  "pdf_url": "https://arxiv.org/pdf/1802.07687v2",
  "html_url": "https://arxiv.org/html/1802.07687v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.27
 },
 {
  "id": "1802.07569",
  "slug": "continual-lifelong-learning-with-neural-networks-a-review",
  "title": "Continual Lifelong Learning with Neural Networks: A Review",
  "abstract": "Humans and animals have the ability to continually acquire, fine-tune, and transfer knowledge and skills throughout their lifespan. This ability, referred to as lifelong learning, is mediated by a rich set of neurocognitive mechanisms that together contribute to the development and specialization of our sensorimotor skills as well as to long-term memory consolidation and retrieval. Consequently, lifelong learning capabilities are crucial for autonomous agents interacting in the real world and processing continuous streams of information. However, lifelong learning remains a long-standing challenge for machine learning and neural network models since the continual acquisition of incrementally available information from non-stationary data distributions generally leads to catastrophic forgetting or interference. This limitation represents a major drawback for state-of-the-art deep neural network models that typically learn representations from stationary batches of training data, thus without accounting for situations in which information becomes incrementally available over time. In this review, we critically summarize the main challenges linked to lifelong learning for artificial learning systems and compare existing neural network approaches that alleviate, to different extents, catastrophic forgetting. We discuss well-established and emerging research motivated by lifelong learning factors in biological systems such as structural plasticity, memory replay, curriculum and transfer learning, intrinsic motivation, and multisensory integration.",
  "published": "2018-02-21",
  "updated": "2019-02-11",
  "year": "2018",
  "authors": [
   "German I. Parisi",
   "Ronald Kemker",
   "Jose L. Part",
   "Christopher Kanan",
   "Stefan Wermter"
  ],
  "author_count": 5,
  "categories": [
   "cs.LG",
   "q-bio.NC",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "Neural Networks",
  "venue_source": "semantic-scholar",
  "citations": 3707,
  "influential_citations": 160,
  "tldr": "This review critically summarize the main challenges linked to lifelong learning for artificial learning systems and compare existing neural network approaches that alleviate, to different extents, catastrophic forgetting.",
  "doi": "10.1016/j.neunet.2019.01.012.",
  "oa_pdf": "https://www.sciencedirect.com/science/article/pii/S0893608019300231/pdf",
  "s2_authors": [
   {
    "name": "G. I. Parisi",
    "id": "2988592",
    "h_index": 18,
    "papers": 57
   },
   {
    "name": "Ronald Kemker",
    "id": "9629738",
    "h_index": 12,
    "papers": 16
   },
   {
    "name": "Jose L. Part",
    "id": "22654433",
    "h_index": 7,
    "papers": 18
   },
   {
    "name": "Christopher Kanan",
    "id": "3290098",
    "h_index": 39,
    "papers": 112
   },
   {
    "name": "Stefan Wermter",
    "id": "1736513",
    "h_index": 47,
    "papers": 522
   }
  ],
  "comment": "",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1802.07569v4",
  "pdf_url": "https://arxiv.org/pdf/1802.07569v4",
  "html_url": "https://arxiv.org/html/1802.07569v4",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "1802.06070",
  "slug": "diversity-is-all-you-need-learning-skills-without-a-reward-function",
  "title": "Diversity is All You Need: Learning Skills without a Reward Function",
  "abstract": "Intelligent creatures can explore their environments and learn useful skills without supervision. In this paper, we propose DIAYN ('Diversity is All You Need'), a method for learning useful skills without a reward function. Our proposed method learns skills by maximizing an information theoretic objective using a maximum entropy policy. On a variety of simulated robotic tasks, we show that this simple objective results in the unsupervised emergence of diverse skills, such as walking and jumping. In a number of reinforcement learning benchmark environments, our method is able to learn a skill that solves the benchmark task despite never receiving the true task reward. We show how pretrained skills can provide a good parameter initialization for downstream tasks, and can be composed hierarchically to solve complex, sparse reward tasks. Our results suggest that unsupervised discovery of skills can serve as an effective pretraining mechanism for overcoming challenges of exploration and data efficiency in reinforcement learning.",
  "published": "2018-02-16",
  "updated": "2018-10-09",
  "year": "2018",
  "authors": [
   "Benjamin Eysenbach",
   "Abhishek Gupta",
   "Julian Ibarz",
   "Sergey Levine"
  ],
  "author_count": 4,
  "categories": [
   "cs.AI",
   "cs.RO"
  ],
  "primary_category": "cs.AI",
  "venue": "ICLR",
  "venue_source": "semantic-scholar",
  "citations": 1323,
  "influential_citations": 200,
  "tldr": "The proposed DIAYN (\"Diversity is All You Need\"), a method for learning useful skills without a reward function, learns skills by maximizing an information theoretic objective using a maximum entropy policy.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Benjamin Eysenbach",
    "id": "8140754",
    "h_index": 34,
    "papers": 75
   },
   {
    "name": "Abhishek Gupta",
    "id": "2129458064",
    "h_index": 57,
    "papers": 156
   },
   {
    "name": "Julian Ibarz",
    "id": "46920727",
    "h_index": 22,
    "papers": 35
   },
   {
    "name": "S. Levine",
    "id": "1736651",
    "h_index": 167,
    "papers": 551
   }
  ],
  "comment": "Videos and code for our experiments are available at: https://sites.google.com/view/diayn",
  "topics": [
   "rl-control",
   "navigation",
   "foundation-pretraining"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1802.06070v6",
  "pdf_url": "https://arxiv.org/pdf/1802.06070v6",
  "html_url": "https://arxiv.org/html/1802.06070v6",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "1802.03006",
  "slug": "learning-and-querying-fast-generative-models-for-reinforcement-learnin",
  "title": "Learning and Querying Fast Generative Models for Reinforcement Learning",
  "abstract": "A key challenge in model-based reinforcement learning (RL) is to synthesize computationally efficient and accurate environment models. We show that carefully designed generative models that learn and operate on compact state representations, so-called state-space models, substantially reduce the computational costs for predicting outcomes of sequences of actions. Extensive experiments establish that state-space models accurately capture the dynamics of Atari games from the Arcade Learning Environment from raw pixels. The computational speed-up of state-space models while maintaining high accuracy makes their application in RL feasible: We demonstrate that agents which query these models for decision making outperform strong model-free baselines on the game MSPACMAN, demonstrating the potential of using learned environment models for planning.",
  "published": "2018-02-08",
  "updated": "2018-02-08",
  "year": "2018",
  "authors": [
   "Lars Buesing",
   "Theophane Weber",
   "Sebastien Racaniere",
   "S. M. Ali Eslami",
   "Danilo Rezende",
   "David P. Reichert",
   "Fabio Viola",
   "Frederic Besse",
   "Karol Gregor",
   "Demis Hassabis",
   "Daan Wierstra"
  ],
  "author_count": 11,
  "categories": [
   "cs.LG"
  ],
  "primary_category": "cs.LG",
  "venue": "",
  "venue_source": "",
  "citations": 136,
  "influential_citations": 10,
  "tldr": "It is demonstrated that agents which query these models for decision making outperform strong model-free baselines on the game MSPACMAN, demonstrating the potential of using learned environment models for planning.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "L. Buesing",
    "id": "1981334",
    "h_index": 33,
    "papers": 62
   },
   {
    "name": "T. Weber",
    "id": "143947744",
    "h_index": 33,
    "papers": 67
   },
   {
    "name": "S. Racani\u00e8re",
    "id": "2026995",
    "h_index": 28,
    "papers": 47
   },
   {
    "name": "S. Eslami",
    "id": "143648071",
    "h_index": 28,
    "papers": 46
   },
   {
    "name": "Danilo Jimenez Rezende",
    "id": "1748523",
    "h_index": 47,
    "papers": 82
   },
   {
    "name": "David P. Reichert",
    "id": "40634311",
    "h_index": 16,
    "papers": 23
   },
   {
    "name": "Fabio Viola",
    "id": "47963165",
    "h_index": 19,
    "papers": 32
   },
   {
    "name": "F. Besse",
    "id": "143923544",
    "h_index": 12,
    "papers": 19
   },
   {
    "name": "Karol Gregor",
    "id": "144717963",
    "h_index": 25,
    "papers": 40
   },
   {
    "name": "D. Hassabis",
    "id": "48987704",
    "h_index": 92,
    "papers": 160
   },
   {
    "name": "Daan Wierstra",
    "id": "1688276",
    "h_index": 45,
    "papers": 66
   }
  ],
  "comment": "",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1802.03006v1",
  "pdf_url": "https://arxiv.org/pdf/1802.03006v1",
  "html_url": "https://arxiv.org/html/1802.03006v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 2.14
 },
 {
  "id": "1802.01561",
  "slug": "impala-scalable-distributed-deep-rl-with-importance-weighted-actor-lea",
  "title": "IMPALA: Scalable Distributed Deep-RL with Importance Weighted Actor-Learner Architectures",
  "abstract": "In this work we aim to solve a large collection of tasks using a single reinforcement learning agent with a single set of parameters. A key challenge is to handle the increased amount of data and extended training time. We have developed a new distributed agent IMPALA (Importance Weighted Actor-Learner Architecture) that not only uses resources more efficiently in single-machine training but also scales to thousands of machines without sacrificing data efficiency or resource utilisation. We achieve stable learning at high throughput by combining decoupled acting and learning with a novel off-policy correction method called V-trace. We demonstrate the effectiveness of IMPALA for multi-task reinforcement learning on DMLab-30 (a set of 30 tasks from the DeepMind Lab environment (Beattie et al., 2016)) and Atari-57 (all available Atari games in Arcade Learning Environment (Bellemare et al., 2013a)). Our results show that IMPALA is able to achieve better performance than previous agents with less data, and crucially exhibits positive transfer between tasks as a result of its multi-task approach.",
  "published": "2018-02-05",
  "updated": "2018-06-28",
  "year": "2018",
  "authors": [
   "Lasse Espeholt",
   "Hubert Soyer",
   "Remi Munos",
   "Karen Simonyan",
   "Volodymir Mnih",
   "Tom Ward",
   "Yotam Doron",
   "Vlad Firoiu",
   "Tim Harley",
   "Iain Dunning",
   "Shane Legg",
   "Koray Kavukcuoglu"
  ],
  "author_count": 12,
  "categories": [
   "cs.LG",
   "cs.AI"
  ],
  "primary_category": "cs.LG",
  "venue": "ICML",
  "venue_source": "semantic-scholar",
  "citations": 1910,
  "influential_citations": 269,
  "tldr": "A new distributed agent IMPALA (Importance Weighted Actor-Learner Architecture) is developed that not only uses resources more efficiently in single-machine training but also scales to thousands of machines without sacrificing data efficiency or resource utilisation.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "L. Espeholt",
    "id": "2311318",
    "h_index": 12,
    "papers": 21
   },
   {
    "name": "Hubert Soyer",
    "id": "2794457",
    "h_index": 15,
    "papers": 27
   },
   {
    "name": "R. Munos",
    "id": "1708654",
    "h_index": 90,
    "papers": 246
   },
   {
    "name": "K. Simonyan",
    "id": "34838386",
    "h_index": 65,
    "papers": 108
   },
   {
    "name": "Volodymyr Mnih",
    "id": "3255983",
    "h_index": 34,
    "papers": 47
   },
   {
    "name": "Tom Ward",
    "id": "2056968992",
    "h_index": 4,
    "papers": 4
   },
   {
    "name": "Yotam Doron",
    "id": "2895238",
    "h_index": 7,
    "papers": 9
   },
   {
    "name": "Vlad Firoiu",
    "id": "9559485",
    "h_index": 11,
    "papers": 17
   },
   {
    "name": "Tim Harley",
    "id": "3367786",
    "h_index": 13,
    "papers": 17
   },
   {
    "name": "Iain Dunning",
    "id": "2768462",
    "h_index": 16,
    "papers": 23
   },
   {
    "name": "S. Legg",
    "id": "34313265",
    "h_index": 32,
    "papers": 66
   },
   {
    "name": "K. Kavukcuoglu",
    "id": "2645384",
    "h_index": 76,
    "papers": 124
   }
  ],
  "comment": "",
  "topics": [
   "rl-control"
  ],
  "orgs": [
   "Google DeepMind"
  ],
  "abs_url": "https://arxiv.org/abs/1802.01561v3",
  "pdf_url": "https://arxiv.org/pdf/1802.01561v3",
  "html_url": "https://arxiv.org/html/1802.01561v3",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 5.0
 },
 {
  "id": "1801.03924",
  "slug": "the-unreasonable-effectiveness-of-deep-features-as-a-perceptual-metric",
  "title": "The Unreasonable Effectiveness of Deep Features as a Perceptual Metric",
  "abstract": "While it is nearly effortless for humans to quickly assess the perceptual similarity between two images, the underlying processes are thought to be quite complex. Despite this, the most widely used perceptual metrics today, such as PSNR and SSIM, are simple, shallow functions, and fail to account for many nuances of human perception. Recently, the deep learning community has found that features of the VGG network trained on ImageNet classification has been remarkably useful as a training loss for image synthesis. But how perceptual are these so-called \"perceptual losses\"? What elements are critical for their success? To answer these questions, we introduce a new dataset of human perceptual similarity judgments. We systematically evaluate deep features across different architectures and tasks and compare them with classic metrics. We find that deep features outperform all previous metrics by large margins on our dataset. More surprisingly, this result is not restricted to ImageNet-trained VGG features, but holds across different deep architectures and levels of supervision (supervised, self-supervised, or even unsupervised). Our results suggest that perceptual similarity is an emergent property shared across deep visual representations.",
  "published": "2018-01-11",
  "updated": "2018-04-10",
  "year": "2018",
  "authors": [
   "Richard Zhang",
   "Phillip Isola",
   "Alexei A. Efros",
   "Eli Shechtman",
   "Oliver Wang"
  ],
  "author_count": 5,
  "categories": [
   "cs.CV",
   "cs.GR"
  ],
  "primary_category": "cs.CV",
  "venue": "CVPR",
  "venue_source": "semantic-scholar",
  "citations": 19458,
  "influential_citations": 2369,
  "tldr": "A new dataset of human perceptual similarity judgments is introduced and it is found that deep features outperform all previous metrics by large margins on this dataset, and suggests that perceptual similarity is an emergent property shared across deep visual representations.",
  "doi": "10.1109/CVPR.2018.00068",
  "oa_pdf": "https://arxiv.org/pdf/1801.03924",
  "s2_authors": [
   {
    "name": "Richard Zhang",
    "id": "2844849",
    "h_index": 23,
    "papers": 39
   },
   {
    "name": "Phillip Isola",
    "id": "2094770",
    "h_index": 57,
    "papers": 115
   },
   {
    "name": "Alexei A. Efros",
    "id": "1763086",
    "h_index": 111,
    "papers": 241
   },
   {
    "name": "Eli Shechtman",
    "id": "2177801",
    "h_index": 73,
    "papers": 162
   },
   {
    "name": "Oliver Wang",
    "id": "39231399",
    "h_index": 49,
    "papers": 119
   }
  ],
  "comment": "Accepted to CVPR 2018; Code and data available at https://www.github.com/richzhang/PerceptualSimilarity",
  "topics": [],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1801.03924v2",
  "pdf_url": "https://arxiv.org/pdf/1801.03924v2",
  "html_url": "https://arxiv.org/html/1801.03924v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "1801.01290",
  "slug": "soft-actor-critic-off-policy-maximum-entropy-deep-reinforcement-learni",
  "title": "Soft Actor-Critic: Off-Policy Maximum Entropy Deep Reinforcement Learning with a Stochastic Actor",
  "abstract": "Model-free deep reinforcement learning (RL) algorithms have been demonstrated on a range of challenging decision making and control tasks. However, these methods typically suffer from two major challenges: very high sample complexity and brittle convergence properties, which necessitate meticulous hyperparameter tuning. Both of these challenges severely limit the applicability of such methods to complex, real-world domains. In this paper, we propose soft actor-critic, an off-policy actor-critic deep RL algorithm based on the maximum entropy reinforcement learning framework. In this framework, the actor aims to maximize expected reward while also maximizing entropy. That is, to succeed at the task while acting as randomly as possible. Prior deep RL methods based on this framework have been formulated as Q-learning methods. By combining off-policy updates with a stable stochastic actor-critic formulation, our method achieves state-of-the-art performance on a range of continuous control benchmark tasks, outperforming prior on-policy and off-policy methods. Furthermore, we demonstrate that, in contrast to other off-policy algorithms, our approach is very stable, achieving very similar performance across different random seeds.",
  "published": "2018-01-04",
  "updated": "2018-08-08",
  "year": "2018",
  "authors": [
   "Tuomas Haarnoja",
   "Aurick Zhou",
   "Pieter Abbeel",
   "Sergey Levine"
  ],
  "author_count": 4,
  "categories": [
   "cs.LG",
   "cs.AI",
   "stat.ML"
  ],
  "primary_category": "cs.LG",
  "venue": "ICML",
  "venue_source": "semantic-scholar",
  "citations": 12130,
  "influential_citations": 2100,
  "tldr": "This paper proposes soft actor-critic, an off-policy actor-Critic deep RL algorithm based on the maximum entropy reinforcement learning framework, and achieves state-of-the-art performance on a range of continuous control benchmark tasks, outperforming prior on-policy and off- policy methods.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Tuomas Haarnoja",
    "id": "2587648",
    "h_index": 15,
    "papers": 31
   },
   {
    "name": "Aurick Zhou",
    "id": "35499972",
    "h_index": 12,
    "papers": 13
   },
   {
    "name": "P. Abbeel",
    "id": "1689992",
    "h_index": 165,
    "papers": 596
   },
   {
    "name": "S. Levine",
    "id": "1736651",
    "h_index": 167,
    "papers": 551
   }
  ],
  "comment": "ICML 2018 Videos: sites.google.com/view/soft-actor-critic Code: github.com/haarnoja/sac",
  "topics": [
   "rl-control"
  ],
  "orgs": [],
  "abs_url": "https://arxiv.org/abs/1801.01290v2",
  "pdf_url": "https://arxiv.org/pdf/1801.01290v2",
  "html_url": "https://arxiv.org/html/1801.01290v2",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 4.5
 },
 {
  "id": "1801.00690",
  "slug": "deepmind-control-suite",
  "title": "DeepMind Control Suite",
  "abstract": "The DeepMind Control Suite is a set of continuous control tasks with a standardised structure and interpretable rewards, intended to serve as performance benchmarks for reinforcement learning agents. The tasks are written in Python and powered by the MuJoCo physics engine, making them easy to use and modify. We include benchmarks for several learning algorithms. The Control Suite is publicly available at https://www.github.com/deepmind/dm_control . A video summary of all tasks is available at http://youtu.be/rAai4QzcYbs .",
  "published": "2018-01-02",
  "updated": "2018-01-02",
  "year": "2018",
  "authors": [
   "Yuval Tassa",
   "Yotam Doron",
   "Alistair Muldal",
   "Tom Erez",
   "Yazhe Li",
   "Diego de Las Casas",
   "David Budden",
   "Abbas Abdolmaleki",
   "Josh Merel",
   "Andrew Lefrancq",
   "Timothy Lillicrap",
   "Martin Riedmiller"
  ],
  "author_count": 12,
  "categories": [
   "cs.AI"
  ],
  "primary_category": "cs.AI",
  "venue": "",
  "venue_source": "",
  "citations": 1471,
  "influential_citations": 210,
  "tldr": "The DeepMind Control Suite is a set of continuous control tasks with a standardised structure and interpretable rewards, intended to serve as performance benchmarks for reinforcement learning agents.",
  "doi": "",
  "oa_pdf": "",
  "s2_authors": [
   {
    "name": "Yuval Tassa",
    "id": "2109481",
    "h_index": 39,
    "papers": 67
   },
   {
    "name": "Yotam Doron",
    "id": "2895238",
    "h_index": 7,
    "papers": 9
   },
   {
    "name": "Alistair Muldal",
    "id": "50654556",
    "h_index": 12,
    "papers": 15
   },
   {
    "name": "Tom Erez",
    "id": "1968210",
    "h_index": 26,
    "papers": 52
   },
   {
    "name": "Yazhe Li",
    "id": "2144417088",
    "h_index": 12,
    "papers": 24
   },
   {
    "name": "Diego de Las Casas",
    "id": "40550616",
    "h_index": 13,
    "papers": 18
   },
   {
    "name": "D. Budden",
    "id": "2508525",
    "h_index": 27,
    "papers": 73
   },
   {
    "name": "A. Abdolmaleki",
    "id": "2799799",
    "h_index": 29,
    "papers": 95
   },
   {
    "name": "J. Merel",
    "id": "1879232",
    "h_index": 28,
    "papers": 51
   },
   {
    "name": "Andrew Lefrancq",
    "id": "8455031",
    "h_index": 3,
    "papers": 3
   },
   {
    "name": "T. Lillicrap",
    "id": "2542999",
    "h_index": 69,
    "papers": 154
   },
   {
    "name": "Martin A. Riedmiller",
    "id": "3137672",
    "h_index": 55,
    "papers": 219
   }
  ],
  "comment": "24 pages, 7 figures, 2 tables",
  "topics": [
   "rl-control"
  ],
  "orgs": [
   "Google DeepMind"
  ],
  "abs_url": "https://arxiv.org/abs/1801.00690v1",
  "pdf_url": "https://arxiv.org/pdf/1801.00690v1",
  "html_url": "https://arxiv.org/html/1801.00690v1",
  "code_url": null,
  "presented": [],
  "club_note": "",
  "featured": false,
  "signal": 3.5
 }
]