{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:T6EWCNPGWMZL43XEVKYFQRLBKQ","short_pith_number":"pith:T6EWCNPG","schema_version":"1.0","canonical_sha256":"9f896135e6b332be6ee4aab05845615405bead61ebe66bb21bec44be3d3030b5","source":{"kind":"arxiv","id":"2411.09153","version":1},"attestation_state":"computed","paper":{"title":"VidMan: Exploiting Implicit Dynamics from Video Diffusion Model for Effective Robot Manipulation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.CV","authors_text":"Hang Xu, Jianhua Han, Junfan Lin, Shen Zhao, Xiaodan Liang, Yi Zhu, Youpeng Wen","submitted_at":"2024-11-14T03:13:26Z","abstract_excerpt":"Recent advancements utilizing large-scale video data for learning video generation models demonstrate significant potential in understanding complex physical dynamics. It suggests the feasibility of leveraging diverse robot trajectory data to develop a unified, dynamics-aware model to enhance robot manipulation. However, given the relatively small amount of available robot data, directly fitting data without considering the relationship between visual observations and actions could lead to suboptimal data utilization. To this end, we propose VidMan (Video Diffusion for Robot Manipulation), a n"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.09153","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-11-14T03:13:26Z","cross_cats_sorted":["cs.RO"],"title_canon_sha256":"128c8781d862d434181a83db0440c95e79ef0380b4f21a513602827c207b5177","abstract_canon_sha256":"d41afa48c002e51d22d8867e99aaf39d354b6bd9aa00df0cd8b83415768118cb"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:35:21.373191Z","signature_b64":"bKpxUrjB3J7u78GER57YdLgVTog9sxLKiX37l1471ota/EraS4H1P/hHgZIem/YYPBD2QknQh+sZb1WB2D9yBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9f896135e6b332be6ee4aab05845615405bead61ebe66bb21bec44be3d3030b5","last_reissued_at":"2026-07-05T09:35:21.372642Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:35:21.372642Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VidMan: Exploiting Implicit Dynamics from Video Diffusion Model for Effective Robot Manipulation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.CV","authors_text":"Hang Xu, Jianhua Han, Junfan Lin, Shen Zhao, Xiaodan Liang, Yi Zhu, Youpeng Wen","submitted_at":"2024-11-14T03:13:26Z","abstract_excerpt":"Recent advancements utilizing large-scale video data for learning video generation models demonstrate significant potential in understanding complex physical dynamics. It suggests the feasibility of leveraging diverse robot trajectory data to develop a unified, dynamics-aware model to enhance robot manipulation. However, given the relatively small amount of available robot data, directly fitting data without considering the relationship between visual observations and actions could lead to suboptimal data utilization. To this end, we propose VidMan (Video Diffusion for Robot Manipulation), a n"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.09153","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.09153/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.09153","created_at":"2026-07-05T09:35:21.372720+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.09153v1","created_at":"2026-07-05T09:35:21.372720+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.09153","created_at":"2026-07-05T09:35:21.372720+00:00"},{"alias_kind":"pith_short_12","alias_value":"T6EWCNPGWMZL","created_at":"2026-07-05T09:35:21.372720+00:00"},{"alias_kind":"pith_short_16","alias_value":"T6EWCNPGWMZL43XE","created_at":"2026-07-05T09:35:21.372720+00:00"},{"alias_kind":"pith_short_8","alias_value":"T6EWCNPG","created_at":"2026-07-05T09:35:21.372720+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08639","citing_title":"Native Video-Action Pretraining for Generalizable Robot Control","ref_index":104,"is_internal_anchor":true},{"citing_arxiv_id":"2605.27947","citing_title":"SANTS: A State-Adaptive Scheduler for World Action Models","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04974","citing_title":"From Video to Control: A Survey of Learning Manipulation Interfaces from Temporal Visual Data","ref_index":96,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/T6EWCNPGWMZL43XEVKYFQRLBKQ","json":"https://pith.science/pith/T6EWCNPGWMZL43XEVKYFQRLBKQ.json","graph_json":"https://pith.science/api/pith-number/T6EWCNPGWMZL43XEVKYFQRLBKQ/graph.json","events_json":"https://pith.science/api/pith-number/T6EWCNPGWMZL43XEVKYFQRLBKQ/events.json","paper":"https://pith.science/paper/T6EWCNPG"},"agent_actions":{"view_html":"https://pith.science/pith/T6EWCNPGWMZL43XEVKYFQRLBKQ","download_json":"https://pith.science/pith/T6EWCNPGWMZL43XEVKYFQRLBKQ.json","view_paper":"https://pith.science/paper/T6EWCNPG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.09153&json=true","fetch_graph":"https://pith.science/api/pith-number/T6EWCNPGWMZL43XEVKYFQRLBKQ/graph.json","fetch_events":"https://pith.science/api/pith-number/T6EWCNPGWMZL43XEVKYFQRLBKQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/T6EWCNPGWMZL43XEVKYFQRLBKQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/T6EWCNPGWMZL43XEVKYFQRLBKQ/action/storage_attestation","attest_author":"https://pith.science/pith/T6EWCNPGWMZL43XEVKYFQRLBKQ/action/author_attestation","sign_citation":"https://pith.science/pith/T6EWCNPGWMZL43XEVKYFQRLBKQ/action/citation_signature","submit_replication":"https://pith.science/pith/T6EWCNPGWMZL43XEVKYFQRLBKQ/action/replication_record"}},"created_at":"2026-07-05T09:35:21.372720+00:00","updated_at":"2026-07-05T09:35:21.372720+00:00"}