{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:VIKJ3JYSN6PXW37QV47HLP2CDI","short_pith_number":"pith:VIKJ3JYS","schema_version":"1.0","canonical_sha256":"aa149da7126f9f7b6ff0af3e75bf421a34ed7c8086407ac33e5cecc8e431186a","source":{"kind":"arxiv","id":"2505.10075","version":1},"attestation_state":"computed","paper":{"title":"FlowDreamer: A RGB-D World Model with Flow-based Motion Representations for Robot Manipulation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.RO","authors_text":"Huaping Liu, Jun Guo, Min Yang, Qing Li, Xiaojian Ma, Yikai Wang","submitted_at":"2025-05-15T08:27:16Z","abstract_excerpt":"This paper investigates training better visual world models for robot manipulation, i.e., models that can predict future visual observations by conditioning on past frames and robot actions. Specifically, we consider world models that operate on RGB-D frames (RGB-D world models). As opposed to canonical approaches that handle dynamics prediction mostly implicitly and reconcile it with visual rendering in a single model, we introduce FlowDreamer, which adopts 3D scene flow as explicit motion representations. FlowDreamer first predicts 3D scene flow from past frame and action conditions with a U"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.10075","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.RO","submitted_at":"2025-05-15T08:27:16Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"f8fc11c806c2bcba1a678ea8e050f7b2ef84ef2f4d1e117425239b070bf222b9","abstract_canon_sha256":"d90ec885bf1a292d4fdb302a4fd8977c0e77aa4645b3dfd655cb826d2aa230b3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:03:32.863367Z","signature_b64":"v3ykD5b/xeUOmA7ogHsjbeV4AohpZIXom1vXJ4RqZ4OGX3J8x8JOIdaYgfoPhC1X7A6LCZqOlBOTvn9wFZP4Cw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"aa149da7126f9f7b6ff0af3e75bf421a34ed7c8086407ac33e5cecc8e431186a","last_reissued_at":"2026-07-05T11:03:32.862948Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:03:32.862948Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"FlowDreamer: A RGB-D World Model with Flow-based Motion Representations for Robot Manipulation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.RO","authors_text":"Huaping Liu, Jun Guo, Min Yang, Qing Li, Xiaojian Ma, Yikai Wang","submitted_at":"2025-05-15T08:27:16Z","abstract_excerpt":"This paper investigates training better visual world models for robot manipulation, i.e., models that can predict future visual observations by conditioning on past frames and robot actions. Specifically, we consider world models that operate on RGB-D frames (RGB-D world models). As opposed to canonical approaches that handle dynamics prediction mostly implicitly and reconcile it with visual rendering in a single model, we introduce FlowDreamer, which adopts 3D scene flow as explicit motion representations. FlowDreamer first predicts 3D scene flow from past frame and action conditions with a U"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.10075","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.10075/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.10075","created_at":"2026-07-05T11:03:32.863002+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.10075v1","created_at":"2026-07-05T11:03:32.863002+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.10075","created_at":"2026-07-05T11:03:32.863002+00:00"},{"alias_kind":"pith_short_12","alias_value":"VIKJ3JYSN6PX","created_at":"2026-07-05T11:03:32.863002+00:00"},{"alias_kind":"pith_short_16","alias_value":"VIKJ3JYSN6PXW37Q","created_at":"2026-07-05T11:03:32.863002+00:00"},{"alias_kind":"pith_short_8","alias_value":"VIKJ3JYS","created_at":"2026-07-05T11:03:32.863002+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.20781","citing_title":"World Action Models: A Survey","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00836","citing_title":"From World Models to World Action Models: A Concise Tutorial for Robotics","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07304","citing_title":"CAPE: Contrastive Action-conditioned Parallel Encoding for Embodied Planning","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00836","citing_title":"From World Models to World Action Models: A Concise Tutorial for Robotics","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14598","citing_title":"DSSP: Diffusion State Space Policy with Full-History Encoding","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12090","citing_title":"World Action Models: The Next Frontier in Embodied AI","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06168","citing_title":"Action Images: End-to-End Policy Learning via Multiview Video Generation","ref_index":16,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VIKJ3JYSN6PXW37QV47HLP2CDI","json":"https://pith.science/pith/VIKJ3JYSN6PXW37QV47HLP2CDI.json","graph_json":"https://pith.science/api/pith-number/VIKJ3JYSN6PXW37QV47HLP2CDI/graph.json","events_json":"https://pith.science/api/pith-number/VIKJ3JYSN6PXW37QV47HLP2CDI/events.json","paper":"https://pith.science/paper/VIKJ3JYS"},"agent_actions":{"view_html":"https://pith.science/pith/VIKJ3JYSN6PXW37QV47HLP2CDI","download_json":"https://pith.science/pith/VIKJ3JYSN6PXW37QV47HLP2CDI.json","view_paper":"https://pith.science/paper/VIKJ3JYS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.10075&json=true","fetch_graph":"https://pith.science/api/pith-number/VIKJ3JYSN6PXW37QV47HLP2CDI/graph.json","fetch_events":"https://pith.science/api/pith-number/VIKJ3JYSN6PXW37QV47HLP2CDI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VIKJ3JYSN6PXW37QV47HLP2CDI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VIKJ3JYSN6PXW37QV47HLP2CDI/action/storage_attestation","attest_author":"https://pith.science/pith/VIKJ3JYSN6PXW37QV47HLP2CDI/action/author_attestation","sign_citation":"https://pith.science/pith/VIKJ3JYSN6PXW37QV47HLP2CDI/action/citation_signature","submit_replication":"https://pith.science/pith/VIKJ3JYSN6PXW37QV47HLP2CDI/action/replication_record"}},"created_at":"2026-07-05T11:03:32.863002+00:00","updated_at":"2026-07-05T11:03:32.863002+00:00"}