{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:WQLKEVI6OSEHBB6JLEPPZDCBES","short_pith_number":"pith:WQLKEVI6","schema_version":"1.0","canonical_sha256":"b416a2551e74887087c9591efc8c41249fde504247adbbf313c7750f23102aea","source":{"kind":"arxiv","id":"2504.13152","version":1},"attestation_state":"computed","paper":{"title":"St4RTrack: Simultaneous 4D Reconstruction and Tracking in the World","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Angjoo Kanazawa, Haiwen Feng, Junyi Zhang, Michael J. Black, Pengcheng Yu, Qianqian Wang, Trevor Darrell, Yufei Ye","submitted_at":"2025-04-17T17:55:58Z","abstract_excerpt":"Dynamic 3D reconstruction and point tracking in videos are typically treated as separate tasks, despite their deep connection. We propose St4RTrack, a feed-forward framework that simultaneously reconstructs and tracks dynamic video content in a world coordinate frame from RGB inputs. This is achieved by predicting two appropriately defined pointmaps for a pair of frames captured at different moments. Specifically, we predict both pointmaps at the same moment, in the same world, capturing both static and dynamic scene geometry while maintaining 3D correspondences. Chaining these predictions thr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.13152","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-04-17T17:55:58Z","cross_cats_sorted":[],"title_canon_sha256":"b5d837a46dbf359b2ff9007cedf0f4e98f7ea28d6141d4d2b9a8ad36451e6ebd","abstract_canon_sha256":"dc9d06d52001e7419ddfa9a53862d9e476c860a1db3fd1ddb642df6b8ad01878"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:50:38.483584Z","signature_b64":"rTF2SkS7EinrvO8GcBylE1gXYdKkCd0EYTzE3Qkklc8pk4kTy0l2czYq7t/EM4UT8tOs5aDFLIggoLJyEAemBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b416a2551e74887087c9591efc8c41249fde504247adbbf313c7750f23102aea","last_reissued_at":"2026-07-05T10:50:38.483095Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:50:38.483095Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"St4RTrack: Simultaneous 4D Reconstruction and Tracking in the World","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Angjoo Kanazawa, Haiwen Feng, Junyi Zhang, Michael J. Black, Pengcheng Yu, Qianqian Wang, Trevor Darrell, Yufei Ye","submitted_at":"2025-04-17T17:55:58Z","abstract_excerpt":"Dynamic 3D reconstruction and point tracking in videos are typically treated as separate tasks, despite their deep connection. We propose St4RTrack, a feed-forward framework that simultaneously reconstructs and tracks dynamic video content in a world coordinate frame from RGB inputs. This is achieved by predicting two appropriately defined pointmaps for a pair of frames captured at different moments. Specifically, we predict both pointmaps at the same moment, in the same world, capturing both static and dynamic scene geometry while maintaining 3D correspondences. Chaining these predictions thr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.13152","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.13152/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.13152","created_at":"2026-07-05T10:50:38.483154+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.13152v1","created_at":"2026-07-05T10:50:38.483154+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.13152","created_at":"2026-07-05T10:50:38.483154+00:00"},{"alias_kind":"pith_short_12","alias_value":"WQLKEVI6OSEH","created_at":"2026-07-05T10:50:38.483154+00:00"},{"alias_kind":"pith_short_16","alias_value":"WQLKEVI6OSEHBB6J","created_at":"2026-07-05T10:50:38.483154+00:00"},{"alias_kind":"pith_short_8","alias_value":"WQLKEVI6","created_at":"2026-07-05T10:50:38.483154+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.18558","citing_title":"MolmoMotion: Forecasting Point Trajectories in 3D with Language Instruction","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2511.02830","citing_title":"Densemarks: Learning Canonical Embeddings for Human Heads Images via Point Tracks","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2508.10934","citing_title":"ViPE: Video Pose Engine for 3D Geometric Perception","ref_index":21,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WQLKEVI6OSEHBB6JLEPPZDCBES","json":"https://pith.science/pith/WQLKEVI6OSEHBB6JLEPPZDCBES.json","graph_json":"https://pith.science/api/pith-number/WQLKEVI6OSEHBB6JLEPPZDCBES/graph.json","events_json":"https://pith.science/api/pith-number/WQLKEVI6OSEHBB6JLEPPZDCBES/events.json","paper":"https://pith.science/paper/WQLKEVI6"},"agent_actions":{"view_html":"https://pith.science/pith/WQLKEVI6OSEHBB6JLEPPZDCBES","download_json":"https://pith.science/pith/WQLKEVI6OSEHBB6JLEPPZDCBES.json","view_paper":"https://pith.science/paper/WQLKEVI6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.13152&json=true","fetch_graph":"https://pith.science/api/pith-number/WQLKEVI6OSEHBB6JLEPPZDCBES/graph.json","fetch_events":"https://pith.science/api/pith-number/WQLKEVI6OSEHBB6JLEPPZDCBES/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WQLKEVI6OSEHBB6JLEPPZDCBES/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WQLKEVI6OSEHBB6JLEPPZDCBES/action/storage_attestation","attest_author":"https://pith.science/pith/WQLKEVI6OSEHBB6JLEPPZDCBES/action/author_attestation","sign_citation":"https://pith.science/pith/WQLKEVI6OSEHBB6JLEPPZDCBES/action/citation_signature","submit_replication":"https://pith.science/pith/WQLKEVI6OSEHBB6JLEPPZDCBES/action/replication_record"}},"created_at":"2026-07-05T10:50:38.483154+00:00","updated_at":"2026-07-05T10:50:38.483154+00:00"}