{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:QGN5DBXTSTCNUGPFOPNVLPUYYD","short_pith_number":"pith:QGN5DBXT","schema_version":"1.0","canonical_sha256":"819bd186f394c4da19e573db55be98c0eabf842febf7308adea5c15bb3b4e145","source":{"kind":"arxiv","id":"2410.11831","version":1},"attestation_state":"computed","paper":{"title":"CoTracker3: Simpler and Better Point Tracking by Pseudo-Labelling Real Videos","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Andrea Vedaldi, Christian Rupprecht, Iurii Makarov, Jianyuan Wang, Natalia Neverova, Nikita Karaev","submitted_at":"2024-10-15T17:56:32Z","abstract_excerpt":"Most state-of-the-art point trackers are trained on synthetic data due to the difficulty of annotating real videos for this task. However, this can result in suboptimal performance due to the statistical gap between synthetic and real videos. In order to understand these issues better, we introduce CoTracker3, comprising a new tracking model and a new semi-supervised training recipe. This allows real videos without annotations to be used during training by generating pseudo-labels using off-the-shelf teachers. The new model eliminates or simplifies components from previous trackers, resulting "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.11831","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2024-10-15T17:56:32Z","cross_cats_sorted":[],"title_canon_sha256":"27746d151287476216d9bcfee496f1e96b6a140b58d70d82ebb54403580e23a4","abstract_canon_sha256":"c40747f6faaccadbf2fcc7950f16a6dc3d26396c8146330dfd8b1b44c843ee55"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:20:59.598684Z","signature_b64":"oHeckPKpwgjM1y8zE8SuB9l/2tpeShNkCAErJSNBk9U8fe66V8iJ5zjTf7M4EV4YFmwiUSCjxrbeS7NV+fU9AQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"819bd186f394c4da19e573db55be98c0eabf842febf7308adea5c15bb3b4e145","last_reissued_at":"2026-07-05T09:20:59.598107Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:20:59.598107Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CoTracker3: Simpler and Better Point Tracking by Pseudo-Labelling Real Videos","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Andrea Vedaldi, Christian Rupprecht, Iurii Makarov, Jianyuan Wang, Natalia Neverova, Nikita Karaev","submitted_at":"2024-10-15T17:56:32Z","abstract_excerpt":"Most state-of-the-art point trackers are trained on synthetic data due to the difficulty of annotating real videos for this task. However, this can result in suboptimal performance due to the statistical gap between synthetic and real videos. In order to understand these issues better, we introduce CoTracker3, comprising a new tracking model and a new semi-supervised training recipe. This allows real videos without annotations to be used during training by generating pseudo-labels using off-the-shelf teachers. The new model eliminates or simplifies components from previous trackers, resulting "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.11831","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.11831/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.11831","created_at":"2026-07-05T09:20:59.598189+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.11831v1","created_at":"2026-07-05T09:20:59.598189+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.11831","created_at":"2026-07-05T09:20:59.598189+00:00"},{"alias_kind":"pith_short_12","alias_value":"QGN5DBXTSTCN","created_at":"2026-07-05T09:20:59.598189+00:00"},{"alias_kind":"pith_short_16","alias_value":"QGN5DBXTSTCNUGPF","created_at":"2026-07-05T09:20:59.598189+00:00"},{"alias_kind":"pith_short_8","alias_value":"QGN5DBXT","created_at":"2026-07-05T09:20:59.598189+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":32,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.22160","citing_title":"GenMatter: Perceiving Physical Objects with Generative Matter Models","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2606.25344","citing_title":"Follow Your Track: Precise Skeleton Animation Controlled by 3D Trajectories","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18558","citing_title":"MolmoMotion: Forecasting Point Trajectories in 3D with Language Instruction","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10988","citing_title":"AnimaSpark: A Feed-Forward Method for Animating Arbitrary 3D Objects","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04593","citing_title":"4D Reconstruction from Sparse Dynamic Cameras","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04480","citing_title":"IMPose: Interactive Multi-person Pose Estimation with Dynamic Correction Propagation","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30608","citing_title":"UnfoldArt: Zero-Shot Recovery of Full Articulated 3D Objects from Text or Image","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24630","citing_title":"DexSIM: Real-time Dexterous Simulation with Unified Causal Video Diffusion","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30608","citing_title":"UnfoldArt: Zero-Shot Recovery of Full Articulated 3D Objects from Text or Image","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28128","citing_title":"PhysisForcing: Physics Reinforced World Simulator for Robotic Manipulation","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00321","citing_title":"Training-Free Object-Agnostic Jam Detection in Fulfillment Centers","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23045","citing_title":"The TIME Machine: On The Power of Motion for Efficient Perception","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23856","citing_title":"Point Tracking Improves World Action Models","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2504.18576","citing_title":"DriVerse: Navigation World Model for Driving Simulation via Multimodal Trajectory Prompting and Motion Alignment","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2510.18822","citing_title":"SAM 2++: Tracking Anything at Any Granularity","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18743","citing_title":"WorldString: Actionable World Representation","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18743","citing_title":"WorldString: Actionable World Representation","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2507.00990","citing_title":"Robotic Manipulation by Imitating Generated Videos Without Physical Demonstrations","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2507.22699","citing_title":"Image-Guided Shape-from-Template Using Mesh Inextensibility Constraints","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2511.02830","citing_title":"Densemarks: Learning Canonical Embeddings for Human Heads Images via Point Tracks","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2511.18373","citing_title":"MASS: Motion-Aware Spatial-Temporal Grounding for Physics Reasoning and Comprehension in Vision-Language Models","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2507.04447","citing_title":"DreamVLA: A Vision-Language-Action Model Dreamed with Comprehensive World Knowledge","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14645","citing_title":"Vision-Based Water Level and Flow Estimation","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15185","citing_title":"Quantitative Video World Model Evaluation for Geometric-Consistency","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02586","citing_title":"TrackerSplat: Exploiting Point Tracking for Fast and Robust Dynamic 3D Gaussians Reconstruction","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QGN5DBXTSTCNUGPFOPNVLPUYYD","json":"https://pith.science/pith/QGN5DBXTSTCNUGPFOPNVLPUYYD.json","graph_json":"https://pith.science/api/pith-number/QGN5DBXTSTCNUGPFOPNVLPUYYD/graph.json","events_json":"https://pith.science/api/pith-number/QGN5DBXTSTCNUGPFOPNVLPUYYD/events.json","paper":"https://pith.science/paper/QGN5DBXT"},"agent_actions":{"view_html":"https://pith.science/pith/QGN5DBXTSTCNUGPFOPNVLPUYYD","download_json":"https://pith.science/pith/QGN5DBXTSTCNUGPFOPNVLPUYYD.json","view_paper":"https://pith.science/paper/QGN5DBXT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.11831&json=true","fetch_graph":"https://pith.science/api/pith-number/QGN5DBXTSTCNUGPFOPNVLPUYYD/graph.json","fetch_events":"https://pith.science/api/pith-number/QGN5DBXTSTCNUGPFOPNVLPUYYD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QGN5DBXTSTCNUGPFOPNVLPUYYD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QGN5DBXTSTCNUGPFOPNVLPUYYD/action/storage_attestation","attest_author":"https://pith.science/pith/QGN5DBXTSTCNUGPFOPNVLPUYYD/action/author_attestation","sign_citation":"https://pith.science/pith/QGN5DBXTSTCNUGPFOPNVLPUYYD/action/citation_signature","submit_replication":"https://pith.science/pith/QGN5DBXTSTCNUGPFOPNVLPUYYD/action/replication_record"}},"created_at":"2026-07-05T09:20:59.598189+00:00","updated_at":"2026-07-05T09:20:59.598189+00:00"}