{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:F5GEWC6FLB2HYYOM3JSEJACTAC","short_pith_number":"pith:F5GEWC6F","schema_version":"1.0","canonical_sha256":"2f4c4b0bc558747c61ccda6444805300b73be1319bd603ef37f8ffe1d87bb680","source":{"kind":"arxiv","id":"2411.11922","version":2},"attestation_state":"computed","paper":{"title":"SAMURAI: Adapting Segment Anything Model for Zero-Shot Visual Tracking with Motion-Aware Memory","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Cheng-Yen Yang, Hsiang-Wei Huang, Jenq-Neng Hwang, Wenhao Chai, Zhongyu Jiang","submitted_at":"2024-11-18T05:59:03Z","abstract_excerpt":"The Segment Anything Model 2 (SAM 2) has demonstrated strong performance in object segmentation tasks but faces challenges in visual object tracking, particularly when managing crowded scenes with fast-moving or self-occluding objects. Furthermore, the fixed-window memory approach in the original model does not consider the quality of memories selected to condition the image features for the next frame, leading to error propagation in videos. This paper introduces SAMURAI, an enhanced adaptation of SAM 2 specifically designed for visual object tracking. By incorporating temporal motion cues wi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.11922","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-11-18T05:59:03Z","cross_cats_sorted":[],"title_canon_sha256":"f2f8769f158927f8de4f6d649d163f574219feac00bdc332c57ff9dc250f5538","abstract_canon_sha256":"b6d257af6a7934b32e34cec58ed5dd10db219b0c9d0c7f1bdb3de97a869cedcf"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:42:32.947023Z","signature_b64":"UoxzjWamzoWQ+Ht6YRqhium2ucydRzANY4/nM3KIzi+lOfGpHWEZyYcEhT0qLdQAAUgSH4JMLxGbyjd5xMkjBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2f4c4b0bc558747c61ccda6444805300b73be1319bd603ef37f8ffe1d87bb680","last_reissued_at":"2026-07-05T09:42:32.946498Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:42:32.946498Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SAMURAI: Adapting Segment Anything Model for Zero-Shot Visual Tracking with Motion-Aware Memory","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Cheng-Yen Yang, Hsiang-Wei Huang, Jenq-Neng Hwang, Wenhao Chai, Zhongyu Jiang","submitted_at":"2024-11-18T05:59:03Z","abstract_excerpt":"The Segment Anything Model 2 (SAM 2) has demonstrated strong performance in object segmentation tasks but faces challenges in visual object tracking, particularly when managing crowded scenes with fast-moving or self-occluding objects. Furthermore, the fixed-window memory approach in the original model does not consider the quality of memories selected to condition the image features for the next frame, leading to error propagation in videos. This paper introduces SAMURAI, an enhanced adaptation of SAM 2 specifically designed for visual object tracking. By incorporating temporal motion cues wi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.11922","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.11922/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.11922","created_at":"2026-07-05T09:42:32.946556+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.11922v2","created_at":"2026-07-05T09:42:32.946556+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.11922","created_at":"2026-07-05T09:42:32.946556+00:00"},{"alias_kind":"pith_short_12","alias_value":"F5GEWC6FLB2H","created_at":"2026-07-05T09:42:32.946556+00:00"},{"alias_kind":"pith_short_16","alias_value":"F5GEWC6FLB2HYYOM","created_at":"2026-07-05T09:42:32.946556+00:00"},{"alias_kind":"pith_short_8","alias_value":"F5GEWC6F","created_at":"2026-07-05T09:42:32.946556+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":13,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08688","citing_title":"SAM-MT: Real-Time Interactive Multi-Target Video Segmentation","ref_index":13,"is_internal_anchor":true},{"citing_arxiv_id":"2606.29861","citing_title":"SUMO: Segment and Track Any Motion with Nonlinear State Space Models","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27655","citing_title":"Temporal-Emerged Prompting for Segment Anything in Multiframe Infrared Small Target Detection","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22538","citing_title":"Segment Anything with Motion, Geometry, and Semantic Adaptation for Complex Nonlinear Visual Object Tracking","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2511.16719","citing_title":"SAM 3: Segment Anything with Concepts","ref_index":147,"is_internal_anchor":false},{"citing_arxiv_id":"2507.01925","citing_title":"A Survey on Vision-Language-Action Models: An Action Tokenization Perspective","ref_index":91,"is_internal_anchor":false},{"citing_arxiv_id":"2601.08831","citing_title":"3AM: 3egment Anything with Geometric Consistency in Videos","ref_index":98,"is_internal_anchor":false},{"citing_arxiv_id":"2602.11183","citing_title":"Mitigating Error Accumulation in Continuous Navigation via Memory-Augmented Kalman Filtering","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03305","citing_title":"HVG-3D: Bridging Real and Simulation Domains for 3D-Conditional Hand-Object Interaction Video Synthesis","ref_index":75,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04016","citing_title":"HOIGS: Human-Object Interaction Gaussian Splatting","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2501.03575","citing_title":"Cosmos World Foundation Model Platform for Physical AI","ref_index":230,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06671","citing_title":"4D Vessel Reconstruction for Benchtop Thrombectomy Analysis","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02638","citing_title":"ViewSAM: Learning View-aware Cross-modal Semantics for Weakly Supervised Cross-view Referring Multi-Object Tracking","ref_index":34,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/F5GEWC6FLB2HYYOM3JSEJACTAC","json":"https://pith.science/pith/F5GEWC6FLB2HYYOM3JSEJACTAC.json","graph_json":"https://pith.science/api/pith-number/F5GEWC6FLB2HYYOM3JSEJACTAC/graph.json","events_json":"https://pith.science/api/pith-number/F5GEWC6FLB2HYYOM3JSEJACTAC/events.json","paper":"https://pith.science/paper/F5GEWC6F"},"agent_actions":{"view_html":"https://pith.science/pith/F5GEWC6FLB2HYYOM3JSEJACTAC","download_json":"https://pith.science/pith/F5GEWC6FLB2HYYOM3JSEJACTAC.json","view_paper":"https://pith.science/paper/F5GEWC6F","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.11922&json=true","fetch_graph":"https://pith.science/api/pith-number/F5GEWC6FLB2HYYOM3JSEJACTAC/graph.json","fetch_events":"https://pith.science/api/pith-number/F5GEWC6FLB2HYYOM3JSEJACTAC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/F5GEWC6FLB2HYYOM3JSEJACTAC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/F5GEWC6FLB2HYYOM3JSEJACTAC/action/storage_attestation","attest_author":"https://pith.science/pith/F5GEWC6FLB2HYYOM3JSEJACTAC/action/author_attestation","sign_citation":"https://pith.science/pith/F5GEWC6FLB2HYYOM3JSEJACTAC/action/citation_signature","submit_replication":"https://pith.science/pith/F5GEWC6FLB2HYYOM3JSEJACTAC/action/replication_record"}},"created_at":"2026-07-05T09:42:32.946556+00:00","updated_at":"2026-07-05T09:42:32.946556+00:00"}