{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:K2WUEM3BIKMJFGZJ7OHGZPZEYL","short_pith_number":"pith:K2WUEM3B","schema_version":"1.0","canonical_sha256":"56ad4233614298929b29fb8e6cbf24c2ca208af4785067d0c5116b1e983254c9","source":{"kind":"arxiv","id":"2304.11968","version":2},"attestation_state":"computed","paper":{"title":"Track Anything: Segment Anything Meets Videos","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Fangjing Wang, Feng Zheng, Jinyu Yang, Mingqi Gao, Shang Gao, Zhe Li","submitted_at":"2023-04-24T10:04:06Z","abstract_excerpt":"Recently, the Segment Anything Model (SAM) gains lots of attention rapidly due to its impressive segmentation performance on images. Regarding its strong ability on image segmentation and high interactivity with different prompts, we found that it performs poorly on consistent segmentation in videos. Therefore, in this report, we propose Track Anything Model (TAM), which achieves high-performance interactive tracking and segmentation in videos. To be detailed, given a video sequence, only with very little human participation, i.e., several clicks, people can track anything they are interested "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2304.11968","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-04-24T10:04:06Z","cross_cats_sorted":[],"title_canon_sha256":"1a986b296816a2e351cece0da4334965f82e7346815c829f7d0fd2717b8f53e7","abstract_canon_sha256":"116c2e42de3648ea6349cf1ebd15d073d6bfa6b0e70f95f7fc5c118135d3fa20"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:05:16.621248Z","signature_b64":"Xxo+uHQ+Tz7hkl4A+gnygP+suWMk8FA5kcC18U4JPVQxJG30YKX/1SqEd9295JxQODkiJd9ArFZCTYFMiLZmAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"56ad4233614298929b29fb8e6cbf24c2ca208af4785067d0c5116b1e983254c9","last_reissued_at":"2026-07-05T06:05:16.620814Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:05:16.620814Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Track Anything: Segment Anything Meets Videos","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Fangjing Wang, Feng Zheng, Jinyu Yang, Mingqi Gao, Shang Gao, Zhe Li","submitted_at":"2023-04-24T10:04:06Z","abstract_excerpt":"Recently, the Segment Anything Model (SAM) gains lots of attention rapidly due to its impressive segmentation performance on images. Regarding its strong ability on image segmentation and high interactivity with different prompts, we found that it performs poorly on consistent segmentation in videos. Therefore, in this report, we propose Track Anything Model (TAM), which achieves high-performance interactive tracking and segmentation in videos. To be detailed, given a video sequence, only with very little human participation, i.e., several clicks, people can track anything they are interested "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2304.11968","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2304.11968/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2304.11968","created_at":"2026-07-05T06:05:16.620870+00:00"},{"alias_kind":"arxiv_version","alias_value":"2304.11968v2","created_at":"2026-07-05T06:05:16.620870+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2304.11968","created_at":"2026-07-05T06:05:16.620870+00:00"},{"alias_kind":"pith_short_12","alias_value":"K2WUEM3BIKMJ","created_at":"2026-07-05T06:05:16.620870+00:00"},{"alias_kind":"pith_short_16","alias_value":"K2WUEM3BIKMJFGZJ","created_at":"2026-07-05T06:05:16.620870+00:00"},{"alias_kind":"pith_short_8","alias_value":"K2WUEM3B","created_at":"2026-07-05T06:05:16.620870+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":16,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2410.04960","citing_title":"On Efficient Variants of Segment Anything Model: A Survey","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2412.03077","citing_title":"RoDyGS: Robust Dynamic Gaussian Splatting for Casual Videos","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2505.23617","citing_title":"One Trajectory, One Token: Grounded Video Tokenization via Panoptic Sub-object Trajectory","ref_index":67,"is_internal_anchor":false},{"citing_arxiv_id":"2510.14244","citing_title":"Reinforcement Learning for Unsupervised Domain Adaptation in Spatio-Temporal Echocardiography Segmentation","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2306.14289","citing_title":"Faster Segment Anything: Towards Lightweight SAM for Mobile Applications","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2511.18264","citing_title":"SatSAM2: Motion-Constrained Video Object Tracking in Satellite Imagery using Promptable SAM2 and Kalman Priors","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2512.17445","citing_title":"LangDriveCTRL: Natural Language Controllable Driving Scene Editing with Multi-modal Agents","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2602.21668","citing_title":"Space-Time Forecasting of Dynamic Scenes with Motion-aware Gaussian Grouping","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09904","citing_title":"TOC-Bench: A Temporal Object Consistency Benchmark for Video Large Language Models","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10106","citing_title":"ViSRA: A Video-based Spatial Reasoning Agent for Multi-modal Large Language Models","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09904","citing_title":"TOC-Bench: A Temporal Object Consistency Benchmark for Video Large Language Models","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09677","citing_title":"VFM-SDM: A vision foundation model-based framework for training-free, marker-free, and calibration-free structural displacement measurement","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23173","citing_title":"One Identity, Many Roles: Multimodal Entity Coreference for Enhanced Video Situation Recognition","ref_index":74,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11170","citing_title":"Do Instance Priors Help Weakly Supervised Semantic Segmentation?","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07986","citing_title":"DP-DeGauss: Dynamic Probabilistic Gaussian Decomposition for Egocentric 4D Scene Reconstruction","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05415","citing_title":"Learning to Synergize Semantic and Geometric Priors for Limited-Data Wheat Disease Segmentation","ref_index":33,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/K2WUEM3BIKMJFGZJ7OHGZPZEYL","json":"https://pith.science/pith/K2WUEM3BIKMJFGZJ7OHGZPZEYL.json","graph_json":"https://pith.science/api/pith-number/K2WUEM3BIKMJFGZJ7OHGZPZEYL/graph.json","events_json":"https://pith.science/api/pith-number/K2WUEM3BIKMJFGZJ7OHGZPZEYL/events.json","paper":"https://pith.science/paper/K2WUEM3B"},"agent_actions":{"view_html":"https://pith.science/pith/K2WUEM3BIKMJFGZJ7OHGZPZEYL","download_json":"https://pith.science/pith/K2WUEM3BIKMJFGZJ7OHGZPZEYL.json","view_paper":"https://pith.science/paper/K2WUEM3B","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2304.11968&json=true","fetch_graph":"https://pith.science/api/pith-number/K2WUEM3BIKMJFGZJ7OHGZPZEYL/graph.json","fetch_events":"https://pith.science/api/pith-number/K2WUEM3BIKMJFGZJ7OHGZPZEYL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/K2WUEM3BIKMJFGZJ7OHGZPZEYL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/K2WUEM3BIKMJFGZJ7OHGZPZEYL/action/storage_attestation","attest_author":"https://pith.science/pith/K2WUEM3BIKMJFGZJ7OHGZPZEYL/action/author_attestation","sign_citation":"https://pith.science/pith/K2WUEM3BIKMJFGZJ7OHGZPZEYL/action/citation_signature","submit_replication":"https://pith.science/pith/K2WUEM3BIKMJFGZJ7OHGZPZEYL/action/replication_record"}},"created_at":"2026-07-05T06:05:16.620870+00:00","updated_at":"2026-07-05T06:05:16.620870+00:00"}