{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:AP7TAPTL4BAO2LT3GRSEQGGBUG","short_pith_number":"pith:AP7TAPTL","schema_version":"1.0","canonical_sha256":"03ff303e6be040ed2e7b34644818c1a19d0b5d07bd15e270a6c2ca8ada0bfd95","source":{"kind":"arxiv","id":"2410.08781","version":1},"attestation_state":"computed","paper":{"title":"VideoSAM: Open-World Video Segmentation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chongruo Wu, Jianxiong Gao, Pinxue Guo, Tianjun Xiao, Tong He, Wenqiang Zhang, Zheng Zhang, Zixu Zhao","submitted_at":"2024-10-11T12:56:32Z","abstract_excerpt":"Video segmentation is essential for advancing robotics and autonomous driving, particularly in open-world settings where continuous perception and object association across video frames are critical. While the Segment Anything Model (SAM) has excelled in static image segmentation, extending its capabilities to video segmentation poses significant challenges. We tackle two major hurdles: a) SAM's embedding limitations in associating objects across frames, and b) granularity inconsistencies in object segmentation. To this end, we introduce VideoSAM, an end-to-end framework designed to address th"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.08781","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-10-11T12:56:32Z","cross_cats_sorted":[],"title_canon_sha256":"1795d635bb7fe6a1cd837bee3cf82e91ea33d13160d49f0b1b009af0caa7a931","abstract_canon_sha256":"af436d29d24ea9da700dcaae57905b74085f9093fa337625a089273403cf8f52"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:19:17.572178Z","signature_b64":"jaxFT8dSirDChKPWso/v3LB6ASPDkgrN2Xng4ZWK+qVvtsMDn2lh54ooybqi8w14HOne0X4gbN++ovJwcK8sBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"03ff303e6be040ed2e7b34644818c1a19d0b5d07bd15e270a6c2ca8ada0bfd95","last_reissued_at":"2026-07-05T09:19:17.571768Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:19:17.571768Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VideoSAM: Open-World Video Segmentation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chongruo Wu, Jianxiong Gao, Pinxue Guo, Tianjun Xiao, Tong He, Wenqiang Zhang, Zheng Zhang, Zixu Zhao","submitted_at":"2024-10-11T12:56:32Z","abstract_excerpt":"Video segmentation is essential for advancing robotics and autonomous driving, particularly in open-world settings where continuous perception and object association across video frames are critical. While the Segment Anything Model (SAM) has excelled in static image segmentation, extending its capabilities to video segmentation poses significant challenges. We tackle two major hurdles: a) SAM's embedding limitations in associating objects across frames, and b) granularity inconsistencies in object segmentation. To this end, we introduce VideoSAM, an end-to-end framework designed to address th"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.08781","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.08781/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.08781","created_at":"2026-07-05T09:19:17.571821+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.08781v1","created_at":"2026-07-05T09:19:17.571821+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.08781","created_at":"2026-07-05T09:19:17.571821+00:00"},{"alias_kind":"pith_short_12","alias_value":"AP7TAPTL4BAO","created_at":"2026-07-05T09:19:17.571821+00:00"},{"alias_kind":"pith_short_16","alias_value":"AP7TAPTL4BAO2LT3","created_at":"2026-07-05T09:19:17.571821+00:00"},{"alias_kind":"pith_short_8","alias_value":"AP7TAPTL","created_at":"2026-07-05T09:19:17.571821+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.21809","citing_title":"VoCap: Video Object Captioning and Segmentation from Any Prompt","ref_index":23,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AP7TAPTL4BAO2LT3GRSEQGGBUG","json":"https://pith.science/pith/AP7TAPTL4BAO2LT3GRSEQGGBUG.json","graph_json":"https://pith.science/api/pith-number/AP7TAPTL4BAO2LT3GRSEQGGBUG/graph.json","events_json":"https://pith.science/api/pith-number/AP7TAPTL4BAO2LT3GRSEQGGBUG/events.json","paper":"https://pith.science/paper/AP7TAPTL"},"agent_actions":{"view_html":"https://pith.science/pith/AP7TAPTL4BAO2LT3GRSEQGGBUG","download_json":"https://pith.science/pith/AP7TAPTL4BAO2LT3GRSEQGGBUG.json","view_paper":"https://pith.science/paper/AP7TAPTL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.08781&json=true","fetch_graph":"https://pith.science/api/pith-number/AP7TAPTL4BAO2LT3GRSEQGGBUG/graph.json","fetch_events":"https://pith.science/api/pith-number/AP7TAPTL4BAO2LT3GRSEQGGBUG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AP7TAPTL4BAO2LT3GRSEQGGBUG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AP7TAPTL4BAO2LT3GRSEQGGBUG/action/storage_attestation","attest_author":"https://pith.science/pith/AP7TAPTL4BAO2LT3GRSEQGGBUG/action/author_attestation","sign_citation":"https://pith.science/pith/AP7TAPTL4BAO2LT3GRSEQGGBUG/action/citation_signature","submit_replication":"https://pith.science/pith/AP7TAPTL4BAO2LT3GRSEQGGBUG/action/replication_record"}},"created_at":"2026-07-05T09:19:17.571821+00:00","updated_at":"2026-07-05T09:19:17.571821+00:00"}