{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:B7W47V7SSEXJEFFULYKY3RDBRD","short_pith_number":"pith:B7W47V7S","schema_version":"1.0","canonical_sha256":"0fedcfd7f2912e9214b45e158dc46188ed1c5d66cd54b28eb772227a490b95e8","source":{"kind":"arxiv","id":"2407.07760","version":2},"attestation_state":"computed","paper":{"title":"Learning Spatial-Semantic Features for Robust Video Object Segmentation","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Deshui Miao, Huchuan Lu, Ming-Hsuan Yang, Xin Li, Yaowei Wang, Zhenyu He","submitted_at":"2024-07-10T15:36:00Z","abstract_excerpt":"Tracking and segmenting multiple similar objects with distinct or complex parts in long-term videos is particularly challenging due to the ambiguity in identifying target components and the confusion caused by occlusion, background clutter, and changes in appearance or environment over time. In this paper, we propose a robust video object segmentation framework that learns spatial-semantic features and discriminative object queries to address the above issues. Specifically, we construct a spatial-semantic block comprising a semantic embedding component and a spatial dependency modeling part fo"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.07760","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2024-07-10T15:36:00Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"009ecd56ff13c31625b4c64915dd622c258aed04c523a4de835c77dece5b0311","abstract_canon_sha256":"a3059824eee8978a959e729fd07d9aa6a42a151f1bc4430ec7ff3a014d3840f7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:45:10.020929Z","signature_b64":"q6GqSxOVeMBRCvpeiYbqxYCXPL3kDZtwinrhoG7N3WVFC/un5jmJnX0hBgPBoYqB9npx7dr14CMZgMepL3CLDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0fedcfd7f2912e9214b45e158dc46188ed1c5d66cd54b28eb772227a490b95e8","last_reissued_at":"2026-07-05T10:45:10.020450Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:45:10.020450Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Learning Spatial-Semantic Features for Robust Video Object Segmentation","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Deshui Miao, Huchuan Lu, Ming-Hsuan Yang, Xin Li, Yaowei Wang, Zhenyu He","submitted_at":"2024-07-10T15:36:00Z","abstract_excerpt":"Tracking and segmenting multiple similar objects with distinct or complex parts in long-term videos is particularly challenging due to the ambiguity in identifying target components and the confusion caused by occlusion, background clutter, and changes in appearance or environment over time. In this paper, we propose a robust video object segmentation framework that learns spatial-semantic features and discriminative object queries to address the above issues. Specifically, we construct a spatial-semantic block comprising a semantic embedding component and a spatial dependency modeling part fo"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.07760","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.07760/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.07760","created_at":"2026-07-05T10:45:10.020507+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.07760v2","created_at":"2026-07-05T10:45:10.020507+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.07760","created_at":"2026-07-05T10:45:10.020507+00:00"},{"alias_kind":"pith_short_12","alias_value":"B7W47V7SSEXJ","created_at":"2026-07-05T10:45:10.020507+00:00"},{"alias_kind":"pith_short_16","alias_value":"B7W47V7SSEXJEFFU","created_at":"2026-07-05T10:45:10.020507+00:00"},{"alias_kind":"pith_short_8","alias_value":"B7W47V7S","created_at":"2026-07-05T10:45:10.020507+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2411.17576","citing_title":"A Distractor-Aware Memory for Visual Object Tracking with SAM2","ref_index":26,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/B7W47V7SSEXJEFFULYKY3RDBRD","json":"https://pith.science/pith/B7W47V7SSEXJEFFULYKY3RDBRD.json","graph_json":"https://pith.science/api/pith-number/B7W47V7SSEXJEFFULYKY3RDBRD/graph.json","events_json":"https://pith.science/api/pith-number/B7W47V7SSEXJEFFULYKY3RDBRD/events.json","paper":"https://pith.science/paper/B7W47V7S"},"agent_actions":{"view_html":"https://pith.science/pith/B7W47V7SSEXJEFFULYKY3RDBRD","download_json":"https://pith.science/pith/B7W47V7SSEXJEFFULYKY3RDBRD.json","view_paper":"https://pith.science/paper/B7W47V7S","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.07760&json=true","fetch_graph":"https://pith.science/api/pith-number/B7W47V7SSEXJEFFULYKY3RDBRD/graph.json","fetch_events":"https://pith.science/api/pith-number/B7W47V7SSEXJEFFULYKY3RDBRD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/B7W47V7SSEXJEFFULYKY3RDBRD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/B7W47V7SSEXJEFFULYKY3RDBRD/action/storage_attestation","attest_author":"https://pith.science/pith/B7W47V7SSEXJEFFULYKY3RDBRD/action/author_attestation","sign_citation":"https://pith.science/pith/B7W47V7SSEXJEFFULYKY3RDBRD/action/citation_signature","submit_replication":"https://pith.science/pith/B7W47V7SSEXJEFFULYKY3RDBRD/action/replication_record"}},"created_at":"2026-07-05T10:45:10.020507+00:00","updated_at":"2026-07-05T10:45:10.020507+00:00"}