{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:K7JO3QHNFMRMO7UJI5RRWLBK36","short_pith_number":"pith:K7JO3QHN","schema_version":"1.0","canonical_sha256":"57d2edc0ed2b22c77e8947631b2c2adf9e230f2438ba05d6a6427f4968a5f938","source":{"kind":"arxiv","id":"2408.16431","version":1},"attestation_state":"computed","paper":{"title":"Discriminative Spatial-Semantic VOS Solution: 1st Place Solution for 6th LSVOS","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Deshui Miao, Ming-Hsuan Yang, Xin Li, Yameng Gu, Yaowei Wang, Zhenyu He","submitted_at":"2024-08-29T10:47:17Z","abstract_excerpt":"Video object segmentation (VOS) is a crucial task in computer vision, but current VOS methods struggle with complex scenes and prolonged object motions. To address these challenges, the MOSE dataset aims to enhance object recognition and differentiation in complex environments, while the LVOS dataset focuses on segmenting objects exhibiting long-term, intricate movements. This report introduces a discriminative spatial-temporal VOS model that utilizes discriminative object features as query representations. The semantic understanding of spatial-semantic modules enables it to recognize object p"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.16431","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CV","submitted_at":"2024-08-29T10:47:17Z","cross_cats_sorted":[],"title_canon_sha256":"8f15b1641284dfac82e09b5dec4d6eb48573fa0ce4aa3996c24bb921a838b3eb","abstract_canon_sha256":"fd3e33b93f924568190d688e42ce6e982c81d56a431b08594bd86de30085613c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:00:41.279442Z","signature_b64":"gg1luJ/CMkfNWABOyho7t930+UZo7D6u1NJPcYCTctU1fJQyTj/26om5VZFBVdbME3q7zfm+BnRZxDmkcN0bBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"57d2edc0ed2b22c77e8947631b2c2adf9e230f2438ba05d6a6427f4968a5f938","last_reissued_at":"2026-07-05T09:00:41.279026Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:00:41.279026Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Discriminative Spatial-Semantic VOS Solution: 1st Place Solution for 6th LSVOS","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Deshui Miao, Ming-Hsuan Yang, Xin Li, Yameng Gu, Yaowei Wang, Zhenyu He","submitted_at":"2024-08-29T10:47:17Z","abstract_excerpt":"Video object segmentation (VOS) is a crucial task in computer vision, but current VOS methods struggle with complex scenes and prolonged object motions. To address these challenges, the MOSE dataset aims to enhance object recognition and differentiation in complex environments, while the LVOS dataset focuses on segmenting objects exhibiting long-term, intricate movements. This report introduces a discriminative spatial-temporal VOS model that utilizes discriminative object features as query representations. The semantic understanding of spatial-semantic modules enables it to recognize object p"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.16431","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.16431/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.16431","created_at":"2026-07-05T09:00:41.279093+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.16431v1","created_at":"2026-07-05T09:00:41.279093+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.16431","created_at":"2026-07-05T09:00:41.279093+00:00"},{"alias_kind":"pith_short_12","alias_value":"K7JO3QHNFMRM","created_at":"2026-07-05T09:00:41.279093+00:00"},{"alias_kind":"pith_short_16","alias_value":"K7JO3QHNFMRMO7UJ","created_at":"2026-07-05T09:00:41.279093+00:00"},{"alias_kind":"pith_short_8","alias_value":"K7JO3QHN","created_at":"2026-07-05T09:00:41.279093+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07230","citing_title":"`Attention-Guided Cross-Temporal Clustering for Self-Supervised Video Object Segmentation","ref_index":1,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/K7JO3QHNFMRMO7UJI5RRWLBK36","json":"https://pith.science/pith/K7JO3QHNFMRMO7UJI5RRWLBK36.json","graph_json":"https://pith.science/api/pith-number/K7JO3QHNFMRMO7UJI5RRWLBK36/graph.json","events_json":"https://pith.science/api/pith-number/K7JO3QHNFMRMO7UJI5RRWLBK36/events.json","paper":"https://pith.science/paper/K7JO3QHN"},"agent_actions":{"view_html":"https://pith.science/pith/K7JO3QHNFMRMO7UJI5RRWLBK36","download_json":"https://pith.science/pith/K7JO3QHNFMRMO7UJI5RRWLBK36.json","view_paper":"https://pith.science/paper/K7JO3QHN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.16431&json=true","fetch_graph":"https://pith.science/api/pith-number/K7JO3QHNFMRMO7UJI5RRWLBK36/graph.json","fetch_events":"https://pith.science/api/pith-number/K7JO3QHNFMRMO7UJI5RRWLBK36/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/K7JO3QHNFMRMO7UJI5RRWLBK36/action/timestamp_anchor","attest_storage":"https://pith.science/pith/K7JO3QHNFMRMO7UJI5RRWLBK36/action/storage_attestation","attest_author":"https://pith.science/pith/K7JO3QHNFMRMO7UJI5RRWLBK36/action/author_attestation","sign_citation":"https://pith.science/pith/K7JO3QHNFMRMO7UJI5RRWLBK36/action/citation_signature","submit_replication":"https://pith.science/pith/K7JO3QHNFMRMO7UJI5RRWLBK36/action/replication_record"}},"created_at":"2026-07-05T09:00:41.279093+00:00","updated_at":"2026-07-05T09:00:41.279093+00:00"}