{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:CGLJUTYJ6NZ2V42F3IXLK7Z3FR","short_pith_number":"pith:CGLJUTYJ","schema_version":"1.0","canonical_sha256":"11969a4f09f373aaf345da2eb57f3b2c5a081c2097a1569e12122cadd9bde4c4","source":{"kind":"arxiv","id":"2408.12447","version":1},"attestation_state":"computed","paper":{"title":"The 2nd Solution for LSVOS Challenge RVOS Track: Spatial-temporal Refinement for Consistent Semantic Segmentation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Tuyen Tran","submitted_at":"2024-08-22T14:43:02Z","abstract_excerpt":"Referring Video Object Segmentation (RVOS) is a challenging task due to its requirement for temporal understanding. Due to the obstacle of computational complexity, many state-of-the-art models are trained on short time intervals. During testing, while these models can effectively process information over short time steps, they struggle to maintain consistent perception over prolonged time sequences, leading to inconsistencies in the resulting semantic segmentation masks. To address this challenge, we take a step further in this work by leveraging the tracking capabilities of the newly introdu"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.12447","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-08-22T14:43:02Z","cross_cats_sorted":[],"title_canon_sha256":"b7302a51e1813e4241339c7b3f5492f1887cd680e437f0933bb39aa797144b29","abstract_canon_sha256":"d9c0146518c00ab0298369d3adae45deb87a5cc7439fb17fed9197acd0ed7fe3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:58:10.208478Z","signature_b64":"t5eOVWU8qs9qgaLnH+mLSQw0nGqBmaVk9fdlV/1hWvu816Kt8GoFq8ITAnx4zrEXQrW5QVzrqAca0sHPx0n/Bg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"11969a4f09f373aaf345da2eb57f3b2c5a081c2097a1569e12122cadd9bde4c4","last_reissued_at":"2026-07-05T08:58:10.207965Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:58:10.207965Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The 2nd Solution for LSVOS Challenge RVOS Track: Spatial-temporal Refinement for Consistent Semantic Segmentation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Tuyen Tran","submitted_at":"2024-08-22T14:43:02Z","abstract_excerpt":"Referring Video Object Segmentation (RVOS) is a challenging task due to its requirement for temporal understanding. Due to the obstacle of computational complexity, many state-of-the-art models are trained on short time intervals. During testing, while these models can effectively process information over short time steps, they struggle to maintain consistent perception over prolonged time sequences, leading to inconsistencies in the resulting semantic segmentation masks. To address this challenge, we take a step further in this work by leveraging the tracking capabilities of the newly introdu"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.12447","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.12447/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.12447","created_at":"2026-07-05T08:58:10.208030+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.12447v1","created_at":"2026-07-05T08:58:10.208030+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.12447","created_at":"2026-07-05T08:58:10.208030+00:00"},{"alias_kind":"pith_short_12","alias_value":"CGLJUTYJ6NZ2","created_at":"2026-07-05T08:58:10.208030+00:00"},{"alias_kind":"pith_short_16","alias_value":"CGLJUTYJ6NZ2V42F","created_at":"2026-07-05T08:58:10.208030+00:00"},{"alias_kind":"pith_short_8","alias_value":"CGLJUTYJ","created_at":"2026-07-05T08:58:10.208030+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2410.04960","citing_title":"On Efficient Variants of Segment Anything Model: A Survey","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10950","citing_title":"Bootstrapping Video Semantic Segmentation Model via Distillation-assisted Test-Time Adaptation","ref_index":49,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CGLJUTYJ6NZ2V42F3IXLK7Z3FR","json":"https://pith.science/pith/CGLJUTYJ6NZ2V42F3IXLK7Z3FR.json","graph_json":"https://pith.science/api/pith-number/CGLJUTYJ6NZ2V42F3IXLK7Z3FR/graph.json","events_json":"https://pith.science/api/pith-number/CGLJUTYJ6NZ2V42F3IXLK7Z3FR/events.json","paper":"https://pith.science/paper/CGLJUTYJ"},"agent_actions":{"view_html":"https://pith.science/pith/CGLJUTYJ6NZ2V42F3IXLK7Z3FR","download_json":"https://pith.science/pith/CGLJUTYJ6NZ2V42F3IXLK7Z3FR.json","view_paper":"https://pith.science/paper/CGLJUTYJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.12447&json=true","fetch_graph":"https://pith.science/api/pith-number/CGLJUTYJ6NZ2V42F3IXLK7Z3FR/graph.json","fetch_events":"https://pith.science/api/pith-number/CGLJUTYJ6NZ2V42F3IXLK7Z3FR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CGLJUTYJ6NZ2V42F3IXLK7Z3FR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CGLJUTYJ6NZ2V42F3IXLK7Z3FR/action/storage_attestation","attest_author":"https://pith.science/pith/CGLJUTYJ6NZ2V42F3IXLK7Z3FR/action/author_attestation","sign_citation":"https://pith.science/pith/CGLJUTYJ6NZ2V42F3IXLK7Z3FR/action/citation_signature","submit_replication":"https://pith.science/pith/CGLJUTYJ6NZ2V42F3IXLK7Z3FR/action/replication_record"}},"created_at":"2026-07-05T08:58:10.208030+00:00","updated_at":"2026-07-05T08:58:10.208030+00:00"}