{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:OE7B57UL3SWUDXCHHMFXNCPUKP","short_pith_number":"pith:OE7B57UL","schema_version":"1.0","canonical_sha256":"713e1efe8bdcad41dc473b0b7689f453e5d6a27610ea45cb38b32a5bafc4f7e1","source":{"kind":"arxiv","id":"2509.05751","version":1},"attestation_state":"computed","paper":{"title":"Unleashing Hierarchical Reasoning: An LLM-Driven Framework for Training-Free Referring Video Object Segmentation","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Bingrui Zhao, Deyin Liu, Jialie Shen, Lin Yuanbo Wu, Lu Zhang, Ruyi He, Xiangtian Fan, Ximing Li","submitted_at":"2025-09-06T15:46:23Z","abstract_excerpt":"Referring Video Object Segmentation (RVOS) aims to segment an object of interest throughout a video based on a language description. The prominent challenge lies in aligning static text with dynamic visual content, particularly when objects exhibiting similar appearances with inconsistent motion and poses. However, current methods often rely on a holistic visual-language fusion that struggles with complex, compositional descriptions. In this paper, we propose \\textbf{PARSE-VOS}, a novel, training-free framework powered by Large Language Models (LLMs), for a hierarchical, coarse-to-fine reasoni"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2509.05751","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/publicdomain/zero/1.0/","primary_cat":"cs.CV","submitted_at":"2025-09-06T15:46:23Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"c291854a454ab14f79529f1f429c9e04845b575f0c1ab0fbc8beb8572056d3c1","abstract_canon_sha256":"58af6b8a3a0af5944981957c308e88972d3eac2a0ac8f580544f9801fb4ef67c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:06:23.517756Z","signature_b64":"fjTK5WyK0ZvgktDbs+kX9q0ohWUFTJOAbJRxdrve2PFzfKQZZ+R5+UbY4EY45rWHSlrqNDp8BzEtLYjk7IYyAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"713e1efe8bdcad41dc473b0b7689f453e5d6a27610ea45cb38b32a5bafc4f7e1","last_reissued_at":"2026-07-05T12:06:23.517256Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:06:23.517256Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Unleashing Hierarchical Reasoning: An LLM-Driven Framework for Training-Free Referring Video Object Segmentation","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Bingrui Zhao, Deyin Liu, Jialie Shen, Lin Yuanbo Wu, Lu Zhang, Ruyi He, Xiangtian Fan, Ximing Li","submitted_at":"2025-09-06T15:46:23Z","abstract_excerpt":"Referring Video Object Segmentation (RVOS) aims to segment an object of interest throughout a video based on a language description. The prominent challenge lies in aligning static text with dynamic visual content, particularly when objects exhibiting similar appearances with inconsistent motion and poses. However, current methods often rely on a holistic visual-language fusion that struggles with complex, compositional descriptions. In this paper, we propose \\textbf{PARSE-VOS}, a novel, training-free framework powered by Large Language Models (LLMs), for a hierarchical, coarse-to-fine reasoni"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2509.05751","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2509.05751/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2509.05751","created_at":"2026-07-05T12:06:23.517309+00:00"},{"alias_kind":"arxiv_version","alias_value":"2509.05751v1","created_at":"2026-07-05T12:06:23.517309+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2509.05751","created_at":"2026-07-05T12:06:23.517309+00:00"},{"alias_kind":"pith_short_12","alias_value":"OE7B57UL3SWU","created_at":"2026-07-05T12:06:23.517309+00:00"},{"alias_kind":"pith_short_16","alias_value":"OE7B57UL3SWUDXCH","created_at":"2026-07-05T12:06:23.517309+00:00"},{"alias_kind":"pith_short_8","alias_value":"OE7B57UL","created_at":"2026-07-05T12:06:23.517309+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OE7B57UL3SWUDXCHHMFXNCPUKP","json":"https://pith.science/pith/OE7B57UL3SWUDXCHHMFXNCPUKP.json","graph_json":"https://pith.science/api/pith-number/OE7B57UL3SWUDXCHHMFXNCPUKP/graph.json","events_json":"https://pith.science/api/pith-number/OE7B57UL3SWUDXCHHMFXNCPUKP/events.json","paper":"https://pith.science/paper/OE7B57UL"},"agent_actions":{"view_html":"https://pith.science/pith/OE7B57UL3SWUDXCHHMFXNCPUKP","download_json":"https://pith.science/pith/OE7B57UL3SWUDXCHHMFXNCPUKP.json","view_paper":"https://pith.science/paper/OE7B57UL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2509.05751&json=true","fetch_graph":"https://pith.science/api/pith-number/OE7B57UL3SWUDXCHHMFXNCPUKP/graph.json","fetch_events":"https://pith.science/api/pith-number/OE7B57UL3SWUDXCHHMFXNCPUKP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OE7B57UL3SWUDXCHHMFXNCPUKP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OE7B57UL3SWUDXCHHMFXNCPUKP/action/storage_attestation","attest_author":"https://pith.science/pith/OE7B57UL3SWUDXCHHMFXNCPUKP/action/author_attestation","sign_citation":"https://pith.science/pith/OE7B57UL3SWUDXCHHMFXNCPUKP/action/citation_signature","submit_replication":"https://pith.science/pith/OE7B57UL3SWUDXCHHMFXNCPUKP/action/replication_record"}},"created_at":"2026-07-05T12:06:23.517309+00:00","updated_at":"2026-07-05T12:06:23.517309+00:00"}