{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:B7EYZACT4UO6P7FWNWZWELZV5J","short_pith_number":"pith:B7EYZACT","schema_version":"1.0","canonical_sha256":"0fc98c8053e51de7fcb66db3622f35ea42519dc9976523cb0d8c62df1db91081","source":{"kind":"arxiv","id":"2506.21233","version":2},"attestation_state":"computed","paper":{"title":"ReME: A Data-Centric Framework for Training-Free Open-Vocabulary Segmentation","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Kwan-Liu Ma, Xiwei Xuan, Ziquan Deng","submitted_at":"2025-06-26T13:22:03Z","abstract_excerpt":"Training-free open-vocabulary semantic segmentation (OVS) aims to segment images given a set of arbitrary textual categories without costly model fine-tuning. Existing solutions often explore attention mechanisms of pre-trained models, such as CLIP, or generate synthetic data and design complex retrieval processes to perform OVS. However, their performance is limited by the capability of reliant models or the suboptimal quality of reference sets. In this work, we investigate the largely overlooked data quality problem for this challenging dense scene understanding task, and identify that a hig"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.21233","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2025-06-26T13:22:03Z","cross_cats_sorted":[],"title_canon_sha256":"2edd3ace576bf1e66f5d87c37f8f6c331d6a7d230aa148a58ee022fb8ceb5d3f","abstract_canon_sha256":"f6937f2a395cd3faa5127cc25ab664b9ae5592fd68ae5b51235982da0db3eb7e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:28:18.251930Z","signature_b64":"xMxRtHr+/d4IkBarbVbo6dy9aDAQyvEwUCqOSu24KmFyG+KLXiWXN7RUcItU8DtXnwr+yA0BImLRTjANpipGAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0fc98c8053e51de7fcb66db3622f35ea42519dc9976523cb0d8c62df1db91081","last_reissued_at":"2026-07-05T11:28:18.251436Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:28:18.251436Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ReME: A Data-Centric Framework for Training-Free Open-Vocabulary Segmentation","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Kwan-Liu Ma, Xiwei Xuan, Ziquan Deng","submitted_at":"2025-06-26T13:22:03Z","abstract_excerpt":"Training-free open-vocabulary semantic segmentation (OVS) aims to segment images given a set of arbitrary textual categories without costly model fine-tuning. Existing solutions often explore attention mechanisms of pre-trained models, such as CLIP, or generate synthetic data and design complex retrieval processes to perform OVS. However, their performance is limited by the capability of reliant models or the suboptimal quality of reference sets. In this work, we investigate the largely overlooked data quality problem for this challenging dense scene understanding task, and identify that a hig"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.21233","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.21233/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.21233","created_at":"2026-07-05T11:28:18.251496+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.21233v2","created_at":"2026-07-05T11:28:18.251496+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.21233","created_at":"2026-07-05T11:28:18.251496+00:00"},{"alias_kind":"pith_short_12","alias_value":"B7EYZACT4UO6","created_at":"2026-07-05T11:28:18.251496+00:00"},{"alias_kind":"pith_short_16","alias_value":"B7EYZACT4UO6P7FW","created_at":"2026-07-05T11:28:18.251496+00:00"},{"alias_kind":"pith_short_8","alias_value":"B7EYZACT","created_at":"2026-07-05T11:28:18.251496+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.09090","citing_title":"Investigating Anisotropy in Visual Grounding under Controlled Counterfactual Perturbations","ref_index":38,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/B7EYZACT4UO6P7FWNWZWELZV5J","json":"https://pith.science/pith/B7EYZACT4UO6P7FWNWZWELZV5J.json","graph_json":"https://pith.science/api/pith-number/B7EYZACT4UO6P7FWNWZWELZV5J/graph.json","events_json":"https://pith.science/api/pith-number/B7EYZACT4UO6P7FWNWZWELZV5J/events.json","paper":"https://pith.science/paper/B7EYZACT"},"agent_actions":{"view_html":"https://pith.science/pith/B7EYZACT4UO6P7FWNWZWELZV5J","download_json":"https://pith.science/pith/B7EYZACT4UO6P7FWNWZWELZV5J.json","view_paper":"https://pith.science/paper/B7EYZACT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.21233&json=true","fetch_graph":"https://pith.science/api/pith-number/B7EYZACT4UO6P7FWNWZWELZV5J/graph.json","fetch_events":"https://pith.science/api/pith-number/B7EYZACT4UO6P7FWNWZWELZV5J/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/B7EYZACT4UO6P7FWNWZWELZV5J/action/timestamp_anchor","attest_storage":"https://pith.science/pith/B7EYZACT4UO6P7FWNWZWELZV5J/action/storage_attestation","attest_author":"https://pith.science/pith/B7EYZACT4UO6P7FWNWZWELZV5J/action/author_attestation","sign_citation":"https://pith.science/pith/B7EYZACT4UO6P7FWNWZWELZV5J/action/citation_signature","submit_replication":"https://pith.science/pith/B7EYZACT4UO6P7FWNWZWELZV5J/action/replication_record"}},"created_at":"2026-07-05T11:28:18.251496+00:00","updated_at":"2026-07-05T11:28:18.251496+00:00"}