{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:YXO427HCB2HEDY7SHNNGOXMPZC","short_pith_number":"pith:YXO427HC","schema_version":"1.0","canonical_sha256":"c5ddcd7ce20e8e41e3f23b5a675d8fc88f1fccf4b5f4fc80a4ffc236050fdb8e","source":{"kind":"arxiv","id":"2505.11838","version":1},"attestation_state":"computed","paper":{"title":"RVTBench: A Benchmark for Visual Reasoning Tasks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chenjia Li, Chenxiao Fan, Mathias Unberath, Yiqing Shen","submitted_at":"2025-05-17T04:58:09Z","abstract_excerpt":"Visual reasoning, the capability to interpret visual input in response to implicit text query through multi-step reasoning, remains a challenge for deep learning models due to the lack of relevant benchmarks. Previous work in visual reasoning has primarily focused on reasoning segmentation, where models aim to segment objects based on implicit text queries. This paper introduces reasoning visual tasks (RVTs), a unified formulation that extends beyond traditional video reasoning segmentation to a diverse family of visual language reasoning problems, which can therefore accommodate multiple outp"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.11838","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-05-17T04:58:09Z","cross_cats_sorted":[],"title_canon_sha256":"1cfa28d62f9a670d920da7a7083266113260d551b7352c545240e5c503d94946","abstract_canon_sha256":"dc96de86ea15211c3d3c5a704081372b3974ac5220b8f89bbe66199b805fd24b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:04:27.389460Z","signature_b64":"cnTSLBFsNHk8sFVFmjG0ti7T7wwtCU9lvcBTZMzl1t4F+BfAGeh7X+xM4fkkjlLvw3WfG2nQxUI/jX4b8YzeCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c5ddcd7ce20e8e41e3f23b5a675d8fc88f1fccf4b5f4fc80a4ffc236050fdb8e","last_reissued_at":"2026-07-05T11:04:27.388891Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:04:27.388891Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"RVTBench: A Benchmark for Visual Reasoning Tasks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chenjia Li, Chenxiao Fan, Mathias Unberath, Yiqing Shen","submitted_at":"2025-05-17T04:58:09Z","abstract_excerpt":"Visual reasoning, the capability to interpret visual input in response to implicit text query through multi-step reasoning, remains a challenge for deep learning models due to the lack of relevant benchmarks. Previous work in visual reasoning has primarily focused on reasoning segmentation, where models aim to segment objects based on implicit text queries. This paper introduces reasoning visual tasks (RVTs), a unified formulation that extends beyond traditional video reasoning segmentation to a diverse family of visual language reasoning problems, which can therefore accommodate multiple outp"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.11838","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.11838/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.11838","created_at":"2026-07-05T11:04:27.388957+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.11838v1","created_at":"2026-07-05T11:04:27.388957+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.11838","created_at":"2026-07-05T11:04:27.388957+00:00"},{"alias_kind":"pith_short_12","alias_value":"YXO427HCB2HE","created_at":"2026-07-05T11:04:27.388957+00:00"},{"alias_kind":"pith_short_16","alias_value":"YXO427HCB2HEDY7S","created_at":"2026-07-05T11:04:27.388957+00:00"},{"alias_kind":"pith_short_8","alias_value":"YXO427HC","created_at":"2026-07-05T11:04:27.388957+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.16718","citing_title":"Temporally-Constrained Video Reasoning Segmentation and Automated Benchmark Construction","ref_index":9,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YXO427HCB2HEDY7SHNNGOXMPZC","json":"https://pith.science/pith/YXO427HCB2HEDY7SHNNGOXMPZC.json","graph_json":"https://pith.science/api/pith-number/YXO427HCB2HEDY7SHNNGOXMPZC/graph.json","events_json":"https://pith.science/api/pith-number/YXO427HCB2HEDY7SHNNGOXMPZC/events.json","paper":"https://pith.science/paper/YXO427HC"},"agent_actions":{"view_html":"https://pith.science/pith/YXO427HCB2HEDY7SHNNGOXMPZC","download_json":"https://pith.science/pith/YXO427HCB2HEDY7SHNNGOXMPZC.json","view_paper":"https://pith.science/paper/YXO427HC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.11838&json=true","fetch_graph":"https://pith.science/api/pith-number/YXO427HCB2HEDY7SHNNGOXMPZC/graph.json","fetch_events":"https://pith.science/api/pith-number/YXO427HCB2HEDY7SHNNGOXMPZC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YXO427HCB2HEDY7SHNNGOXMPZC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YXO427HCB2HEDY7SHNNGOXMPZC/action/storage_attestation","attest_author":"https://pith.science/pith/YXO427HCB2HEDY7SHNNGOXMPZC/action/author_attestation","sign_citation":"https://pith.science/pith/YXO427HCB2HEDY7SHNNGOXMPZC/action/citation_signature","submit_replication":"https://pith.science/pith/YXO427HCB2HEDY7SHNNGOXMPZC/action/replication_record"}},"created_at":"2026-07-05T11:04:27.388957+00:00","updated_at":"2026-07-05T11:04:27.388957+00:00"}