{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:6PYB7QHF6XA4DNX5VU3QCJC7MI","short_pith_number":"pith:6PYB7QHF","schema_version":"1.0","canonical_sha256":"f3f01fc0e5f5c1c1b6fdad3701245f620a080307864aed69a911368705c6176d","source":{"kind":"arxiv","id":"2509.01656","version":1},"attestation_state":"computed","paper":{"title":"Reinforced Visual Perception with Tools","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Dongping Chen, Mingyang Fu, Ranjay Krishna, Sinan Wang, Yao Wan, Zetong Zhou, Zhihan Hu, Zhou Zhao, Zixian Ma","submitted_at":"2025-09-01T17:57:49Z","abstract_excerpt":"Visual reasoning, a cornerstone of human intelligence, encompasses complex perceptual and logical processes essential for solving diverse visual problems. While advances in computer vision have produced powerful models for various perceptual tasks, leveraging these for general visual reasoning remains challenging. Prior work demonstrates that augmenting LLMs with vision models via supervised finetuning improves performance, but faces key limitations such as expensive data generation, reliance on careful data filtering, and poor generalization. To address these issues, we propose ReVPT to enhan"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2509.01656","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-09-01T17:57:49Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"92c0565b85098db6333372e853675a5900580b8f130f9aa975b8568e937a32cd","abstract_canon_sha256":"1660d75470e376dc3808bb5b7f3c1774b5eebb9a27fe6eb33418d7efc49b4a93"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:03:05.310677Z","signature_b64":"EjC1BUAW9tQALw5hx2NnabJMF9412y3Bu1Dgv3R5HxdXuQT/5quZizlrZznH9dvtGtQkVk913lNqtGNcLDzQCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f3f01fc0e5f5c1c1b6fdad3701245f620a080307864aed69a911368705c6176d","last_reissued_at":"2026-07-05T12:03:05.310088Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:03:05.310088Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Reinforced Visual Perception with Tools","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Dongping Chen, Mingyang Fu, Ranjay Krishna, Sinan Wang, Yao Wan, Zetong Zhou, Zhihan Hu, Zhou Zhao, Zixian Ma","submitted_at":"2025-09-01T17:57:49Z","abstract_excerpt":"Visual reasoning, a cornerstone of human intelligence, encompasses complex perceptual and logical processes essential for solving diverse visual problems. While advances in computer vision have produced powerful models for various perceptual tasks, leveraging these for general visual reasoning remains challenging. Prior work demonstrates that augmenting LLMs with vision models via supervised finetuning improves performance, but faces key limitations such as expensive data generation, reliance on careful data filtering, and poor generalization. To address these issues, we propose ReVPT to enhan"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2509.01656","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2509.01656/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2509.01656","created_at":"2026-07-05T12:03:05.310170+00:00"},{"alias_kind":"arxiv_version","alias_value":"2509.01656v1","created_at":"2026-07-05T12:03:05.310170+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2509.01656","created_at":"2026-07-05T12:03:05.310170+00:00"},{"alias_kind":"pith_short_12","alias_value":"6PYB7QHF6XA4","created_at":"2026-07-05T12:03:05.310170+00:00"},{"alias_kind":"pith_short_16","alias_value":"6PYB7QHF6XA4DNX5","created_at":"2026-07-05T12:03:05.310170+00:00"},{"alias_kind":"pith_short_8","alias_value":"6PYB7QHF","created_at":"2026-07-05T12:03:05.310170+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":15,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08434","citing_title":"DeltaV: Thinking with Visual State Updates in Unified Large Multimodal Models","ref_index":24,"is_internal_anchor":true},{"citing_arxiv_id":"2606.23557","citing_title":"Dense Reward for Multi-View 3D Reasoning with Global Maps and Local Views","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00881","citing_title":"OmniView-Space: Reinforcing Spatial Reasoning via Multi-Perspective Spatial Mapping","ref_index":71,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00384","citing_title":"VESTA: Visual Exploration with Statistical Tool Agents","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00096","citing_title":"Diversity Over Frequency: Rethinking Tool Use in Visual Chain-of-Thought Agents","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26520","citing_title":"InterSketch: An Interleaved Reasoning Model with Self-correcting Visual Sketch and Stepwise Reward","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23281","citing_title":"DepthAgent: Towards Better Universal Depth Estimation via Sample-wise Expert Selection","ref_index":83,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18641","citing_title":"Leveraging Latent Visual Reasoning in Silence","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19528","citing_title":"Towards Camera-Robust 3D Localization: Equation-Anchored Tool-Use for MLLMs","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2512.16300","citing_title":"Code-in-the-Loop Forensics: Agentic Tool Use for Image Forgery Detection","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12163","citing_title":"Self-Consistent Latent Reasoning: Long Latent Sequence Reasoning for Vision-Language Model","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12163","citing_title":"Self-Consistent Latent Reasoning: Long Latent Sequence Reasoning for Vision-Language Model","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19945","citing_title":"Visual Reasoning through Tool-supervised Reinforcement Learning","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12896","citing_title":"Don't Show Pixels, Show Cues: Unlocking Visual Tool Reasoning in Language Models via Perception Programs","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09712","citing_title":"LAST: Leveraging Tools as Hints to Enhance Spatial Reasoning for Multimodal Large Language Models","ref_index":51,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6PYB7QHF6XA4DNX5VU3QCJC7MI","json":"https://pith.science/pith/6PYB7QHF6XA4DNX5VU3QCJC7MI.json","graph_json":"https://pith.science/api/pith-number/6PYB7QHF6XA4DNX5VU3QCJC7MI/graph.json","events_json":"https://pith.science/api/pith-number/6PYB7QHF6XA4DNX5VU3QCJC7MI/events.json","paper":"https://pith.science/paper/6PYB7QHF"},"agent_actions":{"view_html":"https://pith.science/pith/6PYB7QHF6XA4DNX5VU3QCJC7MI","download_json":"https://pith.science/pith/6PYB7QHF6XA4DNX5VU3QCJC7MI.json","view_paper":"https://pith.science/paper/6PYB7QHF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2509.01656&json=true","fetch_graph":"https://pith.science/api/pith-number/6PYB7QHF6XA4DNX5VU3QCJC7MI/graph.json","fetch_events":"https://pith.science/api/pith-number/6PYB7QHF6XA4DNX5VU3QCJC7MI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6PYB7QHF6XA4DNX5VU3QCJC7MI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6PYB7QHF6XA4DNX5VU3QCJC7MI/action/storage_attestation","attest_author":"https://pith.science/pith/6PYB7QHF6XA4DNX5VU3QCJC7MI/action/author_attestation","sign_citation":"https://pith.science/pith/6PYB7QHF6XA4DNX5VU3QCJC7MI/action/citation_signature","submit_replication":"https://pith.science/pith/6PYB7QHF6XA4DNX5VU3QCJC7MI/action/replication_record"}},"created_at":"2026-07-05T12:03:05.310170+00:00","updated_at":"2026-07-05T12:03:05.310170+00:00"}