{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:AFURICIKY5WHKJFRFROELDVVUZ","short_pith_number":"pith:AFURICIK","schema_version":"1.0","canonical_sha256":"016914090ac76c7524b12c5c458eb5a678939811a8650910948d460ea5edc707","source":{"kind":"arxiv","id":"2503.07523","version":2},"attestation_state":"computed","paper":{"title":"VisRL: Intention-Driven Visual Perception via Reinforced Reasoning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Dongsheng Li, Xufang Luo, Zhangquan Chen","submitted_at":"2025-03-10T16:49:35Z","abstract_excerpt":"Visual understanding is inherently intention-driven - humans selectively focus on different regions of a scene based on their goals. Recent advances in large multimodal models (LMMs) enable flexible expression of such intentions through natural language, allowing queries to guide visual reasoning processes. Frameworks like Visual Chain-of-Thought have demonstrated the benefit of incorporating explicit reasoning steps, where the model predicts a focus region before answering a query. However, existing approaches rely heavily on supervised training with annotated intermediate bounding boxes, whi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.07523","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-03-10T16:49:35Z","cross_cats_sorted":[],"title_canon_sha256":"3c259dc7eea252bce1bc596c7e07886980fb2ff94f7e9ed2beee8225e76d881e","abstract_canon_sha256":"861c55359fb42db4fda91b5edaf9d6ddb454e08a9896c8059980f98f7648ff8b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:42:15.594663Z","signature_b64":"EwDIsl6ihG+edokEIaFDxeEusr82fJf0ju+OY61b7dbzZ5RG+lzym/ezmv4HCg30ecauHYTwaJAIRPUd7CrGAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"016914090ac76c7524b12c5c458eb5a678939811a8650910948d460ea5edc707","last_reissued_at":"2026-07-05T10:42:15.594194Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:42:15.594194Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VisRL: Intention-Driven Visual Perception via Reinforced Reasoning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Dongsheng Li, Xufang Luo, Zhangquan Chen","submitted_at":"2025-03-10T16:49:35Z","abstract_excerpt":"Visual understanding is inherently intention-driven - humans selectively focus on different regions of a scene based on their goals. Recent advances in large multimodal models (LMMs) enable flexible expression of such intentions through natural language, allowing queries to guide visual reasoning processes. Frameworks like Visual Chain-of-Thought have demonstrated the benefit of incorporating explicit reasoning steps, where the model predicts a focus region before answering a query. However, existing approaches rely heavily on supervised training with annotated intermediate bounding boxes, whi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.07523","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.07523/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.07523","created_at":"2026-07-05T10:42:15.594251+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.07523v2","created_at":"2026-07-05T10:42:15.594251+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.07523","created_at":"2026-07-05T10:42:15.594251+00:00"},{"alias_kind":"pith_short_12","alias_value":"AFURICIKY5WH","created_at":"2026-07-05T10:42:15.594251+00:00"},{"alias_kind":"pith_short_16","alias_value":"AFURICIKY5WHKJFR","created_at":"2026-07-05T10:42:15.594251+00:00"},{"alias_kind":"pith_short_8","alias_value":"AFURICIK","created_at":"2026-07-05T10:42:15.594251+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26196","citing_title":"From Structure to Synergy: A Survey of Vision-Language Perception Paradigm Evolution in Multimodal Large Language Models","ref_index":203,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22873","citing_title":"SingGuard: A Policy-Adaptive Multimodal LLM Guardrail with Dynamic Reasoning","ref_index":264,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22873","citing_title":"SingGuard: A Policy-Adaptive Multimodal LLM Guardrail with Dynamic Reasoning","ref_index":263,"is_internal_anchor":false},{"citing_arxiv_id":"2509.08827","citing_title":"A Survey of Reinforcement Learning for Large Reasoning Models","ref_index":80,"is_internal_anchor":false},{"citing_arxiv_id":"2511.19820","citing_title":"CropVLM: Learning to Zoom for Fine-Grained Vision-Language Perception","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2511.20162","citing_title":"Action Without Interaction: Probing the Physical Foundations of Video LMMs via Contact-Release Detection","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2512.03963","citing_title":"TempR1: Improving Temporal Understanding of MLLMs via Temporal-Aware Multi-Task Reinforcement Learning","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2503.21620","citing_title":"UI-R1: Enhancing Efficient Action Prediction of GUI Agents by Reinforcement Learning","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2503.12605","citing_title":"Multimodal Chain-of-Thought Reasoning: A Comprehensive Survey","ref_index":249,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08378","citing_title":"Reinforcement Learning for Scalable and Trustworthy Intelligent Systems","ref_index":162,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09330","citing_title":"VAG: Dual-Stream Video-Action Generation for Embodied Data Synthesis","ref_index":14,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AFURICIKY5WHKJFRFROELDVVUZ","json":"https://pith.science/pith/AFURICIKY5WHKJFRFROELDVVUZ.json","graph_json":"https://pith.science/api/pith-number/AFURICIKY5WHKJFRFROELDVVUZ/graph.json","events_json":"https://pith.science/api/pith-number/AFURICIKY5WHKJFRFROELDVVUZ/events.json","paper":"https://pith.science/paper/AFURICIK"},"agent_actions":{"view_html":"https://pith.science/pith/AFURICIKY5WHKJFRFROELDVVUZ","download_json":"https://pith.science/pith/AFURICIKY5WHKJFRFROELDVVUZ.json","view_paper":"https://pith.science/paper/AFURICIK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.07523&json=true","fetch_graph":"https://pith.science/api/pith-number/AFURICIKY5WHKJFRFROELDVVUZ/graph.json","fetch_events":"https://pith.science/api/pith-number/AFURICIKY5WHKJFRFROELDVVUZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AFURICIKY5WHKJFRFROELDVVUZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AFURICIKY5WHKJFRFROELDVVUZ/action/storage_attestation","attest_author":"https://pith.science/pith/AFURICIKY5WHKJFRFROELDVVUZ/action/author_attestation","sign_citation":"https://pith.science/pith/AFURICIKY5WHKJFRFROELDVVUZ/action/citation_signature","submit_replication":"https://pith.science/pith/AFURICIKY5WHKJFRFROELDVVUZ/action/replication_record"}},"created_at":"2026-07-05T10:42:15.594251+00:00","updated_at":"2026-07-05T10:42:15.594251+00:00"}